249 lines
8.7 KiB
Python
249 lines
8.7 KiB
Python
"""Voice actions — the editing decisions produced from a voice timeline.
|
|
|
|
This is the contract between *deciding* and *applying*. Whoever makes the
|
|
editorial call — the deterministic rules engine, or a model reading the
|
|
voice timeline JSON — emits the same list of actions, and one applier turns
|
|
it into FCPXML. Nothing that produces actions ever touches XML.
|
|
|
|
Every action's ``start``/``end`` is in **original source seconds**, matching
|
|
the voice timeline. That matters: cuts shift everything after them, so if
|
|
decisions were expressed in post-cut time they would silently land in the
|
|
wrong place the moment a cut was added. Keeping one origin and resolving the
|
|
shift at apply time (:func:`shift_after_cuts`) removes that whole class of bug.
|
|
|
|
Actions arriving from a model are untrusted input: :func:`parse_actions`
|
|
validates and reports what it rejected rather than raising, so one malformed
|
|
row never discards a whole edit.
|
|
"""
|
|
|
|
from dataclasses import dataclass, field
|
|
from typing import Any, List, Optional, Sequence, Tuple
|
|
|
|
# What an action can ask for. Deliberately small — each maps onto one
|
|
# existing writer capability, so no new XML knowledge lives here.
|
|
ACTION_KINDS = ("cut", "zoom", "text", "marker")
|
|
|
|
# Bounds for a zoom's scale factor. Below 1.0 is a pull-back, not a punch-in;
|
|
# above 3x the image falls apart on any normal footage.
|
|
MIN_ZOOM_SCALE = 1.0
|
|
MAX_ZOOM_SCALE = 3.0
|
|
|
|
MAX_TEXT_LENGTH = 120
|
|
|
|
|
|
@dataclass
|
|
class VoiceAction:
|
|
"""One editing decision, in original source time."""
|
|
|
|
kind: str
|
|
start: float
|
|
end: float
|
|
params: dict = field(default_factory=dict)
|
|
reason: str = ""
|
|
speaker: str = ""
|
|
|
|
@property
|
|
def duration(self) -> float:
|
|
return max(0.0, self.end - self.start)
|
|
|
|
def as_dict(self) -> dict:
|
|
return {
|
|
"kind": self.kind,
|
|
"start": round(self.start, 3),
|
|
"end": round(self.end, 3),
|
|
"params": self.params,
|
|
"reason": self.reason,
|
|
"speaker": self.speaker,
|
|
}
|
|
|
|
|
|
def _validate_one(raw: Any, index: int) -> Tuple[Optional[VoiceAction], str]:
|
|
"""Turn one raw row into a VoiceAction, or explain why it can't be."""
|
|
where = f"action[{index}]"
|
|
if not isinstance(raw, dict):
|
|
return None, f"{where}: expected an object, got {type(raw).__name__}"
|
|
|
|
kind = str(raw.get("kind", "")).strip().lower()
|
|
if kind not in ACTION_KINDS:
|
|
return None, f"{where}: unknown kind {raw.get('kind')!r} (expected one of {', '.join(ACTION_KINDS)})"
|
|
|
|
try:
|
|
start = float(raw.get("start"))
|
|
end = float(raw.get("end"))
|
|
except (TypeError, ValueError):
|
|
return None, f"{where}: start/end must be numbers (seconds)"
|
|
|
|
if start < 0:
|
|
return None, f"{where}: start is negative ({start})"
|
|
if end <= start:
|
|
return None, f"{where}: end ({end}) must be after start ({start})"
|
|
|
|
params = raw.get("params")
|
|
params = dict(params) if isinstance(params, dict) else {}
|
|
|
|
if kind == "zoom":
|
|
try:
|
|
scale = float(params.get("scale", 1.3))
|
|
except (TypeError, ValueError):
|
|
return None, f"{where}: zoom scale must be a number"
|
|
if not (MIN_ZOOM_SCALE <= scale <= MAX_ZOOM_SCALE):
|
|
return None, (
|
|
f"{where}: zoom scale {scale} outside {MIN_ZOOM_SCALE}-{MAX_ZOOM_SCALE}"
|
|
)
|
|
params["scale"] = scale
|
|
|
|
if kind == "text":
|
|
content = str(params.get("content", "")).strip()
|
|
if not content:
|
|
return None, f"{where}: text action needs params.content"
|
|
params["content"] = content[:MAX_TEXT_LENGTH]
|
|
|
|
return (
|
|
VoiceAction(
|
|
kind=kind,
|
|
start=start,
|
|
end=end,
|
|
params=params,
|
|
reason=str(raw.get("reason", "")),
|
|
speaker=str(raw.get("speaker", "")),
|
|
),
|
|
"",
|
|
)
|
|
|
|
|
|
def parse_actions(data: Any) -> Tuple[List[VoiceAction], List[str]]:
|
|
"""Validate a decision list into actions, collecting rejections.
|
|
|
|
Accepts either a bare list of actions or ``{"actions": [...]}`` — the
|
|
shape a model is most likely to return. Returns ``(actions, errors)``;
|
|
a row that fails validation is reported and skipped, never fatal.
|
|
"""
|
|
if isinstance(data, dict):
|
|
data = data.get("actions", [])
|
|
if not isinstance(data, Sequence) or isinstance(data, (str, bytes)):
|
|
return [], ["expected a list of actions, or an object with an 'actions' list"]
|
|
|
|
actions: List[VoiceAction] = []
|
|
errors: List[str] = []
|
|
for i, raw in enumerate(data):
|
|
action, error = _validate_one(raw, i)
|
|
if action is not None:
|
|
actions.append(action)
|
|
else:
|
|
errors.append(error)
|
|
return actions, errors
|
|
|
|
|
|
def speaker_cut_actions(
|
|
timeline: dict,
|
|
speaker_ids: Sequence[str],
|
|
padding: float = 0.15,
|
|
) -> List[VoiceAction]:
|
|
"""Cut actions removing everything the given speakers say.
|
|
|
|
The everyday case on a testimonial shoot: an interviewer or a crew
|
|
member talks over the take, and only the subject should survive the
|
|
edit. ``padding`` trims slightly *inside* each segment rather than
|
|
around it — speech boundaries from a transcript are approximate, and
|
|
eating into the neighbouring silence is far safer than clipping the
|
|
first syllable of the person being kept.
|
|
"""
|
|
wanted = {str(s) for s in speaker_ids}
|
|
actions: List[VoiceAction] = []
|
|
for segment in timeline.get("segments", []):
|
|
if str(segment.get("speaker", "")) not in wanted:
|
|
continue
|
|
start = float(segment.get("start", 0.0)) + padding
|
|
end = float(segment.get("end", 0.0)) - padding
|
|
if end <= start:
|
|
continue
|
|
actions.append(
|
|
VoiceAction(
|
|
kind="cut",
|
|
start=start,
|
|
end=end,
|
|
reason=f"fala de {segment.get('speaker')}",
|
|
speaker=str(segment.get("speaker", "")),
|
|
)
|
|
)
|
|
return actions
|
|
|
|
|
|
def merge_cut_ranges(actions: Sequence[VoiceAction]) -> List[Tuple[float, float]]:
|
|
"""The cut actions as merged, sorted, non-overlapping source ranges."""
|
|
cuts = sorted((a.start, a.end) for a in actions if a.kind == "cut")
|
|
merged: List[Tuple[float, float]] = []
|
|
for start, end in cuts:
|
|
if merged and start <= merged[-1][1]:
|
|
merged[-1] = (merged[-1][0], max(merged[-1][1], end))
|
|
else:
|
|
merged.append((start, end))
|
|
return merged
|
|
|
|
|
|
def shift_after_cuts(
|
|
time: float, cuts: Sequence[Tuple[float, float]]
|
|
) -> Optional[float]:
|
|
"""Where source ``time`` lands once ``cuts`` are removed.
|
|
|
|
Returns ``None`` when the time falls *inside* a cut — the material it
|
|
referred to no longer exists, so the action that pointed at it must be
|
|
dropped rather than silently slid onto neighbouring content.
|
|
``cuts`` must be merged and sorted (see :func:`merge_cut_ranges`).
|
|
"""
|
|
shift = 0.0
|
|
for start, end in cuts:
|
|
if time < start:
|
|
break
|
|
if time < end:
|
|
return None
|
|
shift += end - start
|
|
return time - shift
|
|
|
|
|
|
def resolve_actions(
|
|
actions: Sequence[VoiceAction],
|
|
) -> Tuple[List[Tuple[float, float]], List[VoiceAction], List[VoiceAction]]:
|
|
"""Split a decision list into what to cut and what to place afterwards.
|
|
|
|
Returns ``(cut_ranges, placed, dropped)``. Non-cut actions are moved onto
|
|
their post-cut times; any that pointed into removed material land in
|
|
``dropped`` so the caller can report them instead of losing them quietly.
|
|
"""
|
|
cut_ranges = merge_cut_ranges(actions)
|
|
placed: List[VoiceAction] = []
|
|
dropped: List[VoiceAction] = []
|
|
|
|
for action in actions:
|
|
if action.kind == "cut":
|
|
continue
|
|
new_start = shift_after_cuts(action.start, cut_ranges)
|
|
if new_start is None:
|
|
dropped.append(action)
|
|
continue
|
|
if action.kind == "marker":
|
|
# A marker is a point, not a span: it survives as long as its own
|
|
# instant does. Requiring its nominal end to survive too would
|
|
# drop exactly the markers worth keeping — the ones flagging a
|
|
# join, which sit right against a cut edge by definition.
|
|
new_end = new_start + action.duration
|
|
else:
|
|
new_end = shift_after_cuts(action.end, cut_ranges)
|
|
if new_end is None:
|
|
dropped.append(action)
|
|
continue
|
|
if new_end <= new_start:
|
|
dropped.append(action)
|
|
continue
|
|
placed.append(
|
|
VoiceAction(
|
|
kind=action.kind,
|
|
start=new_start,
|
|
end=new_end,
|
|
params=action.params,
|
|
reason=action.reason,
|
|
speaker=action.speaker,
|
|
)
|
|
)
|
|
return cut_ranges, placed, dropped
|