"""Voice actions — the editing decisions produced from a voice timeline. This is the contract between *deciding* and *applying*. Whoever makes the editorial call — the deterministic rules engine, or a model reading the voice timeline JSON — emits the same list of actions, and one applier turns it into FCPXML. Nothing that produces actions ever touches XML. Every action's ``start``/``end`` is in **original source seconds**, matching the voice timeline. That matters: cuts shift everything after them, so if decisions were expressed in post-cut time they would silently land in the wrong place the moment a cut was added. Keeping one origin and resolving the shift at apply time (:func:`shift_after_cuts`) removes that whole class of bug. Actions arriving from a model are untrusted input: :func:`parse_actions` validates and reports what it rejected rather than raising, so one malformed row never discards a whole edit. """ from dataclasses import dataclass, field from typing import Any, List, Optional, Sequence, Tuple # What an action can ask for. Deliberately small — each maps onto one # existing writer capability, so no new XML knowledge lives here. ACTION_KINDS = ("cut", "zoom", "text", "marker") # Bounds for a zoom's scale factor. Below 1.0 is a pull-back, not a punch-in; # above 3x the image falls apart on any normal footage. MIN_ZOOM_SCALE = 1.0 MAX_ZOOM_SCALE = 3.0 MAX_TEXT_LENGTH = 120 @dataclass class VoiceAction: """One editing decision, in original source time.""" kind: str start: float end: float params: dict = field(default_factory=dict) reason: str = "" speaker: str = "" @property def duration(self) -> float: return max(0.0, self.end - self.start) def as_dict(self) -> dict: return { "kind": self.kind, "start": round(self.start, 3), "end": round(self.end, 3), "params": self.params, "reason": self.reason, "speaker": self.speaker, } def _validate_one(raw: Any, index: int) -> Tuple[Optional[VoiceAction], str]: """Turn one raw row into a VoiceAction, or explain why it can't be.""" where = f"action[{index}]" if not isinstance(raw, dict): return None, f"{where}: expected an object, got {type(raw).__name__}" kind = str(raw.get("kind", "")).strip().lower() if kind not in ACTION_KINDS: return None, f"{where}: unknown kind {raw.get('kind')!r} (expected one of {', '.join(ACTION_KINDS)})" try: start = float(raw.get("start")) end = float(raw.get("end")) except (TypeError, ValueError): return None, f"{where}: start/end must be numbers (seconds)" if start < 0: return None, f"{where}: start is negative ({start})" if end <= start: return None, f"{where}: end ({end}) must be after start ({start})" params = raw.get("params") params = dict(params) if isinstance(params, dict) else {} if kind == "zoom": if "scale" in params and params.get("scale") is not None: try: scale = float(params["scale"]) except (TypeError, ValueError): return None, f"{where}: zoom scale must be a number" if not (MIN_ZOOM_SCALE <= scale <= MAX_ZOOM_SCALE): return None, ( f"{where}: zoom scale {scale} outside {MIN_ZOOM_SCALE}-{MAX_ZOOM_SCALE}" ) params["scale"] = scale if kind == "text": content = str(params.get("content", "")).strip() if not content: return None, f"{where}: text action needs params.content" params["content"] = content[:MAX_TEXT_LENGTH] # Style is optional — omitted fields fall back to the "Legendas # Dinâmicas" emphasis style at apply time (see _apply_placed_action), # so a callout matches the captions' look without the caller having # to know or repeat that configuration. Anything given here wins. for key in ("font", "font_color", "face"): if key in params and not isinstance(params[key], str): del params[key] if "font_size" in params: try: params["font_size"] = int(params["font_size"]) except (TypeError, ValueError): del params["font_size"] if "bold" in params: params["bold"] = bool(params["bold"]) return ( VoiceAction( kind=kind, start=start, end=end, params=params, reason=str(raw.get("reason", "")), speaker=str(raw.get("speaker", "")), ), "", ) def parse_actions(data: Any) -> Tuple[List[VoiceAction], List[str]]: """Validate a decision list into actions, collecting rejections. Accepts either a bare list of actions or ``{"actions": [...]}`` — the shape a model is most likely to return. Returns ``(actions, errors)``; a row that fails validation is reported and skipped, never fatal. """ if isinstance(data, dict): data = data.get("actions", []) if not isinstance(data, Sequence) or isinstance(data, (str, bytes)): return [], ["expected a list of actions, or an object with an 'actions' list"] actions: List[VoiceAction] = [] errors: List[str] = [] for i, raw in enumerate(data): action, error = _validate_one(raw, i) if action is not None: actions.append(action) else: errors.append(error) return actions, errors def speaker_cut_actions( timeline: dict, speaker_ids: Sequence[str], padding: float = 0.15, ) -> List[VoiceAction]: """Cut actions removing everything the given speakers say. The everyday case on a testimonial shoot: an interviewer or a crew member talks over the take, and only the subject should survive the edit. ``padding`` trims slightly *inside* each segment rather than around it — speech boundaries from a transcript are approximate, and eating into the neighbouring silence is far safer than clipping the first syllable of the person being kept. """ wanted = {str(s) for s in speaker_ids} actions: List[VoiceAction] = [] for segment in timeline.get("segments", []): if str(segment.get("speaker", "")) not in wanted: continue start = float(segment.get("start", 0.0)) + padding end = float(segment.get("end", 0.0)) - padding if end <= start: continue actions.append( VoiceAction( kind="cut", start=start, end=end, reason=f"fala de {segment.get('speaker')}", speaker=str(segment.get("speaker", "")), ) ) return actions def merge_cut_ranges(actions: Sequence[VoiceAction]) -> List[Tuple[float, float]]: """The cut actions as merged, sorted, non-overlapping source ranges.""" cuts = sorted((a.start, a.end) for a in actions if a.kind == "cut") merged: List[Tuple[float, float]] = [] for start, end in cuts: if merged and start <= merged[-1][1]: merged[-1] = (merged[-1][0], max(merged[-1][1], end)) else: merged.append((start, end)) return merged def shift_after_cuts( time: float, cuts: Sequence[Tuple[float, float]] ) -> Optional[float]: """Where source ``time`` lands once ``cuts`` are removed. Returns ``None`` when the time falls *inside* a cut — the material it referred to no longer exists, so the action that pointed at it must be dropped rather than silently slid onto neighbouring content. ``cuts`` must be merged and sorted (see :func:`merge_cut_ranges`). """ shift = 0.0 for start, end in cuts: if time < start: break if time < end: return None shift += end - start return time - shift def resolve_actions( actions: Sequence[VoiceAction], ) -> Tuple[List[Tuple[float, float]], List[VoiceAction], List[VoiceAction]]: """Split a decision list into what to cut and what to place afterwards. Returns ``(cut_ranges, placed, dropped)``. Non-cut actions are moved onto their post-cut times; any that pointed into removed material land in ``dropped`` so the caller can report them instead of losing them quietly. """ cut_ranges = merge_cut_ranges(actions) placed: List[VoiceAction] = [] dropped: List[VoiceAction] = [] for action in actions: if action.kind == "cut": continue new_start = shift_after_cuts(action.start, cut_ranges) if new_start is None: dropped.append(action) continue if action.kind == "marker": # A marker is a point, not a span: it survives as long as its own # instant does. Requiring its nominal end to survive too would # drop exactly the markers worth keeping — the ones flagging a # join, which sit right against a cut edge by definition. new_end = new_start + action.duration else: new_end = shift_after_cuts(action.end, cut_ranges) if new_end is None: dropped.append(action) continue if new_end <= new_start: dropped.append(action) continue placed.append( VoiceAction( kind=action.kind, start=new_start, end=new_end, params=action.params, reason=action.reason, speaker=action.speaker, ) ) return cut_ranges, placed, dropped