chore: atualização geral
This commit is contained in:
@@ -0,0 +1,248 @@
|
||||
"""Voice actions — the editing decisions produced from a voice timeline.
|
||||
|
||||
This is the contract between *deciding* and *applying*. Whoever makes the
|
||||
editorial call — the deterministic rules engine, or a model reading the
|
||||
voice timeline JSON — emits the same list of actions, and one applier turns
|
||||
it into FCPXML. Nothing that produces actions ever touches XML.
|
||||
|
||||
Every action's ``start``/``end`` is in **original source seconds**, matching
|
||||
the voice timeline. That matters: cuts shift everything after them, so if
|
||||
decisions were expressed in post-cut time they would silently land in the
|
||||
wrong place the moment a cut was added. Keeping one origin and resolving the
|
||||
shift at apply time (:func:`shift_after_cuts`) removes that whole class of bug.
|
||||
|
||||
Actions arriving from a model are untrusted input: :func:`parse_actions`
|
||||
validates and reports what it rejected rather than raising, so one malformed
|
||||
row never discards a whole edit.
|
||||
"""
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any, List, Optional, Sequence, Tuple
|
||||
|
||||
# What an action can ask for. Deliberately small — each maps onto one
|
||||
# existing writer capability, so no new XML knowledge lives here.
|
||||
ACTION_KINDS = ("cut", "zoom", "text", "marker")
|
||||
|
||||
# Bounds for a zoom's scale factor. Below 1.0 is a pull-back, not a punch-in;
|
||||
# above 3x the image falls apart on any normal footage.
|
||||
MIN_ZOOM_SCALE = 1.0
|
||||
MAX_ZOOM_SCALE = 3.0
|
||||
|
||||
MAX_TEXT_LENGTH = 120
|
||||
|
||||
|
||||
@dataclass
|
||||
class VoiceAction:
|
||||
"""One editing decision, in original source time."""
|
||||
|
||||
kind: str
|
||||
start: float
|
||||
end: float
|
||||
params: dict = field(default_factory=dict)
|
||||
reason: str = ""
|
||||
speaker: str = ""
|
||||
|
||||
@property
|
||||
def duration(self) -> float:
|
||||
return max(0.0, self.end - self.start)
|
||||
|
||||
def as_dict(self) -> dict:
|
||||
return {
|
||||
"kind": self.kind,
|
||||
"start": round(self.start, 3),
|
||||
"end": round(self.end, 3),
|
||||
"params": self.params,
|
||||
"reason": self.reason,
|
||||
"speaker": self.speaker,
|
||||
}
|
||||
|
||||
|
||||
def _validate_one(raw: Any, index: int) -> Tuple[Optional[VoiceAction], str]:
|
||||
"""Turn one raw row into a VoiceAction, or explain why it can't be."""
|
||||
where = f"action[{index}]"
|
||||
if not isinstance(raw, dict):
|
||||
return None, f"{where}: expected an object, got {type(raw).__name__}"
|
||||
|
||||
kind = str(raw.get("kind", "")).strip().lower()
|
||||
if kind not in ACTION_KINDS:
|
||||
return None, f"{where}: unknown kind {raw.get('kind')!r} (expected one of {', '.join(ACTION_KINDS)})"
|
||||
|
||||
try:
|
||||
start = float(raw.get("start"))
|
||||
end = float(raw.get("end"))
|
||||
except (TypeError, ValueError):
|
||||
return None, f"{where}: start/end must be numbers (seconds)"
|
||||
|
||||
if start < 0:
|
||||
return None, f"{where}: start is negative ({start})"
|
||||
if end <= start:
|
||||
return None, f"{where}: end ({end}) must be after start ({start})"
|
||||
|
||||
params = raw.get("params")
|
||||
params = dict(params) if isinstance(params, dict) else {}
|
||||
|
||||
if kind == "zoom":
|
||||
try:
|
||||
scale = float(params.get("scale", 1.3))
|
||||
except (TypeError, ValueError):
|
||||
return None, f"{where}: zoom scale must be a number"
|
||||
if not (MIN_ZOOM_SCALE <= scale <= MAX_ZOOM_SCALE):
|
||||
return None, (
|
||||
f"{where}: zoom scale {scale} outside {MIN_ZOOM_SCALE}-{MAX_ZOOM_SCALE}"
|
||||
)
|
||||
params["scale"] = scale
|
||||
|
||||
if kind == "text":
|
||||
content = str(params.get("content", "")).strip()
|
||||
if not content:
|
||||
return None, f"{where}: text action needs params.content"
|
||||
params["content"] = content[:MAX_TEXT_LENGTH]
|
||||
|
||||
return (
|
||||
VoiceAction(
|
||||
kind=kind,
|
||||
start=start,
|
||||
end=end,
|
||||
params=params,
|
||||
reason=str(raw.get("reason", "")),
|
||||
speaker=str(raw.get("speaker", "")),
|
||||
),
|
||||
"",
|
||||
)
|
||||
|
||||
|
||||
def parse_actions(data: Any) -> Tuple[List[VoiceAction], List[str]]:
|
||||
"""Validate a decision list into actions, collecting rejections.
|
||||
|
||||
Accepts either a bare list of actions or ``{"actions": [...]}`` — the
|
||||
shape a model is most likely to return. Returns ``(actions, errors)``;
|
||||
a row that fails validation is reported and skipped, never fatal.
|
||||
"""
|
||||
if isinstance(data, dict):
|
||||
data = data.get("actions", [])
|
||||
if not isinstance(data, Sequence) or isinstance(data, (str, bytes)):
|
||||
return [], ["expected a list of actions, or an object with an 'actions' list"]
|
||||
|
||||
actions: List[VoiceAction] = []
|
||||
errors: List[str] = []
|
||||
for i, raw in enumerate(data):
|
||||
action, error = _validate_one(raw, i)
|
||||
if action is not None:
|
||||
actions.append(action)
|
||||
else:
|
||||
errors.append(error)
|
||||
return actions, errors
|
||||
|
||||
|
||||
def speaker_cut_actions(
|
||||
timeline: dict,
|
||||
speaker_ids: Sequence[str],
|
||||
padding: float = 0.15,
|
||||
) -> List[VoiceAction]:
|
||||
"""Cut actions removing everything the given speakers say.
|
||||
|
||||
The everyday case on a testimonial shoot: an interviewer or a crew
|
||||
member talks over the take, and only the subject should survive the
|
||||
edit. ``padding`` trims slightly *inside* each segment rather than
|
||||
around it — speech boundaries from a transcript are approximate, and
|
||||
eating into the neighbouring silence is far safer than clipping the
|
||||
first syllable of the person being kept.
|
||||
"""
|
||||
wanted = {str(s) for s in speaker_ids}
|
||||
actions: List[VoiceAction] = []
|
||||
for segment in timeline.get("segments", []):
|
||||
if str(segment.get("speaker", "")) not in wanted:
|
||||
continue
|
||||
start = float(segment.get("start", 0.0)) + padding
|
||||
end = float(segment.get("end", 0.0)) - padding
|
||||
if end <= start:
|
||||
continue
|
||||
actions.append(
|
||||
VoiceAction(
|
||||
kind="cut",
|
||||
start=start,
|
||||
end=end,
|
||||
reason=f"fala de {segment.get('speaker')}",
|
||||
speaker=str(segment.get("speaker", "")),
|
||||
)
|
||||
)
|
||||
return actions
|
||||
|
||||
|
||||
def merge_cut_ranges(actions: Sequence[VoiceAction]) -> List[Tuple[float, float]]:
|
||||
"""The cut actions as merged, sorted, non-overlapping source ranges."""
|
||||
cuts = sorted((a.start, a.end) for a in actions if a.kind == "cut")
|
||||
merged: List[Tuple[float, float]] = []
|
||||
for start, end in cuts:
|
||||
if merged and start <= merged[-1][1]:
|
||||
merged[-1] = (merged[-1][0], max(merged[-1][1], end))
|
||||
else:
|
||||
merged.append((start, end))
|
||||
return merged
|
||||
|
||||
|
||||
def shift_after_cuts(
|
||||
time: float, cuts: Sequence[Tuple[float, float]]
|
||||
) -> Optional[float]:
|
||||
"""Where source ``time`` lands once ``cuts`` are removed.
|
||||
|
||||
Returns ``None`` when the time falls *inside* a cut — the material it
|
||||
referred to no longer exists, so the action that pointed at it must be
|
||||
dropped rather than silently slid onto neighbouring content.
|
||||
``cuts`` must be merged and sorted (see :func:`merge_cut_ranges`).
|
||||
"""
|
||||
shift = 0.0
|
||||
for start, end in cuts:
|
||||
if time < start:
|
||||
break
|
||||
if time < end:
|
||||
return None
|
||||
shift += end - start
|
||||
return time - shift
|
||||
|
||||
|
||||
def resolve_actions(
|
||||
actions: Sequence[VoiceAction],
|
||||
) -> Tuple[List[Tuple[float, float]], List[VoiceAction], List[VoiceAction]]:
|
||||
"""Split a decision list into what to cut and what to place afterwards.
|
||||
|
||||
Returns ``(cut_ranges, placed, dropped)``. Non-cut actions are moved onto
|
||||
their post-cut times; any that pointed into removed material land in
|
||||
``dropped`` so the caller can report them instead of losing them quietly.
|
||||
"""
|
||||
cut_ranges = merge_cut_ranges(actions)
|
||||
placed: List[VoiceAction] = []
|
||||
dropped: List[VoiceAction] = []
|
||||
|
||||
for action in actions:
|
||||
if action.kind == "cut":
|
||||
continue
|
||||
new_start = shift_after_cuts(action.start, cut_ranges)
|
||||
if new_start is None:
|
||||
dropped.append(action)
|
||||
continue
|
||||
if action.kind == "marker":
|
||||
# A marker is a point, not a span: it survives as long as its own
|
||||
# instant does. Requiring its nominal end to survive too would
|
||||
# drop exactly the markers worth keeping — the ones flagging a
|
||||
# join, which sit right against a cut edge by definition.
|
||||
new_end = new_start + action.duration
|
||||
else:
|
||||
new_end = shift_after_cuts(action.end, cut_ranges)
|
||||
if new_end is None:
|
||||
dropped.append(action)
|
||||
continue
|
||||
if new_end <= new_start:
|
||||
dropped.append(action)
|
||||
continue
|
||||
placed.append(
|
||||
VoiceAction(
|
||||
kind=action.kind,
|
||||
start=new_start,
|
||||
end=new_end,
|
||||
params=action.params,
|
||||
reason=action.reason,
|
||||
speaker=action.speaker,
|
||||
)
|
||||
)
|
||||
return cut_ranges, placed, dropped
|
||||
Reference in New Issue
Block a user