chore: atualização geral

This commit is contained in:
João Henrique
2026-08-19 16:35:29 -04:00
parent 8fca456ceb
commit e7748c2c58
66 changed files with 13037 additions and 4237 deletions
+266
View File
@@ -0,0 +1,266 @@
"""Tests for collision detection and subtitle-layout validation (Fase 1).
Covers the pure functions in ``fcpxml.collision`` (spatial/temporal overlap,
area/ratio classification, distance, separation suggestion, box measurement)
and the integration through ``FCPXMLModifier.validate_subtitle_layout`` over a
generated document — the post-generation guarantee the layout engine only
provides by construction.
"""
import shutil
import tempfile
from pathlib import Path
import pytest
from fcpxml.collision import (
FONT_MISSING,
FONT_TOO_SMALL,
OUTSIDE_FRAME,
OUTSIDE_SAFE_AREA,
OVERLAP_PROBABLE,
OVERLAP_RENDER_TOLERANCE,
OVERLAP_SEVERE,
SPATIAL_COLLISION,
Box,
blocking,
classify_overlap,
distance_between,
measure_title_box,
overlap_metrics,
separation_suggestion,
temporal_overlap,
validate_titles,
)
from fcpxml.models import DynamicSubtitleConfig
from fcpxml.writer import FCPXMLModifier
SAMPLE = Path(__file__).parent.parent / "examples" / "sample.fcpxml"
WORD_MODE = DynamicSubtitleConfig(granularity="word")
WORDS = [
{"word": "Hello", "start": 0.0, "end": 0.4},
{"word": "there", "start": 0.4, "end": 0.8},
{"word": "friend", "start": 0.8, "end": 1.3},
]
@pytest.fixture
def temp_fcpxml():
with tempfile.NamedTemporaryFile(suffix=".fcpxml", delete=False) as f:
shutil.copy(SAMPLE, f.name)
yield f.name
Path(f.name).unlink(missing_ok=True)
def _title(text, x=0.0, y=0.0, *, start=0.0, end=1.0, font="Helvetica Neue",
face=None, font_size=100.0, kerning=0.0, group=0):
return {
"text": text,
"x": x,
"y": y,
"start": start,
"end": end,
"font": font,
"face": face,
"font_size": font_size,
"kerning": kerning,
"group": group,
}
class TestBoxOverlap:
def test_touching_boxes_do_not_overlap(self):
a = Box(0, 100, 0, 50)
b = Box(100, 200, 0, 50) # shares the right edge
assert not a.overlaps(b)
assert overlap_metrics(a, b)["overlap_area"] == 0
def test_overlap_by_one_pixel_detected(self):
a = Box(0, 100, 0, 50)
b = Box(99, 200, 0, 50) # one-pixel horizontal overlap
assert a.overlaps(b)
metrics = overlap_metrics(a, b)
assert metrics["overlap_width"] == 1
assert metrics["overlap_area"] == 50
def test_vertical_only_overlap(self):
a = Box(0, 100, 0, 50)
b = Box(0, 100, 49, 100) # one-pixel vertical overlap
assert a.overlaps(b)
assert overlap_metrics(a, b)["overlap_height"] == 1
class TestTemporalOverlap:
def test_adjacent_intervals_are_not_simultaneous(self):
# [0, 1) and [1, 2) share no instant.
assert not temporal_overlap(0.0, 1.0, 1.0, 2.0)
assert not temporal_overlap(1.0, 2.0, 0.0, 1.0)
def test_interleaved_intervals_overlap(self):
assert temporal_overlap(0.0, 2.0, 1.0, 3.0)
def test_contained_interval_overlaps(self):
assert temporal_overlap(0.0, 5.0, 1.0, 2.0)
def test_boundary_survives_float_noise_from_the_writer(self):
"""Found on real footage: the writer sets one block's title duration
to make its end land EXACTLY on the next block's start (same exact
FCPXML fraction), but end here is re-derived as start + duration —
two independently-rounded floats — which isn't bit-identical to the
other title's start read as a single division of that same
fraction. 7 of 8 "severe" collisions from one real clip were this,
off by ~1e-13s, far below any frame boundary."""
start_a, duration_a = 74883609 / 24000, 43043 / 24000
end_a = start_a + duration_a # float addition, like validate_titles does
start_b = 74926652 / 24000 # the exact same instant, read directly
assert end_a != start_b # the float noise is real
assert abs(end_a - start_b) < 1e-6 # ...and far below one frame
assert not temporal_overlap(start_a, end_a, start_b, start_b + 1.0)
class TestClassifyOverlap:
def test_zero_area_is_no_conflict(self):
assert classify_overlap(
{"overlap_area": 0, "overlap_height": 0, "overlap_ratio": 0}
) == "none"
def test_ratio_dominates_small_height(self):
# A sliver of 5px that covers most of a tiny box is still severe.
metrics = {"overlap_area": 50, "overlap_height": 5, "overlap_ratio": 0.9}
assert classify_overlap(metrics) == OVERLAP_SEVERE
def test_height_buckets(self):
def metrics(height):
return {"overlap_area": 1, "overlap_height": height,
"overlap_ratio": 0.0}
assert classify_overlap(metrics(5)) == OVERLAP_RENDER_TOLERANCE
assert classify_overlap(metrics(6)) == "warning"
assert classify_overlap(metrics(21)) == OVERLAP_PROBABLE
assert classify_overlap(metrics(51)) == OVERLAP_SEVERE
class TestDistanceAndSeparation:
def test_horizontal_distance(self):
a = Box(0, 100, 0, 50)
b = Box(150, 200, 0, 50)
d = distance_between(a, b)
assert d["distance_x"] == 50
assert d["distance_y"] == 0
assert d["distance"] == 50
def test_vertical_distance(self):
a = Box(0, 100, 0, 50)
b = Box(0, 100, 80, 130)
d = distance_between(a, b)
assert d["distance_y"] == 30
assert d["distance_x"] == 0
def test_separation_picks_smallest_axis(self):
a = Box(0, 100, 0, 100)
b = Box(90, 190, 95, 195) # 10px horizontal, 5px vertical penetration
suggestion = separation_suggestion(a, b)
assert suggestion["axis"] == "vertical"
assert suggestion["minimum_movement"] == 5
def test_blocking_severities(self):
assert blocking(OVERLAP_SEVERE)
assert blocking(OVERLAP_PROBABLE)
assert not blocking("warning")
assert not blocking("none")
class TestValidateTitles:
def test_clean_layout_has_no_issues(self):
report = validate_titles(
[_title("um", x=-200), _title("dois", x=200)],
2160, 3840,
)
assert report["severity"] == "none"
assert report["issues"] == []
def test_simultaneous_collision_reported(self):
titles = [_title("A", x=0), _title("B", x=0)]
report = validate_titles(titles, 2160, 3840)
collisions = [
i for i in report["issues"] if i["type"] == SPATIAL_COLLISION
]
assert len(collisions) == 1
assert collisions[0]["severity"] == OVERLAP_SEVERE
assert collisions[0]["suggested_correction"]["axis"] in (
"vertical", "horizontal",
)
def test_collision_across_times_ignored(self):
titles = [
_title("A", x=0, start=0.0, end=1.0),
_title("B", x=0, start=1.0, end=2.0),
]
report = validate_titles(titles, 2160, 3840)
assert not any(
i["type"] == SPATIAL_COLLISION for i in report["issues"]
)
def test_outside_frame_is_error(self):
report = validate_titles([_title("fora", x=5000)], 2160, 3840)
assert any(i["type"] == OUTSIDE_FRAME for i in report["issues"])
def test_outside_safe_area_is_warning_not_frame(self):
# Near the right edge: inside the frame, outside the 5% safe area.
report = validate_titles(
[_title("a", x=1000, font_size=40)], 2160, 3840,
)
assert any(i["type"] == OUTSIDE_SAFE_AREA for i in report["issues"])
assert not any(i["type"] == OUTSIDE_FRAME for i in report["issues"])
def test_resolution_changes_safe_area(self):
# Same x is fine on a wide frame but out of the safe area on a narrow one.
narrow = validate_titles([_title("a", x=1000, font_size=40)], 1920, 1080)
assert any(i["type"] == OUTSIDE_FRAME for i in narrow["issues"])
def test_font_missing_reported(self):
report = validate_titles(
[_title("oi", font="Comic Sans MS")], 2160, 3840,
)
assert any(i["type"] == FONT_MISSING for i in report["issues"])
def test_font_too_small_reported(self):
report = validate_titles(
[_title("oi", font_size=10)], 2160, 3840, min_font_size=20,
)
assert any(i["type"] == FONT_TOO_SMALL for i in report["issues"])
def test_measure_title_box_uses_real_width(self):
box = measure_title_box("ii", 100, x=0, y=0, font="Helvetica Neue")
# "ii" is the narrowest glyph; a single "W" is much wider.
wide = measure_title_box("W", 100, x=0, y=0, font="Helvetica Neue")
assert wide.width > box.width
class TestIntegration:
def test_generated_layout_is_clean(self, temp_fcpxml):
modifier = FCPXMLModifier(temp_fcpxml)
modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE)
report = modifier.validate_subtitle_layout()
assert report["summary"]["title_count"] >= 3
assert report["summary"]["spatial_collision"] == 0
assert not blocking(report["severity"])
def test_hand_edited_position_is_caught(self, temp_fcpxml):
modifier = FCPXMLModifier(temp_fcpxml)
titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE)
def position(el):
for p in el.findall("param"):
if p.get("name") == "Position":
return p
return None
p0 = position(titles[0])
position(titles[1]).set("value", p0.get("value"))
report = modifier.validate_subtitle_layout()
assert report["summary"]["spatial_collision"] >= 1
assert blocking(report["severity"])
+108
View File
@@ -0,0 +1,108 @@
"""Tests for the diarize_media MCP tool (server.handle_diarize_media).
Diarization itself (pyannote.audio) is monkeypatched so these tests run
without the optional [diarization] extra or a HuggingFace token — matching
the existing TestDetectBeatsHandler pattern in test_media_intel.py.
"""
import json
import pytest
def _write_tiny_wav(path: str, seconds: float = 1.0) -> None:
import struct
import wave
n_frames = int(44100 * seconds)
with wave.open(path, "w") as f:
f.setnchannels(1)
f.setsampwidth(2)
f.setframerate(44100)
f.writeframes(struct.pack("<%dh" % n_frames, *([0] * n_frames)))
class TestDiarizeMediaHandler:
async def test_reports_when_pyannote_unavailable(self, tmp_path, monkeypatch):
import server_tools.voice as server_mod
from server import handle_diarize_media
wav = tmp_path / "clip.wav"
_write_tiny_wav(str(wav))
monkeypatch.setattr(
server_mod, "diarization_capability", lambda token: (False, "Diarização indisponível: componente pyannote.audio ausente.")
)
result = await handle_diarize_media({"media_path": str(wav)})
text = result[0].text
assert "indisponível" in text.lower() or "unavailable" in text.lower()
assert "diarization" in text.lower()
async def test_rejects_disallowed_extension(self, tmp_path):
from server import handle_diarize_media
bad = tmp_path / "clip.txt"
bad.write_text("not audio")
with pytest.raises(ValueError):
await handle_diarize_media({"media_path": str(bad)})
async def test_writes_diarization_json_and_reports(self, tmp_path, monkeypatch):
import server_tools._shared as _shared_mod
import server_tools.voice as server_mod
from server import handle_diarize_media
wav = tmp_path / "clip.wav"
_write_tiny_wav(str(wav), seconds=2.0)
fake_transcript = {
"language": "en",
"duration": 2.0,
"text": "hello world",
"segments": [
{"text": "hello", "start": 0.0, "end": 1.0},
{"text": "world", "start": 1.0, "end": 2.0},
],
"words": [
{"word": "hello", "start": 0.0, "end": 0.5, "confidence": 0.9},
{"word": "world", "start": 1.0, "end": 1.5, "confidence": 0.9},
],
}
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: fake_transcript)
monkeypatch.setattr(server_mod, "diarization_capability", lambda token: (True, "ok"))
monkeypatch.setattr(
server_mod,
"diarize",
lambda path, token, num_speakers="": [(0.0, 1.0, "A"), (1.0, 2.0, "B")],
)
result = await handle_diarize_media({"media_path": str(wav), "hf_token": "fake-token"})
text = result[0].text
assert "Speaker" in text or "speaker" in text.lower()
json_path = tmp_path / "clip_diarization.json"
assert str(json_path) in text
data = json.loads(json_path.read_text())
assert len(data["speakers"]) == 2
assert data["words"][0]["speaker_id"] == "SPEAKER_00"
assert data["words"][1]["speaker_id"] == "SPEAKER_01"
async def test_reports_when_diarization_fails(self, tmp_path, monkeypatch):
import server_tools._shared as _shared_mod
import server_tools.voice as server_mod
from server import handle_diarize_media
wav = tmp_path / "clip.wav"
_write_tiny_wav(str(wav))
fake_transcript = {
"language": "en",
"duration": 1.0,
"text": "hi",
"segments": [{"text": "hi", "start": 0.0, "end": 1.0}],
"words": [{"word": "hi", "start": 0.0, "end": 0.5, "confidence": 0.9}],
}
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: fake_transcript)
monkeypatch.setattr(server_mod, "diarization_capability", lambda token: (True, "ok"))
monkeypatch.setattr(server_mod, "diarize", lambda *a, **k: None)
result = await handle_diarize_media({"media_path": str(wav), "hf_token": "fake-token"})
assert "failed" in result[0].text.lower()
+168 -7
View File
@@ -17,12 +17,27 @@ from pathlib import Path
import pytest
from fcpxml.models import DynamicSubtitleConfig, TimeValue, WordStyle
from fcpxml.models import DynamicSubtitleConfig, TimeValue, WordLook, WordStyle
from fcpxml.parser import parse_fcpxml
from fcpxml.text_layout import POINT_SCALE, REFERENCE_CANVAS_HEIGHT, ink_extent
from fcpxml.text_layout import (
POINT_SCALE,
REFERENCE_BLOCK_LINE_GAP,
REFERENCE_CANVAS_HEIGHT,
TEXT_TEMPLATE_FONT_SCALE,
LayoutBox,
compose_sentence,
ink_extent,
)
from fcpxml.writer import FCPXMLModifier
SAMPLE = Path(__file__).parent.parent / "examples" / "sample.fcpxml"
def font_points(style) -> float:
"""The style's size back in CANVAS POINTS.
The emitted fontSize lives in the template's own space, which is
TEXT_TEMPLATE_FONT_SCALE times bigger than the space positions use, so any
check that mixes the two has to convert first."""
return float(style.get("fontSize")) / TEXT_TEMPLATE_FONT_SCALE
# The earlier rhythm: one title per WORD. The default is now the
# progressive composition (one title per LINE), covered in
@@ -106,7 +121,7 @@ class TestGenerateDynamicSubtitles:
param_names = [p.get("name") for p in title.findall("param")]
assert param_names == [
"Position", "Layout Method", "Left Margin", "Right Margin",
"Position", "Build Out", "Layout Method", "Left Margin", "Right Margin",
"Top Margin", "Bottom Margin", "Alignment", "Line Spacing",
"Auto-Shrink", "Alignment", "Opacity", "Speed", "Custom Speed",
"Apply Speed",
@@ -132,7 +147,8 @@ class TestGenerateDynamicSubtitles:
assert style_def.get("fontFace") == first.face
scale = (1080 * POINT_SCALE) / REFERENCE_CANVAS_HEIGHT
assert style_def.get("fontSize") == str(round(first.font_size * scale))
expected = round(first.font_size * scale) * TEXT_TEMPLATE_FONT_SCALE
assert float(style_def.get("fontSize")) == expected
def test_font_size_scales_with_the_frame(self, temp_fcpxml):
"""A vertical 2160x3840 timeline must get the reference sizes back
@@ -144,7 +160,9 @@ class TestGenerateDynamicSubtitles:
title = modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE)[0]
run = title.find("text/text-style")
style_def = title.find(f"text-style-def[@id='{run.get('ref')}']/text-style")
assert style_def.get("fontSize") == str(WordStyle().rhythm[0].font_size)
assert float(style_def.get("fontSize")) == (
WordStyle().rhythm[0].font_size * TEXT_TEMPLATE_FONT_SCALE
)
def test_position_is_keyframed_constant_hold(self, temp_fcpxml):
"""Regression (2026-08-17): position is a STATIC param in the "Text"
@@ -685,7 +703,7 @@ class TestProgressiveComposition:
style = self._style(t)
top, bottom = ink_extent(
t.find("text/text-style").text,
float(style.get("fontSize")),
font_points(style),
font=style.get("font"),
face=style.get("fontFace"),
)
@@ -718,7 +736,7 @@ class TestProgressiveComposition:
style = self._style(t)
top, bottom = ink_extent(
t.find("text/text-style").text,
float(style.get("fontSize")),
font_points(style),
font=style.get("font"),
face=style.get("fontFace"),
)
@@ -762,3 +780,146 @@ class TestProgressiveComposition:
spoken = " ".join(w["word"] for w in words)
emitted = " ".join(t.find("text/text-style").text for t in titles)
assert emitted == spoken, "no spoken word may be dropped or reordered"
class TestTemplateFontScale:
"""The "Text" template sizes type in frame pixels but positions in canvas
points, so the emitted fontSize must be converted or the block renders in
the right place at half the chosen size."""
STYLE = WordStyle(
emphasis_look=WordLook(200, "1 1 1 1", font="Georgia", kerning=3.0),
body_look=WordLook(100, "1 1 1 1", font="Helvetica Neue", kerning=3.0),
)
def _styles(self, titles):
return [t.find(".//text-style-def/text-style") for t in titles]
def _generate(self, path, **kwargs):
modifier = FCPXMLModifier(path)
titles = modifier.generate_dynamic_subtitles(
"Interview_A", WORDS,
DynamicSubtitleConfig(style=self.STYLE, **kwargs),
)
return titles
def test_emitted_size_is_the_layout_size_times_the_template_scale(self, temp_fcpxml):
unscaled = self._generate(temp_fcpxml, text_scale=1.0)
scaled = self._generate(temp_fcpxml, text_scale=TEXT_TEMPLATE_FONT_SCALE)
for plain, big in zip(self._styles(unscaled), self._styles(scaled)):
assert float(big.get("fontSize")) == (
float(plain.get("fontSize")) * TEXT_TEMPLATE_FONT_SCALE
)
def test_default_config_applies_the_template_scale(self, temp_fcpxml):
assert DynamicSubtitleConfig().text_scale == TEXT_TEMPLATE_FONT_SCALE
default = self._styles(self._generate(temp_fcpxml))
unscaled = self._styles(self._generate(temp_fcpxml, text_scale=1.0))
assert [s.get("fontSize") for s in default] != [
s.get("fontSize") for s in unscaled
]
def test_kerning_scales_with_the_font_size(self, temp_fcpxml):
"""Kerning is in font units too — leaving it behind would tighten the
letter spacing to half as the type doubled."""
unscaled = self._styles(self._generate(temp_fcpxml, text_scale=1.0))
scaled = self._styles(self._generate(temp_fcpxml, text_scale=2.0))
for plain, big in zip(unscaled, scaled):
if plain.get("kerning"):
assert float(big.get("kerning")) == float(plain.get("kerning")) * 2
def _positions(self, titles):
return [
tuple(float(v) for v in t.find(
"param[@name='Position']").get("value").split())
for t in titles
]
def test_position_is_converted_with_the_type(self, temp_fcpxml):
"""The template reads fontSize and Position in the SAME space, so the
conversion has to reach both. Scaling only the type leaves the block at
the old spread with twice the type in it, and the lines collide."""
plain = self._generate(temp_fcpxml, text_scale=1.0)
big = self._generate(temp_fcpxml, text_scale=2.0)
for (x, y), (x2, y2) in zip(self._positions(plain), self._positions(big)):
assert (x2, y2) == pytest.approx((x * 2, y * 2), rel=1e-4, abs=0.01)
def test_type_and_spacing_keep_their_ratio_at_any_scale(self, temp_fcpxml):
"""The invariant that broke in the field: the distance between two
lines, measured in font sizes, must not depend on the scale."""
ratios = []
for scale in (1.0, 2.0, 3.5):
titles = self._generate(temp_fcpxml, text_scale=scale)
ys = [y for _, y in self._positions(titles)]
sizes = [float(s.get("fontSize")) for s in self._styles(titles)]
ratios.append([
(a - b) / size
for a, b, size in zip(ys, ys[1:], sizes)
])
for other in ratios[1:]:
assert other == pytest.approx(ratios[0], rel=1e-4, abs=1e-4)
class TestLineGap:
"""The air between stacked lines is a design choice, negative included."""
STYLE = WordStyle(
emphasis_look=WordLook(200, "1 1 1 1", font="Georgia", kerning=0.0),
body_look=WordLook(100, "1 1 1 1", font="Helvetica Neue", kerning=0.0),
)
PHRASE = "eu tinha muita dificuldade de encontrar roupa"
def _blocks(self, gap):
"""Compose in a band tall enough to hold every line, so the gap is the
only thing that changes — a short band would also change how many
lines fit, which is a different effect."""
words = [
{"word": w, "start": i * 0.3, "end": i * 0.3 + 0.3}
for i, w in enumerate(self.PHRASE.split())
]
box = LayoutBox(width=1080 * 0.92, height=100_000, center_y=0)
return compose_sentence(words, self.STYLE, box, line_gap=gap).blocks
def test_default_matches_the_reference_gap(self):
assert DynamicSubtitleConfig().line_gap == REFERENCE_BLOCK_LINE_GAP
def test_each_step_changes_by_exactly_the_gap(self):
"""The stack places ink boxes edge to edge, so the gap is the whole
distance between two lines beyond their own ink."""
zero = [b.y for b in self._blocks(0.0)]
loose = [b.y for b in self._blocks(50.0)]
assert len(zero) == len(loose) >= 2
for plain, spaced in zip(
[a - b for a, b in zip(zero, zero[1:])],
[a - b for a, b in zip(loose, loose[1:])],
):
assert spaced == pytest.approx(plain + 50.0)
def test_a_negative_gap_overlaps_by_exactly_that_much(self):
"""Negative is a supported look, not a failure: -40 tucks each line 40
points into the one above rather than colliding by some amount the
caller cannot predict."""
zero = [b.y for b in self._blocks(0.0)]
tucked = [b.y for b in self._blocks(-40.0)]
assert len(zero) == len(tucked) >= 2
for plain, tight in zip(
[a - b for a, b in zip(zero, zero[1:])],
[a - b for a, b in zip(tucked, tucked[1:])],
):
assert tight == pytest.approx(plain - 40.0)
def test_the_gap_reaches_the_generated_titles(self, temp_fcpxml):
"""The config field has to survive the trip to the XML."""
def spread(gap):
modifier = FCPXMLModifier(temp_fcpxml)
titles = modifier.generate_dynamic_subtitles(
"Interview_A", WORDS,
DynamicSubtitleConfig(style=self.STYLE, line_gap=gap, text_scale=1.0),
)
ys = [
float(t.find("param[@name='Position']").get("value").split()[1])
for t in titles
]
return max(ys) - min(ys)
assert spread(0.0) < spread(80.0)
+112
View File
@@ -0,0 +1,112 @@
"""Tests for fcpxml/emphasis.py — the emphasis index (pure, no audio needed)."""
from fcpxml.emphasis import (
EmphasisWeights,
annotate_emphasis,
compute_emphasis,
pause_weight,
)
def test_zero_input_gives_zero_score():
assert compute_emphasis(0.0, 0.0, 0.0, 0.0, 0.0) == 0.0
def test_max_input_gives_max_score():
# pause sits inside the "dramatic beat" window, not a scene-change gap
score = compute_emphasis(
energy=1.0, pitch_delta=1.0, rate_delta=1.0, pause_before=1.5, word_duration=10.0
)
assert score == 1.0
def test_negative_pitch_delta_uses_magnitude():
a = compute_emphasis(0.0, pitch_delta=0.5, rate_delta=0.0, pause_before=0.0, word_duration=0.0)
b = compute_emphasis(0.0, pitch_delta=-0.5, rate_delta=0.0, pause_before=0.0, word_duration=0.0)
assert a == b > 0.0
def test_duration_saturates_beyond_cap():
at_cap = compute_emphasis(0.0, 0.0, 0.0, 0.0, word_duration=1.0, max_duration=1.0)
beyond = compute_emphasis(0.0, 0.0, 0.0, 0.0, word_duration=30.0, max_duration=1.0)
assert at_cap == beyond
class TestPauseWeight:
"""A long gap is a scene change, not emphasis — it must not outrank a
word the speaker actually hit hard. Grounded in real footage where 6-9s
gaps were topping the emphasis ranking."""
def test_dramatic_beat_counts_fully(self):
assert pause_weight(1.5, max_pause=1.5) == 1.0
def test_short_beat_counts_proportionally(self):
assert pause_weight(0.75, max_pause=1.5) == 0.5
def test_long_gap_is_ignored(self):
assert pause_weight(8.7, max_pause=1.5, ignore_above=3.0) == 0.0
def test_no_pause_is_zero(self):
assert pause_weight(0.0) == 0.0
def test_cutoff_can_be_disabled(self):
assert pause_weight(30.0, max_pause=1.5, ignore_above=0.0) == 1.0
def test_scene_change_scores_below_a_loud_word(self):
gap = compute_emphasis(0.25, 0.04, 0.0, pause_before=6.2, word_duration=0.5)
loud = compute_emphasis(1.00, 0.28, 0.0, pause_before=1.9, word_duration=0.2)
assert loud > gap
def test_higher_energy_weight_increases_energy_contribution():
low_weight = EmphasisWeights(energy=0.1, pitch_variation=0.0, rate_variation=0.0, pause_before=0.0, duration=0.0)
high_weight = EmphasisWeights(energy=1.0, pitch_variation=0.0, rate_variation=0.0, pause_before=0.0, duration=0.0)
# energy is the only nonzero factor for both weight sets, so normalized
# score should be identical regardless of the absolute weight value.
score_low = compute_emphasis(0.6, 0.0, 0.0, 0.0, 0.0, weights=low_weight)
score_high = compute_emphasis(0.6, 0.0, 0.0, 0.0, 0.0, weights=high_weight)
assert score_low == score_high
def test_all_zero_weights_returns_zero_not_error():
zero_weights = EmphasisWeights(0.0, 0.0, 0.0, 0.0, 0.0)
assert compute_emphasis(1.0, 1.0, 1.0, 1.0, 1.0, weights=zero_weights) == 0.0
def test_weights_round_trip_dict():
w = EmphasisWeights(energy=0.4, pitch_variation=0.3, rate_variation=0.1, pause_before=0.1, duration=0.1)
restored = EmphasisWeights.from_dict(w.as_dict())
assert restored == w
def test_from_dict_fills_missing_with_defaults():
restored = EmphasisWeights.from_dict({"energy": 0.9})
defaults = EmphasisWeights()
assert restored.energy == 0.9
assert restored.pitch_variation == defaults.pitch_variation
def test_annotate_emphasis_adds_score_per_word():
words = [
{"word": "hi", "start": 0.0, "end": 0.3, "energy": 0.2, "pitch_delta": 0.1, "rate_delta": 0.1, "pause_before": 0.0},
{"word": "WOW", "start": 1.0, "end": 1.5, "energy": 0.9, "pitch_delta": 0.8, "rate_delta": 0.7, "pause_before": 1.0},
]
annotated = annotate_emphasis(words)
assert len(annotated) == 2
assert all("emphasis" in w for w in annotated)
assert annotated[1]["emphasis"] > annotated[0]["emphasis"]
def test_annotate_emphasis_does_not_mutate_input():
words = [{"word": "hi", "start": 0.0, "end": 0.3, "energy": 0.5}]
annotate_emphasis(words)
assert "emphasis" not in words[0]
def test_annotate_emphasis_derives_duration_from_start_end():
"""A word with no explicit "duration" key gets it from end - start."""
base = {"energy": 0.0, "pitch_delta": 0.0, "rate_delta": 0.0, "pause_before": 0.0}
derived = annotate_emphasis([{"word": "hi", "start": 1.0, "end": 1.5, **base}], max_duration=0.5)
explicit = annotate_emphasis([{"word": "hi", "start": 0.0, "end": 0.0, "duration": 0.5, **base}], max_duration=0.5)
# duration is the only nonzero factor in both, so scores must match
assert derived[0]["emphasis"] == explicit[0]["emphasis"] > 0.0
+1 -1
View File
@@ -518,7 +518,7 @@ class TestDetectBeatsHandler:
await handle_detect_beats({"media_path": str(bad)})
async def test_reports_when_librosa_unavailable(self, tmp_path, monkeypatch):
import server as server_mod
import server_tools.qc as server_mod
from server import handle_detect_beats
wav = tmp_path / "song.wav"
+63
View File
@@ -0,0 +1,63 @@
"""Tests for `output_dir` routing — the app's "Pasta do projeto" promise.
The setting is documented in the UI as "everything generated is saved in
here". It used to be applied as a sandbox anchor only, while the filename
was still derived in the INPUT's directory — so any call whose output_dir
differed from the input's folder failed its own anchor check.
"""
import pytest
from server_tools._shared import _resolve_io_paths
@pytest.fixture
def project(tmp_path):
"""An .fcpxml in one folder, with a separate chosen output folder."""
source_dir = tmp_path / "media"
source_dir.mkdir()
fcpxml = source_dir / "Projeto.fcpxml"
fcpxml.write_text("<fcpxml version='1.13'/>")
chosen = tmp_path / "pasta do projeto"
chosen.mkdir()
return fcpxml, chosen
class TestOutputDirRouting:
def test_output_lands_in_the_chosen_folder(self, project):
fcpxml, chosen = project
_, output_path = _resolve_io_paths({"filepath": str(fcpxml), "output_dir": str(chosen)}, "_voice_edit")
assert output_path.startswith(str(chosen))
def test_filename_keeps_the_suffix_convention(self, project):
fcpxml, chosen = project
_, output_path = _resolve_io_paths({"filepath": str(fcpxml), "output_dir": str(chosen)}, "_voice_edit")
assert output_path.endswith("Projeto_voice_edit.fcpxml")
def test_cross_directory_call_does_not_raise(self, project):
"""The regression: output_dir different from the input folder used to
raise "output path escapes allowed directory" every single time."""
fcpxml, chosen = project
_resolve_io_paths({"filepath": str(fcpxml), "output_dir": str(chosen)}, "_dynamic_subtitles")
def test_without_output_dir_it_still_writes_beside_the_input(self, project):
fcpxml, _ = project
_, output_path = _resolve_io_paths({"filepath": str(fcpxml)}, "_modified")
assert output_path == str(fcpxml.parent / "Projeto_modified.fcpxml")
def test_explicit_output_path_still_wins(self, project):
fcpxml, chosen = project
target = chosen / "nome escolhido.fcpxml"
_, output_path = _resolve_io_paths(
{"filepath": str(fcpxml), "output_dir": str(chosen), "output_path": str(target)}, "_voice_edit"
)
assert output_path == str(target)
def test_explicit_output_path_outside_the_anchor_is_rejected(self, project, tmp_path):
"""The anchor must keep constraining explicit paths, not just names."""
fcpxml, chosen = project
with pytest.raises(ValueError):
_resolve_io_paths(
{"filepath": str(fcpxml), "output_dir": str(chosen),
"output_path": str(tmp_path / "fora.fcpxml")}, "_voice_edit"
)
+83
View File
@@ -0,0 +1,83 @@
"""Tests for the last-project settings — the folder/file the app reopens with.
The config file (~/.fcp-mcp-server/config.json) is redirected to a tmp_path
so these never touch the developer's real settings.
"""
import json
import pytest
from fcpxml import model_manager
@pytest.fixture(autouse=True)
def isolated_config(tmp_path, monkeypatch):
"""Point model_manager's config file at a throwaway directory."""
monkeypatch.setattr(model_manager, "_CONFIG_DIR", tmp_path)
monkeypatch.setattr(model_manager, "_CONFIG_FILE", tmp_path / "config.json")
return tmp_path / "config.json"
class TestLoadProjectConfig:
def test_defaults_when_nothing_stored(self):
assert model_manager.load_project_config() == model_manager.DEFAULT_PROJECT_CONFIG
def test_non_dict_stored_value_falls_back(self, isolated_config):
isolated_config.write_text(json.dumps({"project": "nonsense"}))
assert model_manager.load_project_config() == model_manager.DEFAULT_PROJECT_CONFIG
def test_reads_back_what_was_stored(self, tmp_path):
folder = tmp_path / "03 - Mastopexia"
folder.mkdir()
model_manager.save_project_config(folder=str(folder))
assert model_manager.load_project_config()["folder"] == str(folder)
def test_path_that_no_longer_exists_comes_back_empty(self, isolated_config, tmp_path):
"""An unmounted volume must degrade to "nothing selected", not a dead path."""
isolated_config.write_text(json.dumps({"project": {"folder": str(tmp_path / "gone")}}))
assert model_manager.load_project_config()["folder"] == ""
def test_non_string_stored_value_is_ignored(self, isolated_config):
isolated_config.write_text(json.dumps({"project": {"folder": 42}}))
assert model_manager.load_project_config()["folder"] == ""
class TestSaveProjectConfig:
def test_omitted_field_keeps_its_current_value(self, tmp_path):
folder = tmp_path / "projeto"
folder.mkdir()
project = tmp_path / "projeto" / "Mastopexia.fcpxml"
project.write_text("<fcpxml/>")
model_manager.save_project_config(folder=str(folder), file=str(project))
model_manager.save_project_config(file=str(project))
assert model_manager.load_project_config()["folder"] == str(folder)
def test_empty_string_clears_a_field(self, tmp_path):
folder = tmp_path / "projeto"
folder.mkdir()
model_manager.save_project_config(folder=str(folder))
model_manager.save_project_config(folder="")
assert model_manager.load_project_config()["folder"] == ""
def test_home_relative_path_is_expanded(self, isolated_config):
model_manager.save_project_config(folder="~/Movies")
stored = json.loads(isolated_config.read_text())["project"]["folder"]
assert not stored.startswith("~")
def test_does_not_disturb_other_config_sections(self, isolated_config, tmp_path):
isolated_config.write_text(json.dumps({"language": "pt", "selected_model": "large-v3"}))
folder = tmp_path / "projeto"
folder.mkdir()
model_manager.save_project_config(folder=str(folder))
data = json.loads(isolated_config.read_text())
assert data["language"] == "pt"
assert data["selected_model"] == "large-v3"
def test_returns_the_merged_config(self, tmp_path):
folder = tmp_path / "projeto"
folder.mkdir()
assert model_manager.save_project_config(folder=str(folder)) == {
"folder": str(folder),
"file": "",
}
@@ -0,0 +1,167 @@
"""Tests for the refine_voice_timeline MCP tool.
The tool exists because emphasis is *relative*: cut the loudest moment of a
recording and every surviving score is still measured against something the
viewer never sees. These tests pin the re-normalization actually happening,
and the times staying in original source seconds so the result can be fed
straight back to apply_voice_actions.
"""
import json
import pytest
from tests.test_voice_features_tool import _write_silent_wav
from tests.test_voice_timeline_tool import _TRANSCRIPT, patched, wav # noqa: F401
_ = _write_silent_wav, _TRANSCRIPT # re-exported fixtures need the imports
async def _build(wav_path):
from server import handle_build_voice_timeline
await handle_build_voice_timeline({"media_path": str(wav_path)})
@pytest.fixture
def two_candidates(monkeypatch):
"""A transcript where BOTH lines carry a content word.
The shared fixture's first line is "isso e" — two function words, which
``suggest_zoom_windows`` skips by design, so it can never produce more
than one candidate to cap.
"""
import fcpxml.voice_timeline as vt
import server_tools._shared as _shared_mod
transcript = {
"language": "pt",
"duration": 4.0,
"text": "cirurgia rapida seguranca total",
"segments": [
{"text": "cirurgia rapida", "start": 0.0, "end": 1.0},
{"text": "seguranca total", "start": 2.0, "end": 4.0},
],
"words": [
{"word": "cirurgia", "start": 0.0, "end": 0.4, "confidence": 0.9},
{"word": "rapida", "start": 0.5, "end": 0.7, "confidence": 0.9},
{"word": "seguranca", "start": 2.0, "end": 2.9, "confidence": 0.9},
{"word": "total", "start": 3.0, "end": 3.5, "confidence": 0.9},
],
}
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: transcript)
monkeypatch.setattr(vt, "extract_pitch", lambda *a, **k: [(2.4, 260.0), (0.2, 120.0)])
monkeypatch.setattr(vt, "extract_energy", lambda *a, **k: [(2.4, 0.95), (0.2, 0.10)])
class TestRefineVoiceTimelineHandler:
async def test_requires_an_existing_timeline(self, wav): # noqa: F811
from server import handle_refine_voice_timeline
result = await handle_refine_voice_timeline({"media_path": str(wav), "cuts": []})
assert "build_voice_timeline" in result[0].text
async def test_rejects_disallowed_extension(self, tmp_path):
from server import handle_refine_voice_timeline
bad = tmp_path / "clip.txt"
bad.write_text("not audio")
with pytest.raises(ValueError):
await handle_refine_voice_timeline({"media_path": str(bad), "cuts": []})
async def test_compares_raw_against_survivors(self, wav, patched): # noqa: F811
from server import handle_refine_voice_timeline
await _build(wav)
result = await handle_refine_voice_timeline(
{"media_path": str(wav), "cuts": [{"start": 0.0, "end": 1.0}]}
)
text = result[0].text
assert "Survivors only" in text
assert "Average emphasis" in text
async def test_cut_words_are_excluded(self, wav, patched): # noqa: F811
from server import handle_refine_voice_timeline
await _build(wav)
result = await handle_refine_voice_timeline(
{"media_path": str(wav), "cuts": [{"start": 0.0, "end": 1.0}], "save": True}
)
assert "_voice_timeline_refined.json" in result[0].text
data = json.loads(
(wav.parent / "clip_voice_timeline_refined.json").read_text(encoding="utf-8")
)
words = [w["text"] for s in data["segments"] for w in s["words"]]
assert "isso" not in words
assert "seguranca" in words
async def test_times_stay_in_original_source_seconds(self, wav, patched): # noqa: F811
"""A cut at the head must NOT slide the survivors back to zero."""
from server import handle_refine_voice_timeline
await _build(wav)
await handle_refine_voice_timeline(
{"media_path": str(wav), "cuts": [{"start": 0.0, "end": 1.0}], "save": True}
)
data = json.loads(
(wav.parent / "clip_voice_timeline_refined.json").read_text(encoding="utf-8")
)
first = data["segments"][0]["words"][0]
assert first["start"] == pytest.approx(2.0)
async def test_proposes_zoom_candidates(self, wav, patched): # noqa: F811
from server import handle_refine_voice_timeline
await _build(wav)
result = await handle_refine_voice_timeline({"media_path": str(wav), "cuts": []})
assert "Zoom Candidates" in result[0].text
async def test_max_zooms_caps_the_list(self, wav, two_candidates): # noqa: F811
from server import handle_refine_voice_timeline
await _build(wav)
async def zoom_rows(**extra):
result = await handle_refine_voice_timeline(
{"media_path": str(wav), "cuts": [], "min_gap": 0.0, **extra}
)
section = result[0].text.split("## Zoom Candidates", 1)[1]
return [
ln for ln in section.splitlines()
if ln.startswith("| ") and ln.rstrip().endswith("|") and "Start" not in ln
]
assert len(await zoom_rows()) == 2
assert len(await zoom_rows(max_zooms=1)) == 1
async def test_malformed_cut_is_reported_not_raised(self, wav, patched): # noqa: F811
from server import handle_refine_voice_timeline
await _build(wav)
result = await handle_refine_voice_timeline(
{"media_path": str(wav), "cuts": [{"start": 3.0, "end": 1.0}]}
)
text = result[0].text
assert "Rejected cuts" in text
assert "must be after start" in text
async def test_cutting_everything_says_so(self, wav, patched): # noqa: F811
from server import handle_refine_voice_timeline
await _build(wav)
result = await handle_refine_voice_timeline(
{"media_path": str(wav), "cuts": [{"start": 0.0, "end": 60.0}]}
)
assert "removed every word" in result[0].text
class TestRefineVoiceTimelineRegistration:
async def test_tool_is_listed(self):
from server import list_tools
assert "refine_voice_timeline" in {t.name for t in await list_tools()}
async def test_tool_is_dispatched(self):
from server import TOOL_HANDLERS, handle_refine_voice_timeline
assert TOOL_HANDLERS["refine_voice_timeline"] is handle_refine_voice_timeline
+227
View File
@@ -0,0 +1,227 @@
"""Tests for fcpxml/voice_actions.py — the decision contract.
Pure functions over untrusted input (a model's decision list), so these
cover the rejection paths as carefully as the happy path.
"""
import pytest
from fcpxml.voice_actions import (
MAX_TEXT_LENGTH,
VoiceAction,
merge_cut_ranges,
parse_actions,
resolve_actions,
shift_after_cuts,
)
class TestParseActions:
def test_accepts_bare_list(self):
actions, errors = parse_actions([{"kind": "cut", "start": 1.0, "end": 2.0}])
assert len(actions) == 1 and errors == []
def test_accepts_actions_envelope(self):
actions, errors = parse_actions({"actions": [{"kind": "cut", "start": 1.0, "end": 2.0}]})
assert len(actions) == 1 and errors == []
def test_rejects_non_list(self):
actions, errors = parse_actions("cortar tudo")
assert actions == [] and len(errors) == 1
def test_one_bad_row_does_not_discard_the_good_ones(self):
actions, errors = parse_actions([
{"kind": "cut", "start": 1.0, "end": 2.0},
{"kind": "teleport", "start": 3.0, "end": 4.0},
{"kind": "zoom", "start": 5.0, "end": 6.0},
])
assert len(actions) == 2
assert len(errors) == 1 and "teleport" in errors[0]
def test_rejects_unknown_kind(self):
_, errors = parse_actions([{"kind": "explode", "start": 0.0, "end": 1.0}])
assert "explode" in errors[0]
def test_rejects_non_numeric_times(self):
_, errors = parse_actions([{"kind": "cut", "start": "início", "end": 2.0}])
assert "numbers" in errors[0]
def test_rejects_negative_start(self):
_, errors = parse_actions([{"kind": "cut", "start": -1.0, "end": 2.0}])
assert "negative" in errors[0]
def test_rejects_end_before_start(self):
_, errors = parse_actions([{"kind": "cut", "start": 5.0, "end": 2.0}])
assert "must be after" in errors[0]
def test_rejects_zero_length(self):
_, errors = parse_actions([{"kind": "cut", "start": 2.0, "end": 2.0}])
assert errors
def test_rejects_row_that_is_not_an_object(self):
_, errors = parse_actions(["cortar aos 5s"])
assert "expected an object" in errors[0]
def test_kind_is_case_insensitive(self):
actions, _ = parse_actions([{"kind": "ZOOM", "start": 1.0, "end": 2.0}])
assert actions[0].kind == "zoom"
def test_preserves_reason_and_speaker(self):
actions, _ = parse_actions([
{"kind": "zoom", "start": 1.0, "end": 2.0,
"reason": "argumento central", "speaker": "SPEAKER_01"}
])
assert actions[0].reason == "argumento central"
assert actions[0].speaker == "SPEAKER_01"
class TestZoomValidation:
def test_default_scale_when_absent(self):
actions, _ = parse_actions([{"kind": "zoom", "start": 1.0, "end": 2.0}])
assert actions[0].params["scale"] == 1.3
def test_rejects_scale_below_one(self):
_, errors = parse_actions([
{"kind": "zoom", "start": 1.0, "end": 2.0, "params": {"scale": 0.5}}
])
assert "outside" in errors[0]
def test_rejects_absurd_scale(self):
_, errors = parse_actions([
{"kind": "zoom", "start": 1.0, "end": 2.0, "params": {"scale": 50}}
])
assert "outside" in errors[0]
def test_rejects_non_numeric_scale(self):
_, errors = parse_actions([
{"kind": "zoom", "start": 1.0, "end": 2.0, "params": {"scale": "muito"}}
])
assert "must be a number" in errors[0]
class TestTextValidation:
def test_requires_content(self):
_, errors = parse_actions([{"kind": "text", "start": 1.0, "end": 2.0}])
assert "params.content" in errors[0]
def test_rejects_blank_content(self):
_, errors = parse_actions([
{"kind": "text", "start": 1.0, "end": 2.0, "params": {"content": " "}}
])
assert "params.content" in errors[0]
def test_truncates_overlong_content(self):
actions, _ = parse_actions([
{"kind": "text", "start": 1.0, "end": 2.0, "params": {"content": "A" * 500}}
])
assert len(actions[0].params["content"]) == MAX_TEXT_LENGTH
class TestMergeCutRanges:
def test_sorts_and_merges_overlaps(self):
actions = [
VoiceAction("cut", 5.0, 7.0),
VoiceAction("cut", 1.0, 3.0),
VoiceAction("cut", 2.0, 4.0),
]
assert merge_cut_ranges(actions) == [(1.0, 4.0), (5.0, 7.0)]
def test_merges_touching_ranges(self):
actions = [VoiceAction("cut", 1.0, 2.0), VoiceAction("cut", 2.0, 3.0)]
assert merge_cut_ranges(actions) == [(1.0, 3.0)]
def test_ignores_non_cut_actions(self):
assert merge_cut_ranges([VoiceAction("zoom", 1.0, 2.0)]) == []
class TestShiftAfterCuts:
def test_time_before_any_cut_is_unchanged(self):
assert shift_after_cuts(0.5, [(2.0, 4.0)]) == 0.5
def test_time_after_a_cut_moves_earlier(self):
assert shift_after_cuts(6.0, [(2.0, 4.0)]) == pytest.approx(4.0)
def test_time_inside_a_cut_is_dropped(self):
assert shift_after_cuts(3.0, [(2.0, 4.0)]) is None
def test_multiple_cuts_accumulate(self):
cuts = [(1.0, 2.0), (5.0, 7.0)]
assert shift_after_cuts(10.0, cuts) == pytest.approx(7.0)
def test_no_cuts_is_identity(self):
assert shift_after_cuts(3.0, []) == 3.0
def test_boundary_start_of_cut_is_inside(self):
assert shift_after_cuts(2.0, [(2.0, 4.0)]) is None
def test_boundary_end_of_cut_survives(self):
assert shift_after_cuts(4.0, [(2.0, 4.0)]) == pytest.approx(2.0)
class TestResolveActions:
def test_zoom_after_a_cut_is_moved_earlier(self):
actions = [VoiceAction("cut", 2.0, 4.0), VoiceAction("zoom", 6.0, 7.0)]
cuts, placed, dropped = resolve_actions(actions)
assert cuts == [(2.0, 4.0)]
assert dropped == []
assert placed[0].start == pytest.approx(4.0)
assert placed[0].end == pytest.approx(5.0)
def test_zoom_inside_a_cut_is_dropped_not_slid(self):
actions = [VoiceAction("cut", 2.0, 8.0), VoiceAction("zoom", 3.0, 4.0)]
_, placed, dropped = resolve_actions(actions)
assert placed == []
assert len(dropped) == 1
def test_zoom_straddling_a_cut_edge_is_dropped(self):
actions = [VoiceAction("cut", 4.0, 8.0), VoiceAction("zoom", 3.0, 5.0)]
_, placed, dropped = resolve_actions(actions)
assert placed == [] and len(dropped) == 1
def test_cuts_are_not_returned_as_placed(self):
_, placed, _ = resolve_actions([VoiceAction("cut", 1.0, 2.0)])
assert placed == []
def test_without_cuts_everything_keeps_its_time(self):
actions = [VoiceAction("zoom", 3.0, 4.0), VoiceAction("text", 5.0, 6.0)]
cuts, placed, dropped = resolve_actions(actions)
assert cuts == [] and dropped == []
assert [(a.start, a.end) for a in placed] == [(3.0, 4.0), (5.0, 6.0)]
def test_params_survive_the_shift(self):
actions = [
VoiceAction("cut", 1.0, 2.0),
VoiceAction("text", 5.0, 6.0, params={"content": "SEGURANÇA"}),
]
_, placed, _ = resolve_actions(actions)
assert placed[0].params["content"] == "SEGURANÇA"
class TestMarkersSurviveCutEdges:
"""A marker is a point in time, not a span. The markers worth keeping are
precisely the ones flagging a join, which sit against a cut edge — so
requiring their nominal end to survive would drop exactly those."""
def test_marker_at_a_cut_edge_survives(self):
actions = [VoiceAction("cut", 21.9, 127.6), VoiceAction("marker", 21.85, 22.0)]
_, placed, dropped = resolve_actions(actions)
assert dropped == []
assert placed[0].kind == "marker"
assert placed[0].start == pytest.approx(21.85)
def test_marker_inside_removed_material_is_still_dropped(self):
actions = [VoiceAction("cut", 20.0, 100.0), VoiceAction("marker", 50.0, 50.2)]
_, placed, dropped = resolve_actions(actions)
assert placed == [] and len(dropped) == 1
def test_marker_keeps_its_length_after_shifting(self):
actions = [VoiceAction("cut", 0.0, 10.0), VoiceAction("marker", 20.0, 20.5)]
_, placed, _ = resolve_actions(actions)
assert placed[0].start == pytest.approx(10.0)
assert placed[0].duration == pytest.approx(0.5)
def test_zoom_straddling_an_edge_is_still_dropped(self):
"""Only markers get the point-action treatment — a span must fit."""
actions = [VoiceAction("cut", 21.9, 127.6), VoiceAction("zoom", 21.0, 22.5)]
_, placed, dropped = resolve_actions(actions)
assert placed == [] and len(dropped) == 1
+267
View File
@@ -0,0 +1,267 @@
"""Tests for the apply_voice_actions MCP tool — decisions -> real FCPXML.
Uses an inline fixture rather than examples/sample.fcpxml so the source
windows are explicit and the assertions can be exact.
"""
import shutil
import pytest
from fcpxml.safe_xml import safe_parse
_FIXTURE = """<?xml version="1.0" encoding="UTF-8"?>
<fcpxml version="1.13">
<resources>
<format id="r1" name="FFVideoFormat1080p30" frameDuration="100/3000s" width="1920" height="1080"/>
<asset id="a1" name="entrevista" start="0s" duration="600/30s" hasVideo="1" hasAudio="1" format="r1">
<media-rep kind="original-media" src="file:///media/entrevista.mov"/>
</asset>
</resources>
<library>
<event name="Ev">
<project name="Proj">
<sequence format="r1" duration="600/30s" tcStart="0s">
<spine>
<asset-clip name="entrevista" ref="a1" offset="0s" start="0s" duration="600/30s"/>
</spine>
</sequence>
</project>
</event>
</library>
</fcpxml>
"""
@pytest.fixture
def project(tmp_path):
path = tmp_path / "proj.fcpxml"
path.write_text(_FIXTURE)
return path
def _out(project):
return project.with_name("proj_voice_edit.fcpxml")
class TestApplyVoiceActionsHandler:
async def test_no_actions_reports_instead_of_writing(self, project):
from server import handle_apply_voice_actions
result = await handle_apply_voice_actions({"filepath": str(project)})
assert "nothing to apply" in result[0].text.lower()
assert not _out(project).exists()
async def test_all_invalid_actions_writes_nothing(self, project):
from server import handle_apply_voice_actions
result = await handle_apply_voice_actions({
"filepath": str(project),
"actions": [{"kind": "teleport", "start": 1.0, "end": 2.0}],
})
assert "No valid actions" in result[0].text
assert "teleport" in result[0].text
assert not _out(project).exists()
async def test_applies_zoom_into_the_hosting_clip(self, project):
from server import handle_apply_voice_actions
result = await handle_apply_voice_actions({
"filepath": str(project),
"actions": [{
"kind": "zoom", "start": 5.0, "end": 6.0,
"params": {"scale": 1.4}, "reason": "argumento central",
}],
})
assert "argumento central" in result[0].text
tree = safe_parse(str(_out(project)))
transforms = tree.getroot().findall(".//adjust-transform")
assert len(transforms) == 1
async def test_applies_text_title(self, project):
from server import handle_apply_voice_actions
await handle_apply_voice_actions({
"filepath": str(project),
"actions": [{
"kind": "text", "start": 3.0, "end": 4.0,
"params": {"content": "SEGURANÇA"},
}],
})
titles = safe_parse(str(_out(project))).getroot().findall(".//title")
assert len(titles) == 1
texts = [t.text for t in titles[0].iter() if t.text]
assert any("SEGURANÇA" in t for t in texts)
async def test_applies_marker(self, project):
from server import handle_apply_voice_actions
await handle_apply_voice_actions({
"filepath": str(project),
"actions": [{
"kind": "marker", "start": 2.0, "end": 2.5, "reason": "virada",
}],
})
markers = safe_parse(str(_out(project))).getroot().findall(".//marker")
assert len(markers) == 1
async def test_cut_shortens_the_timeline(self, project):
from server import handle_apply_voice_actions
before = safe_parse(str(project)).getroot().find(".//asset-clip").get("duration")
result = await handle_apply_voice_actions({
"filepath": str(project),
"actions": [{"kind": "cut", "start": 5.0, "end": 10.0, "reason": "digressão"}],
})
assert "Cuts applied" in result[0].text
clips = safe_parse(str(_out(project))).getroot().findall(".//asset-clip")
total = sum(
int(c.get("duration").split("/")[0]) / int(c.get("duration").split("/")[1].rstrip("s"))
for c in clips
)
original = int(before.split("/")[0]) / int(before.split("/")[1].rstrip("s"))
assert total < original
async def test_action_inside_a_cut_is_dropped_and_reported(self, project):
from server import handle_apply_voice_actions
result = await handle_apply_voice_actions({
"filepath": str(project),
"actions": [
{"kind": "cut", "start": 4.0, "end": 12.0},
{"kind": "zoom", "start": 6.0, "end": 7.0, "reason": "some no material cortado"},
],
})
text = result[0].text
assert "Dropped" in text
assert "some no material cortado" in text
assert safe_parse(str(_out(project))).getroot().findall(".//adjust-transform") == []
async def test_action_beyond_the_media_is_reported_not_silent(self, project):
from server import handle_apply_voice_actions
result = await handle_apply_voice_actions({
"filepath": str(project),
"actions": [{"kind": "zoom", "start": 500.0, "end": 501.0}],
})
assert "Not placed" in result[0].text
assert "outside the edited timeline" in result[0].text
async def test_original_file_is_untouched(self, project):
from server import handle_apply_voice_actions
original = project.read_text()
await handle_apply_voice_actions({
"filepath": str(project),
"actions": [{"kind": "zoom", "start": 5.0, "end": 6.0}],
})
assert project.read_text() == original
async def test_mixed_valid_and_invalid_applies_the_valid_ones(self, project):
from server import handle_apply_voice_actions
result = await handle_apply_voice_actions({
"filepath": str(project),
"actions": [
{"kind": "zoom", "start": 5.0, "end": 6.0},
{"kind": "zoom", "start": 8.0, "end": 9.0, "params": {"scale": 99}},
],
})
assert "Rejected" in result[0].text
assert len(safe_parse(str(_out(project))).getroot().findall(".//adjust-transform")) == 1
async def test_respects_explicit_output_path(self, project, tmp_path):
from server import handle_apply_voice_actions
target = tmp_path / "custom.fcpxml"
await handle_apply_voice_actions({
"filepath": str(project),
"actions": [{"kind": "marker", "start": 1.0, "end": 2.0}],
"output_path": str(target),
})
assert target.exists()
async def test_output_is_valid_parseable_fcpxml(self, project):
from server import handle_apply_voice_actions
await handle_apply_voice_actions({
"filepath": str(project),
"actions": [
{"kind": "cut", "start": 2.0, "end": 4.0},
{"kind": "zoom", "start": 10.0, "end": 11.0},
{"kind": "text", "start": 12.0, "end": 13.0, "params": {"content": "OK"}},
],
})
root = safe_parse(str(_out(project))).getroot()
assert root.tag == "fcpxml"
assert root.find(".//spine") is not None
class TestSampleFixtureStillParses:
"""The applier must not corrupt a real-world document."""
async def test_real_sample_survives_a_zoom(self, tmp_path):
from server import handle_apply_voice_actions
src = "examples/sample.fcpxml"
target = tmp_path / "sample.fcpxml"
shutil.copy(src, target)
result = await handle_apply_voice_actions({
"filepath": str(target),
"actions": [{"kind": "marker", "start": 1.0, "end": 2.0, "reason": "teste"}],
})
assert "Voice Actions Applied" in result[0].text
class TestPlacementsLandOnTheRightPieceAfterCuts:
"""Cutting splits a clip into same-named pieces. Placing before cutting
duplicated the zoom onto every piece and lost markers outright; a
name-based lookup afterwards would always resolve to the first piece.
Both bugs shipped past the suite and only showed up on real footage."""
async def test_zoom_lands_on_exactly_one_piece(self, project):
from server import handle_apply_voice_actions
await handle_apply_voice_actions({
"filepath": str(project),
"actions": [
{"kind": "cut", "start": 2.0, "end": 5.0},
{"kind": "cut", "start": 8.0, "end": 12.0},
{"kind": "zoom", "start": 15.0, "end": 16.0, "params": {"scale": 1.2}},
],
})
root = safe_parse(str(_out(project))).getroot()
assert len(root.findall(".//spine/asset-clip")) == 3
assert len(root.findall(".//adjust-transform")) == 1
async def test_zoom_lands_on_the_last_piece_not_the_first(self, project):
from server import handle_apply_voice_actions
await handle_apply_voice_actions({
"filepath": str(project),
"actions": [
{"kind": "cut", "start": 2.0, "end": 5.0},
{"kind": "zoom", "start": 15.0, "end": 16.0},
],
})
clips = safe_parse(str(_out(project))).getroot().findall(".//spine/asset-clip")
# the zoom is at 15s source -> 12s after a 3s cut, i.e. the 2nd piece
assert clips[0].find("adjust-transform") is None
assert clips[1].find("adjust-transform") is not None
async def test_markers_survive_the_cut(self, project):
from server import handle_apply_voice_actions
result = await handle_apply_voice_actions({
"filepath": str(project),
"actions": [
{"kind": "cut", "start": 5.0, "end": 10.0},
{"kind": "marker", "start": 4.9, "end": 5.05, "reason": "emenda"},
{"kind": "marker", "start": 15.0, "end": 15.2, "reason": "depois"},
],
})
assert "Dropped" not in result[0].text
markers = safe_parse(str(_out(project))).getroot().findall(".//marker")
assert len(markers) == 2
+111
View File
@@ -0,0 +1,111 @@
"""Tests for the Voice Analysis settings — persistence + MCP config tools.
The config file (~/.fcp-mcp-server/config.json) is redirected to a tmp_path
so these never touch the developer's real settings.
"""
import json
import pytest
from fcpxml import model_manager
@pytest.fixture(autouse=True)
def isolated_config(tmp_path, monkeypatch):
"""Point model_manager's config file at a throwaway directory."""
monkeypatch.setattr(model_manager, "_CONFIG_DIR", tmp_path)
monkeypatch.setattr(model_manager, "_CONFIG_FILE", tmp_path / "config.json")
return tmp_path / "config.json"
class TestLoadVoiceAnalysisConfig:
def test_defaults_when_nothing_stored(self):
cfg = model_manager.load_voice_analysis_config()
assert cfg == model_manager.DEFAULT_VOICE_ANALYSIS_CONFIG
def test_defaults_are_not_shared_mutable_state(self):
cfg = model_manager.load_voice_analysis_config()
cfg["emphasis_weights"]["energy"] = 0.99
fresh = model_manager.load_voice_analysis_config()
assert fresh["emphasis_weights"]["energy"] == 0.30
def test_malformed_stored_values_fall_back_to_defaults(self, isolated_config):
isolated_config.write_text(json.dumps({"voice_analysis": {"energy_threshold": "loud"}}))
cfg = model_manager.load_voice_analysis_config()
assert cfg["energy_threshold"] == 0.5
def test_non_dict_stored_value_falls_back(self, isolated_config):
isolated_config.write_text(json.dumps({"voice_analysis": "nonsense"}))
assert model_manager.load_voice_analysis_config() == model_manager.DEFAULT_VOICE_ANALYSIS_CONFIG
def test_thresholds_are_clamped_to_unit_range(self, isolated_config):
isolated_config.write_text(
json.dumps({"voice_analysis": {"energy_threshold": 5.0, "emphasis_floor": -2.0}})
)
cfg = model_manager.load_voice_analysis_config()
assert cfg["energy_threshold"] == 1.0
assert cfg["emphasis_floor"] == 0.0
class TestSaveVoiceAnalysisConfig:
def test_saves_and_reloads(self):
model_manager.save_voice_analysis_config(energy_threshold=0.7, emotion_enabled=True)
cfg = model_manager.load_voice_analysis_config()
assert cfg["energy_threshold"] == 0.7
assert cfg["emotion_enabled"] is True
def test_omitted_fields_keep_current_value(self):
model_manager.save_voice_analysis_config(energy_threshold=0.7)
model_manager.save_voice_analysis_config(emphasis_floor=0.9)
cfg = model_manager.load_voice_analysis_config()
assert cfg["energy_threshold"] == 0.7
assert cfg["emphasis_floor"] == 0.9
def test_partial_weight_update_keeps_other_weights(self):
model_manager.save_voice_analysis_config(emphasis_weights={"energy": 0.55})
weights = model_manager.load_voice_analysis_config()["emphasis_weights"]
assert weights["energy"] == 0.55
assert weights["pitch_variation"] == 0.25
def test_unknown_weight_key_is_ignored(self):
model_manager.save_voice_analysis_config(emphasis_weights={"loudness": 9.0})
weights = model_manager.load_voice_analysis_config()["emphasis_weights"]
assert "loudness" not in weights
def test_does_not_clobber_unrelated_config_keys(self):
model_manager.save_hf_token("tok123")
model_manager.save_voice_analysis_config(energy_threshold=0.7)
assert model_manager.load_hf_token() == "tok123"
def test_returns_full_merged_config(self):
returned = model_manager.save_voice_analysis_config(energy_threshold=0.7)
assert returned == model_manager.load_voice_analysis_config()
class TestVoiceAnalysisConfigTools:
async def test_get_reports_current_settings(self):
from server import handle_get_voice_analysis_config
result = await handle_get_voice_analysis_config({})
text = result[0].text
assert "Voice Analysis Settings" in text
assert "Emphasis Weights" in text
async def test_save_persists_and_echoes_back(self):
from server import handle_save_voice_analysis_config
result = await handle_save_voice_analysis_config(
{"energy_threshold": 0.8, "emotion_enabled": True}
)
assert "0.80" in result[0].text
cfg = model_manager.load_voice_analysis_config()
assert cfg["energy_threshold"] == 0.8
assert cfg["emotion_enabled"] is True
async def test_save_with_no_arguments_is_a_noop(self):
from server import handle_save_voice_analysis_config
before = model_manager.load_voice_analysis_config()
await handle_save_voice_analysis_config({})
assert model_manager.load_voice_analysis_config() == before
+137
View File
@@ -0,0 +1,137 @@
"""Tests for fcpxml/voice_features.py — acoustic features.
The pure helpers (speech rate, pauses, window averaging) need no audio.
The librosa-backed extractors are skipped when the optional [intelligence]
extra is absent, matching the pattern in test_media_intel.py.
"""
import math
import struct
import wave
import pytest
from fcpxml.voice_features import (
compute_pauses,
compute_speech_rate,
extract_energy,
extract_pitch,
features_capability,
word_pitch_energy,
)
try:
import librosa # noqa: F401
LIBROSA = True
except ImportError:
LIBROSA = False
def _write_tone_wav(path: str, hz: float = 220.0, seconds: float = 2.0, rate: int = 22050) -> None:
n = int(rate * seconds)
frames = [int(20000 * math.sin(2 * math.pi * hz * i / rate)) for i in range(n)]
with wave.open(path, "w") as f:
f.setnchannels(1)
f.setsampwidth(2)
f.setframerate(rate)
f.writeframes(struct.pack("<%dh" % n, *frames))
class TestComputePauses:
def test_first_word_pause_is_time_from_zero(self):
words = [{"start": 1.5, "end": 2.0}]
assert compute_pauses(words) == [1.5]
def test_gap_between_words(self):
words = [{"start": 0.0, "end": 1.0}, {"start": 2.5, "end": 3.0}]
assert compute_pauses(words) == [0.0, 1.5]
def test_overlapping_words_clamp_to_zero(self):
words = [{"start": 0.0, "end": 2.0}, {"start": 1.0, "end": 3.0}]
assert compute_pauses(words) == [0.0, 0.0]
def test_empty_words(self):
assert compute_pauses([]) == []
class TestComputeSpeechRate:
def test_rate_counts_words_in_trailing_window(self):
# 3 words within a 3s window -> 1.0 word/sec at the last one
words = [{"start": 0.0}, {"start": 1.0}, {"start": 2.0}]
rates = compute_speech_rate(words, window_seconds=3.0)
assert rates[-1] == pytest.approx(1.0)
def test_old_words_fall_out_of_window(self):
words = [{"start": 0.0}, {"start": 100.0}]
rates = compute_speech_rate(words, window_seconds=3.0)
# only the word itself is in range at t=100
assert rates[-1] == pytest.approx(1 / 3.0)
def test_zero_window_is_not_a_division_error(self):
assert compute_speech_rate([{"start": 0.0}], window_seconds=0.0) == [0.0]
def test_empty_words(self):
assert compute_speech_rate([]) == []
class TestWordPitchEnergy:
def test_averages_track_values_within_word_span(self):
words = [{"word": "a", "start": 0.0, "end": 1.0}]
pitch = [(0.0, 100.0), (0.5, 200.0), (5.0, 999.0)]
energy = [(0.0, 0.2), (1.0, 0.4)]
out = word_pitch_energy(words, pitch, energy)
assert out[0]["pitch_hz"] == pytest.approx(150.0)
assert out[0]["energy"] == pytest.approx(0.3)
def test_none_when_no_frames_in_span(self):
words = [{"word": "a", "start": 10.0, "end": 11.0}]
out = word_pitch_energy(words, [(0.0, 100.0)], [(0.0, 0.5)])
assert out[0]["pitch_hz"] is None
assert out[0]["energy"] is None
def test_none_tracks_degrade_gracefully(self):
out = word_pitch_energy([{"word": "a", "start": 0.0, "end": 1.0}], None, None)
assert out[0]["pitch_hz"] is None
assert out[0]["energy"] is None
def test_does_not_mutate_input(self):
words = [{"word": "a", "start": 0.0, "end": 1.0}]
word_pitch_energy(words, [(0.0, 100.0)], None)
assert "pitch_hz" not in words[0]
def test_word_shorter_than_hop_gets_none_not_crash(self):
"""A word briefer than the frame spacing may contain no frame at all."""
words = [{"word": "a", "start": 0.501, "end": 0.502}]
out = word_pitch_energy(words, [(0.0, 100.0), (1.0, 200.0)], None)
assert out[0]["pitch_hz"] is None
class TestExtractorsDegradeGracefully:
def test_missing_file_returns_none(self):
assert extract_pitch("/nonexistent/audio.wav") is None
assert extract_energy("/nonexistent/audio.wav") is None
@pytest.mark.skipif(not LIBROSA, reason="librosa not installed")
class TestExtractorsWithLibrosa:
def test_capability_is_available(self):
ok, _msg = features_capability()
assert ok is True
def test_extracts_pitch_of_known_tone(self, tmp_path):
wav = tmp_path / "tone.wav"
_write_tone_wav(str(wav), hz=220.0, seconds=2.0)
track = extract_pitch(str(wav))
assert track is not None and len(track) > 0
hz_values = sorted(hz for _t, hz in track)
median = hz_values[len(hz_values) // 2]
assert median == pytest.approx(220.0, rel=0.1)
def test_extracts_energy_track(self, tmp_path):
wav = tmp_path / "tone.wav"
_write_tone_wav(str(wav), seconds=1.0)
track = extract_energy(str(wav))
assert track is not None and len(track) > 0
assert all(rms >= 0 for _t, rms in track)
assert max(rms for _t, rms in track) > 0
+147
View File
@@ -0,0 +1,147 @@
"""Tests for the analyze_voice_features MCP tool.
librosa and Whisper are monkeypatched so these run without the optional
extras, matching the pattern used by TestDetectBeatsHandler.
"""
import json
import struct
import wave
import pytest
def _write_silent_wav(path: str, seconds: float = 2.0) -> None:
n = int(44100 * seconds)
with wave.open(path, "w") as f:
f.setnchannels(1)
f.setsampwidth(2)
f.setframerate(44100)
f.writeframes(struct.pack("<%dh" % n, *([0] * n)))
_FAKE_TRANSCRIPT = {
"language": "pt",
"duration": 3.0,
"text": "isso e seguranca",
"segments": [{"text": "isso e seguranca", "start": 0.0, "end": 3.0}],
"words": [
{"word": "isso", "start": 0.0, "end": 0.4, "confidence": 0.9},
{"word": "e", "start": 0.5, "end": 0.7, "confidence": 0.9},
{"word": "seguranca", "start": 2.0, "end": 2.9, "confidence": 0.9},
],
}
@pytest.fixture
def wav(tmp_path):
path = tmp_path / "clip.wav"
_write_silent_wav(str(path))
return path
@pytest.fixture
def patched_analysis(monkeypatch):
"""Make the tool's transcription + librosa extractors deterministic."""
import server_tools._shared as _shared_mod
import server_tools.voice as server_mod
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: _FAKE_TRANSCRIPT)
monkeypatch.setattr(server_mod, "features_capability", lambda: (True, "ok"))
# "seguranca" (2.0-2.9s) is the loud, high-pitched, emphatic word
monkeypatch.setattr(
server_mod,
"extract_pitch",
lambda *a, **k: [(0.2, 120.0), (0.6, 118.0), (2.4, 260.0)],
)
monkeypatch.setattr(
server_mod,
"extract_energy",
lambda *a, **k: [(0.2, 0.10), (0.6, 0.12), (2.4, 0.95)],
)
class TestAnalyzeVoiceFeaturesHandler:
async def test_reports_when_librosa_unavailable(self, wav, monkeypatch):
import server_tools.voice as server_mod
from server import handle_analyze_voice_features
monkeypatch.setattr(
server_mod, "features_capability", lambda: (False, "componente librosa ausente.")
)
result = await handle_analyze_voice_features({"media_path": str(wav)})
assert "librosa" in result[0].text.lower()
async def test_rejects_disallowed_extension(self, tmp_path):
from server import handle_analyze_voice_features
bad = tmp_path / "clip.txt"
bad.write_text("not audio")
with pytest.raises(ValueError):
await handle_analyze_voice_features({"media_path": str(bad)})
async def test_writes_features_json_with_emphasis_per_word(self, wav, patched_analysis):
from server import handle_analyze_voice_features
result = await handle_analyze_voice_features({"media_path": str(wav)})
text = result[0].text
json_path = wav.parent / "clip_voice_features.json"
assert str(json_path) in text
data = json.loads(json_path.read_text())
assert len(data["words"]) == 3
assert all("emphasis" in w for w in data["words"])
assert all(0.0 <= w["emphasis"] <= 1.0 for w in data["words"])
async def test_loudest_word_scores_highest_emphasis(self, wav, patched_analysis):
from server import handle_analyze_voice_features
await handle_analyze_voice_features({"media_path": str(wav)})
data = json.loads((wav.parent / "clip_voice_features.json").read_text())
by_word = {w["word"]: w["emphasis"] for w in data["words"]}
assert by_word["seguranca"] > by_word["isso"]
assert by_word["seguranca"] > by_word["e"]
async def test_persisted_config_is_embedded_in_output(self, wav, patched_analysis):
from server import handle_analyze_voice_features
await handle_analyze_voice_features({"media_path": str(wav)})
data = json.loads((wav.parent / "clip_voice_features.json").read_text())
assert "energy_threshold" in data["config"]
assert "emphasis_weights" in data["config"]
async def test_empty_transcript_reports_instead_of_crashing(self, wav, monkeypatch):
import server_tools._shared as _shared_mod
import server_tools.voice as server_mod
from server import handle_analyze_voice_features
monkeypatch.setattr(server_mod, "features_capability", lambda: (True, "ok"))
monkeypatch.setattr(
_shared_mod, "transcribe", lambda *a, **k: {**_FAKE_TRANSCRIPT, "words": []}
)
result = await handle_analyze_voice_features({"media_path": str(wav)})
assert "no words" in result[0].text.lower()
async def test_untranscribable_media_reports_install_hint(self, wav, monkeypatch):
import server_tools._shared as _shared_mod
import server_tools.voice as server_mod
from server import handle_analyze_voice_features
monkeypatch.setattr(server_mod, "features_capability", lambda: (True, "ok"))
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: None)
result = await handle_analyze_voice_features({"media_path": str(wav)})
assert "faster-whisper" in result[0].text
async def test_missing_pitch_track_degrades_without_crashing(self, wav, monkeypatch):
import server_tools._shared as _shared_mod
import server_tools.voice as server_mod
from server import handle_analyze_voice_features
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: _FAKE_TRANSCRIPT)
monkeypatch.setattr(server_mod, "features_capability", lambda: (True, "ok"))
monkeypatch.setattr(server_mod, "extract_pitch", lambda *a, **k: None)
monkeypatch.setattr(server_mod, "extract_energy", lambda *a, **k: None)
result = await handle_analyze_voice_features({"media_path": str(wav)})
assert "Voice Feature Analysis" in result[0].text
data = json.loads((wav.parent / "clip_voice_features.json").read_text())
assert all(w["emphasis"] >= 0.0 for w in data["words"])
+338
View File
@@ -0,0 +1,338 @@
"""Tests for fcpxml/voice_timeline.py — the consolidated AI-readable timeline.
The document's shape is the contract downstream consumers (rules engine, a
model reading the JSON) rely on, so these tests pin the shape as much as
the values — including that it survives every analysis layer being absent.
"""
import json
import pytest
from fcpxml.voice_timeline import (
VOICE_TIMELINE_VERSION,
build_voice_timeline,
enrich_words,
load_voice_timeline,
save_voice_timeline,
voice_timeline_path,
)
_TRANSCRIPT = {
"language": "pt",
"duration": 4.0,
"text": "isso e seguranca total",
"segments": [
{"text": "isso e", "start": 0.0, "end": 1.0},
{"text": "seguranca total", "start": 2.0, "end": 4.0},
],
"words": [
{"word": "isso", "start": 0.0, "end": 0.4, "confidence": 0.9},
{"word": "e", "start": 0.5, "end": 0.7, "confidence": 0.9},
{"word": "seguranca", "start": 2.0, "end": 2.9, "confidence": 0.9},
{"word": "total", "start": 3.0, "end": 3.5, "confidence": 0.9},
],
}
# "seguranca" is the loud, high-pitched moment
_PITCH = [(0.2, 120.0), (0.6, 118.0), (2.4, 260.0), (3.2, 130.0)]
_ENERGY = [(0.2, 0.10), (0.6, 0.12), (2.4, 0.95), (3.2, 0.20)]
class TestEnrichWords:
def test_normalizes_energy_against_loudest_word(self):
enriched = enrich_words(_TRANSCRIPT["words"], _PITCH, _ENERGY)
loudest = max(enriched, key=lambda w: w["energy_norm"])
assert loudest["word"] == "seguranca"
assert loudest["energy_norm"] == pytest.approx(1.0)
def test_all_values_stay_within_unit_range(self):
enriched = enrich_words(_TRANSCRIPT["words"], _PITCH, _ENERGY)
for w in enriched:
for key in ("energy_norm", "pitch_delta", "rate_delta", "emphasis"):
assert 0.0 <= w[key] <= 1.0, f"{key} out of range on {w['word']}"
def test_empty_words_returns_empty(self):
assert enrich_words([], _PITCH, _ENERGY) == []
def test_missing_tracks_give_zero_not_crash(self):
enriched = enrich_words(_TRANSCRIPT["words"], None, None)
assert all(w["energy_norm"] == 0.0 for w in enriched)
assert all(w["pitch_delta"] == 0.0 for w in enriched)
class TestBuildVoiceTimeline:
@pytest.fixture
def timeline(self, monkeypatch):
import fcpxml.voice_timeline as vt
monkeypatch.setattr(vt, "extract_pitch", lambda *a, **k: _PITCH)
monkeypatch.setattr(vt, "extract_energy", lambda *a, **k: _ENERGY)
return build_voice_timeline("/tmp/clip.wav", _TRANSCRIPT)
def test_document_has_all_top_level_layers(self, timeline):
for key in ("version", "source", "language", "scales", "summary", "speakers", "segments"):
assert key in timeline
assert timeline["version"] == VOICE_TIMELINE_VERSION
def test_scales_document_every_word_metric(self, timeline):
word = timeline["segments"][0]["words"][0]
for metric in timeline["scales"]["word"]:
assert metric in word, f"{metric} documented in scales but absent from words"
def test_scales_document_every_segment_metric(self, timeline):
segment = timeline["segments"][0]
for metric in timeline["scales"]["segment"]:
assert metric in segment, f"{metric} documented in scales but absent from segments"
def test_take_boundary_flags_a_long_gap(self, timeline):
# the fixture has a 1s gap between its two segments -> not a boundary
assert timeline["segments"][1]["gap_before"] > 0
assert timeline["segments"][1]["take_boundary"] is False
def test_summary_counts_match_the_detail(self, timeline):
summary = timeline["summary"]
assert summary["segment_count"] == len(timeline["segments"])
total_words = sum(len(s["words"]) for s in timeline["segments"])
assert summary["word_count"] == total_words
def test_words_are_grouped_under_their_segment(self, timeline):
first, second = timeline["segments"]
assert [w["text"] for w in first["words"]] == ["isso", "e"]
assert [w["text"] for w in second["words"]] == ["seguranca", "total"]
def test_segment_aggregates_reflect_their_words(self, timeline):
loud_segment = timeline["segments"][1]
quiet_segment = timeline["segments"][0]
assert loud_segment["avg_energy"] > quiet_segment["avg_energy"]
assert loud_segment["peak_emphasis"] >= max(w["emphasis"] for w in loud_segment["words"])
def test_peak_moments_are_sorted_by_emphasis(self, timeline):
peaks = timeline["summary"]["peak_moments"]
assert peaks == sorted(peaks, key=lambda m: m["emphasis"], reverse=True)
def test_defaults_to_single_speaker_without_token(self, timeline):
assert timeline["summary"]["speaker_count"] == 1
assert all(w["speaker"] == "SPEAKER_00" for s in timeline["segments"] for w in s["words"])
def test_is_json_serializable(self, timeline):
# the whole point is handing this to a model / writing it to disk
assert json.loads(json.dumps(timeline, ensure_ascii=False))["version"]
class TestDegradesWithoutAnalysisLayers:
def test_shape_survives_with_no_acoustics(self, monkeypatch):
import fcpxml.voice_timeline as vt
monkeypatch.setattr(vt, "extract_pitch", lambda *a, **k: None)
monkeypatch.setattr(vt, "extract_energy", lambda *a, **k: None)
timeline = build_voice_timeline("/tmp/clip.wav", _TRANSCRIPT)
assert timeline["summary"]["word_count"] == 4
assert timeline["summary"]["avg_emphasis"] >= 0.0
def test_empty_transcript_still_yields_valid_document(self, monkeypatch):
import fcpxml.voice_timeline as vt
monkeypatch.setattr(vt, "extract_pitch", lambda *a, **k: None)
monkeypatch.setattr(vt, "extract_energy", lambda *a, **k: None)
timeline = build_voice_timeline(
"/tmp/clip.wav", {"duration": 0.0, "segments": [], "words": []}
)
assert timeline["segments"] == []
assert timeline["summary"]["word_count"] == 0
assert timeline["summary"]["avg_emphasis"] == 0.0
def test_progress_callback_is_reported(self, monkeypatch):
import fcpxml.voice_timeline as vt
monkeypatch.setattr(vt, "extract_pitch", lambda *a, **k: None)
monkeypatch.setattr(vt, "extract_energy", lambda *a, **k: None)
seen: list[tuple[float, str]] = []
build_voice_timeline("/tmp/clip.wav", _TRANSCRIPT, progress_cb=lambda f, s: seen.append((f, s)))
assert seen and all(0.0 <= f <= 1.0 for f, _ in seen)
class TestPersistence:
def test_round_trip(self, tmp_path):
path = tmp_path / "clip_voice_timeline.json"
timeline = {"version": "1.0", "segments": [], "summary": {}}
save_voice_timeline(timeline, path)
assert load_voice_timeline(path) == timeline
def test_accented_text_stays_readable(self, tmp_path):
path = tmp_path / "t.json"
save_voice_timeline({"segments": [{"text": "segurança"}]}, path)
assert "segurança" in path.read_text(encoding="utf-8")
def test_missing_file_returns_none(self, tmp_path):
assert load_voice_timeline(tmp_path / "absent.json") is None
def test_malformed_json_returns_none(self, tmp_path):
path = tmp_path / "bad.json"
path.write_text("{not json")
assert load_voice_timeline(path) is None
def test_wrong_shape_returns_none(self, tmp_path):
path = tmp_path / "other.json"
path.write_text('{"something": "else"}')
assert load_voice_timeline(path) is None
def test_path_next_to_media_by_default(self):
assert voice_timeline_path("/media/clip.mov").name == "clip_voice_timeline.json"
def test_path_honours_output_dir(self, tmp_path):
path = voice_timeline_path("/media/clip.mov", output_dir=str(tmp_path))
assert path.parent == tmp_path
_RAW_WORDS = [
# a loud outlier that will be cut, plus quieter material that survives
{"text": "GRITO", "start": 1.0, "end": 1.5, "speaker": "SPEAKER_00",
"energy": 0.5, "pitch_delta": 0.5, "rate_delta": 0.0, "pause_before": 0.0,
"emphasis": 0.5, "energy_raw": 1.0, "pitch_hz": 300.0},
{"text": "mastopexia", "start": 10.0, "end": 10.8, "speaker": "SPEAKER_00",
"energy": 0.2, "pitch_delta": 0.1, "rate_delta": 0.0, "pause_before": 0.0,
"emphasis": 0.1, "energy_raw": 0.4, "pitch_hz": 190.0},
{"text": "a", "start": 11.0, "end": 11.1, "speaker": "SPEAKER_00",
"energy": 0.15, "pitch_delta": 0.05, "rate_delta": 0.0, "pause_before": 0.0,
"emphasis": 0.08, "energy_raw": 0.3, "pitch_hz": 185.0},
]
_RESTRICT_TIMELINE = {
"version": "1.0", "source": "x.mp4", "speakers": [],
"segments": [
{"start": 1.0, "end": 1.5, "speaker": "SPEAKER_00", "text": "GRITO",
"gap_before": 0.0, "take_boundary": False, "avg_energy": 0.5,
"peak_emphasis": 0.5, "words": [_RAW_WORDS[0]]},
{"start": 10.0, "end": 11.1, "speaker": "SPEAKER_00",
"text": "mastopexia a", "gap_before": 8.5, "take_boundary": True,
"avg_energy": 0.17, "peak_emphasis": 0.1, "words": _RAW_WORDS[1:]},
],
}
class TestRestrictToKept:
"""Emphasis is relative. Cut the loudest moment out and everything left
is still scored against something the viewer will never see, so the
surviving material has to be re-normalized on its own."""
def test_cut_words_are_dropped(self):
from fcpxml.voice_timeline import restrict_to_kept
r = restrict_to_kept(_RESTRICT_TIMELINE, [(0.0, 5.0)])
texts = [w["text"] for s in r["segments"] for w in s["words"]]
assert "GRITO" not in texts
assert "mastopexia" in texts
def test_survivors_are_rescored_against_each_other(self):
from fcpxml.voice_timeline import restrict_to_kept
r = restrict_to_kept(_RESTRICT_TIMELINE, [(0.0, 5.0)])
word = next(w for s in r["segments"] for w in s["words"] if w["text"] == "mastopexia")
# was 0.2 against the shout's 1.0; alone it becomes the loudest
assert word["energy"] == pytest.approx(1.0)
def test_empty_segments_are_removed(self):
from fcpxml.voice_timeline import restrict_to_kept
r = restrict_to_kept(_RESTRICT_TIMELINE, [(0.0, 5.0)])
assert len(r["segments"]) == 1
def test_no_cuts_keeps_everything(self):
from fcpxml.voice_timeline import restrict_to_kept
r = restrict_to_kept(_RESTRICT_TIMELINE, [])
assert sum(len(s["words"]) for s in r["segments"]) == 3
def test_times_stay_in_original_source_seconds(self):
from fcpxml.voice_timeline import restrict_to_kept
r = restrict_to_kept(_RESTRICT_TIMELINE, [(0.0, 5.0)])
assert r["segments"][0]["start"] == 10.0
class TestSuggestZoomWindows:
def test_skips_function_words(self):
from fcpxml.voice_timeline import restrict_to_kept, suggest_zoom_windows
r = restrict_to_kept(_RESTRICT_TIMELINE, [(0.0, 5.0)])
zooms = suggest_zoom_windows(r)
assert zooms and all(z["word"] != "a" for z in zooms)
def test_window_runs_from_the_word_to_the_end_of_its_line(self):
from fcpxml.voice_timeline import restrict_to_kept, suggest_zoom_windows
r = restrict_to_kept(_RESTRICT_TIMELINE, [(0.0, 5.0)])
z = suggest_zoom_windows(r)[0]
assert z["start"] == 10.0 and z["end"] == 11.1
def test_min_gap_keeps_zooms_apart(self):
from fcpxml.voice_timeline import suggest_zoom_windows
timeline = {"segments": [
{"start": t, "end": t + 1.0, "text": "linha",
"words": [{"text": "palavra", "start": t, "end": t + 0.5, "emphasis": 0.5 - i * 0.01}]}
for i, t in enumerate([0.0, 1.0, 2.0, 30.0])
]}
zooms = suggest_zoom_windows(timeline, min_gap=8.0)
assert len(zooms) == 2
def test_max_zooms_caps_the_result(self):
from fcpxml.voice_timeline import suggest_zoom_windows
timeline = {"segments": [
{"start": t, "end": t + 1.0, "text": "linha",
"words": [{"text": "palavra", "start": t, "end": t + 0.5, "emphasis": 0.5}]}
for t in [0.0, 20.0, 40.0, 60.0]
]}
assert len(suggest_zoom_windows(timeline, min_gap=8.0, max_zooms=2)) == 2
def test_results_are_in_chronological_order(self):
from fcpxml.voice_timeline import suggest_zoom_windows
timeline = {"segments": [
{"start": t, "end": t + 1.0, "text": "linha",
"words": [{"text": "palavra", "start": t, "end": t + 0.5, "emphasis": e}]}
for t, e in [(60.0, 0.9), (0.0, 0.5), (30.0, 0.7)]
]}
zooms = suggest_zoom_windows(timeline, min_gap=8.0)
assert [z["start"] for z in zooms] == sorted(z["start"] for z in zooms)
class TestSentenceEnd:
"""Transcription segments break on breath, not grammar — a sentence
routinely spans several. A zoom ending on a segment boundary releases
mid-thought, which is what makes a punch-in feel arbitrary."""
SEGS = [
{"start": 0.0, "end": 5.0, "text": "Aquela mama com um formato, que dá aquele ar",
"take_boundary": False},
{"start": 5.0, "end": 10.7, "text": "de elegância, isso é desejo de muitas mulheres, né?",
"take_boundary": False},
{"start": 11.0, "end": 14.0, "text": "Com o tempo, o corpo muda.", "take_boundary": False},
]
def test_extends_past_a_segment_that_does_not_end_a_sentence(self):
from fcpxml.voice_timeline import sentence_end
assert sentence_end(self.SEGS, 0) == 10.7
def test_stops_at_terminal_punctuation(self):
from fcpxml.voice_timeline import sentence_end
assert sentence_end(self.SEGS, 2) == 14.0
def test_never_runs_past_a_take_boundary(self):
from fcpxml.voice_timeline import sentence_end
segs = [
{"start": 0.0, "end": 5.0, "text": "frase sem fim", "take_boundary": False},
{"start": 12.0, "end": 15.0, "text": "outra tomada", "take_boundary": True},
]
assert sentence_end(segs, 0) == 5.0
def test_last_segment_without_punctuation_ends_at_itself(self):
from fcpxml.voice_timeline import sentence_end
segs = [{"start": 0.0, "end": 4.0, "text": "sem ponto final", "take_boundary": False}]
assert sentence_end(segs, 0) == 4.0
+107
View File
@@ -0,0 +1,107 @@
"""Tests for the build_voice_timeline MCP tool."""
import json
import pytest
from tests.test_voice_features_tool import _write_silent_wav
_TRANSCRIPT = {
"language": "pt",
"duration": 4.0,
"text": "isso e seguranca total",
"segments": [
{"text": "isso e", "start": 0.0, "end": 1.0},
{"text": "seguranca total", "start": 2.0, "end": 4.0},
],
"words": [
{"word": "isso", "start": 0.0, "end": 0.4, "confidence": 0.9},
{"word": "e", "start": 0.5, "end": 0.7, "confidence": 0.9},
{"word": "seguranca", "start": 2.0, "end": 2.9, "confidence": 0.9},
{"word": "total", "start": 3.0, "end": 3.5, "confidence": 0.9},
],
}
@pytest.fixture
def wav(tmp_path):
path = tmp_path / "clip.wav"
_write_silent_wav(str(path))
return path
@pytest.fixture
def patched(monkeypatch):
"""Deterministic transcription + acoustics, no optional extras needed."""
import fcpxml.voice_timeline as vt
import server_tools._shared as _shared_mod
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: _TRANSCRIPT)
monkeypatch.setattr(vt, "extract_pitch", lambda *a, **k: [(2.4, 260.0), (0.2, 120.0)])
monkeypatch.setattr(vt, "extract_energy", lambda *a, **k: [(2.4, 0.95), (0.2, 0.10)])
class TestBuildVoiceTimelineHandler:
async def test_writes_timeline_json(self, wav, patched):
from server import handle_build_voice_timeline
result = await handle_build_voice_timeline({"media_path": str(wav)})
text = result[0].text
json_path = wav.parent / "clip_voice_timeline.json"
assert str(json_path) in text
data = json.loads(json_path.read_text(encoding="utf-8"))
assert data["summary"]["word_count"] == 4
assert len(data["segments"]) == 2
async def test_reports_which_layers_ran(self, wav, patched):
from server import handle_build_voice_timeline
result = await handle_build_voice_timeline({"media_path": str(wav)})
text = result[0].text
assert "Analysis Layers" in text
assert "Transcript" in text
async def test_rejects_disallowed_extension(self, tmp_path):
from server import handle_build_voice_timeline
bad = tmp_path / "clip.txt"
bad.write_text("not audio")
with pytest.raises(ValueError):
await handle_build_voice_timeline({"media_path": str(bad)})
async def test_untranscribable_media_reports_hint(self, wav, monkeypatch):
import server_tools._shared as _shared_mod
from server import handle_build_voice_timeline
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: None)
result = await handle_build_voice_timeline({"media_path": str(wav)})
assert "faster-whisper" in result[0].text
async def test_uses_persisted_peak_settings(self, wav, patched, monkeypatch):
"""A wider percentile must surface more peak moments."""
import server_tools.voice as server_mod
from server import handle_build_voice_timeline
def config(percentile):
return {
"energy_threshold": 0.5,
"peak_percentile": percentile,
"emphasis_floor": 0.0,
"emphasis_weights": {
"energy": 0.30, "pitch_variation": 0.25, "rate_variation": 0.20,
"pause_before": 0.15, "duration": 0.10,
},
"emotion_enabled": False,
"emotion_sensitivity": 0.5,
}
monkeypatch.setattr(server_mod, "load_voice_analysis_config", lambda: config(0.01))
await handle_build_voice_timeline({"media_path": str(wav)})
strict = json.loads((wav.parent / "clip_voice_timeline.json").read_text())
monkeypatch.setattr(server_mod, "load_voice_analysis_config", lambda: config(1.0))
await handle_build_voice_timeline({"media_path": str(wav)})
loose = json.loads((wav.parent / "clip_voice_timeline.json").read_text())
assert loose["summary"]["peak_count"] > strict["summary"]["peak_count"]
+275 -4
View File
@@ -652,6 +652,7 @@ def test_add_zoom_creates_keyframed_transform(temp_fcpxml):
"""4 keyframes: 100% -> scale -> scale -> 100%, all within [start, end]."""
modifier = FCPXMLModifier(temp_fcpxml)
clip = modifier.add_zoom(clip_id='Broll_Studio', start=1.0, end=3.0, scale=1.3, ease=0.5)
frame = float(modifier.frame_duration_fraction())
transform = clip.find('adjust-transform')
assert transform is not None
@@ -660,9 +661,23 @@ def test_add_zoom_creates_keyframed_transform(temp_fcpxml):
keyframes = param.find('keyframeAnimation').findall('keyframe')
assert len(keyframes) == 4
assert [kf.get('value') for kf in keyframes] == ['1 1', '1.3 1.3', '1.3 1.3', '1 1']
assert all(kf.get('interp') == 'ease' for kf in keyframes)
# Bare keyframes — only time and value — matching a zoom exported from
# FCP itself. It rejects 'interp' on this vector param (discarding the
# whole <param>), and its own export writes no 'curve' either.
assert not any(kf.get('interp') for kf in keyframes)
assert not any(kf.get('curve') for kf in keyframes)
assert all(set(kf.attrib) == {'time', 'value'} for kf in keyframes)
# Keyframe times are anchored in the clip's SOURCE timebase (its own
# `start`), not clip-relative. This fixture starts at 10s, so a zoom over
# clip seconds 1-3 must be written at 11-13s. Writing 1-3s here would put
# the animation outside the clip and FCP imports it as nothing.
origin = modifier._parse_time(clip.get('start', '0s')).to_seconds()
times = [modifier._parse_time(kf.get('time')).to_seconds() for kf in keyframes]
assert times == pytest.approx([1.0, 1.5, 2.5, 3.0], abs=0.05)
assert origin > 0, "fixture must start off zero, or this asserts nothing"
# in over 0.5s, hold, then snap back on the very next frame
assert times == pytest.approx(
[origin + 1.0, origin + 1.5, origin + 3.0 - frame, origin + 3.0], abs=0.05
)
assert times == sorted(times)
@@ -699,8 +714,9 @@ def test_add_zoom_window_outside_clip_duration_raises(temp_fcpxml):
def test_add_zoom_ease_too_long_for_window_raises(temp_fcpxml):
modifier = FCPXMLModifier(temp_fcpxml)
with pytest.raises(ValueError, match="doesn't fit"):
modifier.add_zoom(clip_id='Broll_Studio', start=0.0, end=1.0, ease=1.0)
# start away from the clip head so the ramp-in is actually written
with pytest.raises(ValueError, match="don't fit"):
modifier.add_zoom(clip_id='Broll_Studio', start=1.0, end=2.0, ease=1.5)
def test_change_speed_twice_no_duplicate_elements(temp_fcpxml):
@@ -2067,3 +2083,258 @@ def test_remove_trailing_gaps_noop_without_gap():
assert len(children) == 1
assert children[0].tag == 'asset-clip'
Path(f.name).unlink(missing_ok=True)
class TestNTSCFrameAlignment:
"""23.976/29.97 timebases must not be reported as misaligned.
Regression: the check used int(fps), so an exactly frame-aligned NTSC
duration (a whole multiple of 1001/24000s) was flagged as broken —
every NTSC project produced spurious warnings that buried real ones.
"""
NTSC_DOC = """<?xml version="1.0" encoding="UTF-8"?>
<fcpxml version="1.13">
<resources>
<format id="r1" name="FFVideoFormat1080p2398" frameDuration="1001/24000s" width="1920" height="1080"/>
<asset id="a1" name="v" start="0s" duration="24437413/24000s" hasVideo="1" format="r1">
<media-rep kind="original-media" src="file:///v.mov"/>
</asset>
</resources>
<library><event name="E"><project name="P">
<sequence format="r1" duration="24437413/24000s" tcStart="0s">
<spine><asset-clip name="v" ref="a1" offset="0s" start="0s" duration="24437413/24000s"/></spine>
</sequence>
</project></event></library>
</fcpxml>
"""
def _issues(self, xml):
from fcpxml.safe_xml import safe_fromstring
from fcpxml.writer import validate_fcpxml
root = safe_fromstring(xml)
return [
i for i in (validate_fcpxml(root) or [])
if "frame" in str(i.issue_type.value)
]
def test_aligned_ntsc_duration_is_not_flagged(self):
# 24437413/24000s is exactly 24413 frames of 1001/24000s
assert self._issues(self.NTSC_DOC) == []
def test_genuinely_misaligned_duration_is_still_flagged(self):
broken = self.NTSC_DOC.replace(
'<asset-clip name="v" ref="a1" offset="0s" start="0s" duration="24437413/24000s"/>',
'<asset-clip name="v" ref="a1" offset="0s" start="0s" duration="500/24000s"/>',
)
assert len(self._issues(broken)) == 1
def test_message_names_the_real_rate_not_a_rounded_one(self):
broken = self.NTSC_DOC.replace('duration="24437413/24000s"/>', 'duration="500/24000s"/>')
issues = self._issues(broken)
assert issues and "23.976fps" in issues[0].message
class TestZoomPreservesExistingFraming:
"""A clip may already carry the editor's reframe — rotation for footage
shot sideways, position, a base scale. add_zoom used to delete it, which
on real footage brought the zoomed section back rotated."""
FRAMED = """<?xml version="1.0" encoding="UTF-8"?>
<fcpxml version="1.13">
<resources>
<format id="r1" frameDuration="100/3000s" width="1920" height="1080"/>
<asset id="a1" name="v" start="0s" duration="600/30s" hasVideo="1" format="r1">
<media-rep kind="original-media" src="file:///v.mov"/>
</asset>
</resources>
<library><event name="E"><project name="P">
<sequence format="r1" duration="600/30s" tcStart="0s">
<spine>
<asset-clip name="v" ref="a1" offset="0s" start="0s" duration="600/30s">
<adjust-transform position="0.16 0.66" rotation="90.1" scale="1.77311 1.77311"/>
</asset-clip>
</spine>
</sequence>
</project></event></library>
</fcpxml>
"""
def _zoomed(self, tmp_path, scale=1.2):
from fcpxml.writer import FCPXMLModifier
path = tmp_path / "framed.fcpxml"
path.write_text(self.FRAMED)
m = FCPXMLModifier(str(path))
m.add_zoom(clip_id="v", start=1.0, end=5.0, scale=scale)
return m.root.find(".//adjust-transform")
def test_rotation_and_position_survive(self, tmp_path):
t = self._zoomed(tmp_path)
assert t.get("rotation") == "90.1"
assert t.get("position") == "0.16 0.66"
def test_animation_rests_at_the_existing_scale(self, tmp_path):
kfs = self._zoomed(tmp_path).findall(".//keyframe")
# first and last keyframe return to the clip's own framing, not to 1
assert kfs[0].get("value").split()[0].startswith("1.77")
assert kfs[-1].get("value").split()[0].startswith("1.77")
def test_peak_multiplies_the_existing_scale(self, tmp_path):
kfs = self._zoomed(tmp_path, scale=2.0).findall(".//keyframe")
peak = float(kfs[1].get("value").split()[0])
assert peak == pytest.approx(1.77311 * 2.0, rel=1e-4)
def test_only_one_transform_remains(self, tmp_path):
from fcpxml.writer import FCPXMLModifier
path = tmp_path / "framed.fcpxml"
path.write_text(self.FRAMED)
m = FCPXMLModifier(str(path))
m.add_zoom(clip_id="v", start=1.0, end=5.0, scale=1.2)
m.add_zoom(clip_id="v", start=6.0, end=9.0, scale=1.4)
assert len(m.root.findall(".//adjust-transform")) == 1
def test_second_zoom_on_same_clip_keeps_the_real_base_scale(self, tmp_path):
"""Found on real footage: two zoom actions landing on disjoint
windows of the same post-cut clip. The second add_zoom call used to
see the already-animated <param name="scale"> from the first zoom
instead of a static attribute, read that as "no framing", and
default the base to 1.0 — silently shrinking the shot back to its
unframed size for the whole clip wherever no keyframe applied, and
discarding the first zoom's animation in the process."""
from fcpxml.writer import FCPXMLModifier
path = tmp_path / "framed.fcpxml"
path.write_text(self.FRAMED)
m = FCPXMLModifier(str(path))
m.add_zoom(clip_id="v", start=1.0, end=5.0, scale=1.2)
m.add_zoom(clip_id="v", start=6.0, end=9.0, scale=1.4)
kfs = m.root.findall(".//keyframe")
# Every rest keyframe returns to the clip's real base scale, never 1.0.
rest_values = {kf.get("value") for kf in (kfs[0], kfs[3], kfs[4], kfs[-1])}
assert rest_values == {"1.77311 1.77311"}
# Both peaks survive — the second call didn't erase the first.
peaks = sorted(float(kf.get("value").split()[0]) for kf in (kfs[1], kfs[5]))
assert peaks[0] == pytest.approx(1.77311 * 1.2, rel=1e-4)
assert peaks[1] == pytest.approx(1.77311 * 1.4, rel=1e-4)
def test_overlapping_zoom_on_same_clip_replaces_instead_of_stacking(self, tmp_path):
"""Two windows that OVERLAP mean "redo this zoom", not "add another
one" — the old keyframes are stale and all of them go, matching
test_add_zoom_replaces_existing_zoom's contract."""
from fcpxml.writer import FCPXMLModifier
path = tmp_path / "framed.fcpxml"
path.write_text(self.FRAMED)
m = FCPXMLModifier(str(path))
m.add_zoom(clip_id="v", start=1.0, end=5.0, scale=1.2)
m.add_zoom(clip_id="v", start=3.0, end=7.0, scale=1.5)
values = [kf.get("value") for kf in m.root.findall(".//keyframe")]
assert not any(v.startswith("2.1277") for v in values) # 1.77311*1.2 gone
assert any(v.startswith("2.6596") for v in values) # 1.77311*1.5 present
def test_unframed_clip_still_rests_at_one(self, tmp_path):
from fcpxml.writer import FCPXMLModifier
path = tmp_path / "plain.fcpxml"
path.write_text(self.FRAMED.replace(
'<adjust-transform position="0.16 0.66" rotation="90.1" scale="1.77311 1.77311"/>', ''))
m = FCPXMLModifier(str(path))
m.add_zoom(clip_id="v", start=1.0, end=5.0, scale=1.3)
kfs = m.root.findall(".//keyframe")
assert kfs[0].get("value") == "1 1"
assert kfs[1].get("value") == "1.3 1.3"
class TestZoomShapeIsAsymmetric:
"""The editorial shape: ramp in fast to land with the emphasised word,
hold through the impact phrase, then snap back in a single frame so the
video resumes its normal framing without a drift that draws the eye."""
def _times(self, temp_fcpxml, **kw):
from fcpxml.writer import FCPXMLModifier
m = FCPXMLModifier(temp_fcpxml)
# Broll_Studio is 5s long; end=3.0 keeps clear of the hold-at-cut
# margin so these exercise the ordinary return-to-framing shape.
kw.setdefault('end', 3.0)
clip = m.add_zoom(clip_id='Broll_Studio', start=1.0, **kw)
origin = m._parse_time(clip.get('start', '0s')).to_seconds()
kfs = clip.find('adjust-transform').find('param').find('keyframeAnimation')
return m, origin, [m._parse_time(k.get('time')).to_seconds() - origin
for k in kfs.findall('keyframe')]
def test_return_takes_a_single_frame(self, temp_fcpxml):
m, _, t = self._times(temp_fcpxml, scale=1.2)
frame = float(m.frame_duration_fraction())
assert t[3] - t[2] == pytest.approx(frame, abs=0.005)
def test_ramp_in_is_quick(self, temp_fcpxml):
"""Fast enough to land with the emphasised word rather than drift."""
_, _, t = self._times(temp_fcpxml, scale=1.2)
assert t[1] - t[0] == pytest.approx(0.25, abs=0.05)
def test_peak_is_held_until_the_return(self, temp_fcpxml):
_, _, t = self._times(temp_fcpxml, scale=1.2)
# hold spans from the top of the ramp to one frame before the end
assert t[2] - t[1] > 1.4
def test_zoom_opening_at_a_cut_starts_already_zoomed(self, temp_fcpxml):
"""The cut is the transition — ramping up from it reads as the shot
settling rather than as emphasis."""
from fcpxml.writer import FCPXMLModifier
m = FCPXMLModifier(temp_fcpxml)
clip = m.add_zoom(clip_id='Broll_Studio', start=0.1, end=3.0, scale=1.2)
kfs = clip.find('adjust-transform').find('param').find('keyframeAnimation')
values = [k.get('value') for k in kfs.findall('keyframe')]
assert values[0] != values[-1] # opens zoomed, returns to framing
assert values[0] == values[-2] # ...and was at the peak from frame one
def test_opening_at_peak_can_be_forced_off(self, temp_fcpxml):
from fcpxml.writer import FCPXMLModifier
m = FCPXMLModifier(temp_fcpxml)
clip = m.add_zoom(
clip_id='Broll_Studio', start=0.1, end=3.0, scale=1.2, start_at_peak=False
)
kfs = clip.find('adjust-transform').find('param').find('keyframeAnimation')
assert len(kfs.findall('keyframe')) == 4
def test_ease_out_can_be_made_gradual(self, temp_fcpxml):
_, _, t = self._times(temp_fcpxml, scale=1.2, ease_out=1.0)
assert t[3] - t[2] == pytest.approx(1.0, abs=0.05)
def test_zoom_reaching_the_cut_holds_instead_of_returning(self, temp_fcpxml):
"""Returning right before a cut is wasted motion — the next clip
opens on its own framing, so the move back reads as a twitch."""
_, _, t = self._times(temp_fcpxml, scale=1.2, end=5.0)
assert len(t) == 3 # rest, peak, still peak at the cut
def test_hold_can_be_forced_off_at_a_cut(self, temp_fcpxml):
_, _, t = self._times(temp_fcpxml, scale=1.2, end=5.0, hold_at_end=False)
assert len(t) == 4
def test_hold_can_be_forced_on_mid_clip(self, temp_fcpxml):
_, _, t = self._times(temp_fcpxml, scale=1.2, end=3.0, hold_at_end=True)
assert len(t) == 3
def test_held_zoom_stays_at_the_peak(self, temp_fcpxml):
from fcpxml.writer import FCPXMLModifier
m = FCPXMLModifier(temp_fcpxml)
clip = m.add_zoom(clip_id='Broll_Studio', start=1.0, end=5.0, scale=1.2)
kfs = clip.find('adjust-transform').find('param').find('keyframeAnimation')
values = [k.get('value') for k in kfs.findall('keyframe')]
assert values[-1] == values[-2] != values[0]
def test_window_too_short_for_the_ramp_is_rejected(self, temp_fcpxml):
from fcpxml.writer import FCPXMLModifier
m = FCPXMLModifier(temp_fcpxml)
with pytest.raises(ValueError, match="don't fit"):
m.add_zoom(clip_id='Broll_Studio', start=1.0, end=1.2, scale=1.2, ease=1.5)