chore: atualização geral
This commit is contained in:
@@ -0,0 +1,266 @@
|
||||
"""Tests for collision detection and subtitle-layout validation (Fase 1).
|
||||
|
||||
Covers the pure functions in ``fcpxml.collision`` (spatial/temporal overlap,
|
||||
area/ratio classification, distance, separation suggestion, box measurement)
|
||||
and the integration through ``FCPXMLModifier.validate_subtitle_layout`` over a
|
||||
generated document — the post-generation guarantee the layout engine only
|
||||
provides by construction.
|
||||
"""
|
||||
|
||||
import shutil
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from fcpxml.collision import (
|
||||
FONT_MISSING,
|
||||
FONT_TOO_SMALL,
|
||||
OUTSIDE_FRAME,
|
||||
OUTSIDE_SAFE_AREA,
|
||||
OVERLAP_PROBABLE,
|
||||
OVERLAP_RENDER_TOLERANCE,
|
||||
OVERLAP_SEVERE,
|
||||
SPATIAL_COLLISION,
|
||||
Box,
|
||||
blocking,
|
||||
classify_overlap,
|
||||
distance_between,
|
||||
measure_title_box,
|
||||
overlap_metrics,
|
||||
separation_suggestion,
|
||||
temporal_overlap,
|
||||
validate_titles,
|
||||
)
|
||||
from fcpxml.models import DynamicSubtitleConfig
|
||||
from fcpxml.writer import FCPXMLModifier
|
||||
|
||||
SAMPLE = Path(__file__).parent.parent / "examples" / "sample.fcpxml"
|
||||
|
||||
WORD_MODE = DynamicSubtitleConfig(granularity="word")
|
||||
|
||||
WORDS = [
|
||||
{"word": "Hello", "start": 0.0, "end": 0.4},
|
||||
{"word": "there", "start": 0.4, "end": 0.8},
|
||||
{"word": "friend", "start": 0.8, "end": 1.3},
|
||||
]
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def temp_fcpxml():
|
||||
with tempfile.NamedTemporaryFile(suffix=".fcpxml", delete=False) as f:
|
||||
shutil.copy(SAMPLE, f.name)
|
||||
yield f.name
|
||||
Path(f.name).unlink(missing_ok=True)
|
||||
|
||||
|
||||
def _title(text, x=0.0, y=0.0, *, start=0.0, end=1.0, font="Helvetica Neue",
|
||||
face=None, font_size=100.0, kerning=0.0, group=0):
|
||||
return {
|
||||
"text": text,
|
||||
"x": x,
|
||||
"y": y,
|
||||
"start": start,
|
||||
"end": end,
|
||||
"font": font,
|
||||
"face": face,
|
||||
"font_size": font_size,
|
||||
"kerning": kerning,
|
||||
"group": group,
|
||||
}
|
||||
|
||||
|
||||
class TestBoxOverlap:
|
||||
def test_touching_boxes_do_not_overlap(self):
|
||||
a = Box(0, 100, 0, 50)
|
||||
b = Box(100, 200, 0, 50) # shares the right edge
|
||||
assert not a.overlaps(b)
|
||||
assert overlap_metrics(a, b)["overlap_area"] == 0
|
||||
|
||||
def test_overlap_by_one_pixel_detected(self):
|
||||
a = Box(0, 100, 0, 50)
|
||||
b = Box(99, 200, 0, 50) # one-pixel horizontal overlap
|
||||
assert a.overlaps(b)
|
||||
metrics = overlap_metrics(a, b)
|
||||
assert metrics["overlap_width"] == 1
|
||||
assert metrics["overlap_area"] == 50
|
||||
|
||||
def test_vertical_only_overlap(self):
|
||||
a = Box(0, 100, 0, 50)
|
||||
b = Box(0, 100, 49, 100) # one-pixel vertical overlap
|
||||
assert a.overlaps(b)
|
||||
assert overlap_metrics(a, b)["overlap_height"] == 1
|
||||
|
||||
|
||||
class TestTemporalOverlap:
|
||||
def test_adjacent_intervals_are_not_simultaneous(self):
|
||||
# [0, 1) and [1, 2) share no instant.
|
||||
assert not temporal_overlap(0.0, 1.0, 1.0, 2.0)
|
||||
assert not temporal_overlap(1.0, 2.0, 0.0, 1.0)
|
||||
|
||||
def test_interleaved_intervals_overlap(self):
|
||||
assert temporal_overlap(0.0, 2.0, 1.0, 3.0)
|
||||
|
||||
def test_contained_interval_overlaps(self):
|
||||
assert temporal_overlap(0.0, 5.0, 1.0, 2.0)
|
||||
|
||||
def test_boundary_survives_float_noise_from_the_writer(self):
|
||||
"""Found on real footage: the writer sets one block's title duration
|
||||
to make its end land EXACTLY on the next block's start (same exact
|
||||
FCPXML fraction), but end here is re-derived as start + duration —
|
||||
two independently-rounded floats — which isn't bit-identical to the
|
||||
other title's start read as a single division of that same
|
||||
fraction. 7 of 8 "severe" collisions from one real clip were this,
|
||||
off by ~1e-13s, far below any frame boundary."""
|
||||
start_a, duration_a = 74883609 / 24000, 43043 / 24000
|
||||
end_a = start_a + duration_a # float addition, like validate_titles does
|
||||
start_b = 74926652 / 24000 # the exact same instant, read directly
|
||||
assert end_a != start_b # the float noise is real
|
||||
assert abs(end_a - start_b) < 1e-6 # ...and far below one frame
|
||||
assert not temporal_overlap(start_a, end_a, start_b, start_b + 1.0)
|
||||
|
||||
|
||||
class TestClassifyOverlap:
|
||||
def test_zero_area_is_no_conflict(self):
|
||||
assert classify_overlap(
|
||||
{"overlap_area": 0, "overlap_height": 0, "overlap_ratio": 0}
|
||||
) == "none"
|
||||
|
||||
def test_ratio_dominates_small_height(self):
|
||||
# A sliver of 5px that covers most of a tiny box is still severe.
|
||||
metrics = {"overlap_area": 50, "overlap_height": 5, "overlap_ratio": 0.9}
|
||||
assert classify_overlap(metrics) == OVERLAP_SEVERE
|
||||
|
||||
def test_height_buckets(self):
|
||||
def metrics(height):
|
||||
return {"overlap_area": 1, "overlap_height": height,
|
||||
"overlap_ratio": 0.0}
|
||||
|
||||
assert classify_overlap(metrics(5)) == OVERLAP_RENDER_TOLERANCE
|
||||
assert classify_overlap(metrics(6)) == "warning"
|
||||
assert classify_overlap(metrics(21)) == OVERLAP_PROBABLE
|
||||
assert classify_overlap(metrics(51)) == OVERLAP_SEVERE
|
||||
|
||||
|
||||
class TestDistanceAndSeparation:
|
||||
def test_horizontal_distance(self):
|
||||
a = Box(0, 100, 0, 50)
|
||||
b = Box(150, 200, 0, 50)
|
||||
d = distance_between(a, b)
|
||||
assert d["distance_x"] == 50
|
||||
assert d["distance_y"] == 0
|
||||
assert d["distance"] == 50
|
||||
|
||||
def test_vertical_distance(self):
|
||||
a = Box(0, 100, 0, 50)
|
||||
b = Box(0, 100, 80, 130)
|
||||
d = distance_between(a, b)
|
||||
assert d["distance_y"] == 30
|
||||
assert d["distance_x"] == 0
|
||||
|
||||
def test_separation_picks_smallest_axis(self):
|
||||
a = Box(0, 100, 0, 100)
|
||||
b = Box(90, 190, 95, 195) # 10px horizontal, 5px vertical penetration
|
||||
suggestion = separation_suggestion(a, b)
|
||||
assert suggestion["axis"] == "vertical"
|
||||
assert suggestion["minimum_movement"] == 5
|
||||
|
||||
def test_blocking_severities(self):
|
||||
assert blocking(OVERLAP_SEVERE)
|
||||
assert blocking(OVERLAP_PROBABLE)
|
||||
assert not blocking("warning")
|
||||
assert not blocking("none")
|
||||
|
||||
|
||||
class TestValidateTitles:
|
||||
def test_clean_layout_has_no_issues(self):
|
||||
report = validate_titles(
|
||||
[_title("um", x=-200), _title("dois", x=200)],
|
||||
2160, 3840,
|
||||
)
|
||||
assert report["severity"] == "none"
|
||||
assert report["issues"] == []
|
||||
|
||||
def test_simultaneous_collision_reported(self):
|
||||
titles = [_title("A", x=0), _title("B", x=0)]
|
||||
report = validate_titles(titles, 2160, 3840)
|
||||
collisions = [
|
||||
i for i in report["issues"] if i["type"] == SPATIAL_COLLISION
|
||||
]
|
||||
assert len(collisions) == 1
|
||||
assert collisions[0]["severity"] == OVERLAP_SEVERE
|
||||
assert collisions[0]["suggested_correction"]["axis"] in (
|
||||
"vertical", "horizontal",
|
||||
)
|
||||
|
||||
def test_collision_across_times_ignored(self):
|
||||
titles = [
|
||||
_title("A", x=0, start=0.0, end=1.0),
|
||||
_title("B", x=0, start=1.0, end=2.0),
|
||||
]
|
||||
report = validate_titles(titles, 2160, 3840)
|
||||
assert not any(
|
||||
i["type"] == SPATIAL_COLLISION for i in report["issues"]
|
||||
)
|
||||
|
||||
def test_outside_frame_is_error(self):
|
||||
report = validate_titles([_title("fora", x=5000)], 2160, 3840)
|
||||
assert any(i["type"] == OUTSIDE_FRAME for i in report["issues"])
|
||||
|
||||
def test_outside_safe_area_is_warning_not_frame(self):
|
||||
# Near the right edge: inside the frame, outside the 5% safe area.
|
||||
report = validate_titles(
|
||||
[_title("a", x=1000, font_size=40)], 2160, 3840,
|
||||
)
|
||||
assert any(i["type"] == OUTSIDE_SAFE_AREA for i in report["issues"])
|
||||
assert not any(i["type"] == OUTSIDE_FRAME for i in report["issues"])
|
||||
|
||||
def test_resolution_changes_safe_area(self):
|
||||
# Same x is fine on a wide frame but out of the safe area on a narrow one.
|
||||
narrow = validate_titles([_title("a", x=1000, font_size=40)], 1920, 1080)
|
||||
assert any(i["type"] == OUTSIDE_FRAME for i in narrow["issues"])
|
||||
|
||||
def test_font_missing_reported(self):
|
||||
report = validate_titles(
|
||||
[_title("oi", font="Comic Sans MS")], 2160, 3840,
|
||||
)
|
||||
assert any(i["type"] == FONT_MISSING for i in report["issues"])
|
||||
|
||||
def test_font_too_small_reported(self):
|
||||
report = validate_titles(
|
||||
[_title("oi", font_size=10)], 2160, 3840, min_font_size=20,
|
||||
)
|
||||
assert any(i["type"] == FONT_TOO_SMALL for i in report["issues"])
|
||||
|
||||
def test_measure_title_box_uses_real_width(self):
|
||||
box = measure_title_box("ii", 100, x=0, y=0, font="Helvetica Neue")
|
||||
# "ii" is the narrowest glyph; a single "W" is much wider.
|
||||
wide = measure_title_box("W", 100, x=0, y=0, font="Helvetica Neue")
|
||||
assert wide.width > box.width
|
||||
|
||||
|
||||
class TestIntegration:
|
||||
def test_generated_layout_is_clean(self, temp_fcpxml):
|
||||
modifier = FCPXMLModifier(temp_fcpxml)
|
||||
modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE)
|
||||
report = modifier.validate_subtitle_layout()
|
||||
assert report["summary"]["title_count"] >= 3
|
||||
assert report["summary"]["spatial_collision"] == 0
|
||||
assert not blocking(report["severity"])
|
||||
|
||||
def test_hand_edited_position_is_caught(self, temp_fcpxml):
|
||||
modifier = FCPXMLModifier(temp_fcpxml)
|
||||
titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE)
|
||||
|
||||
def position(el):
|
||||
for p in el.findall("param"):
|
||||
if p.get("name") == "Position":
|
||||
return p
|
||||
return None
|
||||
|
||||
p0 = position(titles[0])
|
||||
position(titles[1]).set("value", p0.get("value"))
|
||||
|
||||
report = modifier.validate_subtitle_layout()
|
||||
assert report["summary"]["spatial_collision"] >= 1
|
||||
assert blocking(report["severity"])
|
||||
@@ -0,0 +1,108 @@
|
||||
"""Tests for the diarize_media MCP tool (server.handle_diarize_media).
|
||||
|
||||
Diarization itself (pyannote.audio) is monkeypatched so these tests run
|
||||
without the optional [diarization] extra or a HuggingFace token — matching
|
||||
the existing TestDetectBeatsHandler pattern in test_media_intel.py.
|
||||
"""
|
||||
|
||||
import json
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
def _write_tiny_wav(path: str, seconds: float = 1.0) -> None:
|
||||
import struct
|
||||
import wave
|
||||
|
||||
n_frames = int(44100 * seconds)
|
||||
with wave.open(path, "w") as f:
|
||||
f.setnchannels(1)
|
||||
f.setsampwidth(2)
|
||||
f.setframerate(44100)
|
||||
f.writeframes(struct.pack("<%dh" % n_frames, *([0] * n_frames)))
|
||||
|
||||
|
||||
class TestDiarizeMediaHandler:
|
||||
async def test_reports_when_pyannote_unavailable(self, tmp_path, monkeypatch):
|
||||
import server_tools.voice as server_mod
|
||||
from server import handle_diarize_media
|
||||
|
||||
wav = tmp_path / "clip.wav"
|
||||
_write_tiny_wav(str(wav))
|
||||
monkeypatch.setattr(
|
||||
server_mod, "diarization_capability", lambda token: (False, "Diarização indisponível: componente pyannote.audio ausente.")
|
||||
)
|
||||
result = await handle_diarize_media({"media_path": str(wav)})
|
||||
text = result[0].text
|
||||
assert "indisponível" in text.lower() or "unavailable" in text.lower()
|
||||
assert "diarization" in text.lower()
|
||||
|
||||
async def test_rejects_disallowed_extension(self, tmp_path):
|
||||
from server import handle_diarize_media
|
||||
|
||||
bad = tmp_path / "clip.txt"
|
||||
bad.write_text("not audio")
|
||||
with pytest.raises(ValueError):
|
||||
await handle_diarize_media({"media_path": str(bad)})
|
||||
|
||||
async def test_writes_diarization_json_and_reports(self, tmp_path, monkeypatch):
|
||||
import server_tools._shared as _shared_mod
|
||||
import server_tools.voice as server_mod
|
||||
from server import handle_diarize_media
|
||||
|
||||
wav = tmp_path / "clip.wav"
|
||||
_write_tiny_wav(str(wav), seconds=2.0)
|
||||
|
||||
fake_transcript = {
|
||||
"language": "en",
|
||||
"duration": 2.0,
|
||||
"text": "hello world",
|
||||
"segments": [
|
||||
{"text": "hello", "start": 0.0, "end": 1.0},
|
||||
{"text": "world", "start": 1.0, "end": 2.0},
|
||||
],
|
||||
"words": [
|
||||
{"word": "hello", "start": 0.0, "end": 0.5, "confidence": 0.9},
|
||||
{"word": "world", "start": 1.0, "end": 1.5, "confidence": 0.9},
|
||||
],
|
||||
}
|
||||
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: fake_transcript)
|
||||
monkeypatch.setattr(server_mod, "diarization_capability", lambda token: (True, "ok"))
|
||||
monkeypatch.setattr(
|
||||
server_mod,
|
||||
"diarize",
|
||||
lambda path, token, num_speakers="": [(0.0, 1.0, "A"), (1.0, 2.0, "B")],
|
||||
)
|
||||
|
||||
result = await handle_diarize_media({"media_path": str(wav), "hf_token": "fake-token"})
|
||||
text = result[0].text
|
||||
assert "Speaker" in text or "speaker" in text.lower()
|
||||
|
||||
json_path = tmp_path / "clip_diarization.json"
|
||||
assert str(json_path) in text
|
||||
data = json.loads(json_path.read_text())
|
||||
assert len(data["speakers"]) == 2
|
||||
assert data["words"][0]["speaker_id"] == "SPEAKER_00"
|
||||
assert data["words"][1]["speaker_id"] == "SPEAKER_01"
|
||||
|
||||
async def test_reports_when_diarization_fails(self, tmp_path, monkeypatch):
|
||||
import server_tools._shared as _shared_mod
|
||||
import server_tools.voice as server_mod
|
||||
from server import handle_diarize_media
|
||||
|
||||
wav = tmp_path / "clip.wav"
|
||||
_write_tiny_wav(str(wav))
|
||||
|
||||
fake_transcript = {
|
||||
"language": "en",
|
||||
"duration": 1.0,
|
||||
"text": "hi",
|
||||
"segments": [{"text": "hi", "start": 0.0, "end": 1.0}],
|
||||
"words": [{"word": "hi", "start": 0.0, "end": 0.5, "confidence": 0.9}],
|
||||
}
|
||||
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: fake_transcript)
|
||||
monkeypatch.setattr(server_mod, "diarization_capability", lambda token: (True, "ok"))
|
||||
monkeypatch.setattr(server_mod, "diarize", lambda *a, **k: None)
|
||||
|
||||
result = await handle_diarize_media({"media_path": str(wav), "hf_token": "fake-token"})
|
||||
assert "failed" in result[0].text.lower()
|
||||
@@ -17,12 +17,27 @@ from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from fcpxml.models import DynamicSubtitleConfig, TimeValue, WordStyle
|
||||
from fcpxml.models import DynamicSubtitleConfig, TimeValue, WordLook, WordStyle
|
||||
from fcpxml.parser import parse_fcpxml
|
||||
from fcpxml.text_layout import POINT_SCALE, REFERENCE_CANVAS_HEIGHT, ink_extent
|
||||
from fcpxml.text_layout import (
|
||||
POINT_SCALE,
|
||||
REFERENCE_BLOCK_LINE_GAP,
|
||||
REFERENCE_CANVAS_HEIGHT,
|
||||
TEXT_TEMPLATE_FONT_SCALE,
|
||||
LayoutBox,
|
||||
compose_sentence,
|
||||
ink_extent,
|
||||
)
|
||||
from fcpxml.writer import FCPXMLModifier
|
||||
|
||||
SAMPLE = Path(__file__).parent.parent / "examples" / "sample.fcpxml"
|
||||
def font_points(style) -> float:
|
||||
"""The style's size back in CANVAS POINTS.
|
||||
|
||||
The emitted fontSize lives in the template's own space, which is
|
||||
TEXT_TEMPLATE_FONT_SCALE times bigger than the space positions use, so any
|
||||
check that mixes the two has to convert first."""
|
||||
return float(style.get("fontSize")) / TEXT_TEMPLATE_FONT_SCALE
|
||||
|
||||
# The earlier rhythm: one title per WORD. The default is now the
|
||||
# progressive composition (one title per LINE), covered in
|
||||
@@ -106,7 +121,7 @@ class TestGenerateDynamicSubtitles:
|
||||
|
||||
param_names = [p.get("name") for p in title.findall("param")]
|
||||
assert param_names == [
|
||||
"Position", "Layout Method", "Left Margin", "Right Margin",
|
||||
"Position", "Build Out", "Layout Method", "Left Margin", "Right Margin",
|
||||
"Top Margin", "Bottom Margin", "Alignment", "Line Spacing",
|
||||
"Auto-Shrink", "Alignment", "Opacity", "Speed", "Custom Speed",
|
||||
"Apply Speed",
|
||||
@@ -132,7 +147,8 @@ class TestGenerateDynamicSubtitles:
|
||||
assert style_def.get("fontFace") == first.face
|
||||
|
||||
scale = (1080 * POINT_SCALE) / REFERENCE_CANVAS_HEIGHT
|
||||
assert style_def.get("fontSize") == str(round(first.font_size * scale))
|
||||
expected = round(first.font_size * scale) * TEXT_TEMPLATE_FONT_SCALE
|
||||
assert float(style_def.get("fontSize")) == expected
|
||||
|
||||
def test_font_size_scales_with_the_frame(self, temp_fcpxml):
|
||||
"""A vertical 2160x3840 timeline must get the reference sizes back
|
||||
@@ -144,7 +160,9 @@ class TestGenerateDynamicSubtitles:
|
||||
title = modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE)[0]
|
||||
run = title.find("text/text-style")
|
||||
style_def = title.find(f"text-style-def[@id='{run.get('ref')}']/text-style")
|
||||
assert style_def.get("fontSize") == str(WordStyle().rhythm[0].font_size)
|
||||
assert float(style_def.get("fontSize")) == (
|
||||
WordStyle().rhythm[0].font_size * TEXT_TEMPLATE_FONT_SCALE
|
||||
)
|
||||
|
||||
def test_position_is_keyframed_constant_hold(self, temp_fcpxml):
|
||||
"""Regression (2026-08-17): position is a STATIC param in the "Text"
|
||||
@@ -685,7 +703,7 @@ class TestProgressiveComposition:
|
||||
style = self._style(t)
|
||||
top, bottom = ink_extent(
|
||||
t.find("text/text-style").text,
|
||||
float(style.get("fontSize")),
|
||||
font_points(style),
|
||||
font=style.get("font"),
|
||||
face=style.get("fontFace"),
|
||||
)
|
||||
@@ -718,7 +736,7 @@ class TestProgressiveComposition:
|
||||
style = self._style(t)
|
||||
top, bottom = ink_extent(
|
||||
t.find("text/text-style").text,
|
||||
float(style.get("fontSize")),
|
||||
font_points(style),
|
||||
font=style.get("font"),
|
||||
face=style.get("fontFace"),
|
||||
)
|
||||
@@ -762,3 +780,146 @@ class TestProgressiveComposition:
|
||||
spoken = " ".join(w["word"] for w in words)
|
||||
emitted = " ".join(t.find("text/text-style").text for t in titles)
|
||||
assert emitted == spoken, "no spoken word may be dropped or reordered"
|
||||
|
||||
|
||||
class TestTemplateFontScale:
|
||||
"""The "Text" template sizes type in frame pixels but positions in canvas
|
||||
points, so the emitted fontSize must be converted or the block renders in
|
||||
the right place at half the chosen size."""
|
||||
|
||||
STYLE = WordStyle(
|
||||
emphasis_look=WordLook(200, "1 1 1 1", font="Georgia", kerning=3.0),
|
||||
body_look=WordLook(100, "1 1 1 1", font="Helvetica Neue", kerning=3.0),
|
||||
)
|
||||
|
||||
def _styles(self, titles):
|
||||
return [t.find(".//text-style-def/text-style") for t in titles]
|
||||
|
||||
def _generate(self, path, **kwargs):
|
||||
modifier = FCPXMLModifier(path)
|
||||
titles = modifier.generate_dynamic_subtitles(
|
||||
"Interview_A", WORDS,
|
||||
DynamicSubtitleConfig(style=self.STYLE, **kwargs),
|
||||
)
|
||||
return titles
|
||||
|
||||
def test_emitted_size_is_the_layout_size_times_the_template_scale(self, temp_fcpxml):
|
||||
unscaled = self._generate(temp_fcpxml, text_scale=1.0)
|
||||
scaled = self._generate(temp_fcpxml, text_scale=TEXT_TEMPLATE_FONT_SCALE)
|
||||
for plain, big in zip(self._styles(unscaled), self._styles(scaled)):
|
||||
assert float(big.get("fontSize")) == (
|
||||
float(plain.get("fontSize")) * TEXT_TEMPLATE_FONT_SCALE
|
||||
)
|
||||
|
||||
def test_default_config_applies_the_template_scale(self, temp_fcpxml):
|
||||
assert DynamicSubtitleConfig().text_scale == TEXT_TEMPLATE_FONT_SCALE
|
||||
default = self._styles(self._generate(temp_fcpxml))
|
||||
unscaled = self._styles(self._generate(temp_fcpxml, text_scale=1.0))
|
||||
assert [s.get("fontSize") for s in default] != [
|
||||
s.get("fontSize") for s in unscaled
|
||||
]
|
||||
|
||||
def test_kerning_scales_with_the_font_size(self, temp_fcpxml):
|
||||
"""Kerning is in font units too — leaving it behind would tighten the
|
||||
letter spacing to half as the type doubled."""
|
||||
unscaled = self._styles(self._generate(temp_fcpxml, text_scale=1.0))
|
||||
scaled = self._styles(self._generate(temp_fcpxml, text_scale=2.0))
|
||||
for plain, big in zip(unscaled, scaled):
|
||||
if plain.get("kerning"):
|
||||
assert float(big.get("kerning")) == float(plain.get("kerning")) * 2
|
||||
|
||||
def _positions(self, titles):
|
||||
return [
|
||||
tuple(float(v) for v in t.find(
|
||||
"param[@name='Position']").get("value").split())
|
||||
for t in titles
|
||||
]
|
||||
|
||||
def test_position_is_converted_with_the_type(self, temp_fcpxml):
|
||||
"""The template reads fontSize and Position in the SAME space, so the
|
||||
conversion has to reach both. Scaling only the type leaves the block at
|
||||
the old spread with twice the type in it, and the lines collide."""
|
||||
plain = self._generate(temp_fcpxml, text_scale=1.0)
|
||||
big = self._generate(temp_fcpxml, text_scale=2.0)
|
||||
for (x, y), (x2, y2) in zip(self._positions(plain), self._positions(big)):
|
||||
assert (x2, y2) == pytest.approx((x * 2, y * 2), rel=1e-4, abs=0.01)
|
||||
|
||||
def test_type_and_spacing_keep_their_ratio_at_any_scale(self, temp_fcpxml):
|
||||
"""The invariant that broke in the field: the distance between two
|
||||
lines, measured in font sizes, must not depend on the scale."""
|
||||
ratios = []
|
||||
for scale in (1.0, 2.0, 3.5):
|
||||
titles = self._generate(temp_fcpxml, text_scale=scale)
|
||||
ys = [y for _, y in self._positions(titles)]
|
||||
sizes = [float(s.get("fontSize")) for s in self._styles(titles)]
|
||||
ratios.append([
|
||||
(a - b) / size
|
||||
for a, b, size in zip(ys, ys[1:], sizes)
|
||||
])
|
||||
for other in ratios[1:]:
|
||||
assert other == pytest.approx(ratios[0], rel=1e-4, abs=1e-4)
|
||||
|
||||
|
||||
class TestLineGap:
|
||||
"""The air between stacked lines is a design choice, negative included."""
|
||||
|
||||
STYLE = WordStyle(
|
||||
emphasis_look=WordLook(200, "1 1 1 1", font="Georgia", kerning=0.0),
|
||||
body_look=WordLook(100, "1 1 1 1", font="Helvetica Neue", kerning=0.0),
|
||||
)
|
||||
PHRASE = "eu tinha muita dificuldade de encontrar roupa"
|
||||
|
||||
def _blocks(self, gap):
|
||||
"""Compose in a band tall enough to hold every line, so the gap is the
|
||||
only thing that changes — a short band would also change how many
|
||||
lines fit, which is a different effect."""
|
||||
words = [
|
||||
{"word": w, "start": i * 0.3, "end": i * 0.3 + 0.3}
|
||||
for i, w in enumerate(self.PHRASE.split())
|
||||
]
|
||||
box = LayoutBox(width=1080 * 0.92, height=100_000, center_y=0)
|
||||
return compose_sentence(words, self.STYLE, box, line_gap=gap).blocks
|
||||
|
||||
def test_default_matches_the_reference_gap(self):
|
||||
assert DynamicSubtitleConfig().line_gap == REFERENCE_BLOCK_LINE_GAP
|
||||
|
||||
def test_each_step_changes_by_exactly_the_gap(self):
|
||||
"""The stack places ink boxes edge to edge, so the gap is the whole
|
||||
distance between two lines beyond their own ink."""
|
||||
zero = [b.y for b in self._blocks(0.0)]
|
||||
loose = [b.y for b in self._blocks(50.0)]
|
||||
assert len(zero) == len(loose) >= 2
|
||||
for plain, spaced in zip(
|
||||
[a - b for a, b in zip(zero, zero[1:])],
|
||||
[a - b for a, b in zip(loose, loose[1:])],
|
||||
):
|
||||
assert spaced == pytest.approx(plain + 50.0)
|
||||
|
||||
def test_a_negative_gap_overlaps_by_exactly_that_much(self):
|
||||
"""Negative is a supported look, not a failure: -40 tucks each line 40
|
||||
points into the one above rather than colliding by some amount the
|
||||
caller cannot predict."""
|
||||
zero = [b.y for b in self._blocks(0.0)]
|
||||
tucked = [b.y for b in self._blocks(-40.0)]
|
||||
assert len(zero) == len(tucked) >= 2
|
||||
for plain, tight in zip(
|
||||
[a - b for a, b in zip(zero, zero[1:])],
|
||||
[a - b for a, b in zip(tucked, tucked[1:])],
|
||||
):
|
||||
assert tight == pytest.approx(plain - 40.0)
|
||||
|
||||
def test_the_gap_reaches_the_generated_titles(self, temp_fcpxml):
|
||||
"""The config field has to survive the trip to the XML."""
|
||||
def spread(gap):
|
||||
modifier = FCPXMLModifier(temp_fcpxml)
|
||||
titles = modifier.generate_dynamic_subtitles(
|
||||
"Interview_A", WORDS,
|
||||
DynamicSubtitleConfig(style=self.STYLE, line_gap=gap, text_scale=1.0),
|
||||
)
|
||||
ys = [
|
||||
float(t.find("param[@name='Position']").get("value").split()[1])
|
||||
for t in titles
|
||||
]
|
||||
return max(ys) - min(ys)
|
||||
|
||||
assert spread(0.0) < spread(80.0)
|
||||
|
||||
@@ -0,0 +1,112 @@
|
||||
"""Tests for fcpxml/emphasis.py — the emphasis index (pure, no audio needed)."""
|
||||
|
||||
from fcpxml.emphasis import (
|
||||
EmphasisWeights,
|
||||
annotate_emphasis,
|
||||
compute_emphasis,
|
||||
pause_weight,
|
||||
)
|
||||
|
||||
|
||||
def test_zero_input_gives_zero_score():
|
||||
assert compute_emphasis(0.0, 0.0, 0.0, 0.0, 0.0) == 0.0
|
||||
|
||||
|
||||
def test_max_input_gives_max_score():
|
||||
# pause sits inside the "dramatic beat" window, not a scene-change gap
|
||||
score = compute_emphasis(
|
||||
energy=1.0, pitch_delta=1.0, rate_delta=1.0, pause_before=1.5, word_duration=10.0
|
||||
)
|
||||
assert score == 1.0
|
||||
|
||||
|
||||
def test_negative_pitch_delta_uses_magnitude():
|
||||
a = compute_emphasis(0.0, pitch_delta=0.5, rate_delta=0.0, pause_before=0.0, word_duration=0.0)
|
||||
b = compute_emphasis(0.0, pitch_delta=-0.5, rate_delta=0.0, pause_before=0.0, word_duration=0.0)
|
||||
assert a == b > 0.0
|
||||
|
||||
|
||||
def test_duration_saturates_beyond_cap():
|
||||
at_cap = compute_emphasis(0.0, 0.0, 0.0, 0.0, word_duration=1.0, max_duration=1.0)
|
||||
beyond = compute_emphasis(0.0, 0.0, 0.0, 0.0, word_duration=30.0, max_duration=1.0)
|
||||
assert at_cap == beyond
|
||||
|
||||
|
||||
class TestPauseWeight:
|
||||
"""A long gap is a scene change, not emphasis — it must not outrank a
|
||||
word the speaker actually hit hard. Grounded in real footage where 6-9s
|
||||
gaps were topping the emphasis ranking."""
|
||||
|
||||
def test_dramatic_beat_counts_fully(self):
|
||||
assert pause_weight(1.5, max_pause=1.5) == 1.0
|
||||
|
||||
def test_short_beat_counts_proportionally(self):
|
||||
assert pause_weight(0.75, max_pause=1.5) == 0.5
|
||||
|
||||
def test_long_gap_is_ignored(self):
|
||||
assert pause_weight(8.7, max_pause=1.5, ignore_above=3.0) == 0.0
|
||||
|
||||
def test_no_pause_is_zero(self):
|
||||
assert pause_weight(0.0) == 0.0
|
||||
|
||||
def test_cutoff_can_be_disabled(self):
|
||||
assert pause_weight(30.0, max_pause=1.5, ignore_above=0.0) == 1.0
|
||||
|
||||
def test_scene_change_scores_below_a_loud_word(self):
|
||||
gap = compute_emphasis(0.25, 0.04, 0.0, pause_before=6.2, word_duration=0.5)
|
||||
loud = compute_emphasis(1.00, 0.28, 0.0, pause_before=1.9, word_duration=0.2)
|
||||
assert loud > gap
|
||||
|
||||
|
||||
def test_higher_energy_weight_increases_energy_contribution():
|
||||
low_weight = EmphasisWeights(energy=0.1, pitch_variation=0.0, rate_variation=0.0, pause_before=0.0, duration=0.0)
|
||||
high_weight = EmphasisWeights(energy=1.0, pitch_variation=0.0, rate_variation=0.0, pause_before=0.0, duration=0.0)
|
||||
# energy is the only nonzero factor for both weight sets, so normalized
|
||||
# score should be identical regardless of the absolute weight value.
|
||||
score_low = compute_emphasis(0.6, 0.0, 0.0, 0.0, 0.0, weights=low_weight)
|
||||
score_high = compute_emphasis(0.6, 0.0, 0.0, 0.0, 0.0, weights=high_weight)
|
||||
assert score_low == score_high
|
||||
|
||||
|
||||
def test_all_zero_weights_returns_zero_not_error():
|
||||
zero_weights = EmphasisWeights(0.0, 0.0, 0.0, 0.0, 0.0)
|
||||
assert compute_emphasis(1.0, 1.0, 1.0, 1.0, 1.0, weights=zero_weights) == 0.0
|
||||
|
||||
|
||||
def test_weights_round_trip_dict():
|
||||
w = EmphasisWeights(energy=0.4, pitch_variation=0.3, rate_variation=0.1, pause_before=0.1, duration=0.1)
|
||||
restored = EmphasisWeights.from_dict(w.as_dict())
|
||||
assert restored == w
|
||||
|
||||
|
||||
def test_from_dict_fills_missing_with_defaults():
|
||||
restored = EmphasisWeights.from_dict({"energy": 0.9})
|
||||
defaults = EmphasisWeights()
|
||||
assert restored.energy == 0.9
|
||||
assert restored.pitch_variation == defaults.pitch_variation
|
||||
|
||||
|
||||
def test_annotate_emphasis_adds_score_per_word():
|
||||
words = [
|
||||
{"word": "hi", "start": 0.0, "end": 0.3, "energy": 0.2, "pitch_delta": 0.1, "rate_delta": 0.1, "pause_before": 0.0},
|
||||
{"word": "WOW", "start": 1.0, "end": 1.5, "energy": 0.9, "pitch_delta": 0.8, "rate_delta": 0.7, "pause_before": 1.0},
|
||||
]
|
||||
annotated = annotate_emphasis(words)
|
||||
assert len(annotated) == 2
|
||||
assert all("emphasis" in w for w in annotated)
|
||||
assert annotated[1]["emphasis"] > annotated[0]["emphasis"]
|
||||
|
||||
|
||||
def test_annotate_emphasis_does_not_mutate_input():
|
||||
words = [{"word": "hi", "start": 0.0, "end": 0.3, "energy": 0.5}]
|
||||
annotate_emphasis(words)
|
||||
assert "emphasis" not in words[0]
|
||||
|
||||
|
||||
def test_annotate_emphasis_derives_duration_from_start_end():
|
||||
"""A word with no explicit "duration" key gets it from end - start."""
|
||||
base = {"energy": 0.0, "pitch_delta": 0.0, "rate_delta": 0.0, "pause_before": 0.0}
|
||||
derived = annotate_emphasis([{"word": "hi", "start": 1.0, "end": 1.5, **base}], max_duration=0.5)
|
||||
explicit = annotate_emphasis([{"word": "hi", "start": 0.0, "end": 0.0, "duration": 0.5, **base}], max_duration=0.5)
|
||||
# duration is the only nonzero factor in both, so scores must match
|
||||
assert derived[0]["emphasis"] == explicit[0]["emphasis"] > 0.0
|
||||
@@ -518,7 +518,7 @@ class TestDetectBeatsHandler:
|
||||
await handle_detect_beats({"media_path": str(bad)})
|
||||
|
||||
async def test_reports_when_librosa_unavailable(self, tmp_path, monkeypatch):
|
||||
import server as server_mod
|
||||
import server_tools.qc as server_mod
|
||||
from server import handle_detect_beats
|
||||
|
||||
wav = tmp_path / "song.wav"
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
"""Tests for `output_dir` routing — the app's "Pasta do projeto" promise.
|
||||
|
||||
The setting is documented in the UI as "everything generated is saved in
|
||||
here". It used to be applied as a sandbox anchor only, while the filename
|
||||
was still derived in the INPUT's directory — so any call whose output_dir
|
||||
differed from the input's folder failed its own anchor check.
|
||||
"""
|
||||
|
||||
import pytest
|
||||
|
||||
from server_tools._shared import _resolve_io_paths
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def project(tmp_path):
|
||||
"""An .fcpxml in one folder, with a separate chosen output folder."""
|
||||
source_dir = tmp_path / "media"
|
||||
source_dir.mkdir()
|
||||
fcpxml = source_dir / "Projeto.fcpxml"
|
||||
fcpxml.write_text("<fcpxml version='1.13'/>")
|
||||
chosen = tmp_path / "pasta do projeto"
|
||||
chosen.mkdir()
|
||||
return fcpxml, chosen
|
||||
|
||||
|
||||
class TestOutputDirRouting:
|
||||
def test_output_lands_in_the_chosen_folder(self, project):
|
||||
fcpxml, chosen = project
|
||||
_, output_path = _resolve_io_paths({"filepath": str(fcpxml), "output_dir": str(chosen)}, "_voice_edit")
|
||||
assert output_path.startswith(str(chosen))
|
||||
|
||||
def test_filename_keeps_the_suffix_convention(self, project):
|
||||
fcpxml, chosen = project
|
||||
_, output_path = _resolve_io_paths({"filepath": str(fcpxml), "output_dir": str(chosen)}, "_voice_edit")
|
||||
assert output_path.endswith("Projeto_voice_edit.fcpxml")
|
||||
|
||||
def test_cross_directory_call_does_not_raise(self, project):
|
||||
"""The regression: output_dir different from the input folder used to
|
||||
raise "output path escapes allowed directory" every single time."""
|
||||
fcpxml, chosen = project
|
||||
_resolve_io_paths({"filepath": str(fcpxml), "output_dir": str(chosen)}, "_dynamic_subtitles")
|
||||
|
||||
def test_without_output_dir_it_still_writes_beside_the_input(self, project):
|
||||
fcpxml, _ = project
|
||||
_, output_path = _resolve_io_paths({"filepath": str(fcpxml)}, "_modified")
|
||||
assert output_path == str(fcpxml.parent / "Projeto_modified.fcpxml")
|
||||
|
||||
def test_explicit_output_path_still_wins(self, project):
|
||||
fcpxml, chosen = project
|
||||
target = chosen / "nome escolhido.fcpxml"
|
||||
_, output_path = _resolve_io_paths(
|
||||
{"filepath": str(fcpxml), "output_dir": str(chosen), "output_path": str(target)}, "_voice_edit"
|
||||
)
|
||||
assert output_path == str(target)
|
||||
|
||||
def test_explicit_output_path_outside_the_anchor_is_rejected(self, project, tmp_path):
|
||||
"""The anchor must keep constraining explicit paths, not just names."""
|
||||
fcpxml, chosen = project
|
||||
with pytest.raises(ValueError):
|
||||
_resolve_io_paths(
|
||||
{"filepath": str(fcpxml), "output_dir": str(chosen),
|
||||
"output_path": str(tmp_path / "fora.fcpxml")}, "_voice_edit"
|
||||
)
|
||||
@@ -0,0 +1,83 @@
|
||||
"""Tests for the last-project settings — the folder/file the app reopens with.
|
||||
|
||||
The config file (~/.fcp-mcp-server/config.json) is redirected to a tmp_path
|
||||
so these never touch the developer's real settings.
|
||||
"""
|
||||
|
||||
import json
|
||||
|
||||
import pytest
|
||||
|
||||
from fcpxml import model_manager
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def isolated_config(tmp_path, monkeypatch):
|
||||
"""Point model_manager's config file at a throwaway directory."""
|
||||
monkeypatch.setattr(model_manager, "_CONFIG_DIR", tmp_path)
|
||||
monkeypatch.setattr(model_manager, "_CONFIG_FILE", tmp_path / "config.json")
|
||||
return tmp_path / "config.json"
|
||||
|
||||
|
||||
class TestLoadProjectConfig:
|
||||
def test_defaults_when_nothing_stored(self):
|
||||
assert model_manager.load_project_config() == model_manager.DEFAULT_PROJECT_CONFIG
|
||||
|
||||
def test_non_dict_stored_value_falls_back(self, isolated_config):
|
||||
isolated_config.write_text(json.dumps({"project": "nonsense"}))
|
||||
assert model_manager.load_project_config() == model_manager.DEFAULT_PROJECT_CONFIG
|
||||
|
||||
def test_reads_back_what_was_stored(self, tmp_path):
|
||||
folder = tmp_path / "03 - Mastopexia"
|
||||
folder.mkdir()
|
||||
model_manager.save_project_config(folder=str(folder))
|
||||
assert model_manager.load_project_config()["folder"] == str(folder)
|
||||
|
||||
def test_path_that_no_longer_exists_comes_back_empty(self, isolated_config, tmp_path):
|
||||
"""An unmounted volume must degrade to "nothing selected", not a dead path."""
|
||||
isolated_config.write_text(json.dumps({"project": {"folder": str(tmp_path / "gone")}}))
|
||||
assert model_manager.load_project_config()["folder"] == ""
|
||||
|
||||
def test_non_string_stored_value_is_ignored(self, isolated_config):
|
||||
isolated_config.write_text(json.dumps({"project": {"folder": 42}}))
|
||||
assert model_manager.load_project_config()["folder"] == ""
|
||||
|
||||
|
||||
class TestSaveProjectConfig:
|
||||
def test_omitted_field_keeps_its_current_value(self, tmp_path):
|
||||
folder = tmp_path / "projeto"
|
||||
folder.mkdir()
|
||||
project = tmp_path / "projeto" / "Mastopexia.fcpxml"
|
||||
project.write_text("<fcpxml/>")
|
||||
model_manager.save_project_config(folder=str(folder), file=str(project))
|
||||
model_manager.save_project_config(file=str(project))
|
||||
assert model_manager.load_project_config()["folder"] == str(folder)
|
||||
|
||||
def test_empty_string_clears_a_field(self, tmp_path):
|
||||
folder = tmp_path / "projeto"
|
||||
folder.mkdir()
|
||||
model_manager.save_project_config(folder=str(folder))
|
||||
model_manager.save_project_config(folder="")
|
||||
assert model_manager.load_project_config()["folder"] == ""
|
||||
|
||||
def test_home_relative_path_is_expanded(self, isolated_config):
|
||||
model_manager.save_project_config(folder="~/Movies")
|
||||
stored = json.loads(isolated_config.read_text())["project"]["folder"]
|
||||
assert not stored.startswith("~")
|
||||
|
||||
def test_does_not_disturb_other_config_sections(self, isolated_config, tmp_path):
|
||||
isolated_config.write_text(json.dumps({"language": "pt", "selected_model": "large-v3"}))
|
||||
folder = tmp_path / "projeto"
|
||||
folder.mkdir()
|
||||
model_manager.save_project_config(folder=str(folder))
|
||||
data = json.loads(isolated_config.read_text())
|
||||
assert data["language"] == "pt"
|
||||
assert data["selected_model"] == "large-v3"
|
||||
|
||||
def test_returns_the_merged_config(self, tmp_path):
|
||||
folder = tmp_path / "projeto"
|
||||
folder.mkdir()
|
||||
assert model_manager.save_project_config(folder=str(folder)) == {
|
||||
"folder": str(folder),
|
||||
"file": "",
|
||||
}
|
||||
@@ -0,0 +1,167 @@
|
||||
"""Tests for the refine_voice_timeline MCP tool.
|
||||
|
||||
The tool exists because emphasis is *relative*: cut the loudest moment of a
|
||||
recording and every surviving score is still measured against something the
|
||||
viewer never sees. These tests pin the re-normalization actually happening,
|
||||
and the times staying in original source seconds so the result can be fed
|
||||
straight back to apply_voice_actions.
|
||||
"""
|
||||
|
||||
import json
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.test_voice_features_tool import _write_silent_wav
|
||||
from tests.test_voice_timeline_tool import _TRANSCRIPT, patched, wav # noqa: F401
|
||||
|
||||
_ = _write_silent_wav, _TRANSCRIPT # re-exported fixtures need the imports
|
||||
|
||||
|
||||
async def _build(wav_path):
|
||||
from server import handle_build_voice_timeline
|
||||
|
||||
await handle_build_voice_timeline({"media_path": str(wav_path)})
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def two_candidates(monkeypatch):
|
||||
"""A transcript where BOTH lines carry a content word.
|
||||
|
||||
The shared fixture's first line is "isso e" — two function words, which
|
||||
``suggest_zoom_windows`` skips by design, so it can never produce more
|
||||
than one candidate to cap.
|
||||
"""
|
||||
import fcpxml.voice_timeline as vt
|
||||
import server_tools._shared as _shared_mod
|
||||
|
||||
transcript = {
|
||||
"language": "pt",
|
||||
"duration": 4.0,
|
||||
"text": "cirurgia rapida seguranca total",
|
||||
"segments": [
|
||||
{"text": "cirurgia rapida", "start": 0.0, "end": 1.0},
|
||||
{"text": "seguranca total", "start": 2.0, "end": 4.0},
|
||||
],
|
||||
"words": [
|
||||
{"word": "cirurgia", "start": 0.0, "end": 0.4, "confidence": 0.9},
|
||||
{"word": "rapida", "start": 0.5, "end": 0.7, "confidence": 0.9},
|
||||
{"word": "seguranca", "start": 2.0, "end": 2.9, "confidence": 0.9},
|
||||
{"word": "total", "start": 3.0, "end": 3.5, "confidence": 0.9},
|
||||
],
|
||||
}
|
||||
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: transcript)
|
||||
monkeypatch.setattr(vt, "extract_pitch", lambda *a, **k: [(2.4, 260.0), (0.2, 120.0)])
|
||||
monkeypatch.setattr(vt, "extract_energy", lambda *a, **k: [(2.4, 0.95), (0.2, 0.10)])
|
||||
|
||||
|
||||
class TestRefineVoiceTimelineHandler:
|
||||
async def test_requires_an_existing_timeline(self, wav): # noqa: F811
|
||||
from server import handle_refine_voice_timeline
|
||||
|
||||
result = await handle_refine_voice_timeline({"media_path": str(wav), "cuts": []})
|
||||
assert "build_voice_timeline" in result[0].text
|
||||
|
||||
async def test_rejects_disallowed_extension(self, tmp_path):
|
||||
from server import handle_refine_voice_timeline
|
||||
|
||||
bad = tmp_path / "clip.txt"
|
||||
bad.write_text("not audio")
|
||||
with pytest.raises(ValueError):
|
||||
await handle_refine_voice_timeline({"media_path": str(bad), "cuts": []})
|
||||
|
||||
async def test_compares_raw_against_survivors(self, wav, patched): # noqa: F811
|
||||
from server import handle_refine_voice_timeline
|
||||
|
||||
await _build(wav)
|
||||
result = await handle_refine_voice_timeline(
|
||||
{"media_path": str(wav), "cuts": [{"start": 0.0, "end": 1.0}]}
|
||||
)
|
||||
text = result[0].text
|
||||
assert "Survivors only" in text
|
||||
assert "Average emphasis" in text
|
||||
|
||||
async def test_cut_words_are_excluded(self, wav, patched): # noqa: F811
|
||||
from server import handle_refine_voice_timeline
|
||||
|
||||
await _build(wav)
|
||||
result = await handle_refine_voice_timeline(
|
||||
{"media_path": str(wav), "cuts": [{"start": 0.0, "end": 1.0}], "save": True}
|
||||
)
|
||||
assert "_voice_timeline_refined.json" in result[0].text
|
||||
data = json.loads(
|
||||
(wav.parent / "clip_voice_timeline_refined.json").read_text(encoding="utf-8")
|
||||
)
|
||||
words = [w["text"] for s in data["segments"] for w in s["words"]]
|
||||
assert "isso" not in words
|
||||
assert "seguranca" in words
|
||||
|
||||
async def test_times_stay_in_original_source_seconds(self, wav, patched): # noqa: F811
|
||||
"""A cut at the head must NOT slide the survivors back to zero."""
|
||||
from server import handle_refine_voice_timeline
|
||||
|
||||
await _build(wav)
|
||||
await handle_refine_voice_timeline(
|
||||
{"media_path": str(wav), "cuts": [{"start": 0.0, "end": 1.0}], "save": True}
|
||||
)
|
||||
data = json.loads(
|
||||
(wav.parent / "clip_voice_timeline_refined.json").read_text(encoding="utf-8")
|
||||
)
|
||||
first = data["segments"][0]["words"][0]
|
||||
assert first["start"] == pytest.approx(2.0)
|
||||
|
||||
async def test_proposes_zoom_candidates(self, wav, patched): # noqa: F811
|
||||
from server import handle_refine_voice_timeline
|
||||
|
||||
await _build(wav)
|
||||
result = await handle_refine_voice_timeline({"media_path": str(wav), "cuts": []})
|
||||
assert "Zoom Candidates" in result[0].text
|
||||
|
||||
async def test_max_zooms_caps_the_list(self, wav, two_candidates): # noqa: F811
|
||||
from server import handle_refine_voice_timeline
|
||||
|
||||
await _build(wav)
|
||||
|
||||
async def zoom_rows(**extra):
|
||||
result = await handle_refine_voice_timeline(
|
||||
{"media_path": str(wav), "cuts": [], "min_gap": 0.0, **extra}
|
||||
)
|
||||
section = result[0].text.split("## Zoom Candidates", 1)[1]
|
||||
return [
|
||||
ln for ln in section.splitlines()
|
||||
if ln.startswith("| ") and ln.rstrip().endswith("|") and "Start" not in ln
|
||||
]
|
||||
|
||||
assert len(await zoom_rows()) == 2
|
||||
assert len(await zoom_rows(max_zooms=1)) == 1
|
||||
|
||||
async def test_malformed_cut_is_reported_not_raised(self, wav, patched): # noqa: F811
|
||||
from server import handle_refine_voice_timeline
|
||||
|
||||
await _build(wav)
|
||||
result = await handle_refine_voice_timeline(
|
||||
{"media_path": str(wav), "cuts": [{"start": 3.0, "end": 1.0}]}
|
||||
)
|
||||
text = result[0].text
|
||||
assert "Rejected cuts" in text
|
||||
assert "must be after start" in text
|
||||
|
||||
async def test_cutting_everything_says_so(self, wav, patched): # noqa: F811
|
||||
from server import handle_refine_voice_timeline
|
||||
|
||||
await _build(wav)
|
||||
result = await handle_refine_voice_timeline(
|
||||
{"media_path": str(wav), "cuts": [{"start": 0.0, "end": 60.0}]}
|
||||
)
|
||||
assert "removed every word" in result[0].text
|
||||
|
||||
|
||||
class TestRefineVoiceTimelineRegistration:
|
||||
async def test_tool_is_listed(self):
|
||||
from server import list_tools
|
||||
|
||||
assert "refine_voice_timeline" in {t.name for t in await list_tools()}
|
||||
|
||||
async def test_tool_is_dispatched(self):
|
||||
from server import TOOL_HANDLERS, handle_refine_voice_timeline
|
||||
|
||||
assert TOOL_HANDLERS["refine_voice_timeline"] is handle_refine_voice_timeline
|
||||
@@ -0,0 +1,227 @@
|
||||
"""Tests for fcpxml/voice_actions.py — the decision contract.
|
||||
|
||||
Pure functions over untrusted input (a model's decision list), so these
|
||||
cover the rejection paths as carefully as the happy path.
|
||||
"""
|
||||
|
||||
import pytest
|
||||
|
||||
from fcpxml.voice_actions import (
|
||||
MAX_TEXT_LENGTH,
|
||||
VoiceAction,
|
||||
merge_cut_ranges,
|
||||
parse_actions,
|
||||
resolve_actions,
|
||||
shift_after_cuts,
|
||||
)
|
||||
|
||||
|
||||
class TestParseActions:
|
||||
def test_accepts_bare_list(self):
|
||||
actions, errors = parse_actions([{"kind": "cut", "start": 1.0, "end": 2.0}])
|
||||
assert len(actions) == 1 and errors == []
|
||||
|
||||
def test_accepts_actions_envelope(self):
|
||||
actions, errors = parse_actions({"actions": [{"kind": "cut", "start": 1.0, "end": 2.0}]})
|
||||
assert len(actions) == 1 and errors == []
|
||||
|
||||
def test_rejects_non_list(self):
|
||||
actions, errors = parse_actions("cortar tudo")
|
||||
assert actions == [] and len(errors) == 1
|
||||
|
||||
def test_one_bad_row_does_not_discard_the_good_ones(self):
|
||||
actions, errors = parse_actions([
|
||||
{"kind": "cut", "start": 1.0, "end": 2.0},
|
||||
{"kind": "teleport", "start": 3.0, "end": 4.0},
|
||||
{"kind": "zoom", "start": 5.0, "end": 6.0},
|
||||
])
|
||||
assert len(actions) == 2
|
||||
assert len(errors) == 1 and "teleport" in errors[0]
|
||||
|
||||
def test_rejects_unknown_kind(self):
|
||||
_, errors = parse_actions([{"kind": "explode", "start": 0.0, "end": 1.0}])
|
||||
assert "explode" in errors[0]
|
||||
|
||||
def test_rejects_non_numeric_times(self):
|
||||
_, errors = parse_actions([{"kind": "cut", "start": "início", "end": 2.0}])
|
||||
assert "numbers" in errors[0]
|
||||
|
||||
def test_rejects_negative_start(self):
|
||||
_, errors = parse_actions([{"kind": "cut", "start": -1.0, "end": 2.0}])
|
||||
assert "negative" in errors[0]
|
||||
|
||||
def test_rejects_end_before_start(self):
|
||||
_, errors = parse_actions([{"kind": "cut", "start": 5.0, "end": 2.0}])
|
||||
assert "must be after" in errors[0]
|
||||
|
||||
def test_rejects_zero_length(self):
|
||||
_, errors = parse_actions([{"kind": "cut", "start": 2.0, "end": 2.0}])
|
||||
assert errors
|
||||
|
||||
def test_rejects_row_that_is_not_an_object(self):
|
||||
_, errors = parse_actions(["cortar aos 5s"])
|
||||
assert "expected an object" in errors[0]
|
||||
|
||||
def test_kind_is_case_insensitive(self):
|
||||
actions, _ = parse_actions([{"kind": "ZOOM", "start": 1.0, "end": 2.0}])
|
||||
assert actions[0].kind == "zoom"
|
||||
|
||||
def test_preserves_reason_and_speaker(self):
|
||||
actions, _ = parse_actions([
|
||||
{"kind": "zoom", "start": 1.0, "end": 2.0,
|
||||
"reason": "argumento central", "speaker": "SPEAKER_01"}
|
||||
])
|
||||
assert actions[0].reason == "argumento central"
|
||||
assert actions[0].speaker == "SPEAKER_01"
|
||||
|
||||
|
||||
class TestZoomValidation:
|
||||
def test_default_scale_when_absent(self):
|
||||
actions, _ = parse_actions([{"kind": "zoom", "start": 1.0, "end": 2.0}])
|
||||
assert actions[0].params["scale"] == 1.3
|
||||
|
||||
def test_rejects_scale_below_one(self):
|
||||
_, errors = parse_actions([
|
||||
{"kind": "zoom", "start": 1.0, "end": 2.0, "params": {"scale": 0.5}}
|
||||
])
|
||||
assert "outside" in errors[0]
|
||||
|
||||
def test_rejects_absurd_scale(self):
|
||||
_, errors = parse_actions([
|
||||
{"kind": "zoom", "start": 1.0, "end": 2.0, "params": {"scale": 50}}
|
||||
])
|
||||
assert "outside" in errors[0]
|
||||
|
||||
def test_rejects_non_numeric_scale(self):
|
||||
_, errors = parse_actions([
|
||||
{"kind": "zoom", "start": 1.0, "end": 2.0, "params": {"scale": "muito"}}
|
||||
])
|
||||
assert "must be a number" in errors[0]
|
||||
|
||||
|
||||
class TestTextValidation:
|
||||
def test_requires_content(self):
|
||||
_, errors = parse_actions([{"kind": "text", "start": 1.0, "end": 2.0}])
|
||||
assert "params.content" in errors[0]
|
||||
|
||||
def test_rejects_blank_content(self):
|
||||
_, errors = parse_actions([
|
||||
{"kind": "text", "start": 1.0, "end": 2.0, "params": {"content": " "}}
|
||||
])
|
||||
assert "params.content" in errors[0]
|
||||
|
||||
def test_truncates_overlong_content(self):
|
||||
actions, _ = parse_actions([
|
||||
{"kind": "text", "start": 1.0, "end": 2.0, "params": {"content": "A" * 500}}
|
||||
])
|
||||
assert len(actions[0].params["content"]) == MAX_TEXT_LENGTH
|
||||
|
||||
|
||||
class TestMergeCutRanges:
|
||||
def test_sorts_and_merges_overlaps(self):
|
||||
actions = [
|
||||
VoiceAction("cut", 5.0, 7.0),
|
||||
VoiceAction("cut", 1.0, 3.0),
|
||||
VoiceAction("cut", 2.0, 4.0),
|
||||
]
|
||||
assert merge_cut_ranges(actions) == [(1.0, 4.0), (5.0, 7.0)]
|
||||
|
||||
def test_merges_touching_ranges(self):
|
||||
actions = [VoiceAction("cut", 1.0, 2.0), VoiceAction("cut", 2.0, 3.0)]
|
||||
assert merge_cut_ranges(actions) == [(1.0, 3.0)]
|
||||
|
||||
def test_ignores_non_cut_actions(self):
|
||||
assert merge_cut_ranges([VoiceAction("zoom", 1.0, 2.0)]) == []
|
||||
|
||||
|
||||
class TestShiftAfterCuts:
|
||||
def test_time_before_any_cut_is_unchanged(self):
|
||||
assert shift_after_cuts(0.5, [(2.0, 4.0)]) == 0.5
|
||||
|
||||
def test_time_after_a_cut_moves_earlier(self):
|
||||
assert shift_after_cuts(6.0, [(2.0, 4.0)]) == pytest.approx(4.0)
|
||||
|
||||
def test_time_inside_a_cut_is_dropped(self):
|
||||
assert shift_after_cuts(3.0, [(2.0, 4.0)]) is None
|
||||
|
||||
def test_multiple_cuts_accumulate(self):
|
||||
cuts = [(1.0, 2.0), (5.0, 7.0)]
|
||||
assert shift_after_cuts(10.0, cuts) == pytest.approx(7.0)
|
||||
|
||||
def test_no_cuts_is_identity(self):
|
||||
assert shift_after_cuts(3.0, []) == 3.0
|
||||
|
||||
def test_boundary_start_of_cut_is_inside(self):
|
||||
assert shift_after_cuts(2.0, [(2.0, 4.0)]) is None
|
||||
|
||||
def test_boundary_end_of_cut_survives(self):
|
||||
assert shift_after_cuts(4.0, [(2.0, 4.0)]) == pytest.approx(2.0)
|
||||
|
||||
|
||||
class TestResolveActions:
|
||||
def test_zoom_after_a_cut_is_moved_earlier(self):
|
||||
actions = [VoiceAction("cut", 2.0, 4.0), VoiceAction("zoom", 6.0, 7.0)]
|
||||
cuts, placed, dropped = resolve_actions(actions)
|
||||
assert cuts == [(2.0, 4.0)]
|
||||
assert dropped == []
|
||||
assert placed[0].start == pytest.approx(4.0)
|
||||
assert placed[0].end == pytest.approx(5.0)
|
||||
|
||||
def test_zoom_inside_a_cut_is_dropped_not_slid(self):
|
||||
actions = [VoiceAction("cut", 2.0, 8.0), VoiceAction("zoom", 3.0, 4.0)]
|
||||
_, placed, dropped = resolve_actions(actions)
|
||||
assert placed == []
|
||||
assert len(dropped) == 1
|
||||
|
||||
def test_zoom_straddling_a_cut_edge_is_dropped(self):
|
||||
actions = [VoiceAction("cut", 4.0, 8.0), VoiceAction("zoom", 3.0, 5.0)]
|
||||
_, placed, dropped = resolve_actions(actions)
|
||||
assert placed == [] and len(dropped) == 1
|
||||
|
||||
def test_cuts_are_not_returned_as_placed(self):
|
||||
_, placed, _ = resolve_actions([VoiceAction("cut", 1.0, 2.0)])
|
||||
assert placed == []
|
||||
|
||||
def test_without_cuts_everything_keeps_its_time(self):
|
||||
actions = [VoiceAction("zoom", 3.0, 4.0), VoiceAction("text", 5.0, 6.0)]
|
||||
cuts, placed, dropped = resolve_actions(actions)
|
||||
assert cuts == [] and dropped == []
|
||||
assert [(a.start, a.end) for a in placed] == [(3.0, 4.0), (5.0, 6.0)]
|
||||
|
||||
def test_params_survive_the_shift(self):
|
||||
actions = [
|
||||
VoiceAction("cut", 1.0, 2.0),
|
||||
VoiceAction("text", 5.0, 6.0, params={"content": "SEGURANÇA"}),
|
||||
]
|
||||
_, placed, _ = resolve_actions(actions)
|
||||
assert placed[0].params["content"] == "SEGURANÇA"
|
||||
|
||||
|
||||
class TestMarkersSurviveCutEdges:
|
||||
"""A marker is a point in time, not a span. The markers worth keeping are
|
||||
precisely the ones flagging a join, which sit against a cut edge — so
|
||||
requiring their nominal end to survive would drop exactly those."""
|
||||
|
||||
def test_marker_at_a_cut_edge_survives(self):
|
||||
actions = [VoiceAction("cut", 21.9, 127.6), VoiceAction("marker", 21.85, 22.0)]
|
||||
_, placed, dropped = resolve_actions(actions)
|
||||
assert dropped == []
|
||||
assert placed[0].kind == "marker"
|
||||
assert placed[0].start == pytest.approx(21.85)
|
||||
|
||||
def test_marker_inside_removed_material_is_still_dropped(self):
|
||||
actions = [VoiceAction("cut", 20.0, 100.0), VoiceAction("marker", 50.0, 50.2)]
|
||||
_, placed, dropped = resolve_actions(actions)
|
||||
assert placed == [] and len(dropped) == 1
|
||||
|
||||
def test_marker_keeps_its_length_after_shifting(self):
|
||||
actions = [VoiceAction("cut", 0.0, 10.0), VoiceAction("marker", 20.0, 20.5)]
|
||||
_, placed, _ = resolve_actions(actions)
|
||||
assert placed[0].start == pytest.approx(10.0)
|
||||
assert placed[0].duration == pytest.approx(0.5)
|
||||
|
||||
def test_zoom_straddling_an_edge_is_still_dropped(self):
|
||||
"""Only markers get the point-action treatment — a span must fit."""
|
||||
actions = [VoiceAction("cut", 21.9, 127.6), VoiceAction("zoom", 21.0, 22.5)]
|
||||
_, placed, dropped = resolve_actions(actions)
|
||||
assert placed == [] and len(dropped) == 1
|
||||
@@ -0,0 +1,267 @@
|
||||
"""Tests for the apply_voice_actions MCP tool — decisions -> real FCPXML.
|
||||
|
||||
Uses an inline fixture rather than examples/sample.fcpxml so the source
|
||||
windows are explicit and the assertions can be exact.
|
||||
"""
|
||||
|
||||
import shutil
|
||||
|
||||
import pytest
|
||||
|
||||
from fcpxml.safe_xml import safe_parse
|
||||
|
||||
_FIXTURE = """<?xml version="1.0" encoding="UTF-8"?>
|
||||
<fcpxml version="1.13">
|
||||
<resources>
|
||||
<format id="r1" name="FFVideoFormat1080p30" frameDuration="100/3000s" width="1920" height="1080"/>
|
||||
<asset id="a1" name="entrevista" start="0s" duration="600/30s" hasVideo="1" hasAudio="1" format="r1">
|
||||
<media-rep kind="original-media" src="file:///media/entrevista.mov"/>
|
||||
</asset>
|
||||
</resources>
|
||||
<library>
|
||||
<event name="Ev">
|
||||
<project name="Proj">
|
||||
<sequence format="r1" duration="600/30s" tcStart="0s">
|
||||
<spine>
|
||||
<asset-clip name="entrevista" ref="a1" offset="0s" start="0s" duration="600/30s"/>
|
||||
</spine>
|
||||
</sequence>
|
||||
</project>
|
||||
</event>
|
||||
</library>
|
||||
</fcpxml>
|
||||
"""
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def project(tmp_path):
|
||||
path = tmp_path / "proj.fcpxml"
|
||||
path.write_text(_FIXTURE)
|
||||
return path
|
||||
|
||||
|
||||
def _out(project):
|
||||
return project.with_name("proj_voice_edit.fcpxml")
|
||||
|
||||
|
||||
class TestApplyVoiceActionsHandler:
|
||||
async def test_no_actions_reports_instead_of_writing(self, project):
|
||||
from server import handle_apply_voice_actions
|
||||
|
||||
result = await handle_apply_voice_actions({"filepath": str(project)})
|
||||
assert "nothing to apply" in result[0].text.lower()
|
||||
assert not _out(project).exists()
|
||||
|
||||
async def test_all_invalid_actions_writes_nothing(self, project):
|
||||
from server import handle_apply_voice_actions
|
||||
|
||||
result = await handle_apply_voice_actions({
|
||||
"filepath": str(project),
|
||||
"actions": [{"kind": "teleport", "start": 1.0, "end": 2.0}],
|
||||
})
|
||||
assert "No valid actions" in result[0].text
|
||||
assert "teleport" in result[0].text
|
||||
assert not _out(project).exists()
|
||||
|
||||
async def test_applies_zoom_into_the_hosting_clip(self, project):
|
||||
from server import handle_apply_voice_actions
|
||||
|
||||
result = await handle_apply_voice_actions({
|
||||
"filepath": str(project),
|
||||
"actions": [{
|
||||
"kind": "zoom", "start": 5.0, "end": 6.0,
|
||||
"params": {"scale": 1.4}, "reason": "argumento central",
|
||||
}],
|
||||
})
|
||||
assert "argumento central" in result[0].text
|
||||
|
||||
tree = safe_parse(str(_out(project)))
|
||||
transforms = tree.getroot().findall(".//adjust-transform")
|
||||
assert len(transforms) == 1
|
||||
|
||||
async def test_applies_text_title(self, project):
|
||||
from server import handle_apply_voice_actions
|
||||
|
||||
await handle_apply_voice_actions({
|
||||
"filepath": str(project),
|
||||
"actions": [{
|
||||
"kind": "text", "start": 3.0, "end": 4.0,
|
||||
"params": {"content": "SEGURANÇA"},
|
||||
}],
|
||||
})
|
||||
titles = safe_parse(str(_out(project))).getroot().findall(".//title")
|
||||
assert len(titles) == 1
|
||||
texts = [t.text for t in titles[0].iter() if t.text]
|
||||
assert any("SEGURANÇA" in t for t in texts)
|
||||
|
||||
async def test_applies_marker(self, project):
|
||||
from server import handle_apply_voice_actions
|
||||
|
||||
await handle_apply_voice_actions({
|
||||
"filepath": str(project),
|
||||
"actions": [{
|
||||
"kind": "marker", "start": 2.0, "end": 2.5, "reason": "virada",
|
||||
}],
|
||||
})
|
||||
markers = safe_parse(str(_out(project))).getroot().findall(".//marker")
|
||||
assert len(markers) == 1
|
||||
|
||||
async def test_cut_shortens_the_timeline(self, project):
|
||||
from server import handle_apply_voice_actions
|
||||
|
||||
before = safe_parse(str(project)).getroot().find(".//asset-clip").get("duration")
|
||||
result = await handle_apply_voice_actions({
|
||||
"filepath": str(project),
|
||||
"actions": [{"kind": "cut", "start": 5.0, "end": 10.0, "reason": "digressão"}],
|
||||
})
|
||||
assert "Cuts applied" in result[0].text
|
||||
|
||||
clips = safe_parse(str(_out(project))).getroot().findall(".//asset-clip")
|
||||
total = sum(
|
||||
int(c.get("duration").split("/")[0]) / int(c.get("duration").split("/")[1].rstrip("s"))
|
||||
for c in clips
|
||||
)
|
||||
original = int(before.split("/")[0]) / int(before.split("/")[1].rstrip("s"))
|
||||
assert total < original
|
||||
|
||||
async def test_action_inside_a_cut_is_dropped_and_reported(self, project):
|
||||
from server import handle_apply_voice_actions
|
||||
|
||||
result = await handle_apply_voice_actions({
|
||||
"filepath": str(project),
|
||||
"actions": [
|
||||
{"kind": "cut", "start": 4.0, "end": 12.0},
|
||||
{"kind": "zoom", "start": 6.0, "end": 7.0, "reason": "some no material cortado"},
|
||||
],
|
||||
})
|
||||
text = result[0].text
|
||||
assert "Dropped" in text
|
||||
assert "some no material cortado" in text
|
||||
assert safe_parse(str(_out(project))).getroot().findall(".//adjust-transform") == []
|
||||
|
||||
async def test_action_beyond_the_media_is_reported_not_silent(self, project):
|
||||
from server import handle_apply_voice_actions
|
||||
|
||||
result = await handle_apply_voice_actions({
|
||||
"filepath": str(project),
|
||||
"actions": [{"kind": "zoom", "start": 500.0, "end": 501.0}],
|
||||
})
|
||||
assert "Not placed" in result[0].text
|
||||
assert "outside the edited timeline" in result[0].text
|
||||
|
||||
async def test_original_file_is_untouched(self, project):
|
||||
from server import handle_apply_voice_actions
|
||||
|
||||
original = project.read_text()
|
||||
await handle_apply_voice_actions({
|
||||
"filepath": str(project),
|
||||
"actions": [{"kind": "zoom", "start": 5.0, "end": 6.0}],
|
||||
})
|
||||
assert project.read_text() == original
|
||||
|
||||
async def test_mixed_valid_and_invalid_applies_the_valid_ones(self, project):
|
||||
from server import handle_apply_voice_actions
|
||||
|
||||
result = await handle_apply_voice_actions({
|
||||
"filepath": str(project),
|
||||
"actions": [
|
||||
{"kind": "zoom", "start": 5.0, "end": 6.0},
|
||||
{"kind": "zoom", "start": 8.0, "end": 9.0, "params": {"scale": 99}},
|
||||
],
|
||||
})
|
||||
assert "Rejected" in result[0].text
|
||||
assert len(safe_parse(str(_out(project))).getroot().findall(".//adjust-transform")) == 1
|
||||
|
||||
async def test_respects_explicit_output_path(self, project, tmp_path):
|
||||
from server import handle_apply_voice_actions
|
||||
|
||||
target = tmp_path / "custom.fcpxml"
|
||||
await handle_apply_voice_actions({
|
||||
"filepath": str(project),
|
||||
"actions": [{"kind": "marker", "start": 1.0, "end": 2.0}],
|
||||
"output_path": str(target),
|
||||
})
|
||||
assert target.exists()
|
||||
|
||||
async def test_output_is_valid_parseable_fcpxml(self, project):
|
||||
from server import handle_apply_voice_actions
|
||||
|
||||
await handle_apply_voice_actions({
|
||||
"filepath": str(project),
|
||||
"actions": [
|
||||
{"kind": "cut", "start": 2.0, "end": 4.0},
|
||||
{"kind": "zoom", "start": 10.0, "end": 11.0},
|
||||
{"kind": "text", "start": 12.0, "end": 13.0, "params": {"content": "OK"}},
|
||||
],
|
||||
})
|
||||
root = safe_parse(str(_out(project))).getroot()
|
||||
assert root.tag == "fcpxml"
|
||||
assert root.find(".//spine") is not None
|
||||
|
||||
|
||||
class TestSampleFixtureStillParses:
|
||||
"""The applier must not corrupt a real-world document."""
|
||||
|
||||
async def test_real_sample_survives_a_zoom(self, tmp_path):
|
||||
from server import handle_apply_voice_actions
|
||||
|
||||
src = "examples/sample.fcpxml"
|
||||
target = tmp_path / "sample.fcpxml"
|
||||
shutil.copy(src, target)
|
||||
result = await handle_apply_voice_actions({
|
||||
"filepath": str(target),
|
||||
"actions": [{"kind": "marker", "start": 1.0, "end": 2.0, "reason": "teste"}],
|
||||
})
|
||||
assert "Voice Actions Applied" in result[0].text
|
||||
|
||||
|
||||
class TestPlacementsLandOnTheRightPieceAfterCuts:
|
||||
"""Cutting splits a clip into same-named pieces. Placing before cutting
|
||||
duplicated the zoom onto every piece and lost markers outright; a
|
||||
name-based lookup afterwards would always resolve to the first piece.
|
||||
Both bugs shipped past the suite and only showed up on real footage."""
|
||||
|
||||
async def test_zoom_lands_on_exactly_one_piece(self, project):
|
||||
from server import handle_apply_voice_actions
|
||||
|
||||
await handle_apply_voice_actions({
|
||||
"filepath": str(project),
|
||||
"actions": [
|
||||
{"kind": "cut", "start": 2.0, "end": 5.0},
|
||||
{"kind": "cut", "start": 8.0, "end": 12.0},
|
||||
{"kind": "zoom", "start": 15.0, "end": 16.0, "params": {"scale": 1.2}},
|
||||
],
|
||||
})
|
||||
root = safe_parse(str(_out(project))).getroot()
|
||||
assert len(root.findall(".//spine/asset-clip")) == 3
|
||||
assert len(root.findall(".//adjust-transform")) == 1
|
||||
|
||||
async def test_zoom_lands_on_the_last_piece_not_the_first(self, project):
|
||||
from server import handle_apply_voice_actions
|
||||
|
||||
await handle_apply_voice_actions({
|
||||
"filepath": str(project),
|
||||
"actions": [
|
||||
{"kind": "cut", "start": 2.0, "end": 5.0},
|
||||
{"kind": "zoom", "start": 15.0, "end": 16.0},
|
||||
],
|
||||
})
|
||||
clips = safe_parse(str(_out(project))).getroot().findall(".//spine/asset-clip")
|
||||
# the zoom is at 15s source -> 12s after a 3s cut, i.e. the 2nd piece
|
||||
assert clips[0].find("adjust-transform") is None
|
||||
assert clips[1].find("adjust-transform") is not None
|
||||
|
||||
async def test_markers_survive_the_cut(self, project):
|
||||
from server import handle_apply_voice_actions
|
||||
|
||||
result = await handle_apply_voice_actions({
|
||||
"filepath": str(project),
|
||||
"actions": [
|
||||
{"kind": "cut", "start": 5.0, "end": 10.0},
|
||||
{"kind": "marker", "start": 4.9, "end": 5.05, "reason": "emenda"},
|
||||
{"kind": "marker", "start": 15.0, "end": 15.2, "reason": "depois"},
|
||||
],
|
||||
})
|
||||
assert "Dropped" not in result[0].text
|
||||
markers = safe_parse(str(_out(project))).getroot().findall(".//marker")
|
||||
assert len(markers) == 2
|
||||
@@ -0,0 +1,111 @@
|
||||
"""Tests for the Voice Analysis settings — persistence + MCP config tools.
|
||||
|
||||
The config file (~/.fcp-mcp-server/config.json) is redirected to a tmp_path
|
||||
so these never touch the developer's real settings.
|
||||
"""
|
||||
|
||||
import json
|
||||
|
||||
import pytest
|
||||
|
||||
from fcpxml import model_manager
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def isolated_config(tmp_path, monkeypatch):
|
||||
"""Point model_manager's config file at a throwaway directory."""
|
||||
monkeypatch.setattr(model_manager, "_CONFIG_DIR", tmp_path)
|
||||
monkeypatch.setattr(model_manager, "_CONFIG_FILE", tmp_path / "config.json")
|
||||
return tmp_path / "config.json"
|
||||
|
||||
|
||||
class TestLoadVoiceAnalysisConfig:
|
||||
def test_defaults_when_nothing_stored(self):
|
||||
cfg = model_manager.load_voice_analysis_config()
|
||||
assert cfg == model_manager.DEFAULT_VOICE_ANALYSIS_CONFIG
|
||||
|
||||
def test_defaults_are_not_shared_mutable_state(self):
|
||||
cfg = model_manager.load_voice_analysis_config()
|
||||
cfg["emphasis_weights"]["energy"] = 0.99
|
||||
fresh = model_manager.load_voice_analysis_config()
|
||||
assert fresh["emphasis_weights"]["energy"] == 0.30
|
||||
|
||||
def test_malformed_stored_values_fall_back_to_defaults(self, isolated_config):
|
||||
isolated_config.write_text(json.dumps({"voice_analysis": {"energy_threshold": "loud"}}))
|
||||
cfg = model_manager.load_voice_analysis_config()
|
||||
assert cfg["energy_threshold"] == 0.5
|
||||
|
||||
def test_non_dict_stored_value_falls_back(self, isolated_config):
|
||||
isolated_config.write_text(json.dumps({"voice_analysis": "nonsense"}))
|
||||
assert model_manager.load_voice_analysis_config() == model_manager.DEFAULT_VOICE_ANALYSIS_CONFIG
|
||||
|
||||
def test_thresholds_are_clamped_to_unit_range(self, isolated_config):
|
||||
isolated_config.write_text(
|
||||
json.dumps({"voice_analysis": {"energy_threshold": 5.0, "emphasis_floor": -2.0}})
|
||||
)
|
||||
cfg = model_manager.load_voice_analysis_config()
|
||||
assert cfg["energy_threshold"] == 1.0
|
||||
assert cfg["emphasis_floor"] == 0.0
|
||||
|
||||
|
||||
class TestSaveVoiceAnalysisConfig:
|
||||
def test_saves_and_reloads(self):
|
||||
model_manager.save_voice_analysis_config(energy_threshold=0.7, emotion_enabled=True)
|
||||
cfg = model_manager.load_voice_analysis_config()
|
||||
assert cfg["energy_threshold"] == 0.7
|
||||
assert cfg["emotion_enabled"] is True
|
||||
|
||||
def test_omitted_fields_keep_current_value(self):
|
||||
model_manager.save_voice_analysis_config(energy_threshold=0.7)
|
||||
model_manager.save_voice_analysis_config(emphasis_floor=0.9)
|
||||
cfg = model_manager.load_voice_analysis_config()
|
||||
assert cfg["energy_threshold"] == 0.7
|
||||
assert cfg["emphasis_floor"] == 0.9
|
||||
|
||||
def test_partial_weight_update_keeps_other_weights(self):
|
||||
model_manager.save_voice_analysis_config(emphasis_weights={"energy": 0.55})
|
||||
weights = model_manager.load_voice_analysis_config()["emphasis_weights"]
|
||||
assert weights["energy"] == 0.55
|
||||
assert weights["pitch_variation"] == 0.25
|
||||
|
||||
def test_unknown_weight_key_is_ignored(self):
|
||||
model_manager.save_voice_analysis_config(emphasis_weights={"loudness": 9.0})
|
||||
weights = model_manager.load_voice_analysis_config()["emphasis_weights"]
|
||||
assert "loudness" not in weights
|
||||
|
||||
def test_does_not_clobber_unrelated_config_keys(self):
|
||||
model_manager.save_hf_token("tok123")
|
||||
model_manager.save_voice_analysis_config(energy_threshold=0.7)
|
||||
assert model_manager.load_hf_token() == "tok123"
|
||||
|
||||
def test_returns_full_merged_config(self):
|
||||
returned = model_manager.save_voice_analysis_config(energy_threshold=0.7)
|
||||
assert returned == model_manager.load_voice_analysis_config()
|
||||
|
||||
|
||||
class TestVoiceAnalysisConfigTools:
|
||||
async def test_get_reports_current_settings(self):
|
||||
from server import handle_get_voice_analysis_config
|
||||
|
||||
result = await handle_get_voice_analysis_config({})
|
||||
text = result[0].text
|
||||
assert "Voice Analysis Settings" in text
|
||||
assert "Emphasis Weights" in text
|
||||
|
||||
async def test_save_persists_and_echoes_back(self):
|
||||
from server import handle_save_voice_analysis_config
|
||||
|
||||
result = await handle_save_voice_analysis_config(
|
||||
{"energy_threshold": 0.8, "emotion_enabled": True}
|
||||
)
|
||||
assert "0.80" in result[0].text
|
||||
cfg = model_manager.load_voice_analysis_config()
|
||||
assert cfg["energy_threshold"] == 0.8
|
||||
assert cfg["emotion_enabled"] is True
|
||||
|
||||
async def test_save_with_no_arguments_is_a_noop(self):
|
||||
from server import handle_save_voice_analysis_config
|
||||
|
||||
before = model_manager.load_voice_analysis_config()
|
||||
await handle_save_voice_analysis_config({})
|
||||
assert model_manager.load_voice_analysis_config() == before
|
||||
@@ -0,0 +1,137 @@
|
||||
"""Tests for fcpxml/voice_features.py — acoustic features.
|
||||
|
||||
The pure helpers (speech rate, pauses, window averaging) need no audio.
|
||||
The librosa-backed extractors are skipped when the optional [intelligence]
|
||||
extra is absent, matching the pattern in test_media_intel.py.
|
||||
"""
|
||||
|
||||
import math
|
||||
import struct
|
||||
import wave
|
||||
|
||||
import pytest
|
||||
|
||||
from fcpxml.voice_features import (
|
||||
compute_pauses,
|
||||
compute_speech_rate,
|
||||
extract_energy,
|
||||
extract_pitch,
|
||||
features_capability,
|
||||
word_pitch_energy,
|
||||
)
|
||||
|
||||
try:
|
||||
import librosa # noqa: F401
|
||||
|
||||
LIBROSA = True
|
||||
except ImportError:
|
||||
LIBROSA = False
|
||||
|
||||
|
||||
def _write_tone_wav(path: str, hz: float = 220.0, seconds: float = 2.0, rate: int = 22050) -> None:
|
||||
n = int(rate * seconds)
|
||||
frames = [int(20000 * math.sin(2 * math.pi * hz * i / rate)) for i in range(n)]
|
||||
with wave.open(path, "w") as f:
|
||||
f.setnchannels(1)
|
||||
f.setsampwidth(2)
|
||||
f.setframerate(rate)
|
||||
f.writeframes(struct.pack("<%dh" % n, *frames))
|
||||
|
||||
|
||||
class TestComputePauses:
|
||||
def test_first_word_pause_is_time_from_zero(self):
|
||||
words = [{"start": 1.5, "end": 2.0}]
|
||||
assert compute_pauses(words) == [1.5]
|
||||
|
||||
def test_gap_between_words(self):
|
||||
words = [{"start": 0.0, "end": 1.0}, {"start": 2.5, "end": 3.0}]
|
||||
assert compute_pauses(words) == [0.0, 1.5]
|
||||
|
||||
def test_overlapping_words_clamp_to_zero(self):
|
||||
words = [{"start": 0.0, "end": 2.0}, {"start": 1.0, "end": 3.0}]
|
||||
assert compute_pauses(words) == [0.0, 0.0]
|
||||
|
||||
def test_empty_words(self):
|
||||
assert compute_pauses([]) == []
|
||||
|
||||
|
||||
class TestComputeSpeechRate:
|
||||
def test_rate_counts_words_in_trailing_window(self):
|
||||
# 3 words within a 3s window -> 1.0 word/sec at the last one
|
||||
words = [{"start": 0.0}, {"start": 1.0}, {"start": 2.0}]
|
||||
rates = compute_speech_rate(words, window_seconds=3.0)
|
||||
assert rates[-1] == pytest.approx(1.0)
|
||||
|
||||
def test_old_words_fall_out_of_window(self):
|
||||
words = [{"start": 0.0}, {"start": 100.0}]
|
||||
rates = compute_speech_rate(words, window_seconds=3.0)
|
||||
# only the word itself is in range at t=100
|
||||
assert rates[-1] == pytest.approx(1 / 3.0)
|
||||
|
||||
def test_zero_window_is_not_a_division_error(self):
|
||||
assert compute_speech_rate([{"start": 0.0}], window_seconds=0.0) == [0.0]
|
||||
|
||||
def test_empty_words(self):
|
||||
assert compute_speech_rate([]) == []
|
||||
|
||||
|
||||
class TestWordPitchEnergy:
|
||||
def test_averages_track_values_within_word_span(self):
|
||||
words = [{"word": "a", "start": 0.0, "end": 1.0}]
|
||||
pitch = [(0.0, 100.0), (0.5, 200.0), (5.0, 999.0)]
|
||||
energy = [(0.0, 0.2), (1.0, 0.4)]
|
||||
out = word_pitch_energy(words, pitch, energy)
|
||||
assert out[0]["pitch_hz"] == pytest.approx(150.0)
|
||||
assert out[0]["energy"] == pytest.approx(0.3)
|
||||
|
||||
def test_none_when_no_frames_in_span(self):
|
||||
words = [{"word": "a", "start": 10.0, "end": 11.0}]
|
||||
out = word_pitch_energy(words, [(0.0, 100.0)], [(0.0, 0.5)])
|
||||
assert out[0]["pitch_hz"] is None
|
||||
assert out[0]["energy"] is None
|
||||
|
||||
def test_none_tracks_degrade_gracefully(self):
|
||||
out = word_pitch_energy([{"word": "a", "start": 0.0, "end": 1.0}], None, None)
|
||||
assert out[0]["pitch_hz"] is None
|
||||
assert out[0]["energy"] is None
|
||||
|
||||
def test_does_not_mutate_input(self):
|
||||
words = [{"word": "a", "start": 0.0, "end": 1.0}]
|
||||
word_pitch_energy(words, [(0.0, 100.0)], None)
|
||||
assert "pitch_hz" not in words[0]
|
||||
|
||||
def test_word_shorter_than_hop_gets_none_not_crash(self):
|
||||
"""A word briefer than the frame spacing may contain no frame at all."""
|
||||
words = [{"word": "a", "start": 0.501, "end": 0.502}]
|
||||
out = word_pitch_energy(words, [(0.0, 100.0), (1.0, 200.0)], None)
|
||||
assert out[0]["pitch_hz"] is None
|
||||
|
||||
|
||||
class TestExtractorsDegradeGracefully:
|
||||
def test_missing_file_returns_none(self):
|
||||
assert extract_pitch("/nonexistent/audio.wav") is None
|
||||
assert extract_energy("/nonexistent/audio.wav") is None
|
||||
|
||||
|
||||
@pytest.mark.skipif(not LIBROSA, reason="librosa not installed")
|
||||
class TestExtractorsWithLibrosa:
|
||||
def test_capability_is_available(self):
|
||||
ok, _msg = features_capability()
|
||||
assert ok is True
|
||||
|
||||
def test_extracts_pitch_of_known_tone(self, tmp_path):
|
||||
wav = tmp_path / "tone.wav"
|
||||
_write_tone_wav(str(wav), hz=220.0, seconds=2.0)
|
||||
track = extract_pitch(str(wav))
|
||||
assert track is not None and len(track) > 0
|
||||
hz_values = sorted(hz for _t, hz in track)
|
||||
median = hz_values[len(hz_values) // 2]
|
||||
assert median == pytest.approx(220.0, rel=0.1)
|
||||
|
||||
def test_extracts_energy_track(self, tmp_path):
|
||||
wav = tmp_path / "tone.wav"
|
||||
_write_tone_wav(str(wav), seconds=1.0)
|
||||
track = extract_energy(str(wav))
|
||||
assert track is not None and len(track) > 0
|
||||
assert all(rms >= 0 for _t, rms in track)
|
||||
assert max(rms for _t, rms in track) > 0
|
||||
@@ -0,0 +1,147 @@
|
||||
"""Tests for the analyze_voice_features MCP tool.
|
||||
|
||||
librosa and Whisper are monkeypatched so these run without the optional
|
||||
extras, matching the pattern used by TestDetectBeatsHandler.
|
||||
"""
|
||||
|
||||
import json
|
||||
import struct
|
||||
import wave
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
def _write_silent_wav(path: str, seconds: float = 2.0) -> None:
|
||||
n = int(44100 * seconds)
|
||||
with wave.open(path, "w") as f:
|
||||
f.setnchannels(1)
|
||||
f.setsampwidth(2)
|
||||
f.setframerate(44100)
|
||||
f.writeframes(struct.pack("<%dh" % n, *([0] * n)))
|
||||
|
||||
|
||||
_FAKE_TRANSCRIPT = {
|
||||
"language": "pt",
|
||||
"duration": 3.0,
|
||||
"text": "isso e seguranca",
|
||||
"segments": [{"text": "isso e seguranca", "start": 0.0, "end": 3.0}],
|
||||
"words": [
|
||||
{"word": "isso", "start": 0.0, "end": 0.4, "confidence": 0.9},
|
||||
{"word": "e", "start": 0.5, "end": 0.7, "confidence": 0.9},
|
||||
{"word": "seguranca", "start": 2.0, "end": 2.9, "confidence": 0.9},
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def wav(tmp_path):
|
||||
path = tmp_path / "clip.wav"
|
||||
_write_silent_wav(str(path))
|
||||
return path
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def patched_analysis(monkeypatch):
|
||||
"""Make the tool's transcription + librosa extractors deterministic."""
|
||||
import server_tools._shared as _shared_mod
|
||||
import server_tools.voice as server_mod
|
||||
|
||||
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: _FAKE_TRANSCRIPT)
|
||||
monkeypatch.setattr(server_mod, "features_capability", lambda: (True, "ok"))
|
||||
# "seguranca" (2.0-2.9s) is the loud, high-pitched, emphatic word
|
||||
monkeypatch.setattr(
|
||||
server_mod,
|
||||
"extract_pitch",
|
||||
lambda *a, **k: [(0.2, 120.0), (0.6, 118.0), (2.4, 260.0)],
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
server_mod,
|
||||
"extract_energy",
|
||||
lambda *a, **k: [(0.2, 0.10), (0.6, 0.12), (2.4, 0.95)],
|
||||
)
|
||||
|
||||
|
||||
class TestAnalyzeVoiceFeaturesHandler:
|
||||
async def test_reports_when_librosa_unavailable(self, wav, monkeypatch):
|
||||
import server_tools.voice as server_mod
|
||||
from server import handle_analyze_voice_features
|
||||
|
||||
monkeypatch.setattr(
|
||||
server_mod, "features_capability", lambda: (False, "componente librosa ausente.")
|
||||
)
|
||||
result = await handle_analyze_voice_features({"media_path": str(wav)})
|
||||
assert "librosa" in result[0].text.lower()
|
||||
|
||||
async def test_rejects_disallowed_extension(self, tmp_path):
|
||||
from server import handle_analyze_voice_features
|
||||
|
||||
bad = tmp_path / "clip.txt"
|
||||
bad.write_text("not audio")
|
||||
with pytest.raises(ValueError):
|
||||
await handle_analyze_voice_features({"media_path": str(bad)})
|
||||
|
||||
async def test_writes_features_json_with_emphasis_per_word(self, wav, patched_analysis):
|
||||
from server import handle_analyze_voice_features
|
||||
|
||||
result = await handle_analyze_voice_features({"media_path": str(wav)})
|
||||
text = result[0].text
|
||||
|
||||
json_path = wav.parent / "clip_voice_features.json"
|
||||
assert str(json_path) in text
|
||||
data = json.loads(json_path.read_text())
|
||||
assert len(data["words"]) == 3
|
||||
assert all("emphasis" in w for w in data["words"])
|
||||
assert all(0.0 <= w["emphasis"] <= 1.0 for w in data["words"])
|
||||
|
||||
async def test_loudest_word_scores_highest_emphasis(self, wav, patched_analysis):
|
||||
from server import handle_analyze_voice_features
|
||||
|
||||
await handle_analyze_voice_features({"media_path": str(wav)})
|
||||
data = json.loads((wav.parent / "clip_voice_features.json").read_text())
|
||||
by_word = {w["word"]: w["emphasis"] for w in data["words"]}
|
||||
assert by_word["seguranca"] > by_word["isso"]
|
||||
assert by_word["seguranca"] > by_word["e"]
|
||||
|
||||
async def test_persisted_config_is_embedded_in_output(self, wav, patched_analysis):
|
||||
from server import handle_analyze_voice_features
|
||||
|
||||
await handle_analyze_voice_features({"media_path": str(wav)})
|
||||
data = json.loads((wav.parent / "clip_voice_features.json").read_text())
|
||||
assert "energy_threshold" in data["config"]
|
||||
assert "emphasis_weights" in data["config"]
|
||||
|
||||
async def test_empty_transcript_reports_instead_of_crashing(self, wav, monkeypatch):
|
||||
import server_tools._shared as _shared_mod
|
||||
import server_tools.voice as server_mod
|
||||
from server import handle_analyze_voice_features
|
||||
|
||||
monkeypatch.setattr(server_mod, "features_capability", lambda: (True, "ok"))
|
||||
monkeypatch.setattr(
|
||||
_shared_mod, "transcribe", lambda *a, **k: {**_FAKE_TRANSCRIPT, "words": []}
|
||||
)
|
||||
result = await handle_analyze_voice_features({"media_path": str(wav)})
|
||||
assert "no words" in result[0].text.lower()
|
||||
|
||||
async def test_untranscribable_media_reports_install_hint(self, wav, monkeypatch):
|
||||
import server_tools._shared as _shared_mod
|
||||
import server_tools.voice as server_mod
|
||||
from server import handle_analyze_voice_features
|
||||
|
||||
monkeypatch.setattr(server_mod, "features_capability", lambda: (True, "ok"))
|
||||
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: None)
|
||||
result = await handle_analyze_voice_features({"media_path": str(wav)})
|
||||
assert "faster-whisper" in result[0].text
|
||||
|
||||
async def test_missing_pitch_track_degrades_without_crashing(self, wav, monkeypatch):
|
||||
import server_tools._shared as _shared_mod
|
||||
import server_tools.voice as server_mod
|
||||
from server import handle_analyze_voice_features
|
||||
|
||||
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: _FAKE_TRANSCRIPT)
|
||||
monkeypatch.setattr(server_mod, "features_capability", lambda: (True, "ok"))
|
||||
monkeypatch.setattr(server_mod, "extract_pitch", lambda *a, **k: None)
|
||||
monkeypatch.setattr(server_mod, "extract_energy", lambda *a, **k: None)
|
||||
result = await handle_analyze_voice_features({"media_path": str(wav)})
|
||||
assert "Voice Feature Analysis" in result[0].text
|
||||
data = json.loads((wav.parent / "clip_voice_features.json").read_text())
|
||||
assert all(w["emphasis"] >= 0.0 for w in data["words"])
|
||||
@@ -0,0 +1,338 @@
|
||||
"""Tests for fcpxml/voice_timeline.py — the consolidated AI-readable timeline.
|
||||
|
||||
The document's shape is the contract downstream consumers (rules engine, a
|
||||
model reading the JSON) rely on, so these tests pin the shape as much as
|
||||
the values — including that it survives every analysis layer being absent.
|
||||
"""
|
||||
|
||||
import json
|
||||
|
||||
import pytest
|
||||
|
||||
from fcpxml.voice_timeline import (
|
||||
VOICE_TIMELINE_VERSION,
|
||||
build_voice_timeline,
|
||||
enrich_words,
|
||||
load_voice_timeline,
|
||||
save_voice_timeline,
|
||||
voice_timeline_path,
|
||||
)
|
||||
|
||||
_TRANSCRIPT = {
|
||||
"language": "pt",
|
||||
"duration": 4.0,
|
||||
"text": "isso e seguranca total",
|
||||
"segments": [
|
||||
{"text": "isso e", "start": 0.0, "end": 1.0},
|
||||
{"text": "seguranca total", "start": 2.0, "end": 4.0},
|
||||
],
|
||||
"words": [
|
||||
{"word": "isso", "start": 0.0, "end": 0.4, "confidence": 0.9},
|
||||
{"word": "e", "start": 0.5, "end": 0.7, "confidence": 0.9},
|
||||
{"word": "seguranca", "start": 2.0, "end": 2.9, "confidence": 0.9},
|
||||
{"word": "total", "start": 3.0, "end": 3.5, "confidence": 0.9},
|
||||
],
|
||||
}
|
||||
|
||||
# "seguranca" is the loud, high-pitched moment
|
||||
_PITCH = [(0.2, 120.0), (0.6, 118.0), (2.4, 260.0), (3.2, 130.0)]
|
||||
_ENERGY = [(0.2, 0.10), (0.6, 0.12), (2.4, 0.95), (3.2, 0.20)]
|
||||
|
||||
|
||||
class TestEnrichWords:
|
||||
def test_normalizes_energy_against_loudest_word(self):
|
||||
enriched = enrich_words(_TRANSCRIPT["words"], _PITCH, _ENERGY)
|
||||
loudest = max(enriched, key=lambda w: w["energy_norm"])
|
||||
assert loudest["word"] == "seguranca"
|
||||
assert loudest["energy_norm"] == pytest.approx(1.0)
|
||||
|
||||
def test_all_values_stay_within_unit_range(self):
|
||||
enriched = enrich_words(_TRANSCRIPT["words"], _PITCH, _ENERGY)
|
||||
for w in enriched:
|
||||
for key in ("energy_norm", "pitch_delta", "rate_delta", "emphasis"):
|
||||
assert 0.0 <= w[key] <= 1.0, f"{key} out of range on {w['word']}"
|
||||
|
||||
def test_empty_words_returns_empty(self):
|
||||
assert enrich_words([], _PITCH, _ENERGY) == []
|
||||
|
||||
def test_missing_tracks_give_zero_not_crash(self):
|
||||
enriched = enrich_words(_TRANSCRIPT["words"], None, None)
|
||||
assert all(w["energy_norm"] == 0.0 for w in enriched)
|
||||
assert all(w["pitch_delta"] == 0.0 for w in enriched)
|
||||
|
||||
|
||||
class TestBuildVoiceTimeline:
|
||||
@pytest.fixture
|
||||
def timeline(self, monkeypatch):
|
||||
import fcpxml.voice_timeline as vt
|
||||
|
||||
monkeypatch.setattr(vt, "extract_pitch", lambda *a, **k: _PITCH)
|
||||
monkeypatch.setattr(vt, "extract_energy", lambda *a, **k: _ENERGY)
|
||||
return build_voice_timeline("/tmp/clip.wav", _TRANSCRIPT)
|
||||
|
||||
def test_document_has_all_top_level_layers(self, timeline):
|
||||
for key in ("version", "source", "language", "scales", "summary", "speakers", "segments"):
|
||||
assert key in timeline
|
||||
assert timeline["version"] == VOICE_TIMELINE_VERSION
|
||||
|
||||
def test_scales_document_every_word_metric(self, timeline):
|
||||
word = timeline["segments"][0]["words"][0]
|
||||
for metric in timeline["scales"]["word"]:
|
||||
assert metric in word, f"{metric} documented in scales but absent from words"
|
||||
|
||||
def test_scales_document_every_segment_metric(self, timeline):
|
||||
segment = timeline["segments"][0]
|
||||
for metric in timeline["scales"]["segment"]:
|
||||
assert metric in segment, f"{metric} documented in scales but absent from segments"
|
||||
|
||||
def test_take_boundary_flags_a_long_gap(self, timeline):
|
||||
# the fixture has a 1s gap between its two segments -> not a boundary
|
||||
assert timeline["segments"][1]["gap_before"] > 0
|
||||
assert timeline["segments"][1]["take_boundary"] is False
|
||||
|
||||
def test_summary_counts_match_the_detail(self, timeline):
|
||||
summary = timeline["summary"]
|
||||
assert summary["segment_count"] == len(timeline["segments"])
|
||||
total_words = sum(len(s["words"]) for s in timeline["segments"])
|
||||
assert summary["word_count"] == total_words
|
||||
|
||||
def test_words_are_grouped_under_their_segment(self, timeline):
|
||||
first, second = timeline["segments"]
|
||||
assert [w["text"] for w in first["words"]] == ["isso", "e"]
|
||||
assert [w["text"] for w in second["words"]] == ["seguranca", "total"]
|
||||
|
||||
def test_segment_aggregates_reflect_their_words(self, timeline):
|
||||
loud_segment = timeline["segments"][1]
|
||||
quiet_segment = timeline["segments"][0]
|
||||
assert loud_segment["avg_energy"] > quiet_segment["avg_energy"]
|
||||
assert loud_segment["peak_emphasis"] >= max(w["emphasis"] for w in loud_segment["words"])
|
||||
|
||||
def test_peak_moments_are_sorted_by_emphasis(self, timeline):
|
||||
peaks = timeline["summary"]["peak_moments"]
|
||||
assert peaks == sorted(peaks, key=lambda m: m["emphasis"], reverse=True)
|
||||
|
||||
def test_defaults_to_single_speaker_without_token(self, timeline):
|
||||
assert timeline["summary"]["speaker_count"] == 1
|
||||
assert all(w["speaker"] == "SPEAKER_00" for s in timeline["segments"] for w in s["words"])
|
||||
|
||||
def test_is_json_serializable(self, timeline):
|
||||
# the whole point is handing this to a model / writing it to disk
|
||||
assert json.loads(json.dumps(timeline, ensure_ascii=False))["version"]
|
||||
|
||||
|
||||
class TestDegradesWithoutAnalysisLayers:
|
||||
def test_shape_survives_with_no_acoustics(self, monkeypatch):
|
||||
import fcpxml.voice_timeline as vt
|
||||
|
||||
monkeypatch.setattr(vt, "extract_pitch", lambda *a, **k: None)
|
||||
monkeypatch.setattr(vt, "extract_energy", lambda *a, **k: None)
|
||||
timeline = build_voice_timeline("/tmp/clip.wav", _TRANSCRIPT)
|
||||
assert timeline["summary"]["word_count"] == 4
|
||||
assert timeline["summary"]["avg_emphasis"] >= 0.0
|
||||
|
||||
def test_empty_transcript_still_yields_valid_document(self, monkeypatch):
|
||||
import fcpxml.voice_timeline as vt
|
||||
|
||||
monkeypatch.setattr(vt, "extract_pitch", lambda *a, **k: None)
|
||||
monkeypatch.setattr(vt, "extract_energy", lambda *a, **k: None)
|
||||
timeline = build_voice_timeline(
|
||||
"/tmp/clip.wav", {"duration": 0.0, "segments": [], "words": []}
|
||||
)
|
||||
assert timeline["segments"] == []
|
||||
assert timeline["summary"]["word_count"] == 0
|
||||
assert timeline["summary"]["avg_emphasis"] == 0.0
|
||||
|
||||
def test_progress_callback_is_reported(self, monkeypatch):
|
||||
import fcpxml.voice_timeline as vt
|
||||
|
||||
monkeypatch.setattr(vt, "extract_pitch", lambda *a, **k: None)
|
||||
monkeypatch.setattr(vt, "extract_energy", lambda *a, **k: None)
|
||||
seen: list[tuple[float, str]] = []
|
||||
build_voice_timeline("/tmp/clip.wav", _TRANSCRIPT, progress_cb=lambda f, s: seen.append((f, s)))
|
||||
assert seen and all(0.0 <= f <= 1.0 for f, _ in seen)
|
||||
|
||||
|
||||
class TestPersistence:
|
||||
def test_round_trip(self, tmp_path):
|
||||
path = tmp_path / "clip_voice_timeline.json"
|
||||
timeline = {"version": "1.0", "segments": [], "summary": {}}
|
||||
save_voice_timeline(timeline, path)
|
||||
assert load_voice_timeline(path) == timeline
|
||||
|
||||
def test_accented_text_stays_readable(self, tmp_path):
|
||||
path = tmp_path / "t.json"
|
||||
save_voice_timeline({"segments": [{"text": "segurança"}]}, path)
|
||||
assert "segurança" in path.read_text(encoding="utf-8")
|
||||
|
||||
def test_missing_file_returns_none(self, tmp_path):
|
||||
assert load_voice_timeline(tmp_path / "absent.json") is None
|
||||
|
||||
def test_malformed_json_returns_none(self, tmp_path):
|
||||
path = tmp_path / "bad.json"
|
||||
path.write_text("{not json")
|
||||
assert load_voice_timeline(path) is None
|
||||
|
||||
def test_wrong_shape_returns_none(self, tmp_path):
|
||||
path = tmp_path / "other.json"
|
||||
path.write_text('{"something": "else"}')
|
||||
assert load_voice_timeline(path) is None
|
||||
|
||||
def test_path_next_to_media_by_default(self):
|
||||
assert voice_timeline_path("/media/clip.mov").name == "clip_voice_timeline.json"
|
||||
|
||||
def test_path_honours_output_dir(self, tmp_path):
|
||||
path = voice_timeline_path("/media/clip.mov", output_dir=str(tmp_path))
|
||||
assert path.parent == tmp_path
|
||||
|
||||
|
||||
_RAW_WORDS = [
|
||||
# a loud outlier that will be cut, plus quieter material that survives
|
||||
{"text": "GRITO", "start": 1.0, "end": 1.5, "speaker": "SPEAKER_00",
|
||||
"energy": 0.5, "pitch_delta": 0.5, "rate_delta": 0.0, "pause_before": 0.0,
|
||||
"emphasis": 0.5, "energy_raw": 1.0, "pitch_hz": 300.0},
|
||||
{"text": "mastopexia", "start": 10.0, "end": 10.8, "speaker": "SPEAKER_00",
|
||||
"energy": 0.2, "pitch_delta": 0.1, "rate_delta": 0.0, "pause_before": 0.0,
|
||||
"emphasis": 0.1, "energy_raw": 0.4, "pitch_hz": 190.0},
|
||||
{"text": "a", "start": 11.0, "end": 11.1, "speaker": "SPEAKER_00",
|
||||
"energy": 0.15, "pitch_delta": 0.05, "rate_delta": 0.0, "pause_before": 0.0,
|
||||
"emphasis": 0.08, "energy_raw": 0.3, "pitch_hz": 185.0},
|
||||
]
|
||||
|
||||
_RESTRICT_TIMELINE = {
|
||||
"version": "1.0", "source": "x.mp4", "speakers": [],
|
||||
"segments": [
|
||||
{"start": 1.0, "end": 1.5, "speaker": "SPEAKER_00", "text": "GRITO",
|
||||
"gap_before": 0.0, "take_boundary": False, "avg_energy": 0.5,
|
||||
"peak_emphasis": 0.5, "words": [_RAW_WORDS[0]]},
|
||||
{"start": 10.0, "end": 11.1, "speaker": "SPEAKER_00",
|
||||
"text": "mastopexia a", "gap_before": 8.5, "take_boundary": True,
|
||||
"avg_energy": 0.17, "peak_emphasis": 0.1, "words": _RAW_WORDS[1:]},
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
class TestRestrictToKept:
|
||||
"""Emphasis is relative. Cut the loudest moment out and everything left
|
||||
is still scored against something the viewer will never see, so the
|
||||
surviving material has to be re-normalized on its own."""
|
||||
|
||||
def test_cut_words_are_dropped(self):
|
||||
from fcpxml.voice_timeline import restrict_to_kept
|
||||
|
||||
r = restrict_to_kept(_RESTRICT_TIMELINE, [(0.0, 5.0)])
|
||||
texts = [w["text"] for s in r["segments"] for w in s["words"]]
|
||||
assert "GRITO" not in texts
|
||||
assert "mastopexia" in texts
|
||||
|
||||
def test_survivors_are_rescored_against_each_other(self):
|
||||
from fcpxml.voice_timeline import restrict_to_kept
|
||||
|
||||
r = restrict_to_kept(_RESTRICT_TIMELINE, [(0.0, 5.0)])
|
||||
word = next(w for s in r["segments"] for w in s["words"] if w["text"] == "mastopexia")
|
||||
# was 0.2 against the shout's 1.0; alone it becomes the loudest
|
||||
assert word["energy"] == pytest.approx(1.0)
|
||||
|
||||
def test_empty_segments_are_removed(self):
|
||||
from fcpxml.voice_timeline import restrict_to_kept
|
||||
|
||||
r = restrict_to_kept(_RESTRICT_TIMELINE, [(0.0, 5.0)])
|
||||
assert len(r["segments"]) == 1
|
||||
|
||||
def test_no_cuts_keeps_everything(self):
|
||||
from fcpxml.voice_timeline import restrict_to_kept
|
||||
|
||||
r = restrict_to_kept(_RESTRICT_TIMELINE, [])
|
||||
assert sum(len(s["words"]) for s in r["segments"]) == 3
|
||||
|
||||
def test_times_stay_in_original_source_seconds(self):
|
||||
from fcpxml.voice_timeline import restrict_to_kept
|
||||
|
||||
r = restrict_to_kept(_RESTRICT_TIMELINE, [(0.0, 5.0)])
|
||||
assert r["segments"][0]["start"] == 10.0
|
||||
|
||||
|
||||
class TestSuggestZoomWindows:
|
||||
def test_skips_function_words(self):
|
||||
from fcpxml.voice_timeline import restrict_to_kept, suggest_zoom_windows
|
||||
|
||||
r = restrict_to_kept(_RESTRICT_TIMELINE, [(0.0, 5.0)])
|
||||
zooms = suggest_zoom_windows(r)
|
||||
assert zooms and all(z["word"] != "a" for z in zooms)
|
||||
|
||||
def test_window_runs_from_the_word_to_the_end_of_its_line(self):
|
||||
from fcpxml.voice_timeline import restrict_to_kept, suggest_zoom_windows
|
||||
|
||||
r = restrict_to_kept(_RESTRICT_TIMELINE, [(0.0, 5.0)])
|
||||
z = suggest_zoom_windows(r)[0]
|
||||
assert z["start"] == 10.0 and z["end"] == 11.1
|
||||
|
||||
def test_min_gap_keeps_zooms_apart(self):
|
||||
from fcpxml.voice_timeline import suggest_zoom_windows
|
||||
|
||||
timeline = {"segments": [
|
||||
{"start": t, "end": t + 1.0, "text": "linha",
|
||||
"words": [{"text": "palavra", "start": t, "end": t + 0.5, "emphasis": 0.5 - i * 0.01}]}
|
||||
for i, t in enumerate([0.0, 1.0, 2.0, 30.0])
|
||||
]}
|
||||
zooms = suggest_zoom_windows(timeline, min_gap=8.0)
|
||||
assert len(zooms) == 2
|
||||
|
||||
def test_max_zooms_caps_the_result(self):
|
||||
from fcpxml.voice_timeline import suggest_zoom_windows
|
||||
|
||||
timeline = {"segments": [
|
||||
{"start": t, "end": t + 1.0, "text": "linha",
|
||||
"words": [{"text": "palavra", "start": t, "end": t + 0.5, "emphasis": 0.5}]}
|
||||
for t in [0.0, 20.0, 40.0, 60.0]
|
||||
]}
|
||||
assert len(suggest_zoom_windows(timeline, min_gap=8.0, max_zooms=2)) == 2
|
||||
|
||||
def test_results_are_in_chronological_order(self):
|
||||
from fcpxml.voice_timeline import suggest_zoom_windows
|
||||
|
||||
timeline = {"segments": [
|
||||
{"start": t, "end": t + 1.0, "text": "linha",
|
||||
"words": [{"text": "palavra", "start": t, "end": t + 0.5, "emphasis": e}]}
|
||||
for t, e in [(60.0, 0.9), (0.0, 0.5), (30.0, 0.7)]
|
||||
]}
|
||||
zooms = suggest_zoom_windows(timeline, min_gap=8.0)
|
||||
assert [z["start"] for z in zooms] == sorted(z["start"] for z in zooms)
|
||||
|
||||
|
||||
class TestSentenceEnd:
|
||||
"""Transcription segments break on breath, not grammar — a sentence
|
||||
routinely spans several. A zoom ending on a segment boundary releases
|
||||
mid-thought, which is what makes a punch-in feel arbitrary."""
|
||||
|
||||
SEGS = [
|
||||
{"start": 0.0, "end": 5.0, "text": "Aquela mama com um formato, que dá aquele ar",
|
||||
"take_boundary": False},
|
||||
{"start": 5.0, "end": 10.7, "text": "de elegância, isso é desejo de muitas mulheres, né?",
|
||||
"take_boundary": False},
|
||||
{"start": 11.0, "end": 14.0, "text": "Com o tempo, o corpo muda.", "take_boundary": False},
|
||||
]
|
||||
|
||||
def test_extends_past_a_segment_that_does_not_end_a_sentence(self):
|
||||
from fcpxml.voice_timeline import sentence_end
|
||||
|
||||
assert sentence_end(self.SEGS, 0) == 10.7
|
||||
|
||||
def test_stops_at_terminal_punctuation(self):
|
||||
from fcpxml.voice_timeline import sentence_end
|
||||
|
||||
assert sentence_end(self.SEGS, 2) == 14.0
|
||||
|
||||
def test_never_runs_past_a_take_boundary(self):
|
||||
from fcpxml.voice_timeline import sentence_end
|
||||
|
||||
segs = [
|
||||
{"start": 0.0, "end": 5.0, "text": "frase sem fim", "take_boundary": False},
|
||||
{"start": 12.0, "end": 15.0, "text": "outra tomada", "take_boundary": True},
|
||||
]
|
||||
assert sentence_end(segs, 0) == 5.0
|
||||
|
||||
def test_last_segment_without_punctuation_ends_at_itself(self):
|
||||
from fcpxml.voice_timeline import sentence_end
|
||||
|
||||
segs = [{"start": 0.0, "end": 4.0, "text": "sem ponto final", "take_boundary": False}]
|
||||
assert sentence_end(segs, 0) == 4.0
|
||||
@@ -0,0 +1,107 @@
|
||||
"""Tests for the build_voice_timeline MCP tool."""
|
||||
|
||||
import json
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.test_voice_features_tool import _write_silent_wav
|
||||
|
||||
_TRANSCRIPT = {
|
||||
"language": "pt",
|
||||
"duration": 4.0,
|
||||
"text": "isso e seguranca total",
|
||||
"segments": [
|
||||
{"text": "isso e", "start": 0.0, "end": 1.0},
|
||||
{"text": "seguranca total", "start": 2.0, "end": 4.0},
|
||||
],
|
||||
"words": [
|
||||
{"word": "isso", "start": 0.0, "end": 0.4, "confidence": 0.9},
|
||||
{"word": "e", "start": 0.5, "end": 0.7, "confidence": 0.9},
|
||||
{"word": "seguranca", "start": 2.0, "end": 2.9, "confidence": 0.9},
|
||||
{"word": "total", "start": 3.0, "end": 3.5, "confidence": 0.9},
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def wav(tmp_path):
|
||||
path = tmp_path / "clip.wav"
|
||||
_write_silent_wav(str(path))
|
||||
return path
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def patched(monkeypatch):
|
||||
"""Deterministic transcription + acoustics, no optional extras needed."""
|
||||
import fcpxml.voice_timeline as vt
|
||||
import server_tools._shared as _shared_mod
|
||||
|
||||
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: _TRANSCRIPT)
|
||||
monkeypatch.setattr(vt, "extract_pitch", lambda *a, **k: [(2.4, 260.0), (0.2, 120.0)])
|
||||
monkeypatch.setattr(vt, "extract_energy", lambda *a, **k: [(2.4, 0.95), (0.2, 0.10)])
|
||||
|
||||
|
||||
class TestBuildVoiceTimelineHandler:
|
||||
async def test_writes_timeline_json(self, wav, patched):
|
||||
from server import handle_build_voice_timeline
|
||||
|
||||
result = await handle_build_voice_timeline({"media_path": str(wav)})
|
||||
text = result[0].text
|
||||
|
||||
json_path = wav.parent / "clip_voice_timeline.json"
|
||||
assert str(json_path) in text
|
||||
data = json.loads(json_path.read_text(encoding="utf-8"))
|
||||
assert data["summary"]["word_count"] == 4
|
||||
assert len(data["segments"]) == 2
|
||||
|
||||
async def test_reports_which_layers_ran(self, wav, patched):
|
||||
from server import handle_build_voice_timeline
|
||||
|
||||
result = await handle_build_voice_timeline({"media_path": str(wav)})
|
||||
text = result[0].text
|
||||
assert "Analysis Layers" in text
|
||||
assert "Transcript" in text
|
||||
|
||||
async def test_rejects_disallowed_extension(self, tmp_path):
|
||||
from server import handle_build_voice_timeline
|
||||
|
||||
bad = tmp_path / "clip.txt"
|
||||
bad.write_text("not audio")
|
||||
with pytest.raises(ValueError):
|
||||
await handle_build_voice_timeline({"media_path": str(bad)})
|
||||
|
||||
async def test_untranscribable_media_reports_hint(self, wav, monkeypatch):
|
||||
import server_tools._shared as _shared_mod
|
||||
from server import handle_build_voice_timeline
|
||||
|
||||
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: None)
|
||||
result = await handle_build_voice_timeline({"media_path": str(wav)})
|
||||
assert "faster-whisper" in result[0].text
|
||||
|
||||
async def test_uses_persisted_peak_settings(self, wav, patched, monkeypatch):
|
||||
"""A wider percentile must surface more peak moments."""
|
||||
import server_tools.voice as server_mod
|
||||
from server import handle_build_voice_timeline
|
||||
|
||||
def config(percentile):
|
||||
return {
|
||||
"energy_threshold": 0.5,
|
||||
"peak_percentile": percentile,
|
||||
"emphasis_floor": 0.0,
|
||||
"emphasis_weights": {
|
||||
"energy": 0.30, "pitch_variation": 0.25, "rate_variation": 0.20,
|
||||
"pause_before": 0.15, "duration": 0.10,
|
||||
},
|
||||
"emotion_enabled": False,
|
||||
"emotion_sensitivity": 0.5,
|
||||
}
|
||||
|
||||
monkeypatch.setattr(server_mod, "load_voice_analysis_config", lambda: config(0.01))
|
||||
await handle_build_voice_timeline({"media_path": str(wav)})
|
||||
strict = json.loads((wav.parent / "clip_voice_timeline.json").read_text())
|
||||
|
||||
monkeypatch.setattr(server_mod, "load_voice_analysis_config", lambda: config(1.0))
|
||||
await handle_build_voice_timeline({"media_path": str(wav)})
|
||||
loose = json.loads((wav.parent / "clip_voice_timeline.json").read_text())
|
||||
|
||||
assert loose["summary"]["peak_count"] > strict["summary"]["peak_count"]
|
||||
+275
-4
@@ -652,6 +652,7 @@ def test_add_zoom_creates_keyframed_transform(temp_fcpxml):
|
||||
"""4 keyframes: 100% -> scale -> scale -> 100%, all within [start, end]."""
|
||||
modifier = FCPXMLModifier(temp_fcpxml)
|
||||
clip = modifier.add_zoom(clip_id='Broll_Studio', start=1.0, end=3.0, scale=1.3, ease=0.5)
|
||||
frame = float(modifier.frame_duration_fraction())
|
||||
|
||||
transform = clip.find('adjust-transform')
|
||||
assert transform is not None
|
||||
@@ -660,9 +661,23 @@ def test_add_zoom_creates_keyframed_transform(temp_fcpxml):
|
||||
keyframes = param.find('keyframeAnimation').findall('keyframe')
|
||||
assert len(keyframes) == 4
|
||||
assert [kf.get('value') for kf in keyframes] == ['1 1', '1.3 1.3', '1.3 1.3', '1 1']
|
||||
assert all(kf.get('interp') == 'ease' for kf in keyframes)
|
||||
# Bare keyframes — only time and value — matching a zoom exported from
|
||||
# FCP itself. It rejects 'interp' on this vector param (discarding the
|
||||
# whole <param>), and its own export writes no 'curve' either.
|
||||
assert not any(kf.get('interp') for kf in keyframes)
|
||||
assert not any(kf.get('curve') for kf in keyframes)
|
||||
assert all(set(kf.attrib) == {'time', 'value'} for kf in keyframes)
|
||||
# Keyframe times are anchored in the clip's SOURCE timebase (its own
|
||||
# `start`), not clip-relative. This fixture starts at 10s, so a zoom over
|
||||
# clip seconds 1-3 must be written at 11-13s. Writing 1-3s here would put
|
||||
# the animation outside the clip and FCP imports it as nothing.
|
||||
origin = modifier._parse_time(clip.get('start', '0s')).to_seconds()
|
||||
times = [modifier._parse_time(kf.get('time')).to_seconds() for kf in keyframes]
|
||||
assert times == pytest.approx([1.0, 1.5, 2.5, 3.0], abs=0.05)
|
||||
assert origin > 0, "fixture must start off zero, or this asserts nothing"
|
||||
# in over 0.5s, hold, then snap back on the very next frame
|
||||
assert times == pytest.approx(
|
||||
[origin + 1.0, origin + 1.5, origin + 3.0 - frame, origin + 3.0], abs=0.05
|
||||
)
|
||||
assert times == sorted(times)
|
||||
|
||||
|
||||
@@ -699,8 +714,9 @@ def test_add_zoom_window_outside_clip_duration_raises(temp_fcpxml):
|
||||
|
||||
def test_add_zoom_ease_too_long_for_window_raises(temp_fcpxml):
|
||||
modifier = FCPXMLModifier(temp_fcpxml)
|
||||
with pytest.raises(ValueError, match="doesn't fit"):
|
||||
modifier.add_zoom(clip_id='Broll_Studio', start=0.0, end=1.0, ease=1.0)
|
||||
# start away from the clip head so the ramp-in is actually written
|
||||
with pytest.raises(ValueError, match="don't fit"):
|
||||
modifier.add_zoom(clip_id='Broll_Studio', start=1.0, end=2.0, ease=1.5)
|
||||
|
||||
|
||||
def test_change_speed_twice_no_duplicate_elements(temp_fcpxml):
|
||||
@@ -2067,3 +2083,258 @@ def test_remove_trailing_gaps_noop_without_gap():
|
||||
assert len(children) == 1
|
||||
assert children[0].tag == 'asset-clip'
|
||||
Path(f.name).unlink(missing_ok=True)
|
||||
|
||||
|
||||
class TestNTSCFrameAlignment:
|
||||
"""23.976/29.97 timebases must not be reported as misaligned.
|
||||
|
||||
Regression: the check used int(fps), so an exactly frame-aligned NTSC
|
||||
duration (a whole multiple of 1001/24000s) was flagged as broken —
|
||||
every NTSC project produced spurious warnings that buried real ones.
|
||||
"""
|
||||
|
||||
NTSC_DOC = """<?xml version="1.0" encoding="UTF-8"?>
|
||||
<fcpxml version="1.13">
|
||||
<resources>
|
||||
<format id="r1" name="FFVideoFormat1080p2398" frameDuration="1001/24000s" width="1920" height="1080"/>
|
||||
<asset id="a1" name="v" start="0s" duration="24437413/24000s" hasVideo="1" format="r1">
|
||||
<media-rep kind="original-media" src="file:///v.mov"/>
|
||||
</asset>
|
||||
</resources>
|
||||
<library><event name="E"><project name="P">
|
||||
<sequence format="r1" duration="24437413/24000s" tcStart="0s">
|
||||
<spine><asset-clip name="v" ref="a1" offset="0s" start="0s" duration="24437413/24000s"/></spine>
|
||||
</sequence>
|
||||
</project></event></library>
|
||||
</fcpxml>
|
||||
"""
|
||||
|
||||
def _issues(self, xml):
|
||||
from fcpxml.safe_xml import safe_fromstring
|
||||
from fcpxml.writer import validate_fcpxml
|
||||
|
||||
root = safe_fromstring(xml)
|
||||
return [
|
||||
i for i in (validate_fcpxml(root) or [])
|
||||
if "frame" in str(i.issue_type.value)
|
||||
]
|
||||
|
||||
def test_aligned_ntsc_duration_is_not_flagged(self):
|
||||
# 24437413/24000s is exactly 24413 frames of 1001/24000s
|
||||
assert self._issues(self.NTSC_DOC) == []
|
||||
|
||||
def test_genuinely_misaligned_duration_is_still_flagged(self):
|
||||
broken = self.NTSC_DOC.replace(
|
||||
'<asset-clip name="v" ref="a1" offset="0s" start="0s" duration="24437413/24000s"/>',
|
||||
'<asset-clip name="v" ref="a1" offset="0s" start="0s" duration="500/24000s"/>',
|
||||
)
|
||||
assert len(self._issues(broken)) == 1
|
||||
|
||||
def test_message_names_the_real_rate_not_a_rounded_one(self):
|
||||
broken = self.NTSC_DOC.replace('duration="24437413/24000s"/>', 'duration="500/24000s"/>')
|
||||
issues = self._issues(broken)
|
||||
assert issues and "23.976fps" in issues[0].message
|
||||
|
||||
|
||||
class TestZoomPreservesExistingFraming:
|
||||
"""A clip may already carry the editor's reframe — rotation for footage
|
||||
shot sideways, position, a base scale. add_zoom used to delete it, which
|
||||
on real footage brought the zoomed section back rotated."""
|
||||
|
||||
FRAMED = """<?xml version="1.0" encoding="UTF-8"?>
|
||||
<fcpxml version="1.13">
|
||||
<resources>
|
||||
<format id="r1" frameDuration="100/3000s" width="1920" height="1080"/>
|
||||
<asset id="a1" name="v" start="0s" duration="600/30s" hasVideo="1" format="r1">
|
||||
<media-rep kind="original-media" src="file:///v.mov"/>
|
||||
</asset>
|
||||
</resources>
|
||||
<library><event name="E"><project name="P">
|
||||
<sequence format="r1" duration="600/30s" tcStart="0s">
|
||||
<spine>
|
||||
<asset-clip name="v" ref="a1" offset="0s" start="0s" duration="600/30s">
|
||||
<adjust-transform position="0.16 0.66" rotation="90.1" scale="1.77311 1.77311"/>
|
||||
</asset-clip>
|
||||
</spine>
|
||||
</sequence>
|
||||
</project></event></library>
|
||||
</fcpxml>
|
||||
"""
|
||||
|
||||
def _zoomed(self, tmp_path, scale=1.2):
|
||||
from fcpxml.writer import FCPXMLModifier
|
||||
|
||||
path = tmp_path / "framed.fcpxml"
|
||||
path.write_text(self.FRAMED)
|
||||
m = FCPXMLModifier(str(path))
|
||||
m.add_zoom(clip_id="v", start=1.0, end=5.0, scale=scale)
|
||||
return m.root.find(".//adjust-transform")
|
||||
|
||||
def test_rotation_and_position_survive(self, tmp_path):
|
||||
t = self._zoomed(tmp_path)
|
||||
assert t.get("rotation") == "90.1"
|
||||
assert t.get("position") == "0.16 0.66"
|
||||
|
||||
def test_animation_rests_at_the_existing_scale(self, tmp_path):
|
||||
kfs = self._zoomed(tmp_path).findall(".//keyframe")
|
||||
# first and last keyframe return to the clip's own framing, not to 1
|
||||
assert kfs[0].get("value").split()[0].startswith("1.77")
|
||||
assert kfs[-1].get("value").split()[0].startswith("1.77")
|
||||
|
||||
def test_peak_multiplies_the_existing_scale(self, tmp_path):
|
||||
kfs = self._zoomed(tmp_path, scale=2.0).findall(".//keyframe")
|
||||
peak = float(kfs[1].get("value").split()[0])
|
||||
assert peak == pytest.approx(1.77311 * 2.0, rel=1e-4)
|
||||
|
||||
def test_only_one_transform_remains(self, tmp_path):
|
||||
from fcpxml.writer import FCPXMLModifier
|
||||
|
||||
path = tmp_path / "framed.fcpxml"
|
||||
path.write_text(self.FRAMED)
|
||||
m = FCPXMLModifier(str(path))
|
||||
m.add_zoom(clip_id="v", start=1.0, end=5.0, scale=1.2)
|
||||
m.add_zoom(clip_id="v", start=6.0, end=9.0, scale=1.4)
|
||||
assert len(m.root.findall(".//adjust-transform")) == 1
|
||||
|
||||
def test_second_zoom_on_same_clip_keeps_the_real_base_scale(self, tmp_path):
|
||||
"""Found on real footage: two zoom actions landing on disjoint
|
||||
windows of the same post-cut clip. The second add_zoom call used to
|
||||
see the already-animated <param name="scale"> from the first zoom
|
||||
instead of a static attribute, read that as "no framing", and
|
||||
default the base to 1.0 — silently shrinking the shot back to its
|
||||
unframed size for the whole clip wherever no keyframe applied, and
|
||||
discarding the first zoom's animation in the process."""
|
||||
from fcpxml.writer import FCPXMLModifier
|
||||
|
||||
path = tmp_path / "framed.fcpxml"
|
||||
path.write_text(self.FRAMED)
|
||||
m = FCPXMLModifier(str(path))
|
||||
m.add_zoom(clip_id="v", start=1.0, end=5.0, scale=1.2)
|
||||
m.add_zoom(clip_id="v", start=6.0, end=9.0, scale=1.4)
|
||||
kfs = m.root.findall(".//keyframe")
|
||||
|
||||
# Every rest keyframe returns to the clip's real base scale, never 1.0.
|
||||
rest_values = {kf.get("value") for kf in (kfs[0], kfs[3], kfs[4], kfs[-1])}
|
||||
assert rest_values == {"1.77311 1.77311"}
|
||||
|
||||
# Both peaks survive — the second call didn't erase the first.
|
||||
peaks = sorted(float(kf.get("value").split()[0]) for kf in (kfs[1], kfs[5]))
|
||||
assert peaks[0] == pytest.approx(1.77311 * 1.2, rel=1e-4)
|
||||
assert peaks[1] == pytest.approx(1.77311 * 1.4, rel=1e-4)
|
||||
|
||||
def test_overlapping_zoom_on_same_clip_replaces_instead_of_stacking(self, tmp_path):
|
||||
"""Two windows that OVERLAP mean "redo this zoom", not "add another
|
||||
one" — the old keyframes are stale and all of them go, matching
|
||||
test_add_zoom_replaces_existing_zoom's contract."""
|
||||
from fcpxml.writer import FCPXMLModifier
|
||||
|
||||
path = tmp_path / "framed.fcpxml"
|
||||
path.write_text(self.FRAMED)
|
||||
m = FCPXMLModifier(str(path))
|
||||
m.add_zoom(clip_id="v", start=1.0, end=5.0, scale=1.2)
|
||||
m.add_zoom(clip_id="v", start=3.0, end=7.0, scale=1.5)
|
||||
values = [kf.get("value") for kf in m.root.findall(".//keyframe")]
|
||||
assert not any(v.startswith("2.1277") for v in values) # 1.77311*1.2 gone
|
||||
assert any(v.startswith("2.6596") for v in values) # 1.77311*1.5 present
|
||||
|
||||
def test_unframed_clip_still_rests_at_one(self, tmp_path):
|
||||
from fcpxml.writer import FCPXMLModifier
|
||||
|
||||
path = tmp_path / "plain.fcpxml"
|
||||
path.write_text(self.FRAMED.replace(
|
||||
'<adjust-transform position="0.16 0.66" rotation="90.1" scale="1.77311 1.77311"/>', ''))
|
||||
m = FCPXMLModifier(str(path))
|
||||
m.add_zoom(clip_id="v", start=1.0, end=5.0, scale=1.3)
|
||||
kfs = m.root.findall(".//keyframe")
|
||||
assert kfs[0].get("value") == "1 1"
|
||||
assert kfs[1].get("value") == "1.3 1.3"
|
||||
|
||||
|
||||
class TestZoomShapeIsAsymmetric:
|
||||
"""The editorial shape: ramp in fast to land with the emphasised word,
|
||||
hold through the impact phrase, then snap back in a single frame so the
|
||||
video resumes its normal framing without a drift that draws the eye."""
|
||||
|
||||
def _times(self, temp_fcpxml, **kw):
|
||||
from fcpxml.writer import FCPXMLModifier
|
||||
|
||||
m = FCPXMLModifier(temp_fcpxml)
|
||||
# Broll_Studio is 5s long; end=3.0 keeps clear of the hold-at-cut
|
||||
# margin so these exercise the ordinary return-to-framing shape.
|
||||
kw.setdefault('end', 3.0)
|
||||
clip = m.add_zoom(clip_id='Broll_Studio', start=1.0, **kw)
|
||||
origin = m._parse_time(clip.get('start', '0s')).to_seconds()
|
||||
kfs = clip.find('adjust-transform').find('param').find('keyframeAnimation')
|
||||
return m, origin, [m._parse_time(k.get('time')).to_seconds() - origin
|
||||
for k in kfs.findall('keyframe')]
|
||||
|
||||
def test_return_takes_a_single_frame(self, temp_fcpxml):
|
||||
m, _, t = self._times(temp_fcpxml, scale=1.2)
|
||||
frame = float(m.frame_duration_fraction())
|
||||
assert t[3] - t[2] == pytest.approx(frame, abs=0.005)
|
||||
|
||||
def test_ramp_in_is_quick(self, temp_fcpxml):
|
||||
"""Fast enough to land with the emphasised word rather than drift."""
|
||||
_, _, t = self._times(temp_fcpxml, scale=1.2)
|
||||
assert t[1] - t[0] == pytest.approx(0.25, abs=0.05)
|
||||
|
||||
def test_peak_is_held_until_the_return(self, temp_fcpxml):
|
||||
_, _, t = self._times(temp_fcpxml, scale=1.2)
|
||||
# hold spans from the top of the ramp to one frame before the end
|
||||
assert t[2] - t[1] > 1.4
|
||||
|
||||
def test_zoom_opening_at_a_cut_starts_already_zoomed(self, temp_fcpxml):
|
||||
"""The cut is the transition — ramping up from it reads as the shot
|
||||
settling rather than as emphasis."""
|
||||
from fcpxml.writer import FCPXMLModifier
|
||||
|
||||
m = FCPXMLModifier(temp_fcpxml)
|
||||
clip = m.add_zoom(clip_id='Broll_Studio', start=0.1, end=3.0, scale=1.2)
|
||||
kfs = clip.find('adjust-transform').find('param').find('keyframeAnimation')
|
||||
values = [k.get('value') for k in kfs.findall('keyframe')]
|
||||
assert values[0] != values[-1] # opens zoomed, returns to framing
|
||||
assert values[0] == values[-2] # ...and was at the peak from frame one
|
||||
|
||||
def test_opening_at_peak_can_be_forced_off(self, temp_fcpxml):
|
||||
from fcpxml.writer import FCPXMLModifier
|
||||
|
||||
m = FCPXMLModifier(temp_fcpxml)
|
||||
clip = m.add_zoom(
|
||||
clip_id='Broll_Studio', start=0.1, end=3.0, scale=1.2, start_at_peak=False
|
||||
)
|
||||
kfs = clip.find('adjust-transform').find('param').find('keyframeAnimation')
|
||||
assert len(kfs.findall('keyframe')) == 4
|
||||
|
||||
def test_ease_out_can_be_made_gradual(self, temp_fcpxml):
|
||||
_, _, t = self._times(temp_fcpxml, scale=1.2, ease_out=1.0)
|
||||
assert t[3] - t[2] == pytest.approx(1.0, abs=0.05)
|
||||
|
||||
def test_zoom_reaching_the_cut_holds_instead_of_returning(self, temp_fcpxml):
|
||||
"""Returning right before a cut is wasted motion — the next clip
|
||||
opens on its own framing, so the move back reads as a twitch."""
|
||||
_, _, t = self._times(temp_fcpxml, scale=1.2, end=5.0)
|
||||
assert len(t) == 3 # rest, peak, still peak at the cut
|
||||
|
||||
def test_hold_can_be_forced_off_at_a_cut(self, temp_fcpxml):
|
||||
_, _, t = self._times(temp_fcpxml, scale=1.2, end=5.0, hold_at_end=False)
|
||||
assert len(t) == 4
|
||||
|
||||
def test_hold_can_be_forced_on_mid_clip(self, temp_fcpxml):
|
||||
_, _, t = self._times(temp_fcpxml, scale=1.2, end=3.0, hold_at_end=True)
|
||||
assert len(t) == 3
|
||||
|
||||
def test_held_zoom_stays_at_the_peak(self, temp_fcpxml):
|
||||
from fcpxml.writer import FCPXMLModifier
|
||||
|
||||
m = FCPXMLModifier(temp_fcpxml)
|
||||
clip = m.add_zoom(clip_id='Broll_Studio', start=1.0, end=5.0, scale=1.2)
|
||||
kfs = clip.find('adjust-transform').find('param').find('keyframeAnimation')
|
||||
values = [k.get('value') for k in kfs.findall('keyframe')]
|
||||
assert values[-1] == values[-2] != values[0]
|
||||
|
||||
def test_window_too_short_for_the_ramp_is_rejected(self, temp_fcpxml):
|
||||
from fcpxml.writer import FCPXMLModifier
|
||||
|
||||
m = FCPXMLModifier(temp_fcpxml)
|
||||
with pytest.raises(ValueError, match="don't fit"):
|
||||
m.add_zoom(clip_id='Broll_Studio', start=1.0, end=1.2, scale=1.2, ease=1.5)
|
||||
|
||||
Reference in New Issue
Block a user