Files
gart/code/tests/test_voice_actions_tool.py
T
João HenriqueandClaude Opus 5 1bebee4359 feat: etapa 5 do assistente — revisão de ênfases com timeline
Transforma a etapa "colar decisões" numa tela de lapidação: a sugestão da
IA chega carregada e o editor afina frase a frase o que é ênfase e o que
fica fora. Essa marcação é o norte da etapa 6 — só as frases com ênfase
recebem zoom e legenda dinâmica; as demais ficam com legenda comum.

O campo de colar o JSON sobe para a etapa 4, então a numeração das etapas
não muda e a etapa 6 segue intacta.

Backend (fcpxml/phrase_review.py):
- build_phrase_review funde o _voice_timeline.json com as actions da IA
- trim por frase que anda em fronteira de palavra; corte parcial da IA
  chega como trim em vez de ser arredondado fora
- phrase_review_to_actions volta a cuts/zooms + emphasis_spans
- merge_saved_decisions reaplica só as decisões salvas sobre uma revisão
  remontada da análise atual, para reprocessar a voz não ficar mascarado
- resolve_source acha a mídia: o voice timeline guarda só o nome do arquivo

App (SwiftUI):
- layout de sala de edição: preview em cima, inspector à direita, timeline
  atravessando embaixo com seis trilhas rotuladas
- preview enquadra no formato de entrega lido do .fcpxml (fonte horizontal,
  projeto vertical), com alternância para a mídia original
- reprodução pula os trechos removidos e para no fim do trecho
- zoom manual por trecho marcado, sem guardar escala: a forma vem das
  configurações de Análise de Voz no render
- emoção da fala exposta por frase

Correções encontradas no caminho:
- VideoPlayer (AVKit) aborta em runtime no app compilado por swiftc;
  trocado por AVPlayerLayer (ver Engine/docs/05_EXPERIENCIAS.md #22)
- teste que ainda afirmava o default zoom scale=1.3 removido do parser (#21)

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-19 21:29:27 -04:00

294 lines
11 KiB
Python

"""Tests for the apply_voice_actions MCP tool — decisions -> real FCPXML.
Uses an inline fixture rather than examples/sample.fcpxml so the source
windows are explicit and the assertions can be exact.
"""
import shutil
import pytest
from fcpxml.safe_xml import safe_parse
_FIXTURE = """<?xml version="1.0" encoding="UTF-8"?>
<fcpxml version="1.13">
<resources>
<format id="r1" name="FFVideoFormat1080p30" frameDuration="100/3000s" width="1920" height="1080"/>
<asset id="a1" name="entrevista" start="0s" duration="600/30s" hasVideo="1" hasAudio="1" format="r1">
<media-rep kind="original-media" src="file:///media/entrevista.mov"/>
</asset>
</resources>
<library>
<event name="Ev">
<project name="Proj">
<sequence format="r1" duration="600/30s" tcStart="0s">
<spine>
<asset-clip name="entrevista" ref="a1" offset="0s" start="0s" duration="600/30s"/>
</spine>
</sequence>
</project>
</event>
</library>
</fcpxml>
"""
@pytest.fixture
def project(tmp_path):
path = tmp_path / "proj.fcpxml"
path.write_text(_FIXTURE)
return path
def _out(project):
return project.with_name("proj_voice_edit.fcpxml")
class TestApplyVoiceActionsHandler:
async def test_no_actions_reports_instead_of_writing(self, project):
from server import handle_apply_voice_actions
result = await handle_apply_voice_actions({"filepath": str(project)})
assert "nothing to apply" in result[0].text.lower()
assert not _out(project).exists()
async def test_all_invalid_actions_writes_nothing(self, project):
from server import handle_apply_voice_actions
result = await handle_apply_voice_actions({
"filepath": str(project),
"actions": [{"kind": "teleport", "start": 1.0, "end": 2.0}],
})
assert "No valid actions" in result[0].text
assert "teleport" in result[0].text
assert not _out(project).exists()
async def test_applies_zoom_into_the_hosting_clip(self, project):
from server import handle_apply_voice_actions
result = await handle_apply_voice_actions({
"filepath": str(project),
"actions": [{
"kind": "zoom", "start": 5.0, "end": 6.0,
"params": {"scale": 1.4}, "reason": "argumento central",
}],
})
assert "argumento central" in result[0].text
tree = safe_parse(str(_out(project)))
transforms = tree.getroot().findall(".//adjust-transform")
assert len(transforms) == 1
async def test_applies_text_title(self, project):
from server import handle_apply_voice_actions
await handle_apply_voice_actions({
"filepath": str(project),
"actions": [{
"kind": "text", "start": 3.0, "end": 4.0,
"params": {"content": "SEGURANÇA"},
}],
})
titles = safe_parse(str(_out(project))).getroot().findall(".//title")
assert len(titles) == 1
texts = [t.text for t in titles[0].iter() if t.text]
assert any("SEGURANÇA" in t for t in texts)
async def test_text_callout_defaults_fit_the_frame(self, project):
from fcpxml.writer import FCPXMLModifier
from server import handle_apply_voice_actions
await handle_apply_voice_actions({
"filepath": str(project),
"actions": [{
"kind": "text", "start": 3.0, "end": 4.0,
"params": {"content": "PRÓTESES DE SILICONE"},
}],
})
modifier = FCPXMLModifier(str(_out(project)))
report = modifier.validate_subtitle_layout()
assert report["summary"]["outside_frame"] == 0
title = modifier.root.find(".//title")
style = title.find("text-style-def/text-style")
position = next(
p.get("value")
for p in title.findall("param")
if p.get("name") == "Position"
)
assert float(style.get("fontSize")) < 530
assert position != "0 0"
async def test_applies_marker(self, project):
from server import handle_apply_voice_actions
await handle_apply_voice_actions({
"filepath": str(project),
"actions": [{
"kind": "marker", "start": 2.0, "end": 2.5, "reason": "virada",
}],
})
markers = safe_parse(str(_out(project))).getroot().findall(".//marker")
assert len(markers) == 1
async def test_cut_shortens_the_timeline(self, project):
from server import handle_apply_voice_actions
before = safe_parse(str(project)).getroot().find(".//asset-clip").get("duration")
result = await handle_apply_voice_actions({
"filepath": str(project),
"actions": [{"kind": "cut", "start": 5.0, "end": 10.0, "reason": "digressão"}],
})
assert "Cuts applied" in result[0].text
clips = safe_parse(str(_out(project))).getroot().findall(".//asset-clip")
total = sum(
int(c.get("duration").split("/")[0]) / int(c.get("duration").split("/")[1].rstrip("s"))
for c in clips
)
original = int(before.split("/")[0]) / int(before.split("/")[1].rstrip("s"))
assert total < original
async def test_action_inside_a_cut_is_dropped_and_reported(self, project):
from server import handle_apply_voice_actions
result = await handle_apply_voice_actions({
"filepath": str(project),
"actions": [
{"kind": "cut", "start": 4.0, "end": 12.0},
{"kind": "zoom", "start": 6.0, "end": 7.0, "reason": "some no material cortado"},
],
})
text = result[0].text
assert "Dropped" in text
assert "some no material cortado" in text
assert safe_parse(str(_out(project))).getroot().findall(".//adjust-transform") == []
async def test_action_beyond_the_media_is_reported_not_silent(self, project):
from server import handle_apply_voice_actions
result = await handle_apply_voice_actions({
"filepath": str(project),
"actions": [{"kind": "zoom", "start": 500.0, "end": 501.0}],
})
assert "Not placed" in result[0].text
assert "outside the edited timeline" in result[0].text
async def test_original_file_is_untouched(self, project):
from server import handle_apply_voice_actions
original = project.read_text()
await handle_apply_voice_actions({
"filepath": str(project),
"actions": [{"kind": "zoom", "start": 5.0, "end": 6.0}],
})
assert project.read_text() == original
async def test_mixed_valid_and_invalid_applies_the_valid_ones(self, project):
from server import handle_apply_voice_actions
result = await handle_apply_voice_actions({
"filepath": str(project),
"actions": [
{"kind": "zoom", "start": 5.0, "end": 6.0},
{"kind": "zoom", "start": 8.0, "end": 9.0, "params": {"scale": 99}},
],
})
assert "Rejected" in result[0].text
assert len(safe_parse(str(_out(project))).getroot().findall(".//adjust-transform")) == 1
async def test_respects_explicit_output_path(self, project, tmp_path):
from server import handle_apply_voice_actions
target = tmp_path / "custom.fcpxml"
await handle_apply_voice_actions({
"filepath": str(project),
"actions": [{"kind": "marker", "start": 1.0, "end": 2.0}],
"output_path": str(target),
})
assert target.exists()
async def test_output_is_valid_parseable_fcpxml(self, project):
from server import handle_apply_voice_actions
await handle_apply_voice_actions({
"filepath": str(project),
"actions": [
{"kind": "cut", "start": 2.0, "end": 4.0},
{"kind": "zoom", "start": 10.0, "end": 11.0},
{"kind": "text", "start": 12.0, "end": 13.0, "params": {"content": "OK"}},
],
})
root = safe_parse(str(_out(project))).getroot()
assert root.tag == "fcpxml"
assert root.find(".//spine") is not None
class TestSampleFixtureStillParses:
"""The applier must not corrupt a real-world document."""
async def test_real_sample_survives_a_zoom(self, tmp_path):
from server import handle_apply_voice_actions
src = "examples/sample.fcpxml"
target = tmp_path / "sample.fcpxml"
shutil.copy(src, target)
result = await handle_apply_voice_actions({
"filepath": str(target),
"actions": [{"kind": "marker", "start": 1.0, "end": 2.0, "reason": "teste"}],
})
assert "Voice Actions Applied" in result[0].text
class TestPlacementsLandOnTheRightPieceAfterCuts:
"""Cutting splits a clip into same-named pieces. Placing before cutting
duplicated the zoom onto every piece and lost markers outright; a
name-based lookup afterwards would always resolve to the first piece.
Both bugs shipped past the suite and only showed up on real footage."""
async def test_zoom_lands_on_exactly_one_piece(self, project):
from server import handle_apply_voice_actions
await handle_apply_voice_actions({
"filepath": str(project),
"actions": [
{"kind": "cut", "start": 2.0, "end": 5.0},
{"kind": "cut", "start": 8.0, "end": 12.0},
{"kind": "zoom", "start": 15.0, "end": 16.0, "params": {"scale": 1.2}},
],
})
root = safe_parse(str(_out(project))).getroot()
assert len(root.findall(".//spine/asset-clip")) == 3
assert len(root.findall(".//adjust-transform")) == 1
async def test_zoom_lands_on_the_last_piece_not_the_first(self, project):
from server import handle_apply_voice_actions
await handle_apply_voice_actions({
"filepath": str(project),
"actions": [
{"kind": "cut", "start": 2.0, "end": 5.0},
{"kind": "zoom", "start": 15.0, "end": 16.0},
],
})
clips = safe_parse(str(_out(project))).getroot().findall(".//spine/asset-clip")
# the zoom is at 15s source -> 12s after a 3s cut, i.e. the 2nd piece
assert clips[0].find("adjust-transform") is None
assert clips[1].find("adjust-transform") is not None
async def test_markers_survive_the_cut(self, project):
from server import handle_apply_voice_actions
result = await handle_apply_voice_actions({
"filepath": str(project),
"actions": [
{"kind": "cut", "start": 5.0, "end": 10.0},
{"kind": "marker", "start": 4.9, "end": 5.05, "reason": "emenda"},
{"kind": "marker", "start": 15.0, "end": 15.2, "reason": "depois"},
],
})
assert "Dropped" not in result[0].text
markers = safe_parse(str(_out(project))).getroot().findall(".//marker")
assert len(markers) == 2