feat: etapa 5 do assistente — revisão de ênfases com timeline
Transforma a etapa "colar decisões" numa tela de lapidação: a sugestão da IA chega carregada e o editor afina frase a frase o que é ênfase e o que fica fora. Essa marcação é o norte da etapa 6 — só as frases com ênfase recebem zoom e legenda dinâmica; as demais ficam com legenda comum. O campo de colar o JSON sobe para a etapa 4, então a numeração das etapas não muda e a etapa 6 segue intacta. Backend (fcpxml/phrase_review.py): - build_phrase_review funde o _voice_timeline.json com as actions da IA - trim por frase que anda em fronteira de palavra; corte parcial da IA chega como trim em vez de ser arredondado fora - phrase_review_to_actions volta a cuts/zooms + emphasis_spans - merge_saved_decisions reaplica só as decisões salvas sobre uma revisão remontada da análise atual, para reprocessar a voz não ficar mascarado - resolve_source acha a mídia: o voice timeline guarda só o nome do arquivo App (SwiftUI): - layout de sala de edição: preview em cima, inspector à direita, timeline atravessando embaixo com seis trilhas rotuladas - preview enquadra no formato de entrega lido do .fcpxml (fonte horizontal, projeto vertical), com alternância para a mídia original - reprodução pula os trechos removidos e para no fim do trecho - zoom manual por trecho marcado, sem guardar escala: a forma vem das configurações de Análise de Voz no render - emoção da fala exposta por frase Correções encontradas no caminho: - VideoPlayer (AVKit) aborta em runtime no app compilado por swiftc; trocado por AVPlayerLayer (ver Engine/docs/05_EXPERIENCIAS.md #22) - teste que ainda afirmava o default zoom scale=1.3 removido do parser (#21) Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
e7748c2c58
commit
1bebee4359
@@ -29,6 +29,7 @@ from fcpxml.text_layout import (
|
||||
ink_extent,
|
||||
)
|
||||
from fcpxml.writer import FCPXMLModifier
|
||||
from server_tools.subtitles import _words_overlapping_clip
|
||||
|
||||
SAMPLE = Path(__file__).parent.parent / "examples" / "sample.fcpxml"
|
||||
def font_points(style) -> float:
|
||||
@@ -51,6 +52,21 @@ WORDS = [
|
||||
]
|
||||
|
||||
|
||||
def test_words_overlapping_clip_keeps_word_that_starts_just_before_in_point():
|
||||
words = [
|
||||
{"word": "Aquela", "start": 2.03, "end": 2.69},
|
||||
{"word": "mama", "start": 2.69, "end": 2.89},
|
||||
{"word": "fora", "start": 10.0, "end": 10.2},
|
||||
]
|
||||
|
||||
clip_words = _words_overlapping_clip(words, 2.0437166666666666, 3.0)
|
||||
|
||||
assert clip_words == [
|
||||
{"word": "Aquela", "start": 0.0, "end": pytest.approx(0.6462833333333332)},
|
||||
{"word": "mama", "start": pytest.approx(0.6462833333333332), "end": pytest.approx(0.8462833333333334)},
|
||||
]
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def temp_fcpxml():
|
||||
with tempfile.NamedTemporaryFile(suffix=".fcpxml", delete=False) as f:
|
||||
|
||||
@@ -0,0 +1,479 @@
|
||||
"""Tests for the phrase review model (voice timeline + AI actions → editable script)."""
|
||||
|
||||
import json
|
||||
|
||||
import pytest
|
||||
|
||||
from fcpxml.phrase_review import (
|
||||
TRACK_BACKSTAGE,
|
||||
TRACK_SCRIPT,
|
||||
ZOOM_SCALE_BY_LEVEL,
|
||||
build_phrase_review,
|
||||
load_phrase_review,
|
||||
merge_saved_decisions,
|
||||
phrase_review_to_actions,
|
||||
resolve_source,
|
||||
review_paths,
|
||||
save_phrase_review,
|
||||
snap_to_words,
|
||||
)
|
||||
|
||||
|
||||
def _words(spans, emphasis=0.0):
|
||||
return [
|
||||
{
|
||||
"text": f"w{i}",
|
||||
"start": start,
|
||||
"end": end,
|
||||
"energy": 0.5,
|
||||
"emphasis": emphasis,
|
||||
}
|
||||
for i, (start, end) in enumerate(spans)
|
||||
]
|
||||
|
||||
|
||||
def _timeline(segments):
|
||||
return {"source": "/tmp/take.mov", "speakers": ["SPEAKER_00"], "segments": segments}
|
||||
|
||||
|
||||
def _segment(start, end, text="linha", peak=0.1, take_boundary=False, words=None):
|
||||
return {
|
||||
"start": start,
|
||||
"end": end,
|
||||
"text": text,
|
||||
"speaker": "SPEAKER_00",
|
||||
"peak_emphasis": peak,
|
||||
"take_boundary": take_boundary,
|
||||
"gap_before": 0.0,
|
||||
"words": words if words is not None else _words([(start, end)]),
|
||||
}
|
||||
|
||||
|
||||
class TestBuildFromAcoustics:
|
||||
def test_emphasis_levels_follow_peak_thresholds(self):
|
||||
review = build_phrase_review(
|
||||
_timeline(
|
||||
[
|
||||
_segment(0, 1, peak=0.10),
|
||||
_segment(1, 2, peak=0.30),
|
||||
_segment(2, 3, peak=0.50),
|
||||
_segment(3, 4, peak=0.90),
|
||||
]
|
||||
)
|
||||
)
|
||||
assert [p["emphasis"] for p in review["phrases"]] == [0, 1, 2, 3]
|
||||
|
||||
def test_every_phrase_starts_active_without_actions(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 1), _segment(1, 2)]))
|
||||
assert all(p["active"] for p in review["phrases"])
|
||||
assert all(p["track"] == TRACK_SCRIPT for p in review["phrases"])
|
||||
|
||||
def test_carries_text_speaker_and_words(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 2, text=" olá ")]))
|
||||
phrase = review["phrases"][0]
|
||||
assert phrase["text"] == "olá"
|
||||
assert phrase["speaker"] == "SPEAKER_00"
|
||||
assert phrase["words"][0]["text"] == "w0"
|
||||
assert review["duration"] == 2.0
|
||||
|
||||
|
||||
class TestCutsDeactivate:
|
||||
def test_fully_cut_phrase_is_inactive(self):
|
||||
review = build_phrase_review(
|
||||
_timeline([_segment(0, 2), _segment(2, 4)]),
|
||||
{"actions": [{"kind": "cut", "start": 0, "end": 2, "reason": "gaguejou"}]},
|
||||
)
|
||||
assert review["phrases"][0]["active"] is False
|
||||
assert review["phrases"][0]["reason"] == "gaguejou"
|
||||
assert review["phrases"][1]["active"] is True
|
||||
|
||||
def test_small_overlap_keeps_the_phrase(self):
|
||||
# 0.2s off a 2s line is a trim, not a removal.
|
||||
review = build_phrase_review(
|
||||
_timeline([_segment(1, 3, words=_words([(1, 1.2), (1.2, 3)]))]),
|
||||
{"actions": [{"kind": "cut", "start": 0.5, "end": 1.2}]},
|
||||
)
|
||||
assert review["phrases"][0]["active"] is True
|
||||
|
||||
def test_majority_overlap_deactivates(self):
|
||||
review = build_phrase_review(
|
||||
_timeline([_segment(0, 2)]),
|
||||
{"actions": [{"kind": "cut", "start": 0, "end": 1.5}]},
|
||||
)
|
||||
assert review["phrases"][0]["active"] is False
|
||||
|
||||
def test_inactive_after_take_boundary_is_backstage(self):
|
||||
review = build_phrase_review(
|
||||
_timeline([_segment(10, 12, take_boundary=True)]),
|
||||
{"actions": [{"kind": "cut", "start": 10, "end": 12}]},
|
||||
)
|
||||
assert review["phrases"][0]["track"] == TRACK_BACKSTAGE
|
||||
|
||||
|
||||
class TestTrimFromPartialCuts:
|
||||
def test_head_cut_becomes_a_trim_snapped_to_a_word(self):
|
||||
review = build_phrase_review(
|
||||
_timeline([_segment(1, 4, words=_words([(1, 1.4), (1.4, 4)]))]),
|
||||
{"actions": [{"kind": "cut", "start": 0.8, "end": 1.35}]},
|
||||
)
|
||||
phrase = review["phrases"][0]
|
||||
assert phrase["active"] is True
|
||||
assert phrase["trim_start"] == 1.4 # snapped to the second word's start
|
||||
assert phrase["trim_end"] == 4.0
|
||||
|
||||
def test_tail_cut_becomes_a_trim(self):
|
||||
review = build_phrase_review(
|
||||
_timeline([_segment(0, 3, words=_words([(0, 2.5), (2.5, 3)]))]),
|
||||
{"actions": [{"kind": "cut", "start": 2.6, "end": 3.5}]},
|
||||
)
|
||||
phrase = review["phrases"][0]
|
||||
assert phrase["trim_start"] == 0.0
|
||||
assert phrase["trim_end"] == 2.5
|
||||
|
||||
def test_untouched_phrase_trims_to_its_own_bounds(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 2)]))
|
||||
phrase = review["phrases"][0]
|
||||
assert (phrase["trim_start"], phrase["trim_end"]) == (0.0, 2.0)
|
||||
|
||||
|
||||
class TestAIDirectionWins:
|
||||
def test_zoom_action_sets_the_level_over_the_heuristic(self):
|
||||
review = build_phrase_review(
|
||||
_timeline([_segment(0, 2, peak=0.05)]),
|
||||
{
|
||||
"actions": [
|
||||
{
|
||||
"kind": "zoom",
|
||||
"start": 0.5,
|
||||
"end": 0.9,
|
||||
"params": {"scale": 1.5},
|
||||
"reason": "virada da história",
|
||||
}
|
||||
]
|
||||
},
|
||||
)
|
||||
phrase = review["phrases"][0]
|
||||
assert phrase["emphasis"] == 3
|
||||
assert phrase["reason"] == "virada da história"
|
||||
|
||||
def test_text_action_marks_emphasis(self):
|
||||
review = build_phrase_review(
|
||||
_timeline([_segment(0, 2, peak=0.0)]),
|
||||
{
|
||||
"actions": [
|
||||
{
|
||||
"kind": "text",
|
||||
"start": 0.5,
|
||||
"end": 1.0,
|
||||
"params": {"content": "3x mais rápido"},
|
||||
}
|
||||
]
|
||||
},
|
||||
)
|
||||
assert review["phrases"][0]["emphasis"] == 2
|
||||
|
||||
def test_highest_level_wins_when_several_actions_overlap(self):
|
||||
review = build_phrase_review(
|
||||
_timeline([_segment(0, 4)]),
|
||||
{
|
||||
"actions": [
|
||||
{"kind": "zoom", "start": 0.2, "end": 0.5, "params": {"scale": 1.15}},
|
||||
{"kind": "zoom", "start": 2.0, "end": 2.4, "params": {"scale": 1.5}},
|
||||
]
|
||||
},
|
||||
)
|
||||
assert review["phrases"][0]["emphasis"] == 3
|
||||
|
||||
def test_malformed_rows_are_reported_not_fatal(self):
|
||||
review = build_phrase_review(
|
||||
_timeline([_segment(0, 2)]),
|
||||
{"actions": [{"kind": "voar", "start": 0, "end": 1}]},
|
||||
)
|
||||
assert len(review["errors"]) == 1
|
||||
assert review["phrases"][0]["active"] is True
|
||||
|
||||
|
||||
class TestBackToActions:
|
||||
def test_inactive_phrase_becomes_a_cut(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 2), _segment(2, 4)]))
|
||||
review["phrases"][0]["active"] = False
|
||||
result = phrase_review_to_actions(review)
|
||||
cuts = [a for a in result["actions"] if a["kind"] == "cut"]
|
||||
assert len(cuts) == 1
|
||||
assert (cuts[0]["start"], cuts[0]["end"]) == (0.0, 2.0)
|
||||
|
||||
def test_emphasis_becomes_a_zoom_and_a_span(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 2)]))
|
||||
review["phrases"][0]["emphasis"] = 2
|
||||
result = phrase_review_to_actions(review)
|
||||
zooms = [a for a in result["actions"] if a["kind"] == "zoom"]
|
||||
assert zooms[0]["params"]["scale"] == ZOOM_SCALE_BY_LEVEL[2]
|
||||
assert result["emphasis_spans"] == [
|
||||
{"start": 0.0, "end": 2.0, "level": 2, "text": "linha"}
|
||||
]
|
||||
|
||||
def test_level_zero_produces_nothing(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 2)]))
|
||||
review["phrases"][0]["emphasis"] = 0
|
||||
result = phrase_review_to_actions(review)
|
||||
assert result["actions"] == []
|
||||
assert result["emphasis_spans"] == []
|
||||
|
||||
def test_trim_becomes_head_and_tail_cuts(self):
|
||||
review = build_phrase_review(
|
||||
_timeline([_segment(0, 4, words=_words([(0, 1), (1, 3), (3, 4)]))])
|
||||
)
|
||||
review["phrases"][0]["trim_start"] = 1.0
|
||||
review["phrases"][0]["trim_end"] = 3.0
|
||||
result = phrase_review_to_actions(review)
|
||||
spans = [(a["start"], a["end"]) for a in result["actions"] if a["kind"] == "cut"]
|
||||
assert spans == [(0.0, 1.0), (3.0, 4.0)]
|
||||
|
||||
def test_inactive_phrase_is_cut_whole_ignoring_its_trim(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 4)]))
|
||||
review["phrases"][0].update({"active": False, "trim_start": 1.0, "trim_end": 3.0})
|
||||
result = phrase_review_to_actions(review)
|
||||
assert [(a["start"], a["end"]) for a in result["actions"]] == [(0.0, 4.0)]
|
||||
|
||||
def test_zoom_follows_the_trimmed_span(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 4)]))
|
||||
review["phrases"][0].update({"emphasis": 1, "trim_start": 1.0, "trim_end": 3.0})
|
||||
result = phrase_review_to_actions(review)
|
||||
zoom = next(a for a in result["actions"] if a["kind"] == "zoom")
|
||||
assert (zoom["start"], zoom["end"]) == (1.0, 3.0)
|
||||
|
||||
def test_impossible_trim_is_ignored(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 4)]))
|
||||
review["phrases"][0].update({"trim_start": 3.0, "trim_end": 1.0})
|
||||
result = phrase_review_to_actions(review)
|
||||
assert result["actions"] == []
|
||||
|
||||
def test_emphasis_out_of_range_is_clamped(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 2)]))
|
||||
review["phrases"][0]["emphasis"] = 99
|
||||
result = phrase_review_to_actions(review)
|
||||
assert result["actions"][0]["params"]["scale"] == ZOOM_SCALE_BY_LEVEL[3]
|
||||
|
||||
def test_rows_that_make_no_sense_are_skipped(self):
|
||||
result = phrase_review_to_actions(
|
||||
{"phrases": ["nope", {"start": 5, "end": 1}, {"start": 0, "end": 1}]}
|
||||
)
|
||||
assert result["actions"] == []
|
||||
|
||||
|
||||
class TestManualZooms:
|
||||
def test_manual_zoom_becomes_an_action_without_a_scale(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 10)]))
|
||||
review["zooms"] = [{"start": 2.0, "end": 4.0}]
|
||||
result = phrase_review_to_actions(review)
|
||||
zoom = next(a for a in result["actions"] if a["kind"] == "zoom")
|
||||
assert (zoom["start"], zoom["end"]) == (2.0, 4.0)
|
||||
# Sem scale: o aplicador usa o zoom_scale configurado pelo usuário.
|
||||
assert "scale" not in zoom["params"]
|
||||
|
||||
def test_zoom_shorter_than_the_ramp_is_refused(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 10)]))
|
||||
review["zooms"] = [{"start": 2.0, "end": 2.1}]
|
||||
assert phrase_review_to_actions(review)["actions"] == []
|
||||
|
||||
def test_manual_zoom_coexists_with_phrase_emphasis(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 10)]))
|
||||
review["phrases"][0]["emphasis"] = 2
|
||||
review["zooms"] = [{"start": 2.0, "end": 4.0}]
|
||||
zooms = [a for a in phrase_review_to_actions(review)["actions"] if a["kind"] == "zoom"]
|
||||
assert len(zooms) == 2
|
||||
|
||||
def test_malformed_zoom_rows_are_skipped(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 10)]))
|
||||
review["zooms"] = ["nope", {"start": 5}, {"start": 4, "end": 1}]
|
||||
assert phrase_review_to_actions(review)["actions"] == []
|
||||
|
||||
def test_saved_zooms_are_restored(self):
|
||||
review = merge_saved_decisions(
|
||||
build_phrase_review(_timeline([_segment(0, 10)])),
|
||||
{"phrases": [], "zooms": [{"start": 1.0, "end": 3.0}]},
|
||||
)
|
||||
assert review["zooms"] == [{"start": 1.0, "end": 3.0}]
|
||||
|
||||
def test_new_review_starts_with_no_manual_zooms(self):
|
||||
assert build_phrase_review(_timeline([_segment(0, 2)]))["zooms"] == []
|
||||
|
||||
|
||||
class TestRoundTrip:
|
||||
def test_review_survives_actions_and_back(self):
|
||||
timeline = _timeline(
|
||||
[_segment(0, 2, peak=0.9), _segment(2, 4), _segment(4, 6, peak=0.5)]
|
||||
)
|
||||
first = build_phrase_review(timeline)
|
||||
first["phrases"][1]["active"] = False
|
||||
actions = phrase_review_to_actions(first)
|
||||
|
||||
second = build_phrase_review(timeline, actions)
|
||||
assert [p["active"] for p in second["phrases"]] == [True, False, True]
|
||||
assert [p["emphasis"] for p in second["phrases"]] == [3, 0, 2]
|
||||
|
||||
|
||||
class TestSnapToWords:
|
||||
def test_snaps_to_the_nearest_start(self):
|
||||
words = _words([(1.0, 1.5), (1.5, 2.0)])
|
||||
assert snap_to_words(1.6, words, 1.6, "in") == 1.5
|
||||
|
||||
def test_snaps_to_the_nearest_end(self):
|
||||
words = _words([(1.0, 1.5), (1.5, 2.0)])
|
||||
assert snap_to_words(1.9, words, 1.9, "out") == 2.0
|
||||
|
||||
def test_falls_back_without_word_timings(self):
|
||||
assert snap_to_words(1.2, [], 3.4, "in") == 3.4
|
||||
|
||||
|
||||
class TestResolveSource:
|
||||
def test_finds_the_media_beside_its_timeline(self, tmp_path):
|
||||
media = tmp_path / "take.mov"
|
||||
media.write_bytes(b"0")
|
||||
timeline = tmp_path / "take_voice_timeline.json"
|
||||
assert resolve_source("take.mov", str(timeline)) == str(media)
|
||||
|
||||
def test_falls_back_to_the_project_folder(self, tmp_path):
|
||||
media_dir = tmp_path / "midia"
|
||||
media_dir.mkdir()
|
||||
media = media_dir / "take.mov"
|
||||
media.write_bytes(b"0")
|
||||
timeline = tmp_path / "json" / "take_voice_timeline.json"
|
||||
assert resolve_source("take.mov", str(timeline), [str(media_dir)]) == str(media)
|
||||
|
||||
def test_absolute_path_is_used_as_is(self, tmp_path):
|
||||
media = tmp_path / "take.mov"
|
||||
media.write_bytes(b"0")
|
||||
assert resolve_source(str(media), "") == str(media)
|
||||
|
||||
def test_missing_media_resolves_to_empty(self, tmp_path):
|
||||
assert resolve_source("take.mov", str(tmp_path / "x_voice_timeline.json")) == ""
|
||||
|
||||
def test_stale_absolute_path_still_finds_the_file_by_name(self, tmp_path):
|
||||
# The fixture's source is an absolute path that no longer exists (the
|
||||
# everyday case: the project moved). Falling back to the file name next
|
||||
# to the timeline is what keeps the preview working after a move.
|
||||
media = tmp_path / "take.mov"
|
||||
media.write_bytes(b"0")
|
||||
assert resolve_source("/tmp/gone/take.mov", str(tmp_path / "t.json")) == str(media)
|
||||
|
||||
def test_review_carries_the_resolved_path(self, tmp_path):
|
||||
media = tmp_path / "take.mov"
|
||||
media.write_bytes(b"0")
|
||||
review = build_phrase_review(
|
||||
{**_timeline([_segment(0, 1)]), "source": "take.mov"},
|
||||
voice_timeline_path=str(tmp_path / "take_voice_timeline.json"),
|
||||
)
|
||||
assert review["source_path"] == str(media)
|
||||
|
||||
def test_review_without_media_reports_no_path(self, tmp_path):
|
||||
review = build_phrase_review(
|
||||
_timeline([_segment(0, 1)]),
|
||||
voice_timeline_path=str(tmp_path / "take_voice_timeline.json"),
|
||||
)
|
||||
assert review["source_path"] == ""
|
||||
|
||||
|
||||
class TestEmotion:
|
||||
def test_segment_emotion_reaches_the_phrase(self):
|
||||
segment = _segment(0, 2)
|
||||
segment["emotion"] = "excited"
|
||||
segment["emotion_confidence"] = 0.72
|
||||
review = build_phrase_review(_timeline([segment]))
|
||||
assert review["phrases"][0]["emotion"] == "excited"
|
||||
assert review["phrases"][0]["emotion_confidence"] == 0.72
|
||||
|
||||
def test_defaults_to_neutral_when_absent(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 2)]))
|
||||
assert review["phrases"][0]["emotion"] == "neutral"
|
||||
assert review["phrases"][0]["emotion_confidence"] == 0.0
|
||||
|
||||
def test_availability_comes_from_the_analysis_layers(self):
|
||||
assert build_phrase_review(_timeline([_segment(0, 1)]))["emotion_available"] is False
|
||||
timeline = {**_timeline([_segment(0, 1)]), "layers": {"emotion": True}}
|
||||
assert build_phrase_review(timeline)["emotion_available"] is True
|
||||
|
||||
|
||||
class TestMergeSavedDecisions:
|
||||
def test_saved_decisions_win_over_the_derivation(self):
|
||||
timeline = _timeline([_segment(0, 2, peak=0.9), _segment(2, 4)])
|
||||
saved = {
|
||||
"phrases": [
|
||||
{"index": 0, "start": 0.0, "emphasis": 0, "active": False,
|
||||
"track": TRACK_BACKSTAGE, "text": "corrigido"},
|
||||
]
|
||||
}
|
||||
review = merge_saved_decisions(build_phrase_review(timeline), saved)
|
||||
first = review["phrases"][0]
|
||||
assert (first["emphasis"], first["active"]) == (0, False)
|
||||
assert first["track"] == TRACK_BACKSTAGE
|
||||
assert first["text"] == "corrigido"
|
||||
assert review["phrases"][1]["active"] is True
|
||||
|
||||
def test_fresh_analysis_fields_are_not_overwritten(self):
|
||||
segment = _segment(0, 2, peak=0.9)
|
||||
segment["emotion"] = "tense"
|
||||
review = merge_saved_decisions(
|
||||
build_phrase_review(_timeline([segment])),
|
||||
{"phrases": [{"index": 0, "start": 0.0, "emphasis": 1}]},
|
||||
)
|
||||
assert review["phrases"][0]["emotion"] == "tense"
|
||||
assert review["phrases"][0]["peak_emphasis"] == 0.9
|
||||
|
||||
def test_decision_is_dropped_when_the_line_moved(self):
|
||||
review = merge_saved_decisions(
|
||||
build_phrase_review(_timeline([_segment(10, 12, peak=0.9)])),
|
||||
{"phrases": [{"index": 0, "start": 0.0, "active": False}]},
|
||||
)
|
||||
assert review["phrases"][0]["active"] is True
|
||||
|
||||
def test_saved_trim_is_restored(self):
|
||||
review = merge_saved_decisions(
|
||||
build_phrase_review(_timeline([_segment(0, 4)])),
|
||||
{"phrases": [{"index": 0, "start": 0.0, "trim_start": 1.0, "trim_end": 3.0}]},
|
||||
)
|
||||
assert (review["phrases"][0]["trim_start"], review["phrases"][0]["trim_end"]) == (1.0, 3.0)
|
||||
|
||||
def test_impossible_saved_trim_is_ignored(self):
|
||||
review = merge_saved_decisions(
|
||||
build_phrase_review(_timeline([_segment(0, 4)])),
|
||||
{"phrases": [{"index": 0, "start": 0.0, "trim_start": 9.0, "trim_end": 12.0}]},
|
||||
)
|
||||
assert (review["phrases"][0]["trim_start"], review["phrases"][0]["trim_end"]) == (0.0, 4.0)
|
||||
|
||||
def test_no_saved_review_is_a_no_op(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 2)]))
|
||||
assert merge_saved_decisions(review, None) is review
|
||||
|
||||
|
||||
class TestPersistence:
|
||||
def test_paths_are_named_after_the_timeline(self, tmp_path):
|
||||
timeline_path = tmp_path / "take_voice_timeline.json"
|
||||
review_path, actions_path = review_paths(str(timeline_path))
|
||||
assert review_path.name == "take_phrase_review.json"
|
||||
assert actions_path.name == "take_phrase_actions.json"
|
||||
|
||||
def test_save_writes_both_files_and_load_reads_it_back(self, tmp_path):
|
||||
timeline_path = tmp_path / "take_voice_timeline.json"
|
||||
review = build_phrase_review(_timeline([_segment(0, 2)]))
|
||||
review["phrases"][0]["emphasis"] = 3
|
||||
|
||||
review_path, actions_path = save_phrase_review(str(timeline_path), review)
|
||||
assert review_path.is_file() and actions_path.is_file()
|
||||
|
||||
written = json.loads(actions_path.read_text(encoding="utf-8"))
|
||||
assert written["actions"][0]["kind"] == "zoom"
|
||||
|
||||
assert load_phrase_review(str(timeline_path))["phrases"][0]["emphasis"] == 3
|
||||
|
||||
def test_load_returns_none_when_absent_or_broken(self, tmp_path):
|
||||
timeline_path = tmp_path / "take_voice_timeline.json"
|
||||
assert load_phrase_review(str(timeline_path)) is None
|
||||
|
||||
review_path, _ = review_paths(str(timeline_path))
|
||||
review_path.write_text("{ not json", encoding="utf-8")
|
||||
assert load_phrase_review(str(timeline_path)) is None
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
pytest.main([__file__, "-v"])
|
||||
@@ -40,7 +40,7 @@ WORDS = [
|
||||
w("the", 2.5, 2.6),
|
||||
w("show", 2.65, 3.0),
|
||||
w("is", 3.05, 3.15),
|
||||
w("um", 3.2, 3.5),
|
||||
w("uh", 3.2, 3.5),
|
||||
w("great.", 3.6, 4.0),
|
||||
]
|
||||
|
||||
@@ -60,7 +60,7 @@ class TestNormalizeWord:
|
||||
class TestFindPhraseSpans:
|
||||
def test_single_word_multiple_hits(self):
|
||||
spans = find_phrase_spans(WORDS, "um")
|
||||
assert spans == [(0.4, 0.6), (3.2, 3.5)]
|
||||
assert spans == [(0.4, 0.6)]
|
||||
|
||||
def test_multi_word_phrase(self):
|
||||
spans = find_phrase_spans(WORDS, "welcome to the show")
|
||||
@@ -84,9 +84,13 @@ class TestFindPhraseSpans:
|
||||
class TestFindFillerSpans:
|
||||
def test_default_fillers(self):
|
||||
spans = find_filler_spans(WORDS)
|
||||
assert (0.4, 0.6) in spans
|
||||
assert (0.4, 0.6) not in spans
|
||||
assert (3.2, 3.5) in spans
|
||||
|
||||
def test_um_can_still_be_explicit(self):
|
||||
spans = find_filler_spans(WORDS, fillers=("um",))
|
||||
assert spans == [(0.4, 0.6)]
|
||||
|
||||
def test_multi_word_filler(self):
|
||||
spans = find_filler_spans(WORDS, fillers=("you know",))
|
||||
assert spans == [(2.0, 2.4)]
|
||||
@@ -96,7 +100,10 @@ class TestFindFillerSpans:
|
||||
assert spans == sorted(spans)
|
||||
|
||||
def test_defaults_are_conservative(self):
|
||||
# "like" and "so" are speech, not noise — must not be default-cut.
|
||||
# "um", "uma", "like" and "so" are speech, not noise — must not be
|
||||
# default-cut.
|
||||
assert "um" not in DEFAULT_FILLERS
|
||||
assert "uma" not in DEFAULT_FILLERS
|
||||
assert "like" not in DEFAULT_FILLERS
|
||||
assert "so" not in DEFAULT_FILLERS
|
||||
|
||||
@@ -269,11 +276,12 @@ class TestRemoveFillerWordsHandler:
|
||||
assert "_defillered" in result[0].text
|
||||
|
||||
segments = _spine_segments(str(tmp_path / "project_defillered.fcpxml"))
|
||||
# um (0-0.5) trims the clip head, uh (3-3.5) splits -> 2 segments, ~7s total
|
||||
# Only uh (3-3.5) is removed by default; Portuguese "um" is preserved
|
||||
# because it is often grammatical speech ("de um jeito").
|
||||
assert len(segments) == 2
|
||||
total = sum(c.duration.seconds for c in segments)
|
||||
assert total == pytest.approx(7.0, abs=0.1)
|
||||
assert segments[0].source_start.seconds == pytest.approx(0.5, abs=0.05)
|
||||
assert total == pytest.approx(7.5, abs=0.1)
|
||||
assert segments[0].source_start.seconds == pytest.approx(0.0, abs=0.05)
|
||||
assert segments[1].source_start.seconds == pytest.approx(3.5, abs=0.05)
|
||||
|
||||
async def test_no_fillers_found_saves_nothing(self, tmp_path):
|
||||
|
||||
@@ -76,9 +76,14 @@ class TestParseActions:
|
||||
|
||||
|
||||
class TestZoomValidation:
|
||||
def test_default_scale_when_absent(self):
|
||||
def test_absent_scale_is_left_absent(self):
|
||||
# The parser no longer stamps a default: an omitted scale must reach the
|
||||
# applier untouched so it can fall back to the user's configured
|
||||
# `zoom_scale` (see server_tools/_shared.py). Filling one in here would
|
||||
# silently override that setting for every action the model sends
|
||||
# without an explicit scale.
|
||||
actions, _ = parse_actions([{"kind": "zoom", "start": 1.0, "end": 2.0}])
|
||||
assert actions[0].params["scale"] == 1.3
|
||||
assert "scale" not in actions[0].params
|
||||
|
||||
def test_rejects_scale_below_one(self):
|
||||
_, errors = parse_actions([
|
||||
|
||||
@@ -94,6 +94,32 @@ class TestApplyVoiceActionsHandler:
|
||||
texts = [t.text for t in titles[0].iter() if t.text]
|
||||
assert any("SEGURANÇA" in t for t in texts)
|
||||
|
||||
async def test_text_callout_defaults_fit_the_frame(self, project):
|
||||
from fcpxml.writer import FCPXMLModifier
|
||||
from server import handle_apply_voice_actions
|
||||
|
||||
await handle_apply_voice_actions({
|
||||
"filepath": str(project),
|
||||
"actions": [{
|
||||
"kind": "text", "start": 3.0, "end": 4.0,
|
||||
"params": {"content": "PRÓTESES DE SILICONE"},
|
||||
}],
|
||||
})
|
||||
|
||||
modifier = FCPXMLModifier(str(_out(project)))
|
||||
report = modifier.validate_subtitle_layout()
|
||||
assert report["summary"]["outside_frame"] == 0
|
||||
|
||||
title = modifier.root.find(".//title")
|
||||
style = title.find("text-style-def/text-style")
|
||||
position = next(
|
||||
p.get("value")
|
||||
for p in title.findall("param")
|
||||
if p.get("name") == "Position"
|
||||
)
|
||||
assert float(style.get("fontSize")) < 530
|
||||
assert position != "0 0"
|
||||
|
||||
async def test_applies_marker(self, project):
|
||||
from server import handle_apply_voice_actions
|
||||
|
||||
|
||||
@@ -11,6 +11,7 @@ import pytest
|
||||
|
||||
from fcpxml.voice_timeline import (
|
||||
VOICE_TIMELINE_VERSION,
|
||||
annotate_emotions,
|
||||
build_voice_timeline,
|
||||
enrich_words,
|
||||
load_voice_timeline,
|
||||
@@ -60,6 +61,14 @@ class TestEnrichWords:
|
||||
assert all(w["energy_norm"] == 0.0 for w in enriched)
|
||||
assert all(w["pitch_delta"] == 0.0 for w in enriched)
|
||||
|
||||
def test_emotion_labels_are_added_when_enabled(self):
|
||||
enriched = enrich_words(_TRANSCRIPT["words"], _PITCH, _ENERGY)
|
||||
emotional = annotate_emotions(enriched, enabled=True, sensitivity=0.1)
|
||||
loudest = max(emotional, key=lambda w: w["arousal"])
|
||||
assert loudest["word"] == "seguranca"
|
||||
assert loudest["emotion"] in {"excited", "tense", "neutral"}
|
||||
assert 0.0 <= loudest["emotion_confidence"] <= 1.0
|
||||
|
||||
|
||||
class TestBuildVoiceTimeline:
|
||||
@pytest.fixture
|
||||
@@ -74,6 +83,7 @@ class TestBuildVoiceTimeline:
|
||||
for key in ("version", "source", "language", "scales", "summary", "speakers", "segments"):
|
||||
assert key in timeline
|
||||
assert timeline["version"] == VOICE_TIMELINE_VERSION
|
||||
assert "emotion" in timeline["layers"]
|
||||
|
||||
def test_scales_document_every_word_metric(self, timeline):
|
||||
word = timeline["segments"][0]["words"][0]
|
||||
@@ -106,6 +116,8 @@ class TestBuildVoiceTimeline:
|
||||
quiet_segment = timeline["segments"][0]
|
||||
assert loud_segment["avg_energy"] > quiet_segment["avg_energy"]
|
||||
assert loud_segment["peak_emphasis"] >= max(w["emphasis"] for w in loud_segment["words"])
|
||||
assert "emotion" in loud_segment
|
||||
assert "arousal" in loud_segment
|
||||
|
||||
def test_peak_moments_are_sorted_by_emphasis(self, timeline):
|
||||
peaks = timeline["summary"]["peak_moments"]
|
||||
|
||||
@@ -94,6 +94,10 @@ class TestBuildVoiceTimelineHandler:
|
||||
},
|
||||
"emotion_enabled": False,
|
||||
"emotion_sensitivity": 0.5,
|
||||
"zoom_scale": 1.3,
|
||||
"zoom_mode": "in_out",
|
||||
"zoom_ease_in": 0.25,
|
||||
"zoom_ease_out": 0.04,
|
||||
}
|
||||
|
||||
monkeypatch.setattr(server_mod, "load_voice_analysis_config", lambda: config(0.01))
|
||||
|
||||
Reference in New Issue
Block a user