diff --git a/code/Engine/docs/03_SERVER_TOOLS.md b/code/Engine/docs/03_SERVER_TOOLS.md index a514806..ce44b33 100644 --- a/code/Engine/docs/03_SERVER_TOOLS.md +++ b/code/Engine/docs/03_SERVER_TOOLS.md @@ -124,16 +124,18 @@ severidade probable/severe → investigar CADA colisão pela fração exata do XML antes de mudar código (ver checklist abaixo) ``` -**`generate_subtitles_by_emphasis`** gera as duas legendas numa passada só — -mas não divide as palavras entre elas. A comum é gerada **completa, do início -ao fim do clipe**, sempre; a dinâmica é gerada só sobre as frases marcadas -como ênfase na etapa 5 (zoom aplicado, nível ≥ 1); e onde a dinâmica cobre um -trecho, os títulos comuns daquele trecho recebem `enabled="0"` — continuam no -XML (editáveis/reativáveis no Final Cut), só não são desenhados. É a tradução -literal de `10-revisao-humana.md` (skill `editar-por-voz`): "a frase de -ênfase recebe zoom E legenda dinâmica; as demais recebem legenda comum" — -sem nunca deixar um vão sem legenda nenhuma se a ênfase for desativada depois -(a comum já estava lá, só desligada). A decisão vem de +**`generate_subtitles_by_emphasis`** gera as duas legendas numa passada só, +dividindo por palavra: a dinâmica cobre as frases marcadas como ênfase na +etapa 5 (zoom aplicado, nível ≥ 1); a comum cobre **todo o resto** — um bloco +comum simplesmente não é criado onde a dinâmica já cobre. A primeira versão +gerava a comum inteira e desativava (`enabled="0"`) o que ficava sob a +dinâmica, mas um título desativado continua aparecendo como clipe riscado na +timeline do Final Cut mesmo sem renderizar — um corte com bastante ênfase +enchia a trilha de clipes mortos. Trocado por não gerar ali: o preço é que, +se a ênfase for desativada à mão depois, a legenda comum daquele trecho +precisa ser regenerada, não só reativada. É a tradução de `10-revisao-humana.md` +(skill `editar-por-voz`): "a frase de ênfase recebe zoom E legenda dinâmica; +as demais recebem legenda comum". A decisão vem de `_phrase_actions.json["emphasis_spans"]`, escrito por `save_phrase_review` quando o editor termina a etapa 5 — sem esse arquivo (ou sem `zoom`/`text` marcados na revisão), a tool gera só a comum, tudo ligado, diff --git a/code/fcpxml/writer/cut.py b/code/fcpxml/writer/cut.py index 2068c91..0f86c31 100644 --- a/code/fcpxml/writer/cut.py +++ b/code/fcpxml/writer/cut.py @@ -195,23 +195,40 @@ class CutMixin: if cursor < clip_duration: keeps.append((cursor, clip_duration)) - # A keep segment shorter than a couple frames at the very start or - # end of the clip is just leftover cut padding with no neighboring - # kept audio on its outer side (the silence butts against the clip's - # own edge) — not a real clip. Rather than emit it as its own - # near-invisible micro-clip, fold it into the adjacent real segment, - # which simply starts earlier / ends later to absorb it. - min_keep_seconds = 2 * float(self.frame_duration_fraction()) - if len(keeps) > 1: - first_start, first_end = keeps[0] - if (first_end - first_start).to_seconds() < min_keep_seconds: - keeps[1] = (first_start, keeps[1][1]) + # A keep segment shorter than MIN_KEEP_SECONDS is leftover between + # two cuts, not a real clip — at the very start/end of the clip it's + # cut padding with no kept audio on the outer side; in the interior + # it's the pause BETWEEN two things that were both cut (e.g. two + # consecutive deactivated phrases in the voice-editing flow), which + # belongs to neither side by construction. At the edges we fold it + # into the one neighboring KEEP segment there is, which simply starts + # earlier / ends later to absorb it. In the interior both neighbors + # are CUT, not keep, so there is nothing to fold into — it is just + # dropped, extending the surrounding cut across it instead of + # surviving as a third near-invisible micro-clip. + # + # The threshold is bigger than one frame on purpose: measured on a + # real voice-edit (0.07-0.23s residues), a single frame did not catch + # them — this is pause/padding leftover, not intentional short + # content, so treating anything under a third of a second this way + # is safe for this cut path. + min_keep_seconds = max(6 * float(self.frame_duration_fraction()), 0.3) + i = 0 + while len(keeps) > 1 and i < len(keeps): + start, end = keeps[i] + if (end - start).to_seconds() >= min_keep_seconds: + i += 1 + continue + if i == 0: + keeps[1] = (start, keeps[1][1]) keeps.pop(0) - if len(keeps) > 1: - last_start, last_end = keeps[-1] - if (last_end - last_start).to_seconds() < min_keep_seconds: - keeps[-2] = (keeps[-2][0], last_end) - keeps.pop() + elif i == len(keeps) - 1: + keeps[i - 1] = (keeps[i - 1][0], end) + keeps.pop(i) + else: + keeps.pop(i) + # Re-check the same index: the segment now there might itself be + # short enough to absorb again (two short keeps in a row). spine.remove(clip) new_clips: List[ET.Element] = [] diff --git a/code/server_tools/subtitles.py b/code/server_tools/subtitles.py index bb2ca70..36e5411 100644 --- a/code/server_tools/subtitles.py +++ b/code/server_tools/subtitles.py @@ -95,7 +95,7 @@ TOOLS = [ ), Tool( name="generate_subtitles_by_emphasis", - description="Generate BOTH subtitle styles over the FULL clip and let them coexist by visibility, not by splitting words: plain static titles (see generate_plain_subtitles) cover every word from start to end; dynamic progressive-composition titles (see generate_dynamic_subtitles) are additionally generated for whichever whole phrases were marked as emphasis in the phrase-review step (etapa 5, zoom applied, level >= 1). Wherever a dynamic phrase is on screen, the plain titles underneath it are set enabled=\"0\" (still present in the FCPXML, editable/re-enable-able in Final Cut, just not rendered) instead of never being generated there — so disabling emphasis later never leaves a silent gap in the plain track. Reads emphasis spans from the media's cached '_phrase_actions.json' (written by save_phrase_review after the app's etapa 5 review) — run the voice-editing wizard through that step first, or nothing is treated as emphasis and every title stays plain and enabled. Style knobs are the saved 'Legendas Dinâmicas'/plain-subtitle configs (~/.fcp-mcp-server/config.json); this tool does not expose per-call style overrides, only the split logic — use generate_dynamic_subtitles/generate_plain_subtitles directly if you need one-off styling.", + description="Generate BOTH subtitle styles in one pass, split by word so they never coexist on the same range: dynamic progressive-composition titles (see generate_dynamic_subtitles) cover whichever whole phrases were marked as emphasis in the phrase-review step (etapa 5, zoom applied, level >= 1); plain static titles (see generate_plain_subtitles) cover every OTHER word in the clip. A plain block is simply not created where a dynamic phrase already covers — not created-then-disabled — because a disabled title still shows as its own struck-through clip in Final Cut's timeline even though it never renders, and a heavily emphasized edit ended up with dozens of dead clips cluttering the track. Trade-off: if emphasis is turned off by hand later, the plain line under it has to be regenerated, not just re-enabled. Reads emphasis spans from the media's cached '_phrase_actions.json' (written by save_phrase_review after the app's etapa 5 review) — run the voice-editing wizard through that step first, or nothing is treated as emphasis and every word gets a plain title. Style knobs are the saved 'Legendas Dinâmicas'/plain-subtitle configs (~/.fcp-mcp-server/config.json); this tool does not expose per-call style overrides, only the split logic — use generate_dynamic_subtitles/generate_plain_subtitles directly if you need one-off styling.", inputSchema={ "type": "object", "properties": { @@ -234,7 +234,7 @@ def _segments_in_spans(segments: Sequence[dict], spans: Sequence[dict]) -> list[ def _overlaps_any_span(start: float, end: float, spans: Sequence[tuple[float, float]]) -> bool: - """Half-open interval overlap: a plain title under this window must hide.""" + """Half-open interval overlap: a plain title under this window is skipped.""" return any(start < span_end and end > span_start for span_start, span_end in spans) @@ -541,13 +541,16 @@ async def handle_generate_plain_subtitles(arguments: dict) -> Sequence[TextConte async def handle_generate_subtitles_by_emphasis(arguments: dict) -> Sequence[TextContent]: - """Generate plain titles for the whole clip and dynamic titles for the - emphasis phrases on top, then hide (enabled="0") the plain titles that - fall under a dynamic phrase — never split the word list between the two. + """Generate dynamic titles for the emphasis phrases, and plain titles for + every OTHER word — a plain block is simply not created where a dynamic + phrase already covers, rather than created and disabled. - Plain always covers every word, so turning emphasis off later (editing - the phrase review and re-running) never leaves a silent gap: the plain - title was there all along, just disabled. + A disabled ("enabled=0") title still shows as its own struck-through clip + in Final Cut's timeline even though it never renders — a heavily + emphasized edit ended up with dozens of dead clips cluttering the track. + Not generating them there trades that clutter for a smaller gap: if the + emphasis is turned off by hand later, the plain line has to be + regenerated rather than just re-enabled. """ model = arguments.get("model", "base") language = arguments.get("language") @@ -658,10 +661,14 @@ async def handle_generate_subtitles_by_emphasis(arguments: dict) -> Sequence[Tex ) dynamic_word_count = len(clip_emphasis_words) - # Plain covers EVERY word in the clip — never filtered by emphasis. - # Titles landing under a dynamic phrase are disabled below instead of - # never being created, so turning emphasis off later never leaves a - # silent gap where neither style is on screen. + # Plain covers every word OUTSIDE an emphasis span. A block landing + # under a dynamic phrase is simply not created there — generating it + # disabled was tried first, but every disabled title still shows up + # as its own clip in Final Cut's timeline (just struck through), so + # a heavily-emphasized edit ended up with dozens of dead clips + # cluttering the track for no visible benefit. The trade-off: if the + # emphasis is later turned off by hand, the plain line under it has + # to be regenerated rather than just re-enabled. plain_created = 0 plain_hidden = 0 clip_all_words = _words_overlapping_clip(all_words, clip_source_start, window_end) @@ -676,8 +683,11 @@ async def handle_generate_subtitles_by_emphasis(arguments: dict) -> Sequence[Tex continue start = max(0.0, min(float(w.get("start", 0.0)) for w in block)) end = max(float(w.get("end", start)) for w in block) + if _overlaps_any_span(start, end, clip_spans): + plain_hidden += 1 + continue duration = max(end - start, modifier.frame_duration_fraction()) - title = modifier.add_text_title( + modifier.add_text_title( el, text, offset=f"{start:.6f}s", @@ -693,9 +703,6 @@ async def handle_generate_subtitles_by_emphasis(arguments: dict) -> Sequence[Tex size_param=plain_font_size, ) plain_created += 1 - if _overlaps_any_span(start, end, clip_spans): - title.set("enabled", "0") - plain_hidden += 1 if dynamic_lines or plain_created: added.append( @@ -721,12 +728,12 @@ async def handle_generate_subtitles_by_emphasis(arguments: dict) -> Sequence[Tex result += ( f"- **Clips Captioned**: {len(added)}\n" f"- **Dynamic Title Lines (emphasis)**: {total_dynamic}\n" - f"- **Plain Title Blocks (full clip)**: {total_plain}\n" - f"- **Plain Blocks Hidden Under Emphasis (enabled=\"0\")**: {total_hidden}\n" + f"- **Plain Title Blocks**: {total_plain}\n" + f"- **Plain Blocks Skipped Under Emphasis (not created there)**: {total_hidden}\n" f"- **Total Words**: {total_words}\n\n" ) result += _markdown_table( - ["Clip", "Dynamic Lines", "Plain Blocks", "Hidden", "Words"], + ["Clip", "Dynamic Lines", "Plain Blocks", "Skipped", "Words"], [[n, str(d), str(p), str(h), str(w)] for n, d, p, h, w in added], ) if no_review: diff --git a/code/tests/test_media_intel.py b/code/tests/test_media_intel.py index d3c16df..e750f2a 100755 --- a/code/tests/test_media_intel.py +++ b/code/tests/test_media_intel.py @@ -331,6 +331,32 @@ class TestCutClipRanges: assert mod._parse_time(seg.get("duration")).to_seconds() == pytest.approx(4.0) assert removed.to_seconds() == pytest.approx(0.0) + def test_interior_sliver_between_two_cuts_is_dropped_not_kept(self, tmp_path): + """A keep segment under the threshold BETWEEN two cuts (both + neighbors already removed) is the pause between two things that were + cut, not real content — regression for a real voice-edit where + consecutive short cuts left 0.07-0.23s clips surviving between the + real ones. Unlike the edge case, there is no kept neighbor to widen: + the sliver is just dropped, extending the surrounding cut over it.""" + from fcpxml.models import TimeValue + + mod = self._make_modifier(tmp_path) + clip = self._spine_clips(mod)[0] + # 4s clip; cut 0.5..1.5 and 1.6..3.5 -> would-be keeps: [0..0.5], + # [1.5..1.6] (0.1s interior sliver), [3.5..4]. + removed = mod.cut_clip_ranges(clip, [ + (TimeValue(1, 2), TimeValue(3, 2)), + (TimeValue(8, 5), TimeValue(7, 2)), + ]) + + clips = self._spine_clips(mod) + assert len(clips) == 3 # interview x2 real segments + broll; no 0.1s sliver + seg1, seg2, broll = clips + assert broll.get("name") == "broll" + assert mod._parse_time(seg1.get("duration")).to_seconds() == pytest.approx(0.5) + assert mod._parse_time(seg2.get("duration")).to_seconds() == pytest.approx(0.5) + assert removed.to_seconds() == pytest.approx(3.0) + def test_markers_follow_their_segment(self, tmp_path): from fcpxml.models import TimeValue