fix(legendas): impede legenda comum sob composição dinâmica e duplicação ao regerar

Marca cada título gerado (dynamic/plain) em metadata para que regenerar
substitua a saída anterior em vez de empilhar, e usa os spans de ênfase
revisados (não os segmentos brutos do Whisper) como janela da composição
dinâmica, evitando que ela invada o trecho de legenda comum seguinte.
suppress_plain_under_dynamic corta qualquer sobra visível como rede de
segurança.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
João Henrique
2026-09-22 21:51:02 -04:00
co-authored by Claude Sonnet 5
parent 32d78d0f8d
commit 0fdfe33613
5 changed files with 479 additions and 10 deletions
+16 -4
View File
@@ -470,11 +470,13 @@ async def handle_generate_dynamic_subtitles(arguments: dict) -> Sequence[TextCon
# loop to whichever one `self.clips` last indexed, stacking every
# clip's captions onto a single wrong spine element instead of each
# clip's own. See Engine/docs/05_EXPERIENCIAS.md, entry 2026-08-17.
modifier.remove_generated_subtitles(el, ('dynamic',))
lines = modifier.generate_dynamic_subtitles(
el, clip_words, configs=configs, segments=clip_segments,
role=single_role_override,
compound_subphrases=True,
)
modifier.suppress_plain_under_dynamic(el)
added.append((name, len(lines), len(clip_words)))
if not added:
@@ -554,6 +556,7 @@ async def handle_generate_plain_subtitles(arguments: dict) -> Sequence[TextConte
skipped.append((name, "no words in clip's source range"))
continue
modifier.remove_generated_subtitles(el, ('plain',))
blocks = _plain_subtitle_blocks(clip_words, max_words)
created = 0
for block in blocks:
@@ -567,7 +570,7 @@ async def handle_generate_plain_subtitles(arguments: dict) -> Sequence[TextConte
start = max(0.0, min(float(w.get("start", 0.0)) for w in block))
end = max(float(w.get("end", start)) for w in block)
duration = max(end - start, modifier.frame_duration_fraction())
modifier.add_text_title(
title = modifier.add_text_title(
el,
text,
offset=f"{start:.6f}s",
@@ -583,7 +586,9 @@ async def handle_generate_plain_subtitles(arguments: dict) -> Sequence[TextConte
size_param=font_size,
role=saved["role"],
)
modifier.mark_generated_subtitle(title, 'plain')
created += 1
modifier.suppress_plain_under_dynamic(el)
if created:
added.append((name, created, len(clip_words)))
@@ -711,14 +716,17 @@ async def handle_generate_subtitles_by_emphasis(arguments: dict) -> Sequence[Tex
clip_hide_plain_spans = clip_spans + _clip_relative(exclude_spans)
all_words = data.get("words", [])
modifier.remove_generated_subtitles(el, ('dynamic', 'plain'))
dynamic_lines = 0
dynamic_word_count = 0
emphasis_words = _words_in_spans(all_words, spans)
clip_emphasis_words = _words_overlapping_clip(emphasis_words, clip_source_start, window_end)
if clip_emphasis_words:
all_segments = data.get("segments", [])
emphasis_segments = _segments_in_spans(all_segments, spans)
# Reviewed phrases, not broader Whisper segments, define where
# a dynamic composition may live. Separate emphasis windows must
# never hold text over the plain speech between them.
emphasis_segments = spans
clip_segments = [
{
"start": float(s.get("start", 0.0)) - clip_source_start,
@@ -736,6 +744,7 @@ async def handle_generate_subtitles_by_emphasis(arguments: dict) -> Sequence[Tex
el, clip_emphasis_words, configs=dynamic_configs, segments=clip_segments,
role=single_dynamic_role,
compound_subphrases=True,
hold_between_sentences=False,
)
)
dynamic_word_count = len(clip_emphasis_words)
@@ -766,7 +775,7 @@ async def handle_generate_subtitles_by_emphasis(arguments: dict) -> Sequence[Tex
plain_hidden += 1
continue
duration = max(end - start, modifier.frame_duration_fraction())
modifier.add_text_title(
title = modifier.add_text_title(
el,
text,
offset=f"{start:.6f}s",
@@ -782,8 +791,11 @@ async def handle_generate_subtitles_by_emphasis(arguments: dict) -> Sequence[Tex
size_param=plain_font_size,
role=saved_plain["role"],
)
modifier.mark_generated_subtitle(title, 'plain')
plain_created += 1
modifier.suppress_plain_under_dynamic(el)
if dynamic_lines or plain_created:
added.append(
(name, dynamic_lines, plain_created, plain_hidden, dynamic_word_count + len(clip_all_words))