feat(legendas): liga compound_subphrases por padrão no pipeline
generate_dynamic_subtitles e a metade dinâmica de generate_subtitles_by_emphasis passam a empacotar cada sub-frase da legenda dinâmica num compound clip por padrão (compound_subphrases=True), completando o wrap_titles_in_compound e split_into_subphrases do commit anterior — que ainda não tinham chamador em produção. Também torna validate_subtitle_layout ciente de compound clips: media cada grupo (spine principal + cada <media> de compound) no seu próprio espaço de tempo, em vez de uma varredura .//title global — sem isso, âncoras de compounds diferentes liam offset "0s" e acusavam colisão espacial entre frases que nunca dividem a tela, só porque compartilham o mesmo zero de tempo local. Testado ponta a ponta na gravação real (Mastopexia): 12 compounds, 41 títulos todos empacotados, zero soltos, zero IDs duplicados, DTD válida. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
688bdeddb6
commit
c99274895c
+157
-77
@@ -13,7 +13,10 @@ from typing import Sequence
|
||||
from mcp.types import TextContent, Tool
|
||||
|
||||
from fcpxml.media_intel import media_src_to_path
|
||||
from fcpxml.model_manager import load_dynamic_subtitle_config, load_plain_subtitle_config
|
||||
from fcpxml.model_manager import (
|
||||
get_active_dynamic_subtitle_layouts,
|
||||
load_plain_subtitle_config,
|
||||
)
|
||||
from fcpxml.models import DynamicSubtitleConfig, WordLook, WordStyle
|
||||
from fcpxml.writer import FCPXMLModifier
|
||||
from server_tools._shared import (
|
||||
@@ -61,6 +64,7 @@ TOOLS = [
|
||||
"emphasis_size": {"type": "integer", "description": "Key-word size in canvas points, at the 2160x3840 reference frame. Falls back to the saved style (default 265)"},
|
||||
"emphasis_color": {"type": "string", "description": "RGBA (0-1, space-separated) for the key word (phrase mode). Defaults to active_color, so the block reads in a single colour unless the key word is deliberately set apart"},
|
||||
"text_scale": {"type": "number", "description": "Ratio between the title template's fontSize space and the canvas-point space it positions in. The \"Text\" template sizes type in frame pixels, so sizes are doubled on the way out. Falls back to the saved style (default 2.0). Lower it only if a template renders type larger than the chosen point size"},
|
||||
"role": {"type": "string", "description": "Final Cut role for every generated title (a 'titles.*' sub-role, never 'subtitles.*'). Falls back to the saved style (default 'titles.dinamicas'). Groups the clips in the role index and tints their lane."},
|
||||
"font": {"type": "string", "description": "Title font family (supporting lines in phrase mode). Falls back to the saved style (default 'Helvetica Neue')"},
|
||||
"font_size": {"type": "integer", "description": "Supporting-line font size in canvas points, at the 2160x3840 reference frame. Falls back to the saved style (default 104)"},
|
||||
"active_color": {"type": "string", "description": "RGBA (0-1, space-separated) for even-indexed lines. Falls back to the saved style (default '1 1 1 1')"},
|
||||
@@ -88,6 +92,7 @@ TOOLS = [
|
||||
"uppercase": {"type": "boolean", "description": "Render text in uppercase."},
|
||||
"keep_punctuation": {"type": "boolean", "description": "Keep punctuation such as comma and period."},
|
||||
"text_scale": {"type": "number", "description": "Template font-size scale. Falls back to saved plain-subtitle config."},
|
||||
"role": {"type": "string", "description": "Final Cut role for every generated title (a 'titles.*' sub-role, never 'subtitles.*'). Falls back to the saved style (default 'titles.convencionais'). Groups the clips in the role index and tints their lane."},
|
||||
"output_path": {"type": "string", "description": "Output path (default: adds _plain_subtitles suffix)"},
|
||||
},
|
||||
"required": ["filepath"]
|
||||
@@ -116,6 +121,31 @@ TOOLS = [
|
||||
|
||||
|
||||
_PUNCT_RE = re.compile(r"[^\w\sÀ-ÖØ-öø-ÿ]", re.UNICODE)
|
||||
_DEFAULT_DYNAMIC_ROLE = "titles.dinamicas"
|
||||
_DEFAULT_PLAIN_ROLE = "titles.convencionais"
|
||||
|
||||
|
||||
def _title_subrole(value: str | None, fallback: str) -> str:
|
||||
"""Return a Final Cut title sub-role, never a closed-caption role."""
|
||||
role = str(value or "").strip() or fallback
|
||||
if role.startswith("subtitles."):
|
||||
return "titles." + role.removeprefix("subtitles.")
|
||||
if role == "subtitles":
|
||||
return fallback
|
||||
if not role.startswith("titles."):
|
||||
return fallback
|
||||
return role
|
||||
|
||||
|
||||
def _separate_subtitle_roles(dynamic_role: str | None, plain_role: str | None) -> tuple[str, str]:
|
||||
"""Keep normal and dynamic subtitles in distinct Final Cut role lanes."""
|
||||
dynamic = _title_subrole(dynamic_role, _DEFAULT_DYNAMIC_ROLE)
|
||||
plain = _title_subrole(plain_role, _DEFAULT_PLAIN_ROLE)
|
||||
if dynamic == plain:
|
||||
if dynamic != _DEFAULT_DYNAMIC_ROLE:
|
||||
return dynamic, _DEFAULT_PLAIN_ROLE
|
||||
return _DEFAULT_DYNAMIC_ROLE, _DEFAULT_PLAIN_ROLE
|
||||
return dynamic, plain
|
||||
|
||||
|
||||
def _words_overlapping_clip(words: Sequence[dict], start: float, end: float) -> list[dict]:
|
||||
@@ -171,12 +201,14 @@ def _phrase_actions_path(media_path: str) -> Path:
|
||||
return Path(media_path).with_name(f"{stem}_phrase_actions.json")
|
||||
|
||||
|
||||
def _load_emphasis_spans(media_path: str) -> list[dict]:
|
||||
"""Load emphasis spans (source-media time) saved by the etapa-5 phrase review.
|
||||
def _load_review_spans(media_path: str, key: str) -> list[dict]:
|
||||
"""Load one span list (source-media time) saved by the etapa-5 phrase review.
|
||||
|
||||
Returns [] if the review was never run for this media — callers should treat
|
||||
that as "nothing is emphasis yet", not as an error, since the wizard's later
|
||||
steps are optional.
|
||||
``key`` is ``"emphasis_spans"`` (phrases with `subtitle_dynamic` on) or
|
||||
``"plain_exclude_spans"`` (phrases with `subtitle_common` off). Returns []
|
||||
if the review was never run for this media, or saved nothing under that
|
||||
key — callers should treat that as "nothing marked", not as an error,
|
||||
since the wizard's later steps are optional.
|
||||
"""
|
||||
path = _phrase_actions_path(media_path)
|
||||
if not path.is_file():
|
||||
@@ -185,10 +217,25 @@ def _load_emphasis_spans(media_path: str) -> list[dict]:
|
||||
data = json.loads(path.read_text(encoding="utf-8"))
|
||||
except (OSError, json.JSONDecodeError):
|
||||
return []
|
||||
spans = data.get("emphasis_spans", [])
|
||||
spans = data.get(key, [])
|
||||
return [s for s in spans if isinstance(s, dict) and "start" in s and "end" in s]
|
||||
|
||||
|
||||
def _load_emphasis_spans(media_path: str) -> list[dict]:
|
||||
"""Spans (source-media time) whose phrase has `subtitle_dynamic` on."""
|
||||
return _load_review_spans(media_path, "emphasis_spans")
|
||||
|
||||
|
||||
def _load_plain_exclude_spans(media_path: str) -> list[dict]:
|
||||
"""Spans (source-media time) whose phrase has `subtitle_common` off.
|
||||
|
||||
Independent from emphasis spans: a phrase can have `subtitle_common` off
|
||||
without being emphasized, so plain must be hidden there too even though
|
||||
no dynamic line is going to cover the gap.
|
||||
"""
|
||||
return _load_review_spans(media_path, "plain_exclude_spans")
|
||||
|
||||
|
||||
def _word_in_spans(word_start: float, word_end: float, spans: Sequence[dict]) -> bool:
|
||||
"""A word belongs to an emphasis span if its midpoint falls inside it.
|
||||
|
||||
@@ -302,6 +349,46 @@ async def handle_validate_subtitle_layout(arguments: dict) -> Sequence[TextConte
|
||||
return _text_result("\n".join(lines))
|
||||
|
||||
|
||||
def _build_dynamic_subtitle_config(saved: dict, overrides: dict | None = None) -> DynamicSubtitleConfig:
|
||||
"""Build a :class:`DynamicSubtitleConfig` from one registered layout dict.
|
||||
|
||||
``overrides`` (typically the tool call's own ``arguments``) only makes
|
||||
sense to apply when there is a single active layout — callers with 2+
|
||||
active layouts pass ``{}`` so every sampled block uses its layout as
|
||||
registered, unambiguously.
|
||||
"""
|
||||
overrides = overrides or {}
|
||||
body_color = overrides.get("active_color") or saved["active_color"]
|
||||
return DynamicSubtitleConfig(
|
||||
style=WordStyle(
|
||||
font=overrides.get("font") or saved["font"],
|
||||
font_size=int(overrides.get("font_size", saved["font_size"])),
|
||||
active_color=body_color,
|
||||
inactive_color=overrides.get("inactive_color", "0.7 0.7 0.7 1"),
|
||||
emphasis_look=WordLook(
|
||||
int(overrides.get("emphasis_size", saved["emphasis_size"])),
|
||||
overrides.get("emphasis_color") or saved["emphasis_color"] or body_color,
|
||||
font=overrides.get("emphasis_font") or saved["emphasis_font"],
|
||||
face=overrides.get("emphasis_face") or saved["emphasis_face"],
|
||||
kerning=0.0,
|
||||
),
|
||||
body_look=WordLook(
|
||||
int(overrides.get("font_size", saved["font_size"])),
|
||||
body_color,
|
||||
font=overrides.get("font") or saved["font"],
|
||||
face="Bold",
|
||||
kerning=1.2,
|
||||
),
|
||||
),
|
||||
band_height=float(overrides.get("band_height", saved["band_height"])),
|
||||
block_center_y=float(overrides.get("block_center_y", saved["block_center_y"])),
|
||||
granularity=overrides.get("granularity", "phrase"),
|
||||
text_scale=float(overrides.get("text_scale", saved["text_scale"])),
|
||||
line_gap=float(overrides.get("line_gap", saved["line_gap"])),
|
||||
role=_title_subrole(saved.get("role"), _DEFAULT_DYNAMIC_ROLE),
|
||||
)
|
||||
|
||||
|
||||
async def handle_generate_dynamic_subtitles(arguments: dict) -> Sequence[TextContent]:
|
||||
"""Generate per-word subtitle titles laid out as a block per sentence.
|
||||
|
||||
@@ -318,39 +405,21 @@ async def handle_generate_dynamic_subtitles(arguments: dict) -> Sequence[TextCon
|
||||
output_dir = arguments.get("output_dir")
|
||||
clip_filter = arguments.get("clip_name")
|
||||
|
||||
# Anything the caller didn't explicitly pass falls back to the style
|
||||
# Anything the caller didn't explicitly pass falls back to the style(s)
|
||||
# persisted from the "Legendas Dinâmicas" screen (~/.fcp-mcp-server/
|
||||
# config.json), not a hardcoded default — so the UI is the single place
|
||||
# that configures the look, and every caller (app, MCP, this session)
|
||||
# renders the same thing without threading 11 fields through every call.
|
||||
saved = load_dynamic_subtitle_config()
|
||||
body_color = arguments.get("active_color") or saved["active_color"]
|
||||
config = DynamicSubtitleConfig(
|
||||
style=WordStyle(
|
||||
font=arguments.get("font") or saved["font"],
|
||||
font_size=int(arguments.get("font_size", saved["font_size"])),
|
||||
active_color=body_color,
|
||||
inactive_color=arguments.get("inactive_color", "0.7 0.7 0.7 1"),
|
||||
emphasis_look=WordLook(
|
||||
int(arguments.get("emphasis_size", saved["emphasis_size"])),
|
||||
arguments.get("emphasis_color") or saved["emphasis_color"] or body_color,
|
||||
font=arguments.get("emphasis_font") or saved["emphasis_font"],
|
||||
face=arguments.get("emphasis_face") or saved["emphasis_face"],
|
||||
kerning=0.0,
|
||||
),
|
||||
body_look=WordLook(
|
||||
int(arguments.get("font_size", saved["font_size"])),
|
||||
body_color,
|
||||
font=arguments.get("font") or saved["font"],
|
||||
face="Bold",
|
||||
kerning=1.2,
|
||||
),
|
||||
),
|
||||
band_height=float(arguments.get("band_height", saved["band_height"])),
|
||||
block_center_y=float(arguments.get("block_center_y", saved["block_center_y"])),
|
||||
granularity=arguments.get("granularity", "phrase"),
|
||||
text_scale=float(arguments.get("text_scale", saved["text_scale"])),
|
||||
line_gap=float(arguments.get("line_gap", saved["line_gap"])),
|
||||
# config.json) — one or more named, active layouts. With exactly one
|
||||
# active layout, per-call overrides (arguments) still apply, same as
|
||||
# before this screen supported multiple layouts. With 2+ active layouts,
|
||||
# each block below randomly samples one of them, so per-call overrides
|
||||
# are ambiguous (which layout would they apply to?) and are ignored —
|
||||
# register/edit the layouts themselves instead.
|
||||
active_layouts = get_active_dynamic_subtitle_layouts()
|
||||
overrides = arguments if len(active_layouts) == 1 else {}
|
||||
configs = [_build_dynamic_subtitle_config(saved, overrides) for saved in active_layouts]
|
||||
single_role_override = (
|
||||
_title_subrole(arguments.get("role"), configs[0].role)
|
||||
if len(active_layouts) == 1 and arguments.get("role")
|
||||
else None
|
||||
)
|
||||
|
||||
filepath, output_path, modifier = _setup_modifier(arguments, "_dynamic_subtitles")
|
||||
@@ -402,7 +471,9 @@ async def handle_generate_dynamic_subtitles(arguments: dict) -> Sequence[TextCon
|
||||
# clip's captions onto a single wrong spine element instead of each
|
||||
# clip's own. See Engine/docs/05_EXPERIENCIAS.md, entry 2026-08-17.
|
||||
lines = modifier.generate_dynamic_subtitles(
|
||||
el, clip_words, config, segments=clip_segments
|
||||
el, clip_words, configs=configs, segments=clip_segments,
|
||||
role=single_role_override,
|
||||
compound_subphrases=True,
|
||||
)
|
||||
added.append((name, len(lines), len(clip_words)))
|
||||
|
||||
@@ -443,6 +514,10 @@ async def handle_generate_plain_subtitles(arguments: dict) -> Sequence[TextConte
|
||||
clip_filter = arguments.get("clip_name")
|
||||
|
||||
saved = load_plain_subtitle_config()
|
||||
saved["role"] = _title_subrole(
|
||||
arguments.get("role") or saved.get("role"),
|
||||
_DEFAULT_PLAIN_ROLE,
|
||||
)
|
||||
font = arguments.get("font") or saved["font"]
|
||||
font_size = int(arguments.get("font_size", saved["font_size"]))
|
||||
font_color = arguments.get("font_color") or saved["font_color"]
|
||||
@@ -506,6 +581,7 @@ async def handle_generate_plain_subtitles(arguments: dict) -> Sequence[TextConte
|
||||
face=None,
|
||||
font_scale=1.0,
|
||||
size_param=font_size,
|
||||
role=saved["role"],
|
||||
)
|
||||
created += 1
|
||||
if created:
|
||||
@@ -558,37 +634,30 @@ async def handle_generate_subtitles_by_emphasis(arguments: dict) -> Sequence[Tex
|
||||
clip_filter = arguments.get("clip_name")
|
||||
granularity = arguments.get("granularity", "phrase")
|
||||
|
||||
saved_dynamic = load_dynamic_subtitle_config()
|
||||
body_color = saved_dynamic["active_color"]
|
||||
dynamic_config = DynamicSubtitleConfig(
|
||||
style=WordStyle(
|
||||
font=saved_dynamic["font"],
|
||||
font_size=int(saved_dynamic["font_size"]),
|
||||
active_color=body_color,
|
||||
inactive_color="0.7 0.7 0.7 1",
|
||||
emphasis_look=WordLook(
|
||||
int(saved_dynamic["emphasis_size"]),
|
||||
saved_dynamic["emphasis_color"] or body_color,
|
||||
font=saved_dynamic["emphasis_font"],
|
||||
face=saved_dynamic["emphasis_face"],
|
||||
kerning=0.0,
|
||||
),
|
||||
body_look=WordLook(
|
||||
int(saved_dynamic["font_size"]),
|
||||
body_color,
|
||||
font=saved_dynamic["font"],
|
||||
face="Bold",
|
||||
kerning=1.2,
|
||||
),
|
||||
),
|
||||
band_height=float(saved_dynamic["band_height"]),
|
||||
block_center_y=float(saved_dynamic["block_center_y"]),
|
||||
granularity=granularity,
|
||||
text_scale=float(saved_dynamic["text_scale"]),
|
||||
line_gap=float(saved_dynamic["line_gap"]),
|
||||
# One or more named, active layouts — with 2+ active, each emphasis block
|
||||
# below randomly samples one of them (see generate_dynamic_subtitles).
|
||||
active_dynamic_layouts = get_active_dynamic_subtitle_layouts()
|
||||
dynamic_configs = [
|
||||
_build_dynamic_subtitle_config(saved, {"granularity": granularity})
|
||||
for saved in active_dynamic_layouts
|
||||
]
|
||||
single_dynamic_role = (
|
||||
active_dynamic_layouts[0]["role"] if len(active_dynamic_layouts) == 1 else None
|
||||
)
|
||||
|
||||
saved_plain = load_plain_subtitle_config()
|
||||
dynamic_role, plain_role = _separate_subtitle_roles(
|
||||
single_dynamic_role or dynamic_configs[0].role,
|
||||
saved_plain.get("role"),
|
||||
)
|
||||
if len(active_dynamic_layouts) == 1:
|
||||
single_dynamic_role = dynamic_role
|
||||
else:
|
||||
for cfg in dynamic_configs:
|
||||
cfg.role = _title_subrole(cfg.role, _DEFAULT_DYNAMIC_ROLE)
|
||||
if cfg.role == plain_role:
|
||||
cfg.role = _DEFAULT_DYNAMIC_ROLE
|
||||
saved_plain["role"] = plain_role
|
||||
plain_font = saved_plain["font"]
|
||||
plain_font_size = int(saved_plain["font_size"])
|
||||
plain_font_color = saved_plain["font_color"]
|
||||
@@ -618,20 +687,28 @@ async def handle_generate_subtitles_by_emphasis(arguments: dict) -> Sequence[Tex
|
||||
continue
|
||||
|
||||
spans = _load_emphasis_spans(media_path)
|
||||
if not spans:
|
||||
exclude_spans = _load_plain_exclude_spans(media_path)
|
||||
if not spans and not exclude_spans:
|
||||
no_review.append(name)
|
||||
|
||||
clip_source_start = modifier.source_file_start(el).to_seconds()
|
||||
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
|
||||
window_end = clip_source_start + clip_duration
|
||||
|
||||
def _clip_relative(span_list: list[dict]) -> list[tuple[float, float]]:
|
||||
return [
|
||||
(max(0.0, float(s["start"]) - clip_source_start), min(clip_duration, float(s["end"]) - clip_source_start))
|
||||
for s in span_list
|
||||
if float(s["end"]) > clip_source_start and float(s["start"]) < window_end
|
||||
]
|
||||
|
||||
# Clip-relative windows, for deciding which plain titles to hide —
|
||||
# same coordinate space add_text_title's offsets end up in.
|
||||
clip_spans = [
|
||||
(max(0.0, float(s["start"]) - clip_source_start), min(clip_duration, float(s["end"]) - clip_source_start))
|
||||
for s in spans
|
||||
if float(s["end"]) > clip_source_start and float(s["start"]) < window_end
|
||||
]
|
||||
# same coordinate space add_text_title's offsets end up in. Dynamic
|
||||
# spans hide plain (see the trade-off note below); explicit
|
||||
# `subtitle_common: false` spans hide it too, even without a dynamic
|
||||
# line covering the gap.
|
||||
clip_spans = _clip_relative(spans)
|
||||
clip_hide_plain_spans = clip_spans + _clip_relative(exclude_spans)
|
||||
|
||||
all_words = data.get("words", [])
|
||||
|
||||
@@ -656,7 +733,9 @@ async def handle_generate_subtitles_by_emphasis(arguments: dict) -> Sequence[Tex
|
||||
# entry 2026-08-17).
|
||||
dynamic_lines = len(
|
||||
modifier.generate_dynamic_subtitles(
|
||||
el, clip_emphasis_words, dynamic_config, segments=clip_segments
|
||||
el, clip_emphasis_words, configs=dynamic_configs, segments=clip_segments,
|
||||
role=single_dynamic_role,
|
||||
compound_subphrases=True,
|
||||
)
|
||||
)
|
||||
dynamic_word_count = len(clip_emphasis_words)
|
||||
@@ -683,7 +762,7 @@ async def handle_generate_subtitles_by_emphasis(arguments: dict) -> Sequence[Tex
|
||||
continue
|
||||
start = max(0.0, min(float(w.get("start", 0.0)) for w in block))
|
||||
end = max(float(w.get("end", start)) for w in block)
|
||||
if _overlaps_any_span(start, end, clip_spans):
|
||||
if _overlaps_any_span(start, end, clip_hide_plain_spans):
|
||||
plain_hidden += 1
|
||||
continue
|
||||
duration = max(end - start, modifier.frame_duration_fraction())
|
||||
@@ -701,6 +780,7 @@ async def handle_generate_subtitles_by_emphasis(arguments: dict) -> Sequence[Tex
|
||||
face=None,
|
||||
font_scale=1.0,
|
||||
size_param=plain_font_size,
|
||||
role=saved_plain["role"],
|
||||
)
|
||||
plain_created += 1
|
||||
|
||||
|
||||
Reference in New Issue
Block a user