Transforma a etapa "colar decisões" numa tela de lapidação: a sugestão da IA chega carregada e o editor afina frase a frase o que é ênfase e o que fica fora. Essa marcação é o norte da etapa 6 — só as frases com ênfase recebem zoom e legenda dinâmica; as demais ficam com legenda comum. O campo de colar o JSON sobe para a etapa 4, então a numeração das etapas não muda e a etapa 6 segue intacta. Backend (fcpxml/phrase_review.py): - build_phrase_review funde o _voice_timeline.json com as actions da IA - trim por frase que anda em fronteira de palavra; corte parcial da IA chega como trim em vez de ser arredondado fora - phrase_review_to_actions volta a cuts/zooms + emphasis_spans - merge_saved_decisions reaplica só as decisões salvas sobre uma revisão remontada da análise atual, para reprocessar a voz não ficar mascarado - resolve_source acha a mídia: o voice timeline guarda só o nome do arquivo App (SwiftUI): - layout de sala de edição: preview em cima, inspector à direita, timeline atravessando embaixo com seis trilhas rotuladas - preview enquadra no formato de entrega lido do .fcpxml (fonte horizontal, projeto vertical), com alternância para a mídia original - reprodução pula os trechos removidos e para no fim do trecho - zoom manual por trecho marcado, sem guardar escala: a forma vem das configurações de Análise de Voz no render - emoção da fala exposta por frase Correções encontradas no caminho: - VideoPlayer (AVKit) aborta em runtime no app compilado por swiftc; trocado por AVPlayerLayer (ver Engine/docs/05_EXPERIENCIAS.md #22) - teste que ainda afirmava o default zoom scale=1.3 removido do parser (#21) Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
749 lines
26 KiB
Python
749 lines
26 KiB
Python
"""Explicit transcription model management (Hex-inspired design).
|
|
|
|
Adapts the model-management concept from the Hex macOS app to the fcp-mcp-server
|
|
stack (Python + MCP), keeping our conventions: rational, allowlisted model
|
|
names, lazy optional imports, graceful degradation, and I/O confined to the
|
|
model cache directory.
|
|
|
|
This module is the skeleton of the manager. The catalog and cache primitives
|
|
are real; the MCP handlers and wiring into ``transcribe()`` come in a later
|
|
phase (see docs/TRANSCRIPTION-MODELS.md).
|
|
|
|
Model weights are the same Systran/faster-whisper artifacts that
|
|
``transcribe.py`` already loads, so install status here matches what
|
|
``WhisperModel(model_size, ...)`` would download on its own.
|
|
"""
|
|
|
|
import json
|
|
import logging
|
|
import re
|
|
import threading
|
|
from pathlib import Path
|
|
from typing import Callable, List, Optional
|
|
|
|
from .transcribe import ALLOWED_MODELS
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
# faster-whisper (CTranslate2) caches HF snapshots under ~/.cache/huggingface/hub
|
|
# as models--Systran--faster-whisper-<size>. This is the on-disk truth for
|
|
# "is downloaded".
|
|
_HF_REPO = "Systran/faster-whisper"
|
|
|
|
# Default cache root — matches what faster-whisper (HF hub) uses on its own,
|
|
# so a default install lines up with anything already on disk.
|
|
_DEFAULT_MODELS_DIR = Path.home() / ".cache" / "huggingface" / "hub"
|
|
|
|
# Config file for the persisted model selection + models dir.
|
|
_CONFIG_DIR = Path.home() / ".fcp-mcp-server"
|
|
_CONFIG_FILE = _CONFIG_DIR / "config.json"
|
|
|
|
# Path-traversal guard: only allow [A-Za-z0-9._-] in an internal model name
|
|
# (already enforced by ALLOWED_MODELS, but the cache-dir helper is belt + braces).
|
|
_SAFE_NAME_RE = re.compile(r"^[\w.-]+$")
|
|
|
|
# Progress callback signature used across the module.
|
|
ProgressCallback = Callable[[float], None]
|
|
|
|
|
|
def get_models_dir() -> Path:
|
|
"""The configured models root (falls back to the default HF hub cache)."""
|
|
configured = _load_config().get("models_dir")
|
|
if configured:
|
|
return Path(configured)
|
|
return _DEFAULT_MODELS_DIR
|
|
|
|
|
|
def save_models_dir(path: str) -> str:
|
|
"""Persist the models root directory. Returns the stored value."""
|
|
if not path.strip():
|
|
raise ValueError("models_dir cannot be empty")
|
|
_CONFIG_DIR.mkdir(parents=True, exist_ok=True)
|
|
data = _load_config()
|
|
data["models_dir"] = str(Path(path).expanduser())
|
|
_write_config(data)
|
|
return data["models_dir"]
|
|
|
|
|
|
def _load_config() -> dict:
|
|
"""Read the full config JSON (never raises; returns {} on error)."""
|
|
try:
|
|
return json.loads(_CONFIG_FILE.read_text(encoding="utf-8"))
|
|
except (OSError, ValueError):
|
|
return {}
|
|
|
|
|
|
def _write_config(data: dict) -> None:
|
|
_CONFIG_DIR.mkdir(parents=True, exist_ok=True)
|
|
_CONFIG_FILE.write_text(json.dumps(data, indent=2), encoding="utf-8")
|
|
|
|
|
|
def _hf_snapshot_dir(model_size: str) -> Path:
|
|
"""Cache folder for a given model size's HF snapshot, under the models root.
|
|
|
|
``Systran/faster-whisper`` -> ``<root>/models--Systran--faster-whisper-<size>``.
|
|
"""
|
|
name = "-".join(_HF_REPO.replace("/", "--").split("-")) + f"-{model_size}"
|
|
return get_models_dir() / f"models--{name}"
|
|
|
|
|
|
def model_cache_dir(model_size: str) -> Path:
|
|
"""The on-disk cache directory for ``model_size`` (as downloaded on disk)."""
|
|
if not _SAFE_NAME_RE.match(model_size):
|
|
raise ValueError(f"Unsafe model name: {model_size!r}")
|
|
return _hf_snapshot_dir(model_size)
|
|
|
|
|
|
def _load_bundled_catalog() -> Optional[list[dict]]:
|
|
"""Read the curated catalog from the bundled ``models.json`` (lazy, cached)."""
|
|
global _catalog
|
|
if _catalog is not None:
|
|
return _catalog
|
|
path = Path(__file__).with_name("models.json")
|
|
try:
|
|
with path.open(encoding="utf-8") as fh:
|
|
_catalog = json.load(fh)
|
|
except (OSError, ValueError):
|
|
logger.warning("Failed to load bundled models.json (%s)", path)
|
|
_catalog = []
|
|
return _catalog
|
|
|
|
|
|
_catalog: Optional[list[dict]] = None
|
|
|
|
|
|
def load_catalog() -> list[dict]:
|
|
"""All curated models (or ``[]`` if the bundled file is unreadable)."""
|
|
return list(_load_bundled_catalog() or [])
|
|
|
|
|
|
def get_catalog_model(internal_name: str) -> Optional[dict]:
|
|
"""The catalog entry whose ``internal_name`` equals ``internal_name``."""
|
|
for entry in load_catalog():
|
|
if entry.get("internal_name") == internal_name:
|
|
return entry
|
|
return None
|
|
|
|
|
|
def is_model_downloaded(model_size: str) -> bool:
|
|
"""True when the model's cache snapshot exists and isn't an empty dir.
|
|
|
|
Never raises on I/O; reports ``False`` for missing/unreadable dirs so
|
|
callers can offer a download instead of crashing.
|
|
"""
|
|
try:
|
|
d = model_cache_dir(model_size)
|
|
except ValueError:
|
|
return False
|
|
if not d.is_dir():
|
|
return False
|
|
try:
|
|
return any(d.iterdir())
|
|
except OSError:
|
|
return False
|
|
|
|
|
|
def list_installed_models() -> List[str]:
|
|
"""Model sizes present in the cache, filtered to the known allowlist."""
|
|
root = get_models_dir()
|
|
if not root.is_dir():
|
|
return []
|
|
installed: List[str] = []
|
|
for entry in root.iterdir():
|
|
if not entry.is_dir():
|
|
continue
|
|
for size in ALLOWED_MODELS:
|
|
if entry.name == _hf_snapshot_dir(size).name and is_model_downloaded(size):
|
|
installed.append(size)
|
|
break
|
|
# Stable, de-duped order following the allowlist.
|
|
return [s for s in ALLOWED_MODELS if s in installed]
|
|
|
|
|
|
def download_model(
|
|
model_size: str,
|
|
*,
|
|
progress_cb: Optional[ProgressCallback] = None,
|
|
cancel_event: Optional[threading.Event] = None,
|
|
) -> Optional[Path]:
|
|
"""Download a model snapshot to the cache.
|
|
|
|
Validates ``model_size`` against ``ALLOWED_MODELS`` and returns ``None``
|
|
(with a logged install hint) when ``huggingface_hub`` is unavailable —
|
|
the same graceful-degradation contract as ``transcribe()``.
|
|
|
|
``cancel_event`` (a ``threading.Event``) aborts the download on the next
|
|
progress tick if set; the partially-downloaded snapshot is removed.
|
|
"""
|
|
if model_size not in ALLOWED_MODELS:
|
|
raise ValueError(
|
|
f"model_size must be one of {', '.join(ALLOWED_MODELS)}, got {model_size!r}"
|
|
)
|
|
try:
|
|
from huggingface_hub import snapshot_download
|
|
except ImportError:
|
|
logger.info(
|
|
"huggingface_hub not installed; install the [transcribe] extra "
|
|
"to enable model download"
|
|
)
|
|
return None
|
|
|
|
target = model_cache_dir(model_size)
|
|
|
|
# tqdm hook that both reports progress and honors cancellation.
|
|
from tqdm import tqdm
|
|
|
|
class _CancelableTqdm(tqdm):
|
|
def update(self, n=1):
|
|
if cancel_event is not None and cancel_event.is_set():
|
|
raise _DownloadCancelledError(model_size)
|
|
super().update(n)
|
|
|
|
def _on_progress(current: int, total: int) -> None:
|
|
if cancel_event is not None and cancel_event.is_set():
|
|
raise _DownloadCancelledError(model_size)
|
|
if progress_cb is not None and total > 0:
|
|
progress_cb(current / total)
|
|
|
|
try:
|
|
snapshot_download(
|
|
repo_id=_HF_REPO,
|
|
revision=model_size,
|
|
# write into the same snapshot folder faster-whisper expects,
|
|
# under the configured models root.
|
|
cache_dir=str(get_models_dir()),
|
|
local_dir=str(target),
|
|
tqdm_class=_CancelableTqdm,
|
|
local_dir_use_symlinks=False,
|
|
)
|
|
# Local snapshot already landed in `target`; simpler than a temp+move.
|
|
except _DownloadCancelledError:
|
|
logger.info("Download cancelled for model %s", model_size)
|
|
_remove_dir(target)
|
|
return None
|
|
except Exception:
|
|
logger.warning("Failed to download model %s", model_size)
|
|
_remove_dir(target)
|
|
return None
|
|
return target
|
|
|
|
|
|
class _DownloadCancelledError(Exception):
|
|
"""Internal signal raised to abort a model download."""
|
|
|
|
|
|
def _remove_dir(path: Path) -> None:
|
|
import shutil
|
|
|
|
try:
|
|
if path.is_dir():
|
|
shutil.rmtree(path, ignore_errors=True)
|
|
except OSError:
|
|
logger.warning("Failed to clean up partial download %s", path)
|
|
|
|
|
|
def delete_model(model_size: str) -> bool:
|
|
"""Remove a model's cache snapshot from disk. Returns True if something was removed."""
|
|
try:
|
|
d = model_cache_dir(model_size)
|
|
except ValueError:
|
|
return False
|
|
if not d.is_dir():
|
|
return False
|
|
try:
|
|
import shutil
|
|
|
|
shutil.rmtree(d, ignore_errors=True)
|
|
except OSError:
|
|
logger.warning("Failed to delete model cache %s", d)
|
|
return False
|
|
return not d.exists()
|
|
|
|
|
|
def _load_selection() -> str:
|
|
"""The persisted selected model size (or ``""`` when absent/invalid)."""
|
|
return str(_load_config().get("selected_model", ""))
|
|
|
|
|
|
def save_selected_model(model_size: str) -> str:
|
|
"""Persist the selected model size. Validates against ``ALLOWED_MODELS``.
|
|
|
|
Returns the persisted value so callers can confirm it round-tripped.
|
|
"""
|
|
if model_size not in ALLOWED_MODELS:
|
|
raise ValueError(
|
|
f"model_size must be one of {', '.join(ALLOWED_MODELS)}, got {model_size!r}"
|
|
)
|
|
data = _load_config()
|
|
data["selected_model"] = model_size
|
|
_write_config(data)
|
|
return model_size
|
|
|
|
|
|
def load_selected_model() -> str:
|
|
"""The effective selected model, with a safe fallback.
|
|
|
|
- Returns the persisted selection when it's in ``ALLOWED_MODELS``.
|
|
- If the persisted model isn't on disk but another is installed, returns that
|
|
installed one (never clears the user's persisted value on a false-negative
|
|
availability scan, mirroring Hex's rule).
|
|
- Otherwise returns ``""`` so callers can fall back to ``"base"``.
|
|
"""
|
|
selected = _load_selection()
|
|
if selected and selected in ALLOWED_MODELS and is_model_downloaded(selected):
|
|
return selected
|
|
installed = list_installed_models()
|
|
if installed:
|
|
return installed[0]
|
|
return ""
|
|
|
|
|
|
def load_hf_token() -> str:
|
|
"""The persisted HuggingFace token for diarization (or ``""``)."""
|
|
return str(_load_config().get("hf_token", ""))
|
|
|
|
|
|
def save_hf_token(token: str) -> str:
|
|
"""Persist the HuggingFace token used for speaker diarization."""
|
|
data = _load_config()
|
|
data["hf_token"] = str(token or "").strip()
|
|
_write_config(data)
|
|
return data["hf_token"]
|
|
|
|
|
|
# Transcription languages, matching the codes used across the app UI.
|
|
# ``""`` / ``"auto"`` means "detect automatically".
|
|
ALLOWED_LANGUAGES = {
|
|
"auto",
|
|
"pt",
|
|
"en",
|
|
"es",
|
|
"fr",
|
|
"de",
|
|
"it",
|
|
"nl",
|
|
"ja",
|
|
"ko",
|
|
"zh",
|
|
}
|
|
|
|
|
|
def load_transcript_language() -> str:
|
|
"""The persisted transcription language (``""``/``"auto"`` = auto-detect)."""
|
|
lang = str(_load_config().get("language", ""))
|
|
return lang if lang in ALLOWED_LANGUAGES else "auto"
|
|
|
|
|
|
def save_transcript_language(lang: str) -> str:
|
|
"""Persist the transcription language. Returns the stored value."""
|
|
val = str(lang or "auto").strip().lower()
|
|
if val not in ALLOWED_LANGUAGES:
|
|
raise ValueError(
|
|
f"language must be one of {', '.join(sorted(ALLOWED_LANGUAGES))}, got {lang!r}"
|
|
)
|
|
data = _load_config()
|
|
data["language"] = val
|
|
_write_config(data)
|
|
return val
|
|
|
|
|
|
def load_num_speakers() -> str:
|
|
"""The persisted expected participant count (``""`` = auto-detect)."""
|
|
return str(_load_config().get("num_speakers", ""))
|
|
|
|
|
|
def save_num_speakers(num: str) -> str:
|
|
"""Persist the expected participant count (empty string = auto)."""
|
|
val = str(num or "").strip()
|
|
data = _load_config()
|
|
data["num_speakers"] = val
|
|
_write_config(data)
|
|
return val
|
|
|
|
|
|
DEFAULT_VOICE_ANALYSIS_CONFIG: dict = {
|
|
"energy_threshold": 0.5,
|
|
"emphasis_weights": {
|
|
"energy": 0.30,
|
|
"pitch_variation": 0.25,
|
|
"rate_variation": 0.20,
|
|
"pause_before": 0.15,
|
|
"duration": 0.10,
|
|
},
|
|
# Peaks are selected RELATIVELY — the top slice of the distribution —
|
|
# because the emphasis index is a weighted average whose real range
|
|
# depends on the material. Measured on a 17-minute interview the index
|
|
# never passed 0.55, so any absolute cutoff near the spec's 0.85 selects
|
|
# nothing; on punchier material the same cutoff would flood the edit.
|
|
# 2% of words is roughly one highlight every 50 words.
|
|
"peak_percentile": 0.02,
|
|
# Guard for genuinely flat audio, where even the top of the distribution
|
|
# carries no emphasis worth cutting on.
|
|
"emphasis_floor": 0.25,
|
|
"emotion_enabled": False,
|
|
"emotion_sensitivity": 0.5,
|
|
"zoom_scale": 1.30,
|
|
"zoom_mode": "in_out",
|
|
"zoom_ease_in": 0.25,
|
|
"zoom_ease_out": 0.04,
|
|
}
|
|
|
|
|
|
def load_voice_analysis_config() -> dict:
|
|
"""The persisted voice-analysis thresholds/weights, merged over defaults.
|
|
|
|
Backs the "Análise de Voz" settings screen: energy threshold (how loud
|
|
counts as "high energy"), the emphasis-index weights (see
|
|
``emphasis.EmphasisWeights``), the punch-in emphasis cutoff, and the
|
|
emotion-detection toggle/sensitivity. Unknown/malformed stored values
|
|
fall back to the default rather than raising, so a hand-edited or
|
|
partially-written config.json never breaks the settings screen.
|
|
"""
|
|
cfg = {
|
|
**DEFAULT_VOICE_ANALYSIS_CONFIG,
|
|
"emphasis_weights": dict(DEFAULT_VOICE_ANALYSIS_CONFIG["emphasis_weights"]),
|
|
}
|
|
stored = _load_config().get("voice_analysis")
|
|
if not isinstance(stored, dict):
|
|
return cfg
|
|
for key in (
|
|
"energy_threshold", "peak_percentile", "emphasis_floor",
|
|
"emotion_sensitivity", "zoom_scale", "zoom_ease_in", "zoom_ease_out",
|
|
):
|
|
if key in stored:
|
|
try:
|
|
value = float(stored[key])
|
|
if key == "zoom_scale":
|
|
cfg[key] = max(1.0, min(3.0, value))
|
|
elif key.startswith("zoom_ease"):
|
|
cfg[key] = max(0.01, min(5.0, value))
|
|
else:
|
|
cfg[key] = max(0.0, min(1.0, value))
|
|
except (TypeError, ValueError):
|
|
pass
|
|
if "emphasis_threshold" in stored and "emphasis_floor" not in stored:
|
|
try:
|
|
cfg["emphasis_floor"] = max(0.0, min(1.0, float(stored["emphasis_threshold"])))
|
|
except (TypeError, ValueError):
|
|
pass
|
|
if stored.get("zoom_mode") in ("in_out", "in", "out"):
|
|
cfg["zoom_mode"] = stored["zoom_mode"]
|
|
if "emotion_enabled" in stored:
|
|
cfg["emotion_enabled"] = bool(stored["emotion_enabled"])
|
|
weights = stored.get("emphasis_weights")
|
|
if isinstance(weights, dict):
|
|
for key in cfg["emphasis_weights"]:
|
|
if key in weights:
|
|
try:
|
|
cfg["emphasis_weights"][key] = max(0.0, float(weights[key]))
|
|
except (TypeError, ValueError):
|
|
pass
|
|
return cfg
|
|
|
|
|
|
def save_voice_analysis_config(
|
|
energy_threshold: float | None = None,
|
|
emphasis_weights: dict | None = None,
|
|
peak_percentile: float | None = None,
|
|
emphasis_floor: float | None = None,
|
|
emotion_enabled: bool | None = None,
|
|
emotion_sensitivity: float | None = None,
|
|
zoom_scale: float | None = None,
|
|
zoom_mode: str | None = None,
|
|
zoom_ease_in: float | None = None,
|
|
zoom_ease_out: float | None = None,
|
|
) -> dict:
|
|
"""Persist voice-analysis thresholds/weights. Only given fields change.
|
|
|
|
Returns the full merged config (same shape as
|
|
:func:`load_voice_analysis_config`) so callers can render it back
|
|
immediately without a second round-trip.
|
|
"""
|
|
cfg = load_voice_analysis_config()
|
|
if energy_threshold is not None:
|
|
cfg["energy_threshold"] = max(0.0, min(1.0, float(energy_threshold)))
|
|
if peak_percentile is not None:
|
|
cfg["peak_percentile"] = max(0.0, min(1.0, float(peak_percentile)))
|
|
if emphasis_floor is not None:
|
|
cfg["emphasis_floor"] = max(0.0, min(1.0, float(emphasis_floor)))
|
|
if emotion_enabled is not None:
|
|
cfg["emotion_enabled"] = bool(emotion_enabled)
|
|
if emotion_sensitivity is not None:
|
|
cfg["emotion_sensitivity"] = max(0.0, min(1.0, float(emotion_sensitivity)))
|
|
if zoom_scale is not None:
|
|
cfg["zoom_scale"] = max(1.0, min(3.0, float(zoom_scale)))
|
|
if zoom_mode in ("in_out", "in", "out"):
|
|
cfg["zoom_mode"] = zoom_mode
|
|
if zoom_ease_in is not None:
|
|
cfg["zoom_ease_in"] = max(0.01, min(5.0, float(zoom_ease_in)))
|
|
if zoom_ease_out is not None:
|
|
cfg["zoom_ease_out"] = max(0.01, min(5.0, float(zoom_ease_out)))
|
|
if emphasis_weights is not None:
|
|
for key, value in emphasis_weights.items():
|
|
if key in cfg["emphasis_weights"] and value is not None:
|
|
cfg["emphasis_weights"][key] = max(0.0, float(value))
|
|
data = _load_config()
|
|
data["voice_analysis"] = cfg
|
|
_write_config(data)
|
|
return cfg
|
|
|
|
|
|
# Mirrors the "Legendas Dinâmicas" tab's own defaults (MacApp/Sources/
|
|
# CaptionsView.swift), so a fresh install shows the same look in the UI and
|
|
# in what generate_dynamic_subtitles renders when no override is passed.
|
|
DEFAULT_DYNAMIC_SUBTITLE_CONFIG: dict = {
|
|
"band_height": 0.22,
|
|
"block_center_y": -167.0,
|
|
"line_gap": 8.0,
|
|
"font": "Helvetica Neue",
|
|
"font_size": 104,
|
|
"emphasis_font": "Playfair Display",
|
|
"emphasis_face": "Medium Italic",
|
|
"emphasis_size": 265,
|
|
"active_color": "1 1 1 1",
|
|
"emphasis_color": "1 1 1 1",
|
|
"text_scale": 2.0,
|
|
}
|
|
|
|
|
|
def load_dynamic_subtitle_config() -> dict:
|
|
"""The persisted dynamic-subtitle style, merged over defaults.
|
|
|
|
Backs the "Legendas Dinâmicas" settings screen AND is the fallback
|
|
``generate_dynamic_subtitles`` reads for any field the caller doesn't
|
|
explicitly override — so the style configured in the UI is what actually
|
|
renders, without the app having to thread every field through each call.
|
|
Unknown/malformed stored values fall back to the default, same as
|
|
:func:`load_voice_analysis_config`.
|
|
"""
|
|
cfg = dict(DEFAULT_DYNAMIC_SUBTITLE_CONFIG)
|
|
stored = _load_config().get("dynamic_subtitles")
|
|
if not isinstance(stored, dict):
|
|
return cfg
|
|
for key in ("band_height", "block_center_y", "line_gap", "text_scale"):
|
|
if key in stored:
|
|
try:
|
|
cfg[key] = float(stored[key])
|
|
except (TypeError, ValueError):
|
|
pass
|
|
for key in ("font_size", "emphasis_size"):
|
|
if key in stored:
|
|
try:
|
|
cfg[key] = int(stored[key])
|
|
except (TypeError, ValueError):
|
|
pass
|
|
for key in ("font", "emphasis_font", "emphasis_face", "active_color", "emphasis_color"):
|
|
if key in stored and isinstance(stored[key], str) and stored[key]:
|
|
cfg[key] = stored[key]
|
|
return cfg
|
|
|
|
|
|
def save_dynamic_subtitle_config(**fields) -> dict:
|
|
"""Persist dynamic-subtitle style fields. Only given fields change.
|
|
|
|
Accepts the same keys as :data:`DEFAULT_DYNAMIC_SUBTITLE_CONFIG`; unknown
|
|
keys are ignored so a newer app talking to an older config shape degrades
|
|
quietly. Returns the full merged config, mirroring
|
|
:func:`save_voice_analysis_config`.
|
|
"""
|
|
cfg = load_dynamic_subtitle_config()
|
|
for key, value in fields.items():
|
|
if key not in DEFAULT_DYNAMIC_SUBTITLE_CONFIG or value is None:
|
|
continue
|
|
if isinstance(DEFAULT_DYNAMIC_SUBTITLE_CONFIG[key], float):
|
|
try:
|
|
cfg[key] = float(value)
|
|
except (TypeError, ValueError):
|
|
continue
|
|
elif isinstance(DEFAULT_DYNAMIC_SUBTITLE_CONFIG[key], int):
|
|
try:
|
|
cfg[key] = int(value)
|
|
except (TypeError, ValueError):
|
|
continue
|
|
else:
|
|
cfg[key] = str(value)
|
|
data = _load_config()
|
|
data["dynamic_subtitles"] = cfg
|
|
_write_config(data)
|
|
return cfg
|
|
|
|
|
|
DEFAULT_PLAIN_SUBTITLE_CONFIG: dict = {
|
|
"font": "Helvetica Neue",
|
|
"font_size": 82,
|
|
"font_color": "1 1 1 1",
|
|
"max_words": 7,
|
|
"position_y": -820.0,
|
|
"uppercase": False,
|
|
"keep_punctuation": True,
|
|
"text_scale": 2.0,
|
|
}
|
|
|
|
|
|
def load_plain_subtitle_config() -> dict:
|
|
"""Persisted style for simple editable FCPXML title subtitles."""
|
|
cfg = dict(DEFAULT_PLAIN_SUBTITLE_CONFIG)
|
|
stored = _load_config().get("plain_subtitles")
|
|
if not isinstance(stored, dict):
|
|
return cfg
|
|
for key in ("position_y", "text_scale"):
|
|
if key in stored:
|
|
try:
|
|
cfg[key] = float(stored[key])
|
|
except (TypeError, ValueError):
|
|
pass
|
|
for key in ("font_size", "max_words"):
|
|
if key in stored:
|
|
try:
|
|
cfg[key] = int(stored[key])
|
|
except (TypeError, ValueError):
|
|
pass
|
|
for key in ("font", "font_color"):
|
|
if key in stored and isinstance(stored[key], str) and stored[key]:
|
|
cfg[key] = stored[key]
|
|
for key in ("uppercase", "keep_punctuation"):
|
|
if key in stored:
|
|
cfg[key] = bool(stored[key])
|
|
cfg["max_words"] = max(1, int(cfg["max_words"]))
|
|
return cfg
|
|
|
|
|
|
def save_plain_subtitle_config(**fields) -> dict:
|
|
"""Persist simple subtitle style fields. Only given fields change."""
|
|
cfg = load_plain_subtitle_config()
|
|
for key, value in fields.items():
|
|
if key not in DEFAULT_PLAIN_SUBTITLE_CONFIG or value is None:
|
|
continue
|
|
if isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], bool):
|
|
cfg[key] = bool(value)
|
|
elif isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], float):
|
|
try:
|
|
cfg[key] = float(value)
|
|
except (TypeError, ValueError):
|
|
continue
|
|
elif isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], int):
|
|
try:
|
|
cfg[key] = int(value)
|
|
except (TypeError, ValueError):
|
|
continue
|
|
else:
|
|
cfg[key] = str(value)
|
|
cfg["max_words"] = max(1, int(cfg["max_words"]))
|
|
data = _load_config()
|
|
data["plain_subtitles"] = cfg
|
|
_write_config(data)
|
|
return cfg
|
|
|
|
|
|
# Mirrors the silence thresholds the detection/removal handlers use when no
|
|
# argument is passed (server_tools/qc.py). Persisted so the app's slider and
|
|
# any later run agree without threading three fields through every call.
|
|
DEFAULT_SILENCE_CONFIG: dict = {
|
|
# dBFS below which audio counts as silence.
|
|
"noise_db": -30.0,
|
|
# Seconds a quiet stretch must last before it's a cut candidate.
|
|
"min_silence": 0.5,
|
|
# Seconds left inside each cut so speech never gets clipped at the edges.
|
|
"padding": 0.05,
|
|
}
|
|
|
|
|
|
def load_silence_config() -> dict:
|
|
"""The persisted silence-detection thresholds, merged over defaults.
|
|
|
|
Read by ``detect_media_silence``/``remove_media_silence`` as their
|
|
fallback, so the tolerance chosen in the app is what actually runs.
|
|
Malformed stored values fall back to the default rather than raising,
|
|
matching :func:`load_voice_analysis_config`.
|
|
"""
|
|
cfg = dict(DEFAULT_SILENCE_CONFIG)
|
|
stored = _load_config().get("silence")
|
|
if not isinstance(stored, dict):
|
|
return cfg
|
|
for key in cfg:
|
|
if key in stored:
|
|
try:
|
|
cfg[key] = float(stored[key])
|
|
except (TypeError, ValueError):
|
|
pass
|
|
return cfg
|
|
|
|
|
|
def save_silence_config(
|
|
noise_db: float | None = None,
|
|
min_silence: float | None = None,
|
|
padding: float | None = None,
|
|
) -> dict:
|
|
"""Persist silence thresholds. Only the given fields change.
|
|
|
|
Values are clamped to the same ranges the handlers validate against, so
|
|
a bad write here can't produce a config the tools would later reject.
|
|
"""
|
|
cfg = load_silence_config()
|
|
if noise_db is not None:
|
|
try:
|
|
cfg["noise_db"] = max(-120.0, min(0.0, float(noise_db)))
|
|
except (TypeError, ValueError):
|
|
pass
|
|
if min_silence is not None:
|
|
try:
|
|
cfg["min_silence"] = max(0.01, min(3600.0, float(min_silence)))
|
|
except (TypeError, ValueError):
|
|
pass
|
|
if padding is not None:
|
|
try:
|
|
cfg["padding"] = max(0.0, min(5.0, float(padding)))
|
|
except (TypeError, ValueError):
|
|
pass
|
|
data = _load_config()
|
|
data["silence"] = cfg
|
|
_write_config(data)
|
|
return cfg
|
|
|
|
|
|
# Last project worked on, so the app reopens where the user left off instead of
|
|
# making them pick the folder again every launch. Only paths that still exist
|
|
# are handed back — a project on an unmounted volume degrades to "none selected"
|
|
# rather than to a dead path the tools would later fail on.
|
|
DEFAULT_PROJECT_CONFIG: dict = {
|
|
# Folder every generated file (transcript .json, XML, SRT) is written to.
|
|
"folder": "",
|
|
# The .fcpxml/.fcpxmld that was loaded from it.
|
|
"file": "",
|
|
}
|
|
|
|
|
|
def load_project_config() -> dict:
|
|
"""The persisted last project (folder + file), merged over defaults.
|
|
|
|
Paths that no longer exist on disk come back empty, matching what the app
|
|
shows for "nothing selected". Malformed stored values fall back to the
|
|
default rather than raising, same as :func:`load_voice_analysis_config`.
|
|
"""
|
|
cfg = dict(DEFAULT_PROJECT_CONFIG)
|
|
stored = _load_config().get("project")
|
|
if not isinstance(stored, dict):
|
|
return cfg
|
|
for key in cfg:
|
|
value = stored.get(key)
|
|
if isinstance(value, str) and value and Path(value).exists():
|
|
cfg[key] = value
|
|
return cfg
|
|
|
|
|
|
def save_project_config(folder: str | None = None, file: str | None = None) -> dict:
|
|
"""Persist the last project folder/file. Only the given fields change.
|
|
|
|
Passing an empty string clears a field (the app does this when the user
|
|
deselects), while ``None`` leaves it untouched.
|
|
"""
|
|
cfg = load_project_config()
|
|
for key, value in (("folder", folder), ("file", file)):
|
|
if value is None:
|
|
continue
|
|
cfg[key] = str(Path(value).expanduser()) if str(value).strip() else ""
|
|
data = _load_config()
|
|
data["project"] = cfg
|
|
_write_config(data)
|
|
return cfg
|