Files
gart/code/fcpxml/model_manager.py
T
João HenriqueandClaude Sonnet 5 32d78d0f8d fix(qc): padding padrão do remove_media_silence de 0,05s para 0,2s
0,05s existia como margem de segurança contra cortar a palavra em cima,
mas um silêncio que essa ferramenta encontra costuma ser o respiro
natural antes de uma frase nova, não sujeira de edição — e 0,05s raspava
esse respiro quase todo.

Caso real (projeto Mastopexia): a pausa antes de "Com" tinha 0,567s no
áudio original; com padding 0,05 sobrou só ~0,1s no total (0,05 de cada
lado), colando o clipe seguinte a 5ms da palavra em vez de deixar uma
pausa perceptível. 0,2s alinha com a convenção já documentada para folga
em corte de fronteira de frase (editar-por-voz/06-texto-corte-marcador.md).

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-08-26 17:15:44 -04:00

756 lines
26 KiB
Python

"""Explicit transcription model management (Hex-inspired design).
Adapts the model-management concept from the Hex macOS app to the fcp-mcp-server
stack (Python + MCP), keeping our conventions: rational, allowlisted model
names, lazy optional imports, graceful degradation, and I/O confined to the
model cache directory.
This module is the skeleton of the manager. The catalog and cache primitives
are real; the MCP handlers and wiring into ``transcribe()`` come in a later
phase (see docs/TRANSCRIPTION-MODELS.md).
Model weights are the same Systran/faster-whisper artifacts that
``transcribe.py`` already loads, so install status here matches what
``WhisperModel(model_size, ...)`` would download on its own.
"""
import json
import logging
import re
import threading
from pathlib import Path
from typing import Callable, List, Optional
from .transcribe import ALLOWED_MODELS
logger = logging.getLogger(__name__)
# faster-whisper (CTranslate2) caches HF snapshots under ~/.cache/huggingface/hub
# as models--Systran--faster-whisper-<size>. This is the on-disk truth for
# "is downloaded".
_HF_REPO = "Systran/faster-whisper"
# Default cache root — matches what faster-whisper (HF hub) uses on its own,
# so a default install lines up with anything already on disk.
_DEFAULT_MODELS_DIR = Path.home() / ".cache" / "huggingface" / "hub"
# Config file for the persisted model selection + models dir.
_CONFIG_DIR = Path.home() / ".fcp-mcp-server"
_CONFIG_FILE = _CONFIG_DIR / "config.json"
# Path-traversal guard: only allow [A-Za-z0-9._-] in an internal model name
# (already enforced by ALLOWED_MODELS, but the cache-dir helper is belt + braces).
_SAFE_NAME_RE = re.compile(r"^[\w.-]+$")
# Progress callback signature used across the module.
ProgressCallback = Callable[[float], None]
def get_models_dir() -> Path:
"""The configured models root (falls back to the default HF hub cache)."""
configured = _load_config().get("models_dir")
if configured:
return Path(configured)
return _DEFAULT_MODELS_DIR
def save_models_dir(path: str) -> str:
"""Persist the models root directory. Returns the stored value."""
if not path.strip():
raise ValueError("models_dir cannot be empty")
_CONFIG_DIR.mkdir(parents=True, exist_ok=True)
data = _load_config()
data["models_dir"] = str(Path(path).expanduser())
_write_config(data)
return data["models_dir"]
def _load_config() -> dict:
"""Read the full config JSON (never raises; returns {} on error)."""
try:
return json.loads(_CONFIG_FILE.read_text(encoding="utf-8"))
except (OSError, ValueError):
return {}
def _write_config(data: dict) -> None:
_CONFIG_DIR.mkdir(parents=True, exist_ok=True)
_CONFIG_FILE.write_text(json.dumps(data, indent=2), encoding="utf-8")
def _hf_snapshot_dir(model_size: str) -> Path:
"""Cache folder for a given model size's HF snapshot, under the models root.
``Systran/faster-whisper`` -> ``<root>/models--Systran--faster-whisper-<size>``.
"""
name = "-".join(_HF_REPO.replace("/", "--").split("-")) + f"-{model_size}"
return get_models_dir() / f"models--{name}"
def model_cache_dir(model_size: str) -> Path:
"""The on-disk cache directory for ``model_size`` (as downloaded on disk)."""
if not _SAFE_NAME_RE.match(model_size):
raise ValueError(f"Unsafe model name: {model_size!r}")
return _hf_snapshot_dir(model_size)
def _load_bundled_catalog() -> Optional[list[dict]]:
"""Read the curated catalog from the bundled ``models.json`` (lazy, cached)."""
global _catalog
if _catalog is not None:
return _catalog
path = Path(__file__).with_name("models.json")
try:
with path.open(encoding="utf-8") as fh:
_catalog = json.load(fh)
except (OSError, ValueError):
logger.warning("Failed to load bundled models.json (%s)", path)
_catalog = []
return _catalog
_catalog: Optional[list[dict]] = None
def load_catalog() -> list[dict]:
"""All curated models (or ``[]`` if the bundled file is unreadable)."""
return list(_load_bundled_catalog() or [])
def get_catalog_model(internal_name: str) -> Optional[dict]:
"""The catalog entry whose ``internal_name`` equals ``internal_name``."""
for entry in load_catalog():
if entry.get("internal_name") == internal_name:
return entry
return None
def is_model_downloaded(model_size: str) -> bool:
"""True when the model's cache snapshot exists and isn't an empty dir.
Never raises on I/O; reports ``False`` for missing/unreadable dirs so
callers can offer a download instead of crashing.
"""
try:
d = model_cache_dir(model_size)
except ValueError:
return False
if not d.is_dir():
return False
try:
return any(d.iterdir())
except OSError:
return False
def list_installed_models() -> List[str]:
"""Model sizes present in the cache, filtered to the known allowlist."""
root = get_models_dir()
if not root.is_dir():
return []
installed: List[str] = []
for entry in root.iterdir():
if not entry.is_dir():
continue
for size in ALLOWED_MODELS:
if entry.name == _hf_snapshot_dir(size).name and is_model_downloaded(size):
installed.append(size)
break
# Stable, de-duped order following the allowlist.
return [s for s in ALLOWED_MODELS if s in installed]
def download_model(
model_size: str,
*,
progress_cb: Optional[ProgressCallback] = None,
cancel_event: Optional[threading.Event] = None,
) -> Optional[Path]:
"""Download a model snapshot to the cache.
Validates ``model_size`` against ``ALLOWED_MODELS`` and returns ``None``
(with a logged install hint) when ``huggingface_hub`` is unavailable —
the same graceful-degradation contract as ``transcribe()``.
``cancel_event`` (a ``threading.Event``) aborts the download on the next
progress tick if set; the partially-downloaded snapshot is removed.
"""
if model_size not in ALLOWED_MODELS:
raise ValueError(
f"model_size must be one of {', '.join(ALLOWED_MODELS)}, got {model_size!r}"
)
try:
from huggingface_hub import snapshot_download
except ImportError:
logger.info(
"huggingface_hub not installed; install the [transcribe] extra "
"to enable model download"
)
return None
target = model_cache_dir(model_size)
# tqdm hook that both reports progress and honors cancellation.
from tqdm import tqdm
class _CancelableTqdm(tqdm):
def update(self, n=1):
if cancel_event is not None and cancel_event.is_set():
raise _DownloadCancelledError(model_size)
super().update(n)
def _on_progress(current: int, total: int) -> None:
if cancel_event is not None and cancel_event.is_set():
raise _DownloadCancelledError(model_size)
if progress_cb is not None and total > 0:
progress_cb(current / total)
try:
snapshot_download(
repo_id=_HF_REPO,
revision=model_size,
# write into the same snapshot folder faster-whisper expects,
# under the configured models root.
cache_dir=str(get_models_dir()),
local_dir=str(target),
tqdm_class=_CancelableTqdm,
local_dir_use_symlinks=False,
)
# Local snapshot already landed in `target`; simpler than a temp+move.
except _DownloadCancelledError:
logger.info("Download cancelled for model %s", model_size)
_remove_dir(target)
return None
except Exception:
logger.warning("Failed to download model %s", model_size)
_remove_dir(target)
return None
return target
class _DownloadCancelledError(Exception):
"""Internal signal raised to abort a model download."""
def _remove_dir(path: Path) -> None:
import shutil
try:
if path.is_dir():
shutil.rmtree(path, ignore_errors=True)
except OSError:
logger.warning("Failed to clean up partial download %s", path)
def delete_model(model_size: str) -> bool:
"""Remove a model's cache snapshot from disk. Returns True if something was removed."""
try:
d = model_cache_dir(model_size)
except ValueError:
return False
if not d.is_dir():
return False
try:
import shutil
shutil.rmtree(d, ignore_errors=True)
except OSError:
logger.warning("Failed to delete model cache %s", d)
return False
return not d.exists()
def _load_selection() -> str:
"""The persisted selected model size (or ``""`` when absent/invalid)."""
return str(_load_config().get("selected_model", ""))
def save_selected_model(model_size: str) -> str:
"""Persist the selected model size. Validates against ``ALLOWED_MODELS``.
Returns the persisted value so callers can confirm it round-tripped.
"""
if model_size not in ALLOWED_MODELS:
raise ValueError(
f"model_size must be one of {', '.join(ALLOWED_MODELS)}, got {model_size!r}"
)
data = _load_config()
data["selected_model"] = model_size
_write_config(data)
return model_size
def load_selected_model() -> str:
"""The effective selected model, with a safe fallback.
- Returns the persisted selection when it's in ``ALLOWED_MODELS``.
- If the persisted model isn't on disk but another is installed, returns that
installed one (never clears the user's persisted value on a false-negative
availability scan, mirroring Hex's rule).
- Otherwise returns ``""`` so callers can fall back to ``"base"``.
"""
selected = _load_selection()
if selected and selected in ALLOWED_MODELS and is_model_downloaded(selected):
return selected
installed = list_installed_models()
if installed:
return installed[0]
return ""
def load_hf_token() -> str:
"""The persisted HuggingFace token for diarization (or ``""``)."""
return str(_load_config().get("hf_token", ""))
def save_hf_token(token: str) -> str:
"""Persist the HuggingFace token used for speaker diarization."""
data = _load_config()
data["hf_token"] = str(token or "").strip()
_write_config(data)
return data["hf_token"]
# Transcription languages, matching the codes used across the app UI.
# ``""`` / ``"auto"`` means "detect automatically".
ALLOWED_LANGUAGES = {
"auto",
"pt",
"en",
"es",
"fr",
"de",
"it",
"nl",
"ja",
"ko",
"zh",
}
def load_transcript_language() -> str:
"""The persisted transcription language (``""``/``"auto"`` = auto-detect)."""
lang = str(_load_config().get("language", ""))
return lang if lang in ALLOWED_LANGUAGES else "auto"
def save_transcript_language(lang: str) -> str:
"""Persist the transcription language. Returns the stored value."""
val = str(lang or "auto").strip().lower()
if val not in ALLOWED_LANGUAGES:
raise ValueError(
f"language must be one of {', '.join(sorted(ALLOWED_LANGUAGES))}, got {lang!r}"
)
data = _load_config()
data["language"] = val
_write_config(data)
return val
def load_num_speakers() -> str:
"""The persisted expected participant count (``""`` = auto-detect)."""
return str(_load_config().get("num_speakers", ""))
def save_num_speakers(num: str) -> str:
"""Persist the expected participant count (empty string = auto)."""
val = str(num or "").strip()
data = _load_config()
data["num_speakers"] = val
_write_config(data)
return val
DEFAULT_VOICE_ANALYSIS_CONFIG: dict = {
"energy_threshold": 0.5,
"emphasis_weights": {
"energy": 0.30,
"pitch_variation": 0.25,
"rate_variation": 0.20,
"pause_before": 0.15,
"duration": 0.10,
},
# Peaks are selected RELATIVELY — the top slice of the distribution —
# because the emphasis index is a weighted average whose real range
# depends on the material. Measured on a 17-minute interview the index
# never passed 0.55, so any absolute cutoff near the spec's 0.85 selects
# nothing; on punchier material the same cutoff would flood the edit.
# 2% of words is roughly one highlight every 50 words.
"peak_percentile": 0.02,
# Guard for genuinely flat audio, where even the top of the distribution
# carries no emphasis worth cutting on.
"emphasis_floor": 0.25,
"emotion_enabled": False,
"emotion_sensitivity": 0.5,
"zoom_scale": 1.30,
"zoom_mode": "in_out",
"zoom_ease_in": 0.25,
"zoom_ease_out": 0.04,
}
def load_voice_analysis_config() -> dict:
"""The persisted voice-analysis thresholds/weights, merged over defaults.
Backs the "Análise de Voz" settings screen: energy threshold (how loud
counts as "high energy"), the emphasis-index weights (see
``emphasis.EmphasisWeights``), the punch-in emphasis cutoff, and the
emotion-detection toggle/sensitivity. Unknown/malformed stored values
fall back to the default rather than raising, so a hand-edited or
partially-written config.json never breaks the settings screen.
"""
cfg = {
**DEFAULT_VOICE_ANALYSIS_CONFIG,
"emphasis_weights": dict(DEFAULT_VOICE_ANALYSIS_CONFIG["emphasis_weights"]),
}
stored = _load_config().get("voice_analysis")
if not isinstance(stored, dict):
return cfg
for key in (
"energy_threshold", "peak_percentile", "emphasis_floor",
"emotion_sensitivity", "zoom_scale", "zoom_ease_in", "zoom_ease_out",
):
if key in stored:
try:
value = float(stored[key])
if key == "zoom_scale":
cfg[key] = max(1.0, min(3.0, value))
elif key.startswith("zoom_ease"):
cfg[key] = max(0.01, min(5.0, value))
else:
cfg[key] = max(0.0, min(1.0, value))
except (TypeError, ValueError):
pass
if "emphasis_threshold" in stored and "emphasis_floor" not in stored:
try:
cfg["emphasis_floor"] = max(0.0, min(1.0, float(stored["emphasis_threshold"])))
except (TypeError, ValueError):
pass
if stored.get("zoom_mode") in ("in_out", "in", "out"):
cfg["zoom_mode"] = stored["zoom_mode"]
if "emotion_enabled" in stored:
cfg["emotion_enabled"] = bool(stored["emotion_enabled"])
weights = stored.get("emphasis_weights")
if isinstance(weights, dict):
for key in cfg["emphasis_weights"]:
if key in weights:
try:
cfg["emphasis_weights"][key] = max(0.0, float(weights[key]))
except (TypeError, ValueError):
pass
return cfg
def save_voice_analysis_config(
energy_threshold: float | None = None,
emphasis_weights: dict | None = None,
peak_percentile: float | None = None,
emphasis_floor: float | None = None,
emotion_enabled: bool | None = None,
emotion_sensitivity: float | None = None,
zoom_scale: float | None = None,
zoom_mode: str | None = None,
zoom_ease_in: float | None = None,
zoom_ease_out: float | None = None,
) -> dict:
"""Persist voice-analysis thresholds/weights. Only given fields change.
Returns the full merged config (same shape as
:func:`load_voice_analysis_config`) so callers can render it back
immediately without a second round-trip.
"""
cfg = load_voice_analysis_config()
if energy_threshold is not None:
cfg["energy_threshold"] = max(0.0, min(1.0, float(energy_threshold)))
if peak_percentile is not None:
cfg["peak_percentile"] = max(0.0, min(1.0, float(peak_percentile)))
if emphasis_floor is not None:
cfg["emphasis_floor"] = max(0.0, min(1.0, float(emphasis_floor)))
if emotion_enabled is not None:
cfg["emotion_enabled"] = bool(emotion_enabled)
if emotion_sensitivity is not None:
cfg["emotion_sensitivity"] = max(0.0, min(1.0, float(emotion_sensitivity)))
if zoom_scale is not None:
cfg["zoom_scale"] = max(1.0, min(3.0, float(zoom_scale)))
if zoom_mode in ("in_out", "in", "out"):
cfg["zoom_mode"] = zoom_mode
if zoom_ease_in is not None:
cfg["zoom_ease_in"] = max(0.01, min(5.0, float(zoom_ease_in)))
if zoom_ease_out is not None:
cfg["zoom_ease_out"] = max(0.01, min(5.0, float(zoom_ease_out)))
if emphasis_weights is not None:
for key, value in emphasis_weights.items():
if key in cfg["emphasis_weights"] and value is not None:
cfg["emphasis_weights"][key] = max(0.0, float(value))
data = _load_config()
data["voice_analysis"] = cfg
_write_config(data)
return cfg
# Mirrors the "Legendas Dinâmicas" tab's own defaults (MacApp/Sources/
# CaptionsView.swift), so a fresh install shows the same look in the UI and
# in what generate_dynamic_subtitles renders when no override is passed.
DEFAULT_DYNAMIC_SUBTITLE_CONFIG: dict = {
"band_height": 0.22,
"block_center_y": -167.0,
"line_gap": 8.0,
"font": "Helvetica Neue",
"font_size": 104,
"emphasis_font": "Playfair Display",
"emphasis_face": "Medium Italic",
"emphasis_size": 265,
"active_color": "1 1 1 1",
"emphasis_color": "1 1 1 1",
"text_scale": 2.0,
}
def load_dynamic_subtitle_config() -> dict:
"""The persisted dynamic-subtitle style, merged over defaults.
Backs the "Legendas Dinâmicas" settings screen AND is the fallback
``generate_dynamic_subtitles`` reads for any field the caller doesn't
explicitly override — so the style configured in the UI is what actually
renders, without the app having to thread every field through each call.
Unknown/malformed stored values fall back to the default, same as
:func:`load_voice_analysis_config`.
"""
cfg = dict(DEFAULT_DYNAMIC_SUBTITLE_CONFIG)
stored = _load_config().get("dynamic_subtitles")
if not isinstance(stored, dict):
return cfg
for key in ("band_height", "block_center_y", "line_gap", "text_scale"):
if key in stored:
try:
cfg[key] = float(stored[key])
except (TypeError, ValueError):
pass
for key in ("font_size", "emphasis_size"):
if key in stored:
try:
cfg[key] = int(stored[key])
except (TypeError, ValueError):
pass
for key in ("font", "emphasis_font", "emphasis_face", "active_color", "emphasis_color"):
if key in stored and isinstance(stored[key], str) and stored[key]:
cfg[key] = stored[key]
return cfg
def save_dynamic_subtitle_config(**fields) -> dict:
"""Persist dynamic-subtitle style fields. Only given fields change.
Accepts the same keys as :data:`DEFAULT_DYNAMIC_SUBTITLE_CONFIG`; unknown
keys are ignored so a newer app talking to an older config shape degrades
quietly. Returns the full merged config, mirroring
:func:`save_voice_analysis_config`.
"""
cfg = load_dynamic_subtitle_config()
for key, value in fields.items():
if key not in DEFAULT_DYNAMIC_SUBTITLE_CONFIG or value is None:
continue
if isinstance(DEFAULT_DYNAMIC_SUBTITLE_CONFIG[key], float):
try:
cfg[key] = float(value)
except (TypeError, ValueError):
continue
elif isinstance(DEFAULT_DYNAMIC_SUBTITLE_CONFIG[key], int):
try:
cfg[key] = int(value)
except (TypeError, ValueError):
continue
else:
cfg[key] = str(value)
data = _load_config()
data["dynamic_subtitles"] = cfg
_write_config(data)
return cfg
DEFAULT_PLAIN_SUBTITLE_CONFIG: dict = {
"font": "Helvetica Neue",
"font_size": 82,
"font_color": "1 1 1 1",
"max_words": 7,
"position_y": -820.0,
"uppercase": False,
"keep_punctuation": True,
"text_scale": 2.0,
}
def load_plain_subtitle_config() -> dict:
"""Persisted style for simple editable FCPXML title subtitles."""
cfg = dict(DEFAULT_PLAIN_SUBTITLE_CONFIG)
stored = _load_config().get("plain_subtitles")
if not isinstance(stored, dict):
return cfg
for key in ("position_y", "text_scale"):
if key in stored:
try:
cfg[key] = float(stored[key])
except (TypeError, ValueError):
pass
for key in ("font_size", "max_words"):
if key in stored:
try:
cfg[key] = int(stored[key])
except (TypeError, ValueError):
pass
for key in ("font", "font_color"):
if key in stored and isinstance(stored[key], str) and stored[key]:
cfg[key] = stored[key]
for key in ("uppercase", "keep_punctuation"):
if key in stored:
cfg[key] = bool(stored[key])
cfg["max_words"] = max(1, int(cfg["max_words"]))
return cfg
def save_plain_subtitle_config(**fields) -> dict:
"""Persist simple subtitle style fields. Only given fields change."""
cfg = load_plain_subtitle_config()
for key, value in fields.items():
if key not in DEFAULT_PLAIN_SUBTITLE_CONFIG or value is None:
continue
if isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], bool):
cfg[key] = bool(value)
elif isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], float):
try:
cfg[key] = float(value)
except (TypeError, ValueError):
continue
elif isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], int):
try:
cfg[key] = int(value)
except (TypeError, ValueError):
continue
else:
cfg[key] = str(value)
cfg["max_words"] = max(1, int(cfg["max_words"]))
data = _load_config()
data["plain_subtitles"] = cfg
_write_config(data)
return cfg
# Mirrors the silence thresholds the detection/removal handlers use when no
# argument is passed (server_tools/qc.py). Persisted so the app's slider and
# any later run agree without threading three fields through every call.
DEFAULT_SILENCE_CONFIG: dict = {
# dBFS below which audio counts as silence.
"noise_db": -30.0,
# Seconds a quiet stretch must last before it's a cut candidate.
"min_silence": 0.5,
# Seconds left inside each cut so speech never gets clipped at the edges.
# 0.2s matches the breathing-room convention for phrase-boundary cuts
# (see editar-por-voz/criterios/06-texto-corte-marcador.md) — a silence
# span this tool finds is often the natural breath before a new
# sentence, not just editing slop, and 0.05s shaved that breath down to
# almost nothing (real case: Mastopexia project, the pause before "Com"
# went from 0.567s to 0.1s across the cut, landing the next clip only
# 5ms after the word instead of a natural pause before it).
"padding": 0.2,
}
def load_silence_config() -> dict:
"""The persisted silence-detection thresholds, merged over defaults.
Read by ``detect_media_silence``/``remove_media_silence`` as their
fallback, so the tolerance chosen in the app is what actually runs.
Malformed stored values fall back to the default rather than raising,
matching :func:`load_voice_analysis_config`.
"""
cfg = dict(DEFAULT_SILENCE_CONFIG)
stored = _load_config().get("silence")
if not isinstance(stored, dict):
return cfg
for key in cfg:
if key in stored:
try:
cfg[key] = float(stored[key])
except (TypeError, ValueError):
pass
return cfg
def save_silence_config(
noise_db: float | None = None,
min_silence: float | None = None,
padding: float | None = None,
) -> dict:
"""Persist silence thresholds. Only the given fields change.
Values are clamped to the same ranges the handlers validate against, so
a bad write here can't produce a config the tools would later reject.
"""
cfg = load_silence_config()
if noise_db is not None:
try:
cfg["noise_db"] = max(-120.0, min(0.0, float(noise_db)))
except (TypeError, ValueError):
pass
if min_silence is not None:
try:
cfg["min_silence"] = max(0.01, min(3600.0, float(min_silence)))
except (TypeError, ValueError):
pass
if padding is not None:
try:
cfg["padding"] = max(0.0, min(5.0, float(padding)))
except (TypeError, ValueError):
pass
data = _load_config()
data["silence"] = cfg
_write_config(data)
return cfg
# Last project worked on, so the app reopens where the user left off instead of
# making them pick the folder again every launch. Only paths that still exist
# are handed back — a project on an unmounted volume degrades to "none selected"
# rather than to a dead path the tools would later fail on.
DEFAULT_PROJECT_CONFIG: dict = {
# Folder every generated file (transcript .json, XML, SRT) is written to.
"folder": "",
# The .fcpxml/.fcpxmld that was loaded from it.
"file": "",
}
def load_project_config() -> dict:
"""The persisted last project (folder + file), merged over defaults.
Paths that no longer exist on disk come back empty, matching what the app
shows for "nothing selected". Malformed stored values fall back to the
default rather than raising, same as :func:`load_voice_analysis_config`.
"""
cfg = dict(DEFAULT_PROJECT_CONFIG)
stored = _load_config().get("project")
if not isinstance(stored, dict):
return cfg
for key in cfg:
value = stored.get(key)
if isinstance(value, str) and value and Path(value).exists():
cfg[key] = value
return cfg
def save_project_config(folder: str | None = None, file: str | None = None) -> dict:
"""Persist the last project folder/file. Only the given fields change.
Passing an empty string clears a field (the app does this when the user
deselects), while ``None`` leaves it untouched.
"""
cfg = load_project_config()
for key, value in (("folder", folder), ("file", file)):
if value is None:
continue
cfg[key] = str(Path(value).expanduser()) if str(value).strip() else ""
data = _load_config()
data["project"] = cfg
_write_config(data)
return cfg