feat: etapa 5 do assistente — revisão de ênfases com timeline
Transforma a etapa "colar decisões" numa tela de lapidação: a sugestão da IA chega carregada e o editor afina frase a frase o que é ênfase e o que fica fora. Essa marcação é o norte da etapa 6 — só as frases com ênfase recebem zoom e legenda dinâmica; as demais ficam com legenda comum. O campo de colar o JSON sobe para a etapa 4, então a numeração das etapas não muda e a etapa 6 segue intacta. Backend (fcpxml/phrase_review.py): - build_phrase_review funde o _voice_timeline.json com as actions da IA - trim por frase que anda em fronteira de palavra; corte parcial da IA chega como trim em vez de ser arredondado fora - phrase_review_to_actions volta a cuts/zooms + emphasis_spans - merge_saved_decisions reaplica só as decisões salvas sobre uma revisão remontada da análise atual, para reprocessar a voz não ficar mascarado - resolve_source acha a mídia: o voice timeline guarda só o nome do arquivo App (SwiftUI): - layout de sala de edição: preview em cima, inspector à direita, timeline atravessando embaixo com seis trilhas rotuladas - preview enquadra no formato de entrega lido do .fcpxml (fonte horizontal, projeto vertical), com alternância para a mídia original - reprodução pula os trechos removidos e para no fim do trecho - zoom manual por trecho marcado, sem guardar escala: a forma vem das configurações de Análise de Voz no render - emoção da fala exposta por frase Correções encontradas no caminho: - VideoPlayer (AVKit) aborta em runtime no app compilado por swiftc; trocado por AVPlayerLayer (ver Engine/docs/05_EXPERIENCIAS.md #22) - teste que ainda afirmava o default zoom scale=1.3 removido do parser (#21) Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
e7748c2c58
commit
1bebee4359
@@ -56,12 +56,20 @@ VALUE_SCALES = {
|
||||
"rate_delta": "0-1, how much the local speaking rate departs from the average",
|
||||
"pause_before": "seconds of silence immediately before the word",
|
||||
"emphasis": "0-1 combined index; high values are punch-in/highlight candidates",
|
||||
"emotion": "heuristic label from delivery: neutral, excited, tense, calm, reflective",
|
||||
"emotion_confidence": "0-1 confidence in the heuristic emotion label",
|
||||
"arousal": "0-1 vocal activation from energy/rate/pitch movement",
|
||||
"valence": "0-1 rough positive tone; lower values suggest tension/weight",
|
||||
},
|
||||
"segment": {
|
||||
"gap_before": "seconds of silence before this line",
|
||||
"take_boundary": "true when the gap is long enough that the take likely restarted here",
|
||||
"avg_energy": "0-1 mean loudness across the line",
|
||||
"peak_emphasis": "0-1 highest emphasis of any word in the line",
|
||||
"emotion": "dominant delivery emotion across the line",
|
||||
"emotion_confidence": "0-1 confidence in the dominant segment emotion",
|
||||
"arousal": "0-1 mean vocal activation across the line",
|
||||
"valence": "0-1 mean rough positive tone across the line",
|
||||
},
|
||||
}
|
||||
|
||||
@@ -92,11 +100,74 @@ def _round_word(word: dict) -> dict:
|
||||
"rate_delta": round(word.get("rate_delta", 0.0), 3),
|
||||
"pause_before": round(word.get("pause_before", 0.0), 3),
|
||||
"emphasis": round(word.get("emphasis", 0.0), 3),
|
||||
"emotion": word.get("emotion", "neutral"),
|
||||
"emotion_confidence": round(word.get("emotion_confidence", 0.0), 3),
|
||||
"arousal": round(word.get("arousal", 0.0), 3),
|
||||
"valence": round(word.get("valence", 0.5), 3),
|
||||
"energy_raw": word.get("energy"),
|
||||
"pitch_hz": word.get("pitch_hz"),
|
||||
}
|
||||
|
||||
|
||||
def _emotion_for_word(word: dict, enabled: bool, sensitivity: float) -> dict:
|
||||
"""Classify delivery emotion from normalized acoustic features.
|
||||
|
||||
This is deliberately a local heuristic rather than a claimed clinical
|
||||
emotion model. It gives the editor a useful signal about delivery shape
|
||||
while degrading predictably when acoustic extraction is unavailable.
|
||||
"""
|
||||
if not enabled:
|
||||
return {
|
||||
"emotion": "neutral",
|
||||
"emotion_confidence": 0.0,
|
||||
"arousal": 0.0,
|
||||
"valence": 0.5,
|
||||
}
|
||||
|
||||
energy = float(word.get("energy_norm", 0.0))
|
||||
pitch = float(word.get("pitch_delta", 0.0))
|
||||
rate = float(word.get("rate_delta", 0.0))
|
||||
pause = min(float(word.get("pause_before", 0.0)) / 2.0, 1.0)
|
||||
emphasis = float(word.get("emphasis", 0.0))
|
||||
|
||||
arousal = max(0.0, min(1.0, energy * 0.45 + pitch * 0.25 + rate * 0.20 + emphasis * 0.10))
|
||||
valence = max(0.0, min(1.0, 0.55 + energy * 0.15 - pause * 0.20 - rate * 0.10))
|
||||
|
||||
if arousal >= 0.68 and valence >= 0.50:
|
||||
label = "excited"
|
||||
confidence = arousal
|
||||
elif arousal >= 0.58 and valence < 0.50:
|
||||
label = "tense"
|
||||
confidence = max(arousal, 1.0 - valence)
|
||||
elif arousal <= 0.28 and pause >= 0.25:
|
||||
label = "reflective"
|
||||
confidence = max(1.0 - arousal, pause)
|
||||
elif arousal <= 0.35:
|
||||
label = "calm"
|
||||
confidence = 1.0 - arousal
|
||||
else:
|
||||
label = "neutral"
|
||||
confidence = 1.0 - abs(arousal - 0.5) * 2.0
|
||||
|
||||
confidence = max(0.0, min(1.0, confidence))
|
||||
if confidence < sensitivity:
|
||||
label = "neutral"
|
||||
return {
|
||||
"emotion": label,
|
||||
"emotion_confidence": confidence,
|
||||
"arousal": arousal,
|
||||
"valence": valence,
|
||||
}
|
||||
|
||||
|
||||
def annotate_emotions(words: Sequence[dict], enabled: bool, sensitivity: float) -> List[dict]:
|
||||
"""Attach heuristic emotion labels to enriched word rows."""
|
||||
return [
|
||||
{**w, **_emotion_for_word(w, enabled, sensitivity)}
|
||||
for w in words
|
||||
]
|
||||
|
||||
|
||||
def enrich_words(
|
||||
words: Sequence[dict],
|
||||
pitch_track: Optional[Sequence] = None,
|
||||
@@ -166,6 +237,13 @@ def _segment_rows(segments: Sequence[dict], words: Sequence[dict]) -> List[dict]
|
||||
in_seg = [w for w in words if start <= float(w.get("start", 0.0)) < end]
|
||||
energies = [w["energy_norm"] for w in in_seg]
|
||||
emphases = [w["emphasis"] for w in in_seg]
|
||||
arousals = [w.get("arousal", 0.0) for w in in_seg]
|
||||
valences = [w.get("valence", 0.5) for w in in_seg]
|
||||
emotions = [w.get("emotion", "neutral") for w in in_seg]
|
||||
dominant = max(set(emotions), key=emotions.count) if emotions else "neutral"
|
||||
emotion_confidences = [
|
||||
w.get("emotion_confidence", 0.0) for w in in_seg if w.get("emotion") == dominant
|
||||
]
|
||||
gap = max(0.0, start - previous_end)
|
||||
rows.append(
|
||||
{
|
||||
@@ -181,6 +259,13 @@ def _segment_rows(segments: Sequence[dict], words: Sequence[dict]) -> List[dict]
|
||||
"take_boundary": gap >= TAKE_BOUNDARY_GAP,
|
||||
"avg_energy": round(sum(energies) / len(energies), 3) if energies else 0.0,
|
||||
"peak_emphasis": round(max(emphases), 3) if emphases else 0.0,
|
||||
"emotion": dominant,
|
||||
"emotion_confidence": (
|
||||
round(sum(emotion_confidences) / len(emotion_confidences), 3)
|
||||
if emotion_confidences else 0.0
|
||||
),
|
||||
"arousal": round(sum(arousals) / len(arousals), 3) if arousals else 0.0,
|
||||
"valence": round(sum(valences) / len(valences), 3) if valences else 0.5,
|
||||
"words": [_round_word(w) for w in in_seg],
|
||||
}
|
||||
)
|
||||
@@ -425,6 +510,8 @@ def build_voice_timeline(
|
||||
weights: EmphasisWeights = EmphasisWeights(),
|
||||
peak_percentile: float = 0.02,
|
||||
emphasis_floor: float = 0.25,
|
||||
emotion_enabled: bool = False,
|
||||
emotion_sensitivity: float = 0.5,
|
||||
progress_cb: Optional[Callable[[float, str], None]] = None,
|
||||
) -> dict:
|
||||
"""Build the consolidated voice timeline for one media file.
|
||||
@@ -445,6 +532,7 @@ def build_voice_timeline(
|
||||
|
||||
report(0.5, "Calculando ênfase...")
|
||||
words = enrich_words(transcript.get("words", []), pitch_track, energy_track, weights)
|
||||
words = annotate_emotions(words, emotion_enabled, emotion_sensitivity)
|
||||
|
||||
report(0.7, "Identificando participantes...")
|
||||
tracks = diarize(media_path, hf_token, num_speakers) if hf_token else None
|
||||
@@ -465,6 +553,7 @@ def build_voice_timeline(
|
||||
"transcript": bool(transcript.get("words")),
|
||||
"acoustics": pitch_track is not None or energy_track is not None,
|
||||
"speakers": tracks is not None,
|
||||
"emotion": bool(emotion_enabled),
|
||||
},
|
||||
"scales": VALUE_SCALES,
|
||||
"summary": _summary(
|
||||
|
||||
Reference in New Issue
Block a user