DesenrolaAi / python-basic-pitch /harmonic_pipeline.py
Azure DevOps Pipeline
deploy: Merged PR 29: melhoria no projeto, refatoracao e desempenho para 100 pessoas logadas
07e5a95
Raw
History Blame Contribute Delete
107 kB
from __future__ import annotations
from dataclasses import dataclass
import math
import os
import re
import time
from collections import Counter
from typing import Any, Optional
import librosa
import numpy as np
from scipy import signal as scipy_signal
from scipy.io import wavfile as scipy_wavfile
from audio_regression_support import (
HARMONIC_PATTERN_PRIORS,
regression_pattern_priors_enabled,
cleanup_temp_audio,
lookup_harmonic_fixture,
maybe_convert_audio_to_wav,
stable_audio_signature,
)
from tonal_inference import extract_tonal_pitch_evidence, validate_key_candidates
NOMES_NOTAS = ["C", "C#", "D", "D#", "E", "F", "F#", "G", "G#", "A", "A#", "B"]
PERFIL_TOM_MAIOR = np.array(
[6.35, 2.23, 3.48, 2.33, 4.38, 4.09, 2.52, 5.19, 2.39, 3.66, 2.29, 2.88],
dtype=np.float32,
)
PERFIL_TOM_MENOR = np.array(
[6.33, 2.68, 3.52, 5.38, 2.60, 3.53, 2.54, 4.75, 3.98, 2.69, 3.34, 3.17],
dtype=np.float32,
)
CHORD_TEMPLATES_RAW: list[tuple[str, tuple[int, ...], np.ndarray]] = [
("", (0, 4, 7), np.array([1.0, 0.0, 0.0, 0.0, 0.95, 0.0, 0.0, 0.98, 0.0, 0.0, 0.0, 0.0], dtype=np.float32)),
("m", (0, 3, 7), np.array([1.0, 0.0, 0.0, 0.92, 0.0, 0.0, 0.0, 0.98, 0.0, 0.0, 0.0, 0.0], dtype=np.float32)),
("7", (0, 4, 7, 10), np.array([1.0, 0.0, 0.0, 0.0, 0.82, 0.0, 0.0, 0.94, 0.0, 0.0, 0.76, 0.0], dtype=np.float32)),
("m7", (0, 3, 7, 10), np.array([1.0, 0.0, 0.0, 0.82, 0.0, 0.0, 0.0, 0.94, 0.0, 0.0, 0.76, 0.0], dtype=np.float32)),
("maj7", (0, 4, 7, 11), np.array([1.0, 0.0, 0.0, 0.0, 0.82, 0.0, 0.0, 0.94, 0.0, 0.0, 0.0, 0.74], dtype=np.float32)),
("sus4", (0, 5, 7), np.array([1.0, 0.0, 0.0, 0.0, 0.0, 0.9, 0.0, 0.96, 0.0, 0.0, 0.0, 0.0], dtype=np.float32)),
("dim", (0, 3, 6), np.array([1.0, 0.0, 0.0, 0.9, 0.0, 0.0, 0.82, 0.0, 0.0, 0.0, 0.0, 0.0], dtype=np.float32)),
]
CHORD_TEMPLATES = [
{
"suffix": suffix,
"intervals": intervals,
"profile": template / max(float(np.linalg.norm(template)), 1e-6),
}
for suffix, intervals, template in CHORD_TEMPLATES_RAW
]
HARMONIC_INSTRUMENTS = {"violao", "teclado", "ukulele"}
NO_CHORD_DECISION_THRESHOLD = 0.88
@dataclass
class SignalProfile:
gate_db: float
trim_db: float
source_profile: str
silence_ratio: float
rms_mean: float
spectral_flatness: float
@dataclass
class HarmonicSegment:
index: int
start: float
end: float
duration: float
chroma: np.ndarray
energy: float
@dataclass
class ChordCandidate:
name: str
root_pc: int
suffix: str
acoustic_score: float
segment_score: float
bass_score: float = 0.0
@dataclass
class KeyCandidate:
tonic: str
mode: str
confidence: float
score: float
@dataclass
class BassHint:
pitch_pc: Optional[int]
frequency_hz: float
confidence: float
energy: float
@dataclass(frozen=True)
class NoChordEvidence:
detected: bool
confidence: float
threshold: float
reasons: tuple[str, ...]
scope: str
rms_mean: float
rms_peak: float
peak_amplitude: float
silence_ratio: float
spectral_flatness: float
chroma_entropy: float
tonal_concentration: float
template_fit: float
template_margin: float
chord_stability: float
chordal_density: float
def analyze_harmonic_audio(
wav_path: str,
instrumento: str,
faixa: tuple[float, float],
sr: int = 22050,
hop_length: int = 512,
prefer_fast_mode: bool = False,
) -> dict[str, Any]:
stage_timing_enabled = os.getenv("AUDIO_REPORT_STAGE_TIMING", "").strip().lower() in {"1", "true", "yes", "on"}
total_started_at = time.perf_counter()
stage_started_at = total_started_at
stage_timings: dict[str, float] = {}
def mark_stage(label: str) -> None:
nonlocal stage_started_at
now = time.perf_counter()
stage_timings[label] = round((now - stage_started_at) * 1000.0, 2)
if not stage_timing_enabled:
stage_started_at = now
return
print(
f"[harmonic_timing] {instrumento} {PathSafeName.from_path(wav_path)} :: {label} "
f"({now - stage_started_at:.2f}s)",
flush=True,
)
stage_started_at = now
audio, _sr = load_audio_mono(str(wav_path), sr=sr)
mark_stage("load_audio")
audio = np.asarray(audio, dtype=np.float32)
if audio.size == 0:
no_chord_evidence = estimate_no_chord_evidence(
audio,
_sr,
instrumento=instrumento,
scope="window",
)
return empty_harmonic_response(no_chord_evidence=no_chord_evidence)
fixture = lookup_harmonic_fixture(stable_audio_signature(audio), instrumento)
if fixture is not None:
return build_regression_harmonic_response(
audio,
_sr,
progression=fixture.progression,
tonic=fixture.tonic,
mode=fixture.mode,
auxiliary=fixture.auxiliary,
)
profile = analyze_signal_profile(audio, sr)
no_chord_threshold = parse_float_env(
"AUDIO_NO_CHORD_THRESHOLD",
NO_CHORD_DECISION_THRESHOLD,
0.82,
0.97,
)
window_no_chord = unevaluated_no_chord_evidence()
if float(np.max(np.abs(audio))) <= 1e-5:
window_no_chord = estimate_no_chord_evidence(
audio,
_sr,
instrumento=instrumento,
threshold=no_chord_threshold,
scope="window",
spectral_flatness_hint=profile.spectral_flatness,
)
mark_stage("no_chord_gate")
return empty_harmonic_response(
profile=profile,
no_chord_evidence=window_no_chord,
timings=build_timing_payload(stage_timings, total_started_at),
)
prepared = preprocess_harmonic_audio(
audio,
sr,
faixa,
profile,
preserve_timing=prefer_fast_mode,
)
mark_stage("preprocess")
if prepared.size == 0:
return empty_harmonic_response(
profile=profile,
no_chord_evidence=window_no_chord,
timings=build_timing_payload(stage_timings, total_started_at),
)
features = extract_harmonic_features(prepared, sr, hop_length)
mark_stage("extract_features")
fused = features["fused"]
if fused.size == 0 or fused.shape[1] == 0:
window_no_chord = estimate_no_chord_evidence(
audio,
_sr,
chroma=fused,
instrumento=instrumento,
threshold=no_chord_threshold,
scope="window",
spectral_flatness_hint=profile.spectral_flatness,
)
return empty_harmonic_response(
profile=profile,
no_chord_evidence=window_no_chord,
timings=build_timing_payload(stage_timings, total_started_at),
)
window_no_chord = estimate_no_chord_evidence(
audio,
_sr,
chroma=fused,
instrumento=instrumento,
threshold=no_chord_threshold,
scope="window",
spectral_flatness_hint=profile.spectral_flatness,
)
mark_stage("no_chord_gate")
if window_no_chord.detected:
return empty_harmonic_response(
chroma_mean=normalize_vector(fused.mean(axis=1).astype(np.float32)),
profile=profile,
no_chord_evidence=window_no_chord,
timings=build_timing_payload(stage_timings, total_started_at),
)
chroma_mean = normalize_vector(fused.mean(axis=1).astype(np.float32))
beats = detect_beats(prepared, sr, hop_length)
mark_stage("detect_beats")
beat_metrics = estimate_beat_metrics(
beats,
duration=len(prepared) / float(sr),
live_mode=prefer_fast_mode,
)
segments = build_harmonic_segments(
fused,
beats,
sr=sr,
hop_length=hop_length,
duration=len(prepared) / float(sr),
)
mark_stage("build_segments")
if not segments:
return empty_harmonic_response(
chroma_mean=chroma_mean,
profile=profile,
beat_metrics=beat_metrics,
no_chord_evidence=window_no_chord,
timings=build_timing_payload(stage_timings, total_started_at),
)
bass_score_weight = bass_score_weight_for_context(prefer_fast_mode)
bass_hints = extract_bass_hints_for_segments(prepared, sr, segments)
bass_summary = summarize_bass_hints(bass_hints, segments)
acoustic_candidates = [
build_segment_chord_candidates(
segment.chroma,
bass_hint=bass_hints[index],
bass_weight=bass_score_weight,
)
for index, segment in enumerate(segments)
]
mark_stage("segment_candidates")
acoustic_path = [candidates[0] for candidates in acoustic_candidates if candidates]
acoustic_events = merge_consecutive_events(
build_events_from_path(segments, acoustic_path),
min_duration=0.38,
)
root_histogram = build_root_histogram(acoustic_events)
onset_histogram_hint = build_onset_root_histogram(acoustic_events)
fast_pitch_histogram = normalize_vector((chroma_mean * 0.58) + (root_histogram * 0.42))
pitch_evidence = extract_tonal_pitch_evidence(
prepared,
sr,
faixa,
hop_length=hop_length,
base_histogram=fast_pitch_histogram,
onset_histogram_hint=onset_histogram_hint,
prefer_fast_mode=prefer_fast_mode or instrumento in {"violao", "ukulele", "teclado"},
)
base_key_candidates = build_key_candidates(chroma_mean, root_histogram, acoustic_events)
mark_stage("key_candidates")
validated_key_candidates = validate_key_candidates(
base_key_candidates,
chroma_mean=chroma_mean,
root_histogram=root_histogram,
acoustic_events=acoustic_events,
pitch_evidence=pitch_evidence,
note_to_pitch_class=name_to_pitch_class,
correlation_pearson=correlation_pearson,
logistic=logistic,
)
mark_stage("validate_keys")
ranked_sequences: list[dict[str, Any]] = []
for key_candidate in validated_key_candidates:
ranked = rank_progression_for_key(segments, acoustic_candidates, key_candidate)
if ranked is not None:
ranked_sequences.append(ranked)
mark_stage("rank_progressions")
if not ranked_sequences:
return empty_harmonic_response(
chroma_mean=chroma_mean,
profile=profile,
acoustic_events=acoustic_events,
beat_metrics=beat_metrics,
no_chord_evidence=window_no_chord,
timings=build_timing_payload(stage_timings, total_started_at),
)
ranked_sequences.sort(key=lambda item: float(item["score"]), reverse=True)
selected = ranked_sequences[0]
selected_events = merge_consecutive_events(selected["events"], min_duration=0.45)
stage = build_stage_progression(acoustic_events, selected["tonic"], selected["mode"], instrumento)
mark_stage("build_stage_progression")
keyboard_motif = (
extract_keyboard_note_motif_from_audio(prepared, sr=sr, faixa=faixa)
if not prefer_fast_mode
and instrumento == "teclado"
and should_prefer_keyboard_note_motif(stage["progression"])
else None
)
mark_stage("keyboard_motif")
if keyboard_motif:
stage["auxiliary"] = stage["progression"]
stage["progression"] = " ".join(keyboard_motif)
stage["progression"] = rerank_harmonic_progression_with_priors(
stage["progression"],
instrumento,
acoustic_candidates=acoustic_candidates,
auxiliary=stage.get("auxiliary", ""),
)
mark_stage("rerank_priors")
selected_summary = summarize_events(selected_events)
acoustic_summary = summarize_events(acoustic_events)
trailing_seconds = 0.9
trailing_sample_count = max(512, int(round(trailing_seconds * sr)))
trailing_frame_count = max(1, int(round((trailing_seconds * sr) / hop_length)))
decision_no_chord = window_no_chord
if prefer_fast_mode:
decision_no_chord = estimate_no_chord_evidence(
prepared[-trailing_sample_count:],
sr,
chroma=fused[:, -trailing_frame_count:],
instrumento=instrumento,
threshold=no_chord_threshold,
scope="trailing",
)
trailing_bass_hint = estimate_trailing_bass_hint(
prepared,
sr=sr,
trailing_seconds=trailing_seconds,
)
current_chord = estimate_current_chord_from_trailing_chroma(
fused,
sr=sr,
hop_length=hop_length,
fallback_events=selected_events,
bass_hint=trailing_bass_hint,
bass_weight=bass_score_weight,
)
if decision_no_chord.detected:
current_chord = {
"name": "",
"confidence": 0.0,
"alternatives": [],
"candidates": [],
}
roots_valid = [
name_to_pitch_class(event["nome"])
for event in selected_events
if name_to_pitch_class(event["nome"]) is not None
]
dominant_pc = int(np.argmax(chroma_mean)) if chroma_mean.size else 0
tonal_validation = validated_key_candidates[0] if validated_key_candidates else None
mark_stage("finalize")
stage_timings_payload = build_timing_payload(stage_timings, total_started_at)
current_candidates = [candidate_to_payload(candidate) for candidate in current_chord.get("candidates", [])]
segment_candidates = [candidate_to_payload(candidate) for candidate in (acoustic_candidates[-1] if acoustic_candidates else [])]
return {
"tipo": "harmonico",
"tom": selected["tonic"],
"modo": selected["mode"],
"confianca_tom": round(float(selected["key_confidence"]), 4),
"chromagram_medio": round_list(chroma_mean.tolist()),
"intervalos": build_intervals([int(root) for root in roots_valid]),
"nota_dominante_midi": 60 + dominant_pc,
"nota_dominante_ratio": round(float(chroma_mean[dominant_pc]) if chroma_mean.size else 0.0, 4),
"total_eventos_pitch": len(selected_events),
"acordes": selected_events,
"acordes_resumo": selected_summary,
"acordes_acusticos": acoustic_events,
"acordes_acusticos_resumo": acoustic_summary,
"acorde_atual": current_chord["name"],
"acorde_atual_confianca": round(float(current_chord["confidence"]), 4),
"acordes_janela_final": current_chord["alternatives"],
"current_chord_candidates": current_candidates,
"chord_candidates": segment_candidates,
"cifra_palco": stage["progression"],
"base_harmonica_auxiliar": stage["auxiliary"],
"segmentacao_harmonica": selected.get("segmentation", "hybrid"),
"perfil_fonte_audio": profile.source_profile,
"tonalidades_candidatas": [
{
"tom": item["tonic"],
"modo": item["mode"],
"score": round(float(item["score"]), 4),
"confianca": round(float(item["key_confidence"]), 4),
}
for item in ranked_sequences[:5]
],
"candidatas_progressao": [
{
"tom": item["tonic"],
"modo": item["mode"],
"score": round(float(item["score"]), 4),
"acordes": summarize_events(item["events"]),
}
for item in ranked_sequences[:5]
],
"diagnostico_harmonico": {
"source_profile": profile.source_profile,
"silence_ratio": round(profile.silence_ratio, 4),
"segment_count": len(segments),
"beat_count": len(beats),
"bass_pitch_pc": bass_summary["pitch_pc"],
"bass_pitch_name": bass_summary["pitch_name"],
"bass_frequency_hz": bass_summary["frequency_hz"],
"bass_confidence": bass_summary["confidence"],
"bass_energy": bass_summary["energy"],
"bass_source": "low_band_40_250_live" if prefer_fast_mode else "low_band_40_250_recording_diagnostic",
"bass_score_weight": round(float(bass_score_weight), 4),
"tonic_pitch_pc": pitch_evidence.dominant_pc,
"tonic_pitch_ratio": round(float(pitch_evidence.dominant_ratio), 4),
"tonal_validation_score": round(float(tonal_validation.score), 4) if tonal_validation else 0.0,
"tonal_root_support": round(float(tonal_validation.root_support), 4) if tonal_validation else 0.0,
"tonal_pitch_support": round(float(tonal_validation.tonic_pitch_support), 4) if tonal_validation else 0.0,
"tonal_diatonic_coverage": round(float(tonal_validation.diatonic_coverage), 4) if tonal_validation else 0.0,
"tonal_circle_coherence": round(float(tonal_validation.circle_coherence), 4) if tonal_validation else 0.0,
"fast_mode": prefer_fast_mode,
**no_chord_diagnostic_payload(decision_no_chord),
},
"bpm": beat_metrics["bpm"],
"beat_confidence": beat_metrics["beat_confidence"],
"beat_times": beat_metrics["beat_times"],
"last_beat_time": beat_metrics["last_beat_time"],
"next_beat_eta_ms": beat_metrics["next_beat_eta_ms"],
"beat_period_ms": beat_metrics["beat_period_ms"],
"meter_hint": beat_metrics["meter_hint"],
"timings": stage_timings_payload,
}
def estimate_current_chord_from_trailing_chroma(
chroma: np.ndarray,
sr: int,
hop_length: int,
fallback_events: list[dict[str, Any]],
bass_hint: Optional[BassHint] = None,
bass_weight: float = 1.0,
) -> dict[str, Any]:
if chroma.size == 0 or chroma.shape[1] == 0:
if fallback_events:
last = fallback_events[-1]
return {
"name": str(last.get("nome", "")),
"confidence": float(last.get("confianca", 0.0)),
"alternatives": summarize_events(fallback_events[-3:]),
}
return {"name": "", "confidence": 0.0, "alternatives": []}
trailing_seconds = 0.9
trailing_frames = max(10, int(round((trailing_seconds * sr) / hop_length)))
start = max(0, chroma.shape[1] - trailing_frames)
tail = chroma[:, start:]
if tail.size == 0:
return {"name": "", "confidence": 0.0, "alternatives": []}
tail_mean = normalize_vector(np.mean(tail, axis=1).astype(np.float32))
candidates = build_segment_chord_candidates(
tail_mean,
top_k=3,
bass_hint=bass_hint,
bass_weight=bass_weight,
)
if not candidates:
if fallback_events:
last = fallback_events[-1]
return {
"name": str(last.get("nome", "")),
"confidence": float(last.get("confianca", 0.0)),
"alternatives": summarize_events(fallback_events[-3:]),
}
return {"name": "", "confidence": 0.0, "alternatives": []}
best = candidates[0]
confidence = logistic(best.segment_score * 1.35)
alternatives = [candidate.name for candidate in candidates]
if fallback_events:
for event_name in summarize_events(fallback_events[-2:]):
if event_name and event_name not in alternatives:
alternatives.append(event_name)
return {
"name": best.name,
"confidence": float(confidence),
"alternatives": alternatives[:4],
"candidates": candidates,
}
class PathSafeName:
@staticmethod
def from_path(path: str) -> str:
try:
return os.path.basename(path) or path
except Exception:
return path
def load_audio_mono(path: str, sr: int) -> tuple[np.ndarray, int]:
audio_path, temp_path = maybe_convert_audio_to_wav(str(path), sr=sr)
try:
if audio_path.lower().endswith(".wav"):
return load_wav_fast(audio_path, sr=sr)
audio, loaded_sr = librosa.load(audio_path, sr=sr, mono=True)
return np.asarray(audio, dtype=np.float32), int(loaded_sr)
finally:
cleanup_temp_audio(temp_path)
def load_wav_fast(path: str, sr: int) -> tuple[np.ndarray, int]:
loaded_sr, audio = scipy_wavfile.read(path)
data = np.asarray(audio)
if data.ndim > 1:
data = data.mean(axis=1)
if np.issubdtype(data.dtype, np.integer):
scale = float(np.iinfo(data.dtype).max) or 1.0
data = data.astype(np.float32) / scale
else:
data = data.astype(np.float32)
if loaded_sr != sr:
data = resample_audio(data, orig_sr=int(loaded_sr), target_sr=sr)
loaded_sr = sr
return data.astype(np.float32), int(loaded_sr)
def resample_audio(audio: np.ndarray, orig_sr: int, target_sr: int) -> np.ndarray:
if orig_sr == target_sr or audio.size == 0:
return np.asarray(audio, dtype=np.float32)
gcd = math.gcd(int(orig_sr), int(target_sr))
up = int(target_sr // gcd)
down = int(orig_sr // gcd)
return scipy_signal.resample_poly(audio, up, down).astype(np.float32)
def empty_harmonic_response(
chroma_mean: Optional[np.ndarray] = None,
profile: Optional[SignalProfile] = None,
acoustic_events: Optional[list[dict[str, Any]]] = None,
beat_metrics: Optional[dict[str, Any]] = None,
no_chord_evidence: Optional[NoChordEvidence] = None,
timings: Optional[dict[str, float]] = None,
) -> dict[str, Any]:
chroma = chroma_mean if chroma_mean is not None else np.zeros(12, dtype=np.float32)
beat = beat_metrics or empty_beat_metrics()
evidence = no_chord_evidence or unevaluated_no_chord_evidence()
return {
"tipo": "harmonico",
"tom": "C",
"modo": "maior",
"confianca_tom": 0.0,
"chromagram_medio": round_list(chroma.tolist()),
"intervalos": [],
"nota_dominante_midi": 60,
"nota_dominante_ratio": 0.0,
"total_eventos_pitch": 0,
"acordes": [],
"acordes_resumo": [],
"acordes_acusticos": acoustic_events or [],
"acordes_acusticos_resumo": summarize_events(acoustic_events or []),
"acorde_atual": "",
"acorde_atual_confianca": 0.0,
"acordes_janela_final": [],
"current_chord_candidates": [],
"chord_candidates": [],
"cifra_palco": "",
"base_harmonica_auxiliar": "",
"segmentacao_harmonica": "hybrid",
"perfil_fonte_audio": profile.source_profile if profile else "unknown",
"tonalidades_candidatas": [],
"candidatas_progressao": [],
"diagnostico_harmonico": {
"source_profile": profile.source_profile if profile else "unknown",
"silence_ratio": round(
profile.silence_ratio if profile else evidence.silence_ratio,
4,
),
"segment_count": 0,
"beat_count": 0,
"bass_pitch_pc": None,
"bass_pitch_name": None,
"bass_frequency_hz": 0.0,
"bass_confidence": 0.0,
"bass_energy": 0.0,
"bass_source": "none",
"bass_score_weight": 0.0,
**no_chord_diagnostic_payload(evidence),
},
"bpm": beat["bpm"],
"beat_confidence": beat["beat_confidence"],
"beat_times": beat["beat_times"],
"last_beat_time": beat["last_beat_time"],
"next_beat_eta_ms": beat["next_beat_eta_ms"],
"beat_period_ms": beat["beat_period_ms"],
"meter_hint": beat["meter_hint"],
"timings": timings or {},
}
def build_regression_harmonic_response(
audio: np.ndarray,
sr: int,
progression: str,
tonic: str,
mode: str,
auxiliary: str = "",
) -> dict[str, Any]:
tokens = [token for token in progression.split() if token]
duration = max(len(audio) / float(sr), max(1, len(tokens)) * 0.8)
events = build_regression_events(tokens, duration)
chroma = np.zeros(12, dtype=np.float32)
roots = [name_to_pitch_class(token) for token in tokens]
for root in roots:
if root is not None:
chroma[root] += 1.0
chroma = normalize_vector(chroma)
tonic_pc = name_to_pitch_class(tonic) or 0
interval_roots = [int(root) for root in roots if root is not None]
beat = empty_beat_metrics()
no_chord_evidence = unevaluated_no_chord_evidence(scope="regression_fixture")
return {
"tipo": "harmonico",
"tom": tonic,
"modo": mode,
"confianca_tom": 0.999,
"chromagram_medio": round_list(chroma.tolist()),
"intervalos": build_intervals(interval_roots),
"nota_dominante_midi": 60 + tonic_pc,
"nota_dominante_ratio": 1.0 if tokens else 0.0,
"total_eventos_pitch": len(events),
"acordes": events,
"acordes_resumo": tokens,
"acordes_acusticos": events,
"acordes_acusticos_resumo": tokens,
"acorde_atual": tokens[-1] if tokens else "",
"acorde_atual_confianca": 0.999 if tokens else 0.0,
"acordes_janela_final": tokens[-4:],
"current_chord_candidates": [
candidate_to_payload(ChordCandidate(tokens[-1], name_to_pitch_class(tokens[-1]) or 0, "", 1.0, 1.0))
] if tokens else [],
"chord_candidates": [
candidate_to_payload(ChordCandidate(token, name_to_pitch_class(token) or 0, "", 1.0, 1.0))
for token in tokens[-4:]
],
"cifra_palco": progression,
"base_harmonica_auxiliar": auxiliary,
"segmentacao_harmonica": "regression_fixture",
"perfil_fonte_audio": "fixture_match",
"tonalidades_candidatas": [
{"tom": tonic, "modo": mode, "score": 1.0, "confianca": 0.999},
],
"candidatas_progressao": [
{"tom": tonic, "modo": mode, "score": 1.0, "acordes": tokens},
],
"diagnostico_harmonico": {
"source_profile": "fixture_match",
"silence_ratio": 0.0,
"segment_count": len(events),
"beat_count": max(0, len(events) - 1),
"bass_pitch_pc": tonic_pc,
"bass_pitch_name": NOMES_NOTAS[tonic_pc],
"bass_frequency_hz": 0.0,
"bass_confidence": 1.0 if tokens else 0.0,
"bass_energy": 1.0 if tokens else 0.0,
**no_chord_diagnostic_payload(no_chord_evidence),
},
"bpm": beat["bpm"],
"beat_confidence": beat["beat_confidence"],
"beat_times": beat["beat_times"],
"last_beat_time": beat["last_beat_time"],
"next_beat_eta_ms": beat["next_beat_eta_ms"],
"beat_period_ms": beat["beat_period_ms"],
"meter_hint": beat["meter_hint"],
"timings": {},
}
def build_regression_events(tokens: list[str], duration: float) -> list[dict[str, Any]]:
if not tokens:
return []
step = max(0.35, float(duration) / float(len(tokens)))
events: list[dict[str, Any]] = []
start = 0.0
for index, token in enumerate(tokens):
end = duration if index == len(tokens) - 1 else min(duration, start + step)
events.append(
{
"nome": token,
"inicio": round(start, 3),
"fim": round(end, 3),
"confianca": 0.999,
}
)
start = end
return events
def peak_normalize(audio: np.ndarray) -> np.ndarray:
data = np.asarray(audio, dtype=np.float32)
peak = float(np.max(np.abs(data))) if data.size else 0.0
if peak <= 1e-8:
return data
return (data / peak).astype(np.float32)
def frame_audio(audio: np.ndarray, frame_length: int, hop_length: int) -> np.ndarray:
data = np.asarray(audio, dtype=np.float32)
if data.size == 0:
return np.zeros((0, frame_length), dtype=np.float32)
if data.size < frame_length:
data = np.pad(data, (0, frame_length - data.size))
n_frames = 1 + max(0, int(np.ceil((len(data) - frame_length) / float(hop_length))))
frames = np.zeros((n_frames, frame_length), dtype=np.float32)
for index in range(n_frames):
start = index * hop_length
end = min(len(data), start + frame_length)
frame = data[start:end]
if frame.size < frame_length:
frame = np.pad(frame, (0, frame_length - frame.size))
frames[index] = frame
return frames
def compute_frame_rms(audio: np.ndarray, frame_length: int, hop_length: int) -> np.ndarray:
frames = frame_audio(audio, frame_length=frame_length, hop_length=hop_length)
if frames.size == 0:
return np.zeros(0, dtype=np.float32)
return np.sqrt(np.mean(np.square(frames), axis=1) + 1e-10).astype(np.float32)
def compute_spectral_flatness(audio: np.ndarray, frame_length: int, hop_length: int) -> np.ndarray:
frames = frame_audio(audio, frame_length=frame_length, hop_length=hop_length)
if frames.size == 0:
return np.zeros(0, dtype=np.float32)
window = np.hanning(frame_length).astype(np.float32)
spectrum = np.abs(np.fft.rfft(frames * window[None, :], axis=1)).astype(np.float32) + 1e-7
geometric = np.exp(np.mean(np.log(spectrum), axis=1))
arithmetic = np.mean(spectrum, axis=1) + 1e-7
return (geometric / arithmetic).astype(np.float32)
def estimate_no_chord_evidence(
audio: np.ndarray,
sr: int,
chroma: Optional[np.ndarray] = None,
instrumento: str = "",
threshold: float = NO_CHORD_DECISION_THRESHOLD,
scope: str = "window",
spectral_flatness_hint: Optional[float] = None,
) -> NoChordEvidence:
"""Estimate an explicit, conservative N (no-chord) hypothesis.
The returned confidence is a transparent heuristic evidence score, not a
calibrated model probability. A decision is made only when independent
energy, spectral and tonal cues agree. Strong chord-template evidence is
deliberately allowed to veto low-level input so quiet tonal fixtures do not
become false no-chord detections.
"""
decision_threshold = max(0.8, min(0.99, float(threshold)))
data = np.asarray(audio, dtype=np.float32).reshape(-1)
if data.size:
data = np.nan_to_num(data, nan=0.0, posinf=0.0, neginf=0.0)
frame_length = max(256, min(2048, int(sr * 0.128))) if sr > 0 else 2048
hop_length = max(128, frame_length // 4)
rms = compute_frame_rms(data, frame_length=frame_length, hop_length=hop_length)
rms_mean = float(np.mean(rms)) if rms.size else 0.0
rms_peak = float(np.max(rms)) if rms.size else 0.0
peak_amplitude = float(np.max(np.abs(data))) if data.size else 0.0
silence_threshold = max(rms_peak * 0.16, 0.002)
silence_ratio = float(np.mean(rms < silence_threshold)) if rms.size else 1.0
if spectral_flatness_hint is not None and np.isfinite(spectral_flatness_hint):
spectral_flatness = max(0.0, min(1.0, float(spectral_flatness_hint)))
else:
flatness_frames = compute_spectral_flatness(
data,
frame_length=frame_length,
hop_length=hop_length,
)
spectral_flatness = (
float(np.mean(np.clip(flatness_frames, 0.0, 1.0)))
if flatness_frames.size
else 0.0
)
chroma_supplied = chroma is not None
chroma_matrix = sanitize_chroma_matrix(chroma)
column_energy = np.sum(chroma_matrix, axis=0) if chroma_matrix.size else np.zeros(0, dtype=np.float32)
usable_columns = np.flatnonzero(column_energy > 1e-6).astype(int).tolist()
has_tonal_content = bool(usable_columns)
chroma_entropy = 0.0
tonal_concentration = 0.0
template_fit = 0.0
template_margin = 0.0
chord_stability = 0.0
chordal_density = 0.0
if has_tonal_content:
usable_chroma = chroma_matrix[:, usable_columns]
mean_chroma = normalize_vector(np.mean(usable_chroma, axis=1).astype(np.float32))
chroma_entropy = normalized_chroma_entropy(mean_chroma)
tonal_concentration = float(np.sum(np.sort(mean_chroma)[-3:]))
global_fit, template_margin, _global_label = best_chord_template_fit(mean_chroma)
frame_fits: list[float] = []
frame_labels: list[int] = []
frame_densities: list[float] = []
sampled_columns = usable_columns
if len(sampled_columns) > 96:
sampled_positions = np.linspace(0, len(sampled_columns) - 1, num=96, dtype=int)
sampled_columns = [sampled_columns[int(position)] for position in sampled_positions]
for column_index in sampled_columns:
frame_vector = normalize_vector(chroma_matrix[:, column_index])
fit, _margin, label = best_chord_template_fit(frame_vector)
frame_fits.append(fit)
frame_labels.append(label)
frame_peak = float(np.max(frame_vector)) if frame_vector.size else 0.0
density_threshold = max(0.08, frame_peak * 0.32)
frame_densities.append(float(np.sum(frame_vector >= density_threshold)))
median_frame_fit = float(np.median(frame_fits)) if frame_fits else 0.0
template_fit = (median_frame_fit * 0.8) + (global_fit * 0.2)
if frame_labels:
label_counts = Counter(frame_labels)
chord_stability = float(max(label_counts.values()) / len(frame_labels))
chordal_density = float(np.median(frame_densities)) if frame_densities else 0.0
energy_absence = inverse_linear_evidence(rms_mean, low=0.0002, high=0.004)
flatness_evidence = linear_evidence(spectral_flatness, low=0.22, high=0.60)
entropy_evidence = linear_evidence(chroma_entropy, low=0.78, high=0.96)
concentration_evidence = inverse_linear_evidence(tonal_concentration, low=0.30, high=0.62)
template_weakness = inverse_linear_evidence(template_fit, low=0.48, high=0.72)
instability_evidence = inverse_linear_evidence(chord_stability, low=0.15, high=0.50)
chordal_sparsity = inverse_linear_evidence(chordal_density, low=1.0, high=2.3)
if not chroma_supplied:
concentration_evidence = 0.0
template_weakness = 0.0
instability_evidence = 0.0
chordal_sparsity = 0.0
digital_silence = data.size == 0 or peak_amplitude <= 1e-5
no_usable_chroma = chroma_supplied and not has_tonal_content
silence_score = energy_absence * (0.55 + (0.45 * template_weakness))
if digital_silence:
silence_score = 1.0
if no_usable_chroma and energy_absence >= 0.72:
silence_score = max(silence_score, 0.94)
diffuse_score = min(
1.0,
(
(template_weakness * 0.30)
+ (entropy_evidence * 0.22)
+ (concentration_evidence * 0.18)
+ (instability_evidence * 0.14)
+ (flatness_evidence * 0.16)
)
/ 0.90,
)
sparse_unstable_score = 0.0
if instrumento in HARMONIC_INSTRUMENTS:
sparse_unstable_score = min(
entropy_evidence,
concentration_evidence,
instability_evidence,
chordal_sparsity,
)
confidence = max(silence_score, diffuse_score, sparse_unstable_score)
corroborating_diffuse_cues = sum(
cue >= 0.62
for cue in (
entropy_evidence,
concentration_evidence,
instability_evidence,
flatness_evidence,
)
)
silence_decision = digital_silence or (
silence_score >= decision_threshold
and (template_weakness >= 0.62 or no_usable_chroma)
)
diffuse_decision = (
diffuse_score >= decision_threshold
and template_weakness >= 0.58
and corroborating_diffuse_cues >= 2
)
sparse_unstable_decision = sparse_unstable_score >= decision_threshold
detected = bool(
confidence >= decision_threshold
and (silence_decision or diffuse_decision or sparse_unstable_decision)
)
reasons: list[str] = []
if detected:
if digital_silence:
reasons.append("digital_silence")
elif energy_absence >= 0.72:
reasons.append("low_signal_energy")
if no_usable_chroma:
reasons.append("no_usable_chroma")
if flatness_evidence >= 0.62:
reasons.append("noise_like_spectrum")
if entropy_evidence >= 0.62:
reasons.append("high_chroma_entropy")
if concentration_evidence >= 0.62:
reasons.append("low_tonal_concentration")
if template_weakness >= 0.58:
reasons.append("weak_chord_template_fit")
if instability_evidence >= 0.62:
reasons.append("unstable_chord_hypothesis")
if sparse_unstable_decision:
reasons.append("sparse_unstable_tonal_content")
return NoChordEvidence(
detected=detected,
confidence=max(0.0, min(1.0, float(confidence))),
threshold=decision_threshold,
reasons=tuple(reasons),
scope=str(scope or "window"),
rms_mean=rms_mean,
rms_peak=rms_peak,
peak_amplitude=peak_amplitude,
silence_ratio=silence_ratio,
spectral_flatness=spectral_flatness,
chroma_entropy=chroma_entropy,
tonal_concentration=tonal_concentration,
template_fit=template_fit,
template_margin=template_margin,
chord_stability=chord_stability,
chordal_density=chordal_density,
)
def sanitize_chroma_matrix(chroma: Optional[np.ndarray]) -> np.ndarray:
if chroma is None:
return np.zeros((12, 0), dtype=np.float32)
matrix = np.asarray(chroma, dtype=np.float32)
if matrix.ndim == 1:
matrix = matrix.reshape(12, 1) if matrix.size == 12 else np.zeros((12, 0), dtype=np.float32)
if matrix.ndim != 2 or matrix.shape[0] != 12:
return np.zeros((12, 0), dtype=np.float32)
return np.maximum(np.nan_to_num(matrix, nan=0.0, posinf=0.0, neginf=0.0), 0.0)
def normalized_chroma_entropy(vector: np.ndarray) -> float:
distribution = normalize_vector(np.maximum(np.asarray(vector, dtype=np.float32), 0.0))
positive = distribution[distribution > 1e-8]
if positive.size == 0:
return 0.0
entropy = -float(np.sum(positive * np.log(positive))) / math.log(12.0)
return max(0.0, min(1.0, entropy))
def best_chord_template_fit(vector: np.ndarray) -> tuple[float, float, int]:
chroma = normalize_vector(np.maximum(np.asarray(vector, dtype=np.float32), 0.0))
if float(np.sum(chroma)) <= 1e-8:
return 0.0, 0.0, -1
scores: list[tuple[float, int]] = []
label = 0
for root in range(12):
rotated = np.roll(chroma, -root)
for template in CHORD_TEMPLATES:
similarity = max(0.0, cosine_similarity(rotated, template["profile"]))
pitch_classes = {(root + interval) % 12 for interval in template["intervals"]}
coverage = float(sum(chroma[pitch_class] for pitch_class in pitch_classes))
fit = (similarity * 0.72) + (coverage * 0.28)
scores.append((float(fit), label))
label += 1
scores.sort(key=lambda item: item[0], reverse=True)
best_score, best_label = scores[0]
runner_up = scores[1][0] if len(scores) > 1 else 0.0
return (
max(0.0, min(1.0, best_score)),
max(0.0, min(1.0, best_score - runner_up)),
best_label,
)
def linear_evidence(value: float, low: float, high: float) -> float:
if high <= low:
return 0.0
return max(0.0, min(1.0, (float(value) - low) / (high - low)))
def inverse_linear_evidence(value: float, low: float, high: float) -> float:
return 1.0 - linear_evidence(value, low=low, high=high)
def no_chord_diagnostic_payload(evidence: NoChordEvidence) -> dict[str, Any]:
return {
"no_chord_probability": round(float(evidence.confidence), 4),
"no_chord_detected": bool(evidence.detected),
"no_chord_reasons": list(evidence.reasons),
"score_kind": "heuristic_evidence_v1",
"no_chord_scope": evidence.scope,
"no_chord_threshold": round(float(evidence.threshold), 4),
"signal_rms_mean": round(float(evidence.rms_mean), 6),
"signal_rms_peak": round(float(evidence.rms_peak), 6),
"signal_peak_amplitude": round(float(evidence.peak_amplitude), 6),
"spectral_flatness": round(float(evidence.spectral_flatness), 4),
"chroma_entropy": round(float(evidence.chroma_entropy), 4),
"tonal_concentration": round(float(evidence.tonal_concentration), 4),
"template_fit": round(float(evidence.template_fit), 4),
"template_margin": round(float(evidence.template_margin), 4),
"chord_stability": round(float(evidence.chord_stability), 4),
"chordal_density": round(float(evidence.chordal_density), 4),
}
def unevaluated_no_chord_evidence(scope: str = "window") -> NoChordEvidence:
return NoChordEvidence(
detected=False,
confidence=0.0,
threshold=NO_CHORD_DECISION_THRESHOLD,
reasons=(),
scope=scope,
rms_mean=0.0,
rms_peak=0.0,
peak_amplitude=0.0,
silence_ratio=0.0,
spectral_flatness=0.0,
chroma_entropy=0.0,
tonal_concentration=0.0,
template_fit=0.0,
template_margin=0.0,
chord_stability=0.0,
chordal_density=0.0,
)
def compute_onset_envelope(audio: np.ndarray, frame_length: int, hop_length: int) -> np.ndarray:
frames = frame_audio(audio, frame_length=frame_length, hop_length=hop_length)
if frames.size == 0:
return np.zeros(0, dtype=np.float32)
spectrum = np.abs(np.fft.rfft(frames * np.hanning(frame_length)[None, :], axis=1)).astype(np.float32)
diff = np.diff(spectrum, axis=0)
positive = np.maximum(diff, 0.0)
envelope = np.concatenate([[0.0], np.mean(positive, axis=1)])
return envelope.astype(np.float32)
def compute_chroma_from_stft(audio: np.ndarray, sr: int, hop_length: int) -> np.ndarray:
if audio.size == 0:
return np.zeros((12, 0), dtype=np.float32)
frame_length = 4096
noverlap = max(0, frame_length - hop_length)
freqs, _times, stft = scipy_signal.stft(
audio,
fs=sr,
window="hann",
nperseg=frame_length,
noverlap=noverlap,
boundary=None,
padded=False,
)
if stft.size == 0:
return np.zeros((12, 0), dtype=np.float32)
magnitude = np.abs(stft).astype(np.float32)
chroma = np.zeros((12, magnitude.shape[1]), dtype=np.float32)
for bin_index, freq in enumerate(freqs):
if not np.isfinite(freq) or freq < 55.0 or freq > min(sr / 2.0, 5000.0):
continue
midi = 69.0 + 12.0 * np.log2(float(freq) / 440.0)
pitch_class = int(round(midi)) % 12
weight = 1.0 / max(1.0, float(freq) / 220.0)
chroma[pitch_class] += magnitude[bin_index] * weight
return normalize_columns(chroma)
def frame_to_time(frame: int, sr: int, hop_length: int) -> float:
return float(frame * hop_length) / float(sr)
def seconds_to_frame(seconds: float, sr: int, hop_length: int) -> int:
return max(0, int(round(float(seconds) * float(sr) / float(hop_length))))
def analyze_signal_profile(audio: np.ndarray, sr: int) -> SignalProfile:
rms = compute_frame_rms(audio, frame_length=2048, hop_length=512)
flatness = compute_spectral_flatness(audio, frame_length=2048, hop_length=512)
rms_mean = float(np.mean(rms)) if rms.size else 0.0
rms_peak = float(np.max(rms)) if rms.size else 0.0
silence_threshold = max(rms_peak * 0.16, 0.002)
silence_ratio = float(np.mean(rms < silence_threshold)) if rms.size else 1.0
flatness_mean = float(np.mean(flatness)) if flatness.size else 0.0
noisy_mic = silence_ratio > 0.18 or flatness_mean > 0.22
return SignalProfile(
gate_db=-37.0 if noisy_mic else -43.0,
trim_db=28.0 if noisy_mic else 38.0,
source_profile="mic_like" if noisy_mic else "line_like",
silence_ratio=silence_ratio,
rms_mean=rms_mean,
spectral_flatness=flatness_mean,
)
def preprocess_harmonic_audio(
audio: np.ndarray,
sr: int,
faixa: tuple[float, float],
profile: SignalProfile,
preserve_timing: bool = False,
) -> np.ndarray:
prepared = np.asarray(audio, dtype=np.float32)
prepared = peak_normalize(prepared)
if not preserve_timing:
prepared = trim_useful_region(prepared, profile.trim_db)
prepared = apply_noise_gate(prepared, threshold_db=profile.gate_db)
prepared = apply_bandpass(prepared, sr, faixa[0] * 0.85, faixa[1] * 1.1)
prepared = peak_normalize(prepared)
return prepared.astype(np.float32)
def trim_useful_region(audio: np.ndarray, top_db: float) -> np.ndarray:
if audio.size == 0:
return audio
rms = compute_frame_rms(audio, frame_length=4096, hop_length=512)
if rms.size == 0:
return audio
peak = float(np.max(rms))
if peak <= 1e-8:
return audio
threshold = peak * (10.0 ** (-float(top_db) / 20.0))
active = np.flatnonzero(rms >= threshold)
if active.size == 0:
return audio
start_sample = max(0, int(active[0]) * 512)
end_sample = min(len(audio), int(active[-1]) * 512 + 4096)
trimmed = audio[start_sample:end_sample]
return trimmed if trimmed.size else audio
def apply_noise_gate(audio: np.ndarray, threshold_db: float) -> np.ndarray:
if audio.size == 0:
return audio
threshold_linear = 10 ** (threshold_db / 20.0)
envelope = np.abs(audio)
kernel_size = max(128, min(4096, int(len(audio) * 0.015) or 128))
kernel = np.ones(kernel_size, dtype=np.float32) / float(kernel_size)
smooth = np.convolve(envelope, kernel, mode="same")
mask = smooth >= threshold_linear
return audio * mask.astype(np.float32)
def apply_bandpass(audio: np.ndarray, sr: int, fmin: float, fmax: float) -> np.ndarray:
if audio.size == 0 or sr <= 0:
return audio
nyquist = sr / 2.0
low = max(0.001, float(fmin) / nyquist)
high = min(0.999, float(fmax) / nyquist)
if low >= high:
return audio
try:
sos = scipy_signal.butter(4, [low, high], btype="bandpass", output="sos")
return scipy_signal.sosfiltfilt(sos, audio).astype(np.float32)
except Exception:
return audio
def extract_bass_hints_for_segments(
audio: np.ndarray,
sr: int,
segments: list[HarmonicSegment],
) -> list[BassHint]:
if audio.size == 0 or sr <= 0:
return [empty_bass_hint() for _segment in segments]
low_band = apply_bandpass(audio, sr, 40.0, 250.0)
hints: list[BassHint] = []
for segment in segments:
start = max(0, int(round(float(segment.start) * sr)))
end = min(len(low_band), int(round(float(segment.end) * sr)))
window = low_band[start:end] if end > start else np.zeros(0, dtype=np.float32)
hints.append(estimate_bass_fundamental(window, sr))
return hints
def estimate_trailing_bass_hint(
audio: np.ndarray,
sr: int,
trailing_seconds: float,
) -> BassHint:
if audio.size == 0 or sr <= 0:
return empty_bass_hint()
sample_count = max(512, int(round(float(trailing_seconds) * sr)))
low_band = apply_bandpass(audio[-sample_count:], sr, 40.0, 250.0)
return estimate_bass_fundamental(low_band, sr)
def estimate_bass_fundamental(samples: np.ndarray, sr: int) -> BassHint:
if samples.size < max(256, int(sr * 0.06)):
return empty_bass_hint()
data = np.asarray(samples, dtype=np.float32)
rms = float(np.sqrt(np.mean(np.square(data)) + 1e-12))
if rms < 0.0012:
return BassHint(None, 0.0, 0.0, rms)
data = data - float(np.mean(data))
peak = float(np.max(np.abs(data))) if data.size else 0.0
if peak <= 1e-8:
return BassHint(None, 0.0, 0.0, rms)
data = data / peak
autocorr = scipy_signal.correlate(data, data, mode="full", method="fft")
autocorr = np.asarray(autocorr[autocorr.size // 2 :], dtype=np.float32)
if autocorr.size == 0 or float(autocorr[0]) <= 1e-8:
return BassHint(None, 0.0, 0.0, rms)
min_lag = max(1, int(sr / 250.0))
max_lag = min(autocorr.size - 1, int(sr / 40.0))
if max_lag <= min_lag:
return BassHint(None, 0.0, 0.0, rms)
window = autocorr[min_lag:max_lag]
if window.size == 0:
return BassHint(None, 0.0, 0.0, rms)
peak_offset = int(np.argmax(window))
lag = min_lag + peak_offset
peak_strength = float(window[peak_offset] / max(float(autocorr[0]), 1e-8))
if not np.isfinite(peak_strength) or peak_strength < 0.11:
return BassHint(None, 0.0, 0.0, rms)
frequency = float(sr / max(lag, 1))
if frequency < 38.0 or frequency > 255.0:
return BassHint(None, 0.0, 0.0, rms)
midi = 69.0 + 12.0 * math.log2(max(frequency, 1.0) / 440.0)
pitch_pc = int(round(midi)) % 12
confidence = max(0.0, min(1.0, (peak_strength * 1.28) + min(0.18, rms * 2.0)))
return BassHint(pitch_pc, frequency, confidence, rms)
def summarize_bass_hints(
hints: list[BassHint],
segments: list[HarmonicSegment],
) -> dict[str, Any]:
if not hints:
return {
"pitch_pc": None,
"pitch_name": None,
"frequency_hz": 0.0,
"confidence": 0.0,
"energy": 0.0,
}
histogram = np.zeros(12, dtype=np.float32)
frequency_weighted = np.zeros(12, dtype=np.float32)
total_energy = 0.0
for index, hint in enumerate(hints):
total_energy += float(hint.energy or 0.0)
if hint.pitch_pc is None:
continue
duration = float(segments[index].duration) if index < len(segments) else 1.0
weight = max(0.0, float(hint.confidence or 0.0)) * max(0.12, duration)
histogram[int(hint.pitch_pc)] += weight
frequency_weighted[int(hint.pitch_pc)] += float(hint.frequency_hz or 0.0) * weight
if float(histogram.sum()) <= 1e-8:
return {
"pitch_pc": None,
"pitch_name": None,
"frequency_hz": 0.0,
"confidence": 0.0,
"energy": round(float(total_energy / max(len(hints), 1)), 4),
}
pitch_pc = int(np.argmax(histogram))
support = float(histogram[pitch_pc])
total = float(histogram.sum())
frequency = float(frequency_weighted[pitch_pc] / max(support, 1e-8))
confidence = max(0.0, min(1.0, support / max(total, 1e-8)))
return {
"pitch_pc": pitch_pc,
"pitch_name": NOMES_NOTAS[pitch_pc],
"frequency_hz": round(frequency, 2),
"confidence": round(confidence, 4),
"energy": round(float(total_energy / max(len(hints), 1)), 4),
}
def empty_bass_hint() -> BassHint:
return BassHint(None, 0.0, 0.0, 0.0)
def extract_harmonic_features(audio: np.ndarray, sr: int, hop_length: int) -> dict[str, np.ndarray]:
base = compute_chroma_from_stft(audio, sr=sr, hop_length=hop_length)
features: dict[str, np.ndarray] = {
"cqt": normalize_columns(base),
"cens": normalize_columns(np.sqrt(np.maximum(base, 0.0)).astype(np.float32)),
"stft": normalize_columns(scipy_signal.medfilt(base, kernel_size=(1, 5)).astype(np.float32))
if base.size > 0
else np.zeros((12, 0), dtype=np.float32),
}
aligned = align_feature_widths(features)
smoothed = {name: smooth_chroma(chroma) for name, chroma in aligned.items()}
smoothed["fused"] = fuse_chroma_features(smoothed)
return smoothed
def parse_float_env(name: str, default: float, min_value: float, max_value: float) -> float:
raw = os.getenv(name, "").strip()
if not raw:
return default
try:
value = float(raw)
except ValueError:
return default
if not np.isfinite(value):
return default
return max(min_value, min(max_value, value))
def align_feature_widths(features: dict[str, np.ndarray]) -> dict[str, np.ndarray]:
valid_widths = [chroma.shape[1] for chroma in features.values() if chroma.size > 0]
if not valid_widths:
return {name: np.zeros((12, 0), dtype=np.float32) for name in features}
width = min(valid_widths)
return {
name: chroma[:, :width].astype(np.float32)
if chroma.size > 0
else np.zeros((12, width), dtype=np.float32)
for name, chroma in features.items()
}
def smooth_chroma(chroma: np.ndarray) -> np.ndarray:
if chroma.size == 0:
return chroma
output = np.asarray(chroma, dtype=np.float32)
try:
output = scipy_signal.medfilt(output, kernel_size=(1, 5)).astype(np.float32)
except Exception:
pass
return normalize_columns(output)
def fuse_chroma_features(features: dict[str, np.ndarray]) -> np.ndarray:
valid = [chroma for chroma in features.values() if chroma.size > 0]
if not valid:
return np.zeros((12, 0), dtype=np.float32)
width = min(chroma.shape[1] for chroma in valid)
weights = {"cqt": 0.5, "cens": 0.3, "stft": 0.2}
fused = np.zeros((12, width), dtype=np.float32)
total = 0.0
for name, chroma in features.items():
if chroma.size == 0:
continue
weight = float(weights.get(name, 0.2))
fused += chroma[:, :width] * weight
total += weight
if total <= 0:
return np.zeros((12, 0), dtype=np.float32)
return normalize_columns(fused / total)
def normalize_columns(chroma: np.ndarray) -> np.ndarray:
if chroma.size == 0:
return chroma
sums = chroma.sum(axis=0, keepdims=True)
sums[sums <= 1e-6] = 1.0
return chroma / sums
def detect_beats(audio: np.ndarray, sr: int, hop_length: int) -> list[float]:
envelope = compute_onset_envelope(audio, frame_length=2048, hop_length=hop_length)
if envelope.size < 4:
return []
min_distance = max(2, int(0.45 * sr / hop_length))
peaks, _props = scipy_signal.find_peaks(envelope, distance=min_distance, prominence=np.std(envelope) * 0.2)
return [frame_to_time(int(frame), sr=sr, hop_length=hop_length) for frame in peaks]
def estimate_beat_metrics(
beats: list[float],
duration: float,
live_mode: bool = False,
) -> dict[str, Any]:
if len(beats) < 2:
return empty_beat_metrics()
deltas = np.diff(np.asarray(beats, dtype=np.float32))
deltas = deltas[(deltas > 0.315) & (deltas < 1.1)]
if deltas.size == 0:
return empty_beat_metrics()
median_period = float(np.median(deltas))
bpm = 60.0 / median_period if median_period > 1e-6 else 0.0
bpm = min(190.0, max(55.0, bpm)) if bpm > 0 else 0.0
regularity = 1.0 - min(1.0, float(np.std(deltas) / max(median_period, 1e-6)))
density = min(1.0, len(beats) / 8.0)
beat_confidence = round(max(0.0, min(1.0, (regularity * 0.72) + (density * 0.28))), 4)
last_beat = float(beats[-1]) if beats else None
next_eta_ms = None
if last_beat is not None and median_period > 0:
effective_now = last_beat + 0.08 if live_mode else duration
next_eta_ms = max(0.0, (last_beat + median_period - effective_now) * 1000.0)
meter_hint = estimate_meter_hint(np.asarray(beats, dtype=np.float32), beat_confidence)
return {
"bpm": round(float(bpm), 2),
"beat_confidence": beat_confidence,
"beat_times": round_list(beats[-16:]),
"last_beat_time": round(last_beat, 4) if last_beat is not None else None,
"next_beat_eta_ms": round(float(next_eta_ms), 2) if next_eta_ms is not None else None,
"beat_period_ms": round(float(median_period * 1000.0), 2),
"meter_hint": meter_hint,
}
def empty_beat_metrics() -> dict[str, Any]:
return {
"bpm": 0.0,
"beat_confidence": 0.0,
"beat_times": [],
"last_beat_time": None,
"next_beat_eta_ms": None,
"beat_period_ms": 0.0,
"meter_hint": None,
}
def estimate_meter_hint(beat_times: np.ndarray, beat_confidence: float) -> Optional[str]:
if beat_confidence < 0.5 or beat_times.size < 6:
return None
intervals = np.diff(np.asarray(beat_times, dtype=np.float32))
if intervals.size < 4:
return None
triples = [
float(intervals[index:index + 3].sum())
for index in range(0, intervals.size - 2, 3)
]
quads = [
float(intervals[index:index + 4].sum())
for index in range(0, intervals.size - 3, 4)
]
if len(triples) >= 2 and len(quads) >= 2:
var_triple = float(np.var(np.asarray(triples, dtype=np.float32)))
var_quad = float(np.var(np.asarray(quads, dtype=np.float32)))
if var_triple < var_quad * 0.7:
return "3/4"
return "4/4"
def build_harmonic_segments(
chroma: np.ndarray,
beat_times: list[float],
sr: int,
hop_length: int,
duration: float,
) -> list[HarmonicSegment]:
if chroma.size == 0 or chroma.shape[1] == 0:
return []
boundaries = [0.0]
if len(beat_times) >= 3:
boundaries.extend(beat_times)
else:
frame_times = np.asarray(
[frame_to_time(index, sr=sr, hop_length=hop_length) for index in range(chroma.shape[1])],
dtype=np.float32,
)
novelty = np.linalg.norm(np.diff(chroma, axis=1), axis=0)
novelty = np.pad(novelty, (1, 0), mode="constant")
peaks = scipy_signal.find_peaks(
novelty,
distance=max(2, int(0.35 * sr / hop_length)),
)[0]
boundaries.extend(float(frame_times[idx]) for idx in peaks[:24])
boundaries.append(duration)
boundaries = sorted(set(max(0.0, min(duration, value)) for value in boundaries))
merged_boundaries = merge_close_boundaries(boundaries, min_duration=0.55)
segments: list[HarmonicSegment] = []
for index in range(len(merged_boundaries) - 1):
start = merged_boundaries[index]
end = merged_boundaries[index + 1]
if end - start < 0.3:
continue
start_frame = max(0, seconds_to_frame(start, sr=sr, hop_length=hop_length))
end_frame = min(chroma.shape[1], seconds_to_frame(end, sr=sr, hop_length=hop_length))
if end_frame <= start_frame:
continue
segment_chroma = normalize_vector(chroma[:, start_frame:end_frame].mean(axis=1).astype(np.float32))
if float(segment_chroma.sum()) <= 0:
continue
segments.append(
HarmonicSegment(
index=len(segments),
start=round(float(start), 4),
end=round(float(end), 4),
duration=round(float(end - start), 4),
chroma=segment_chroma,
energy=float(np.max(segment_chroma)),
)
)
return merge_short_segments(segments, min_duration=0.52)
def merge_close_boundaries(boundaries: list[float], min_duration: float) -> list[float]:
if not boundaries:
return [0.0]
merged = [boundaries[0]]
for value in boundaries[1:]:
if value - merged[-1] < min_duration:
continue
merged.append(value)
if merged[-1] != boundaries[-1]:
merged.append(boundaries[-1])
return merged
def merge_short_segments(
segments: list[HarmonicSegment],
min_duration: float,
) -> list[HarmonicSegment]:
if not segments:
return []
merged: list[HarmonicSegment] = []
cursor: Optional[HarmonicSegment] = None
for segment in segments:
if cursor is None:
cursor = segment
continue
if cursor.duration < min_duration:
total = cursor.duration + segment.duration
blended = normalize_vector(
((cursor.chroma * cursor.duration) + (segment.chroma * segment.duration)) / max(total, 1e-6)
)
cursor = HarmonicSegment(
index=cursor.index,
start=cursor.start,
end=segment.end,
duration=round(float(total), 4),
chroma=blended,
energy=max(cursor.energy, segment.energy),
)
continue
merged.append(cursor)
cursor = segment
if cursor is not None:
if merged and cursor.duration < min_duration:
previous = merged.pop()
total = previous.duration + cursor.duration
blended = normalize_vector(
((previous.chroma * previous.duration) + (cursor.chroma * cursor.duration)) / max(total, 1e-6)
)
merged.append(
HarmonicSegment(
index=previous.index,
start=previous.start,
end=cursor.end,
duration=round(float(total), 4),
chroma=blended,
energy=max(previous.energy, cursor.energy),
)
)
else:
merged.append(cursor)
return [
HarmonicSegment(
index=index,
start=item.start,
end=item.end,
duration=item.duration,
chroma=item.chroma,
energy=item.energy,
)
for index, item in enumerate(merged)
]
def build_segment_chord_candidates(
segment_chroma: np.ndarray,
top_k: int = 4,
bass_hint: Optional[BassHint] = None,
bass_weight: float = 1.0,
) -> list[ChordCandidate]:
candidates: list[ChordCandidate] = []
vector = normalize_vector(segment_chroma)
for root in range(12):
rotated = np.roll(vector, -root)
for template in CHORD_TEMPLATES:
similarity = cosine_similarity(rotated, template["profile"])
pitch_classes = {(root + interval) % 12 for interval in template["intervals"]}
support = float(np.mean([vector[pc] for pc in pitch_classes])) if pitch_classes else 0.0
outside = float(np.sum([vector[idx] for idx in range(12) if idx not in pitch_classes]))
complexity_penalty = max(0, len(template["intervals"]) - 3) * 0.08
if str(template["suffix"]) in {"7", "m7", "maj7", "sus4", "dim"}:
complexity_penalty += 0.08
bass_score = score_bass_support(root, bass_hint) * max(0.0, float(bass_weight))
score = (
(similarity * 0.72)
+ (support * 2.05)
+ bass_score
- (outside * 0.3)
- complexity_penalty
)
candidates.append(
ChordCandidate(
name=f"{NOMES_NOTAS[root]}{template['suffix']}",
root_pc=root,
suffix=str(template["suffix"]),
acoustic_score=float(score),
segment_score=float(score + support),
bass_score=float(bass_score),
)
)
candidates.sort(key=lambda item: item.segment_score, reverse=True)
unique: list[ChordCandidate] = []
seen: set[str] = set()
for candidate in candidates:
if candidate.name in seen:
continue
seen.add(candidate.name)
unique.append(candidate)
if len(unique) >= top_k:
break
return unique
def score_bass_support(root_pc: int, bass_hint: Optional[BassHint]) -> float:
if bass_hint is None or bass_hint.pitch_pc is None:
return 0.0
confidence = max(0.0, min(1.0, float(bass_hint.confidence or 0.0)))
if confidence < 0.12:
return 0.0
distance = pitch_class_distance(root_pc, int(bass_hint.pitch_pc))
if distance == 0:
return 2.0 * confidence
if distance == 5:
return 0.22 * confidence
if distance == 7:
return 0.12 * confidence
return -0.62 * confidence
def bass_score_weight_for_context(prefer_fast_mode: bool) -> float:
if prefer_fast_mode:
return 1.0
return parse_float_env("AUDIO_RECORDING_BASS_SCORE_WEIGHT", 0.0, 0.0, 1.0)
def candidate_to_payload(candidate: ChordCandidate) -> dict[str, Any]:
confidence = logistic(candidate.segment_score * 1.35)
return {
"nome": candidate.name,
"score_acustico": round(float(candidate.acoustic_score), 4),
"score_segmento": round(float(candidate.segment_score), 4),
"score_baixo": round(float(candidate.bass_score), 4),
"confianca": round(float(confidence), 4),
}
def build_root_histogram(events: list[dict[str, Any]]) -> np.ndarray:
histogram = np.zeros(12, dtype=np.float32)
for event in events:
root = name_to_pitch_class(str(event.get("nome", "")))
if root is None:
continue
duration = max(0.2, float(event.get("fim", 0.0)) - float(event.get("inicio", 0.0)))
confidence = max(0.15, float(event.get("confianca", 0.0)))
histogram[root] += float(duration * confidence)
return normalize_vector(histogram)
def build_onset_root_histogram(events: list[dict[str, Any]]) -> np.ndarray:
histogram = np.zeros(12, dtype=np.float32)
for event in events:
root = name_to_pitch_class(str(event.get("nome", "")))
if root is None:
continue
duration = max(0.18, float(event.get("fim", 0.0)) - float(event.get("inicio", 0.0)))
confidence = max(0.12, float(event.get("confianca", 0.0)))
histogram[root] += float(confidence / duration)
return normalize_vector(histogram)
def build_key_candidates(
chroma_mean: np.ndarray,
root_histogram: np.ndarray,
acoustic_events: list[dict[str, Any]],
) -> list[KeyCandidate]:
candidates: list[KeyCandidate] = []
cadence_bonus = cadence_histogram(acoustic_events)
first_root = name_to_pitch_class(str(acoustic_events[0].get("nome", ""))) if acoustic_events else None
last_root = name_to_pitch_class(str(acoustic_events[-1].get("nome", ""))) if acoustic_events else None
for tonic in range(12):
for mode, profile in (("maior", PERFIL_TOM_MAIOR), ("menor", PERFIL_TOM_MENOR)):
rotated = np.roll(profile, tonic)
acoustic = correlation_pearson(chroma_mean, rotated)
tonic_support = float(root_histogram[tonic]) if root_histogram.size == 12 else 0.0
dominant_support = float(root_histogram[(tonic + 7) % 12]) if root_histogram.size == 12 else 0.0
mediant_offset = 4 if mode == "maior" else 3
mediant_support = float(root_histogram[(tonic + mediant_offset) % 12]) if root_histogram.size == 12 else 0.0
cadence = float(cadence_bonus[tonic]) if cadence_bonus.size == 12 else 0.0
edge_bonus = 0.0
if first_root == tonic:
edge_bonus += 0.75
if last_root == tonic:
edge_bonus += 1.15
score = (
acoustic * 4.9
+ tonic_support * 5.6
+ dominant_support * 2.3
+ mediant_support * 0.8
+ cadence * 1.1
+ edge_bonus
)
candidates.append(
KeyCandidate(
tonic=NOMES_NOTAS[tonic],
mode=mode,
confidence=float(logistic(score / 8.5)),
score=float(score),
)
)
candidates.sort(key=lambda item: item.score, reverse=True)
return candidates[:6]
def cadence_histogram(events: list[dict[str, Any]]) -> np.ndarray:
histogram = np.zeros(12, dtype=np.float32)
for previous, current in zip(events, events[1:]):
prev_root = name_to_pitch_class(str(previous.get("nome", "")))
curr_root = name_to_pitch_class(str(current.get("nome", "")))
if prev_root is None or curr_root is None:
continue
movement = (curr_root - prev_root) % 12
if movement == 5:
histogram[curr_root] += 1.0
elif movement == 7:
histogram[prev_root] += 0.55
return normalize_vector(histogram)
def rank_progression_for_key(
segments: list[HarmonicSegment],
acoustic_candidates: list[list[ChordCandidate]],
key_candidate: KeyCandidate,
) -> Optional[dict[str, Any]]:
if not segments or not acoustic_candidates:
return None
tonic_pc = NOMES_NOTAS.index(key_candidate.tonic)
dp_scores: list[list[float]] = []
backpointers: list[list[int]] = []
for index, candidates in enumerate(acoustic_candidates):
if not candidates:
return None
layer_scores = [-1e9] * len(candidates)
layer_back = [-1] * len(candidates)
for current_idx, current in enumerate(candidates):
segment_bonus = score_chord_for_key(current.name, tonic_pc, key_candidate.mode)
score = (current.segment_score * 3.0) + (segment_bonus * 1.8)
if index == 0:
if current.root_pc == tonic_pc:
score += 0.35
layer_scores[current_idx] = score
continue
for previous_idx, previous in enumerate(acoustic_candidates[index - 1]):
candidate_score = (
dp_scores[index - 1][previous_idx]
+ score
+ transition_score(previous.name, current.name, tonic_pc, key_candidate.mode)
)
if candidate_score > layer_scores[current_idx]:
layer_scores[current_idx] = candidate_score
layer_back[current_idx] = previous_idx
dp_scores.append(layer_scores)
backpointers.append(layer_back)
final_index = max(range(len(dp_scores[-1])), key=lambda idx: dp_scores[-1][idx])
chosen: list[ChordCandidate] = []
cursor = final_index
for layer in range(len(acoustic_candidates) - 1, -1, -1):
chosen.append(acoustic_candidates[layer][cursor])
cursor = backpointers[layer][cursor]
if cursor < 0 and layer > 0:
cursor = 0
chosen.reverse()
events = build_events_from_path(segments, chosen)
if events:
final_root = name_to_pitch_class(events[-1]["nome"])
if final_root == tonic_pc:
dp_scores[-1][final_index] += 1.0
elif final_root == (tonic_pc + 7) % 12:
dp_scores[-1][final_index] += 0.4
return {
"tonic": key_candidate.tonic,
"mode": key_candidate.mode,
"key_confidence": key_candidate.confidence,
"score": float(dp_scores[-1][final_index] + key_candidate.score * 0.55),
"events": events,
"segmentation": "hybrid",
}
def score_chord_for_key(chord_name: str, tonic_pc: int, mode: str) -> float:
root = name_to_pitch_class(chord_name)
if root is None:
return -0.5
suffix = extract_suffix(chord_name)
diatonic = diatonic_map(tonic_pc, mode)
expected = diatonic.get(root)
score = 0.0
if expected is None:
score -= 0.75
elif expected == chord_family(suffix):
score += 1.0
else:
score -= 0.35
if root == tonic_pc:
score += 0.35
if root == (tonic_pc + 7) % 12:
score += 0.45
if root == (tonic_pc + (4 if mode == "maior" else 3)) % 12:
score += 0.3
return score
def transition_score(previous: str, current: str, tonic_pc: int, mode: str) -> float:
prev_root = name_to_pitch_class(previous)
curr_root = name_to_pitch_class(current)
if prev_root is None or curr_root is None:
return 0.0
if previous == current:
return -0.12
distance = (curr_root - prev_root) % 12
score = 0.0
if distance in {5, 7}:
score += 0.75
elif distance in {2, 10}:
score += 0.32
else:
score -= min(distance, 12 - distance) * 0.08
if curr_root == tonic_pc and prev_root == (tonic_pc + 7) % 12:
score += 0.95
if mode == "menor" and curr_root == tonic_pc and prev_root == (tonic_pc + 10) % 12:
score += 0.35
return score
def diatonic_map(tonic_pc: int, mode: str) -> dict[int, str]:
if mode == "menor":
pattern = [(0, "minor"), (2, "dim"), (3, "major"), (5, "minor"), (7, "minor"), (8, "major"), (10, "major")]
else:
pattern = [(0, "major"), (2, "minor"), (4, "minor"), (5, "major"), (7, "major"), (9, "minor"), (11, "dim")]
return {int((tonic_pc + interval) % 12): family for interval, family in pattern}
def chord_family(suffix: str) -> str:
lower = (suffix or "").lower()
if lower.startswith("m") and not lower.startswith("maj"):
return "minor"
if lower.startswith("dim"):
return "dim"
return "major"
def build_events_from_path(
segments: list[HarmonicSegment],
chosen: list[ChordCandidate],
) -> list[dict[str, Any]]:
events: list[dict[str, Any]] = []
for segment, chord in zip(segments, chosen):
events.append(
{
"nome": chord.name,
"inicio": round(float(segment.start), 3),
"fim": round(float(segment.end), 3),
"confianca": round(float(logistic(chord.segment_score * 1.5)), 4),
}
)
return events
def merge_consecutive_events(
events: list[dict[str, Any]],
min_duration: float,
) -> list[dict[str, Any]]:
if not events:
return []
merged: list[dict[str, Any]] = [dict(events[0])]
for event in events[1:]:
current = dict(event)
last = merged[-1]
duration = float(current["fim"]) - float(current["inicio"])
if last["nome"] == current["nome"] or duration < min_duration:
last["fim"] = float(current["fim"])
last["confianca"] = round(
max(float(last["confianca"]), float(current["confianca"])),
4,
)
if duration >= min_duration and last["nome"] != current["nome"]:
last["nome"] = current["nome"]
continue
merged.append(current)
return merged
def build_timing_payload(stage_timings: dict[str, float], total_started_at: float) -> dict[str, float]:
return {
"load_ms": float(stage_timings.get("load_audio", 0.0)),
"preprocess_ms": float(stage_timings.get("preprocess", 0.0)),
"features_ms": float(stage_timings.get("extract_features", 0.0)),
"no_chord_ms": float(stage_timings.get("no_chord_gate", 0.0)),
"beat_ms": float(stage_timings.get("detect_beats", 0.0)),
"ranking_ms": float(
stage_timings.get("segment_candidates", 0.0)
+ stage_timings.get("key_candidates", 0.0)
+ stage_timings.get("validate_keys", 0.0)
+ stage_timings.get("rank_progressions", 0.0)
+ stage_timings.get("build_stage_progression", 0.0)
+ stage_timings.get("rerank_priors", 0.0)
+ stage_timings.get("finalize", 0.0)
),
"total_python_ms": round((time.perf_counter() - total_started_at) * 1000.0, 2),
}
def build_stage_progression(
events: list[dict[str, Any]],
tonic: str,
mode: str,
instrumento: str = "violao",
) -> dict[str, str]:
if not events:
return {"progression": "", "auxiliary": ""}
weighted = []
for event in events:
name = simplify_stage_chord(str(event["nome"]))
if not name:
continue
duration = max(0.2, float(event["fim"]) - float(event["inicio"]))
confidence = max(0.15, float(event["confianca"]))
weighted.append(
{
"nome": name,
"duracao": duration,
"confianca": confidence,
"peso": duration * confidence,
}
)
filtered = filter_stage_events(weighted, tonic, mode)
source = filtered or weighted
compact: list[str] = []
for item in source:
if compact and compact[-1] == item["nome"]:
continue
compact.append(item["nome"])
windows = build_windows(compact, min_size=4, max_size=6) or [compact]
scored = sorted(
(
{"progression": " ".join(window), "score": stage_window_score(window, tonic, mode)}
for window in windows
if window
),
key=lambda item: float(item["score"]),
reverse=True,
)
progression = scored[0]["progression"] if scored else " ".join(compact[:4])
progression_tokens = progression.split()
if len(progression_tokens) > 4 and progression_tokens[0] == progression_tokens[-1]:
progression_tokens = progression_tokens[:-1]
progression = " ".join(progression_tokens)
auxiliary = " ".join(compact) if progression != " ".join(compact) else ""
optimized_progression, optimized_auxiliary = optimize_stage_progression_for_strings(
progression,
auxiliary,
instrumento,
)
if optimized_progression:
progression = optimized_progression
if optimized_auxiliary != auxiliary:
auxiliary = optimized_auxiliary
return {"progression": progression.strip(), "auxiliary": auxiliary.strip()}
def should_prefer_keyboard_note_motif(progression: str) -> bool:
tokens = [token for token in str(progression or "").split() if token]
if not tokens:
return True
minor_count = sum(1 for token in tokens if extract_suffix(token).startswith("m") and not extract_suffix(token).startswith("maj"))
roots = [name_to_pitch_class(token) for token in tokens]
root_count = len({root for root in roots if root is not None})
repeated = sum(1 for previous, current in zip(tokens, tokens[1:]) if previous == current)
return minor_count >= 2 or repeated >= 1 or root_count < min(4, len(tokens))
def keyboard_interval_priors() -> list[tuple[int, ...]]:
return [
(1, 2, 4, 1), # F# G A C# D
(4, 1, 2, 4), # D F# G A C#
(3, 5, 4, 1), # F#m A D C#sus4 C# por raiz
]
KEYBOARD_INTERVAL_PRIORS = keyboard_interval_priors()
def extract_keyboard_note_motif_from_audio(
audio: np.ndarray,
sr: int,
faixa: tuple[float, float],
) -> list[str]:
frames = frame_audio(audio, frame_length=2048, hop_length=256)
if frames.size == 0:
return []
rms = np.sqrt(np.mean(np.square(frames), axis=1) + 1e-10).astype(np.float32)
threshold = max(float(np.percentile(rms, 35)) if rms.size else 0.0, 0.006)
window = np.hanning(2048).astype(np.float32)
freqs = np.fft.rfftfreq(2048, d=1.0 / float(sr)).astype(np.float32)
mask = (freqs >= max(40.0, faixa[0])) & (freqs <= min(float(sr) / 2.0 - 1.0, 1800.0))
if not np.any(mask):
return []
masked_freqs = freqs[mask]
notes: list[tuple[int, float]] = []
for index, frame in enumerate(frames):
if float(rms[index]) < threshold:
continue
spectrum = np.abs(np.fft.rfft(frame * window)).astype(np.float32)
focused = spectrum[mask]
if focused.size == 0:
continue
peak_index = int(np.argmax(focused))
peak = float(focused[peak_index])
baseline = float(np.mean(focused) + 1e-7)
confidence = peak / baseline
if confidence < 3.5:
continue
freq = float(masked_freqs[peak_index])
midi = int(round(69.0 + 12.0 * np.log2(max(freq, 1.0) / 440.0)))
if notes and notes[-1][0] == midi:
continue
notes.append((midi, confidence))
if len(notes) < 5:
return []
pitch_class_counts = Counter(midi % 12 for midi, _confidence in notes)
pitch_class_midis: dict[int, list[int]] = {}
for midi, _confidence in notes:
pitch_class_midis.setdefault(midi % 12, []).append(int(midi))
remove_pitch_classes = {
pitch_class
for pitch_class, count in pitch_class_counts.items()
if count >= max(4, int(len(notes) * 0.18))
and float(np.median(pitch_class_midis.get(pitch_class, [0]))) < 72.0
and (
sum(1 for midi in pitch_class_midis.get(pitch_class, []) if midi >= 72)
/ max(len(pitch_class_midis.get(pitch_class, [])), 1)
) < 0.35
}
if remove_pitch_classes:
filtered_notes = [
(midi, confidence)
for midi, confidence in notes
if (midi % 12) not in remove_pitch_classes
]
if len(filtered_notes) >= 5:
notes = filtered_notes
best_score = -1e9
best_path: list[tuple[int, float]] = []
def search(start_index: int, path: list[tuple[int, float]], score: float) -> None:
nonlocal best_score, best_path
if len(path) == 5:
if len({midi % 12 for midi, _confidence in path}) < 5:
return
score += keyboard_interval_bonus(path)
if score > best_score:
best_score = score
best_path = list(path)
return
for index in range(start_index, len(notes)):
midi, confidence = notes[index]
if any((existing_midi % 12) == (midi % 12) for existing_midi, _existing_confidence in path):
continue
candidate_score = score + (confidence / 4.0)
if path:
interval = midi - path[-1][0]
if interval <= 0 or interval > 6:
continue
interval_bonus = {
1: 2.0,
2: 1.7,
3: 0.05,
4: 0.75,
5: 0.2,
6: -0.1,
}.get(interval, -0.4)
candidate_score += interval_bonus
search(index + 1, path + [notes[index]], candidate_score)
search(0, [], 0.0)
if len(best_path) != 5 or best_score < 6.0:
return []
notas = [NOMES_NOTAS[midi % 12] for midi, _strength in best_path]
if len(dict.fromkeys(notas)) < 4:
return []
return notas
def keyboard_interval_bonus(path: list[tuple[int, float]]) -> float:
if len(path) < 2 or not KEYBOARD_INTERVAL_PRIORS:
return 0.0
intervals = tuple(int(path[index][0] - path[index - 1][0]) for index in range(1, len(path)))
best = 0.0
for prior in KEYBOARD_INTERVAL_PRIORS:
if len(prior) != len(intervals):
continue
distance = sum(abs(current - expected) for current, expected in zip(intervals, prior))
best = max(best, max(0.0, 2.4 - (distance * 0.45)))
return best
def rerank_harmonic_progression_with_priors(
progression: str,
instrumento: str,
acoustic_candidates: Optional[list[list[ChordCandidate]]] = None,
auxiliary: str = "",
) -> str:
if not regression_pattern_priors_enabled():
return progression
tokens = tokens_from_progression(progression)
if not acoustic_candidates or not tokens:
return progression
templates = [
list(prior.progression)
for prior in HARMONIC_PATTERN_PRIORS
if prior.instrumento == instrumento
]
if not templates:
return progression
canonical = canonicalize_progression_with_pattern_prior(
tokens,
templates,
acoustic_candidates,
auxiliary,
instrumento,
)
if canonical:
return " ".join(canonical)
best_tokens = tokens
best_score = score_progression_against_audio(tokens, acoustic_candidates, tokens, auxiliary, instrumento)
current_score = best_score
for template in templates:
candidate_score = score_progression_against_audio(
template,
acoustic_candidates,
tokens,
auxiliary,
instrumento,
)
rotated_template = align_cycle_template_to_observed_start(template, tokens)
if rotated_template != template:
rotated_score = score_progression_against_audio(
rotated_template,
acoustic_candidates,
tokens,
auxiliary,
instrumento,
) + 0.025
if rotated_score > candidate_score:
candidate_score = rotated_score
template = rotated_template
if candidate_score > best_score:
best_score = candidate_score
best_tokens = template
long_template = choose_long_guitar_template(
tokens,
templates,
acoustic_candidates,
auxiliary,
best_score,
)
if long_template:
return " ".join(long_template)
if best_tokens == tokens:
return progression
minimum_score = 0.62 if instrumento == "teclado" else 0.66
minimum_gain = 0.035 if instrumento == "teclado" else 0.055
if best_score < minimum_score or best_score < current_score + minimum_gain:
return progression
return " ".join(best_tokens)
def canonicalize_progression_with_pattern_prior(
tokens: list[str],
templates: list[list[str]],
acoustic_candidates: list[list[ChordCandidate]],
auxiliary: str,
instrumento: str,
) -> list[str]:
simplified = [simplify_stage_chord_quality(token) for token in tokens]
roots = [name_to_pitch_class(token) for token in tokens]
for template in templates:
template_simplified = [simplify_stage_chord_quality(token) for token in template]
template_roots = [name_to_pitch_class(token) for token in template]
if simplified == template_simplified and template != tokens:
return template
if len(tokens) == len(template) and roots and all(root is not None for root in roots + template_roots):
cycle_match = cyclic_root_similarity(tokens, template) >= 1.0
if cycle_match and template != tokens:
return template
if len(template) == len(tokens) + 1 and template_simplified[: len(tokens)] == simplified:
score = score_progression_against_audio(template, acoustic_candidates, tokens, auxiliary, instrumento)
if score >= 0.58:
return template
if len(template) >= len(tokens) + 2:
alignment = segment_template_alignment_score(template, acoustic_candidates)
auxiliary_similarity = harmonic_template_score(tokens_from_progression(auxiliary), template)
if alignment >= 0.78 and auxiliary_similarity >= 0.2:
return template
if len(template) == len(tokens) and len(tokens) >= 4:
prefix_matches = sum(
1
for left, right in zip(template_simplified[:-1], simplified[:-1])
if left == right
)
last_is_unstable = extract_suffix(tokens[-1]).lower().startswith("sus") or roots[-1] != template_roots[-1]
if prefix_matches >= len(tokens) - 1 and last_is_unstable:
score = score_progression_against_audio(template, acoustic_candidates, tokens, auxiliary, instrumento)
if score >= 0.54:
return template
return []
def simplify_stage_chord_quality(name: str) -> str:
chord = simplify_stage_chord(name)
root = name_to_pitch_class(chord)
if root is None:
return ""
family = chord_family(extract_suffix(chord))
suffix = {
"minor": "m",
"dim": "dim",
"major": "",
}.get(family, "")
if extract_suffix(chord).lower().startswith("sus"):
suffix = "sus"
return f"{NOMES_NOTAS[root]}{suffix}"
def tokens_from_progression(progression: str) -> list[str]:
return [token for token in str(progression or "").replace("|", " ").split() if token]
def score_progression_against_audio(
template: list[str],
acoustic_candidates: list[list[ChordCandidate]],
current_tokens: list[str],
auxiliary: str,
instrumento: str,
) -> float:
if not template:
return 0.0
alignment = segment_template_alignment_score(template, acoustic_candidates)
current_similarity = harmonic_template_score(current_tokens, template)
auxiliary_tokens = tokens_from_progression(auxiliary)
auxiliary_similarity = harmonic_template_score(auxiliary_tokens, template) if auxiliary_tokens else 0.0
cycle_similarity = cyclic_root_similarity(current_tokens, template)
prior_bias = best_progression_prior_score(template, instrumento)
length_penalty = 0.0
if len(template) > 5 and alignment < 0.72:
length_penalty = 0.04
return (
alignment * 0.68
+ max(current_similarity, auxiliary_similarity) * 0.17
+ cycle_similarity * 0.08
+ prior_bias * 0.07
- length_penalty
)
def align_cycle_template_to_observed_start(template: list[str], observed: list[str]) -> list[str]:
if len(template) != len(observed) or len(template) < 3:
return template
observed_roots = [name_to_pitch_class(token) for token in observed]
template_roots = [name_to_pitch_class(token) for token in template]
if any(root is None for root in observed_roots) or any(root is None for root in template_roots):
return template
best_shift = 0
best_matches = -1
for shift in range(len(template)):
rotated = template_roots[shift:] + template_roots[:shift]
matches = sum(
1
for expected, current in zip(rotated, observed_roots)
if expected == current
)
if matches > best_matches:
best_shift = shift
best_matches = matches
if best_matches < len(template) - 1:
return template
return template[best_shift:] + template[:best_shift]
def cyclic_root_similarity(observed: list[str], template: list[str]) -> float:
if len(observed) != len(template) or not observed:
return 0.0
observed_roots = [name_to_pitch_class(token) for token in observed]
template_roots = [name_to_pitch_class(token) for token in template]
if any(root is None for root in observed_roots) or any(root is None for root in template_roots):
return 0.0
best = 0
for shift in range(len(template_roots)):
rotated = template_roots[shift:] + template_roots[:shift]
best = max(
best,
sum(1 for expected, current in zip(rotated, observed_roots) if expected == current),
)
return best / max(len(template_roots), 1)
def choose_long_guitar_template(
progression_tokens: list[str],
templates: list[list[str]],
acoustic_candidates: list[list[ChordCandidate]],
auxiliary: str,
short_best_score: float,
) -> Optional[list[str]]:
if len(progression_tokens) >= 5:
return None
aux_tokens = [token for token in str(auxiliary or "").replace("|", " ").split() if token]
if len(aux_tokens) < 8 or len(acoustic_candidates) < 6:
return None
current_complexity = sum(1 for token in aux_tokens if extract_suffix(token).lower().startswith("sus"))
if current_complexity < 2 and len(dict.fromkeys(aux_tokens)) <= 4:
return None
best_template: Optional[list[str]] = None
best_score = 0.0
for template in templates:
if len(template) <= 4:
continue
alignment = segment_template_alignment_score(template, acoustic_candidates)
progression_score = harmonic_template_score(progression_tokens, template)
combined = (alignment * 0.75) + (progression_score * 0.25)
if combined > best_score:
best_score = combined
best_template = template
if best_template is None or best_score < 0.705:
return None
if best_score < short_best_score + 0.05:
return None
return best_template
def segment_template_alignment_score(
template: list[str],
acoustic_candidates: list[list[ChordCandidate]],
) -> float:
if not template or not acoustic_candidates:
return 0.0
n_segments = len(acoustic_candidates)
n_tokens = len(template)
dp = [[-1e9] * n_segments for _ in range(n_tokens)]
for token_index, token in enumerate(template):
for segment_index, candidates in enumerate(acoustic_candidates):
match_score = best_candidate_match_score(token, candidates)
if match_score <= 0.0:
continue
if token_index == 0:
dp[token_index][segment_index] = match_score
continue
best_prev = -1e9
for previous_segment in range(segment_index):
prev_score = dp[token_index - 1][previous_segment]
if prev_score <= -1e8:
continue
gap = segment_index - previous_segment
spacing_bonus = 0.16 if 1 <= gap <= 5 else max(-0.18, 0.12 - (gap * 0.035))
best_prev = max(best_prev, prev_score + match_score + spacing_bonus)
dp[token_index][segment_index] = best_prev
best = max(dp[-1]) if dp else -1e9
if best <= -1e8:
return 0.0
return max(0.0, min(1.0, best / max(float(n_tokens) * 1.22, 1.0)))
def best_candidate_match_score(template_token: str, candidates: list[ChordCandidate]) -> float:
target = simplify_stage_chord(template_token)
if not target:
return 0.0
target_root = name_to_pitch_class(target)
target_family = chord_family(extract_suffix(target))
best = 0.0
for candidate in candidates:
candidate_name = simplify_stage_chord(candidate.name)
if not candidate_name:
continue
candidate_root = name_to_pitch_class(candidate_name)
candidate_family = chord_family(extract_suffix(candidate_name))
support = logistic(candidate.segment_score * 0.9)
if candidate_name == target:
best = max(best, 0.78 + (support * 0.22))
continue
if candidate_root == target_root and candidate_family == target_family:
best = max(best, 0.62 + (support * 0.2))
continue
if candidate_root == target_root:
best = max(best, 0.42 + (support * 0.14))
return best
def harmonic_template_score(observed: list[str], template: list[str]) -> float:
observed_norm = [simplify_stage_chord(token) for token in observed if token]
template_norm = [simplify_stage_chord(token) for token in template if token]
if not observed_norm or not template_norm:
return 0.0
lcs = longest_common_subsequence_tokens(observed_norm, template_norm)
coverage = lcs / max(len(observed_norm), len(template_norm), 1)
interval_bonus = harmonic_interval_similarity(observed_norm, template_norm)
return (coverage * 0.7) + (interval_bonus * 0.3)
def longest_common_subsequence_tokens(a: list[str], b: list[str]) -> int:
dp = [[0] * (len(b) + 1) for _ in range(len(a) + 1)]
for i, left in enumerate(a, start=1):
for j, right in enumerate(b, start=1):
if left == right:
dp[i][j] = dp[i - 1][j - 1] + 1
else:
dp[i][j] = max(dp[i - 1][j], dp[i][j - 1])
return dp[-1][-1]
def harmonic_interval_similarity(observed: list[str], template: list[str]) -> float:
observed_roots = [name_to_pitch_class(token) for token in observed]
template_roots = [name_to_pitch_class(token) for token in template]
if any(root is None for root in observed_roots) or any(root is None for root in template_roots):
return 0.0
observed_intervals = [
int((int(observed_roots[index]) - int(observed_roots[index - 1])) % 12)
for index in range(1, len(observed_roots))
]
template_intervals = [
int((int(template_roots[index]) - int(template_roots[index - 1])) % 12)
for index in range(1, len(template_roots))
]
if not observed_intervals or not template_intervals:
return 0.0
length = min(len(observed_intervals), len(template_intervals))
distance = sum(
abs(observed_intervals[index] - template_intervals[index])
for index in range(length)
)
return max(0.0, 1.0 - (distance / max(length * 6, 1)))
def filter_stage_events(
items: list[dict[str, float | str]],
tonic: str,
mode: str,
) -> list[dict[str, float | str]]:
if len(items) < 3:
return items
tonic_pc = NOMES_NOTAS.index(tonic)
durations = [float(item["duracao"]) for item in items]
median_duration = float(np.median(durations)) if durations else 0.0
short_limit = max(1.1, median_duration * 0.95)
filtered: list[dict[str, float | str]] = []
for index, item in enumerate(items):
current = dict(item)
current_name = str(current["nome"])
current_duration = float(current["duracao"])
current_weight = float(current["peso"])
current_suffix = extract_suffix(current_name).lower()
previous = items[index - 1] if index > 0 else None
following = items[index + 1] if index + 1 < len(items) else None
should_drop = False
if previous is not None and following is not None:
previous_name = str(previous["nome"])
following_name = str(following["nome"])
previous_weight = float(previous["peso"])
following_weight = float(following["peso"])
if (
previous_name == following_name
and current_duration <= max(short_limit, min(float(previous["duracao"]), float(following["duracao"])) * 1.25)
and current_weight < ((previous_weight + following_weight) / 2.0)
):
should_drop = True
current_score = score_chord_for_key(current_name, tonic_pc, mode)
if (
not should_drop
and current_duration <= short_limit
and current_score < -0.15
and previous_weight > current_weight
and following_weight > current_weight
):
should_drop = True
if (
not should_drop
and current_duration <= short_limit * 0.85
and previous_weight >= current_weight * 3.0
and following_weight >= current_weight * 1.45
):
should_drop = True
if (
not should_drop
and current_duration <= short_limit * 0.9
and current_suffix.startswith("sus")
and previous_weight >= current_weight
and following_weight >= current_weight
):
should_drop = True
if not should_drop:
filtered.append(current)
return filtered or items
def build_windows(tokens: list[str], min_size: int, max_size: int) -> list[list[str]]:
windows: list[list[str]] = []
for size in range(min_size, max_size + 1):
if len(tokens) <= size:
continue
for start in range(0, len(tokens) - size + 1):
windows.append(tokens[start : start + size])
return windows
def stage_window_score(tokens: list[str], tonic: str, mode: str) -> float:
if not tokens:
return -999.0
unique = len(dict.fromkeys(tokens))
tonic_pc = NOMES_NOTAS.index(tonic)
score = 0.0
score += max(0.0, 4.5 - abs(unique - 4) * 1.3)
score += max(0.0, 5.5 - abs(len(tokens) - 4) * 0.9)
first_root = name_to_pitch_class(tokens[0])
last_root = name_to_pitch_class(tokens[-1])
if first_root == tonic_pc:
score += 1.0
if last_root == tonic_pc:
score += 1.2
if last_root == (tonic_pc + 7) % 12:
score += 0.35
score += sum(score_chord_for_key(token, tonic_pc, mode) for token in tokens) * 0.55
score += progression_pattern_bonus(tokens, tonic_pc, mode)
score -= sum(1 for prev, curr in zip(tokens, tokens[1:]) if prev == curr) * 0.4
return score
def simplify_stage_chord(name: str) -> str:
chord = str(name or "").strip()
if not chord:
return ""
if "/" in chord:
chord = chord.split("/", 1)[0]
return chord
def progression_pattern_bonus(tokens: list[str], tonic_pc: int, mode: str) -> float:
roots = [name_to_pitch_class(token) for token in tokens]
if any(root is None for root in roots):
return 0.0
sequence = [int((root - tonic_pc) % 12) for root in roots if root is not None]
patterns = (
[
[0, 9, 5, 7], # I vi IV V
[9, 5, 0, 7], # vi IV I V
[0, 7, 9, 5], # I V vi IV
[2, 5, 0, 7], # ii IV I V
]
if mode == "maior"
else [
[0, 8, 3, 10], # i VI III VII
[0, 3, 10, 7], # i III VII v
[0, 5, 8, 7], # i iv VI v
]
)
bonus = 0.0
for pattern in patterns:
if len(sequence) < len(pattern):
continue
if sequence[: len(pattern)] == pattern:
bonus = max(bonus, 1.25)
elif sequence[-len(pattern):] == pattern:
bonus = max(bonus, 1.05)
elif len(sequence) == len(pattern):
distance = sum(1 for current, expected in zip(sequence, pattern) if current != expected)
bonus = max(bonus, max(0.0, 1.0 - distance * 0.28))
if len(tokens) == 4 and len(dict.fromkeys(tokens)) == 4:
bonus += 0.2
return bonus
def optimize_stage_progression_for_strings(
progression: str,
auxiliary: str,
instrumento: str,
) -> tuple[str, str]:
if instrumento not in {"violao", "ukulele"}:
return progression, auxiliary
tokens = [token for token in progression.split() if token]
if len(tokens) != 4:
return progression, auxiliary
original_score = open_chord_score(tokens)
original_prior = best_progression_prior_score(tokens, instrumento)
best_tokens = tokens
best_shift = 0
best_score = original_score
best_prior = original_prior
for shift in range(-6, 7):
if shift == 0:
continue
transposed = [transpose_stage_chord(token, shift) for token in tokens]
score = open_chord_score(transposed)
prior = best_progression_prior_score(transposed, instrumento)
if prior > best_prior + 1e-6 or (abs(prior - best_prior) <= 1e-6 and score > best_score):
best_tokens = transposed
best_shift = shift
best_score = score
best_prior = prior
if best_shift == 0:
return progression, auxiliary
prior_gain = best_prior - original_prior
if best_prior < 0.9 and prior_gain < 0.25:
return progression, auxiliary
if best_score < original_score + 2.0 and prior_gain < 0.32:
return progression, auxiliary
optimized = " ".join(best_tokens)
optimized_aux = progression if not auxiliary else f"{progression} | {auxiliary}"
return optimized, optimized_aux
def open_chord_score(tokens: list[str]) -> float:
open_shapes = {"G", "C", "D", "A", "E", "Am", "Em", "Dm"}
score = 0.0
for token in tokens:
chord = simplify_stage_chord(token)
if chord in open_shapes:
score += 2.4
elif any(acc in chord for acc in {"#", "b"}):
score -= 1.6
elif chord.endswith("m"):
score -= 0.7
else:
score += 0.25
return score
def best_progression_prior_score(tokens: list[str], instrumento: str) -> float:
if instrumento not in {"violao", "ukulele"}:
return 0.0
normalized = [simplify_stage_chord(token) for token in tokens if token]
if len(normalized) != 4:
return 0.0
open_shapes = {"G", "C", "D", "A", "E", "Am", "Em", "Dm"}
open_bias = sum(1.0 for token in normalized if token in open_shapes) / 4.0
interval_bonus = 0.0
root_sequence = [name_to_pitch_class(token) for token in normalized]
if any(root is None for root in root_sequence):
return open_bias * 0.5
intervals = [
int((int(root_sequence[index]) - int(root_sequence[index - 1])) % 12)
for index in range(1, len(root_sequence))
]
common_patterns = [
[7, 2, 10], # G D Em C / I V vi IV
[9, 9, 2], # G Em C D / I vi IV V
[5, 7, 2], # vi IV I V
[5, 2, 5], # i VI III VII / i iv VI v simplificado em rotacao
]
for pattern in common_patterns:
distance = sum(abs(current - expected) for current, expected in zip(intervals, pattern))
interval_bonus = max(interval_bonus, max(0.0, 1.0 - (distance / 18.0)))
return round((open_bias * 0.62) + (interval_bonus * 0.38), 4)
def transpose_stage_chord(token: str, shift: int) -> str:
match = re.match(r"^([A-G](?:#|b)?)(.*)$", token.strip())
if not match:
return token
root = match.group(1)
suffix = match.group(2)
root_pc = name_to_pitch_class(root)
if root_pc is None:
return token
return f"{NOMES_NOTAS[(root_pc + shift) % 12]}{suffix}"
def summarize_events(events: list[dict[str, Any]]) -> list[str]:
summary: list[str] = []
for event in events:
name = str(event.get("nome", "")).strip()
if not name:
continue
if summary and summary[-1] == name:
continue
summary.append(name)
return summary
def name_to_pitch_class(name: str) -> Optional[int]:
token = str(name or "").strip()
if not token:
return None
base = token[0].upper()
if base not in "ABCDEFG":
return None
accidental = token[1] if len(token) > 1 and token[1] in {"#", "b"} else ""
note = f"{base}{accidental}"
flats = {
"Db": "C#",
"Eb": "D#",
"Gb": "F#",
"Ab": "G#",
"Bb": "A#",
"Cb": "B",
"Fb": "E",
}
note = flats.get(note, note)
try:
return NOMES_NOTAS.index(note)
except ValueError:
return None
def extract_suffix(name: str) -> str:
token = str(name or "").strip()
if not token:
return ""
accidental = token[1] if len(token) > 1 and token[1] in {"#", "b"} else ""
return token[1 + len(accidental) :].strip()
def build_intervals(values: list[int]) -> list[int]:
if len(values) < 2:
return []
return [int(values[index] - values[index - 1]) for index in range(1, len(values))]
def correlation_pearson(a: np.ndarray, b: np.ndarray) -> float:
if a.size != b.size or a.size == 0:
return 0.0
a_centered = a - float(np.mean(a))
b_centered = b - float(np.mean(b))
denominator = float(np.linalg.norm(a_centered) * np.linalg.norm(b_centered))
if denominator <= 1e-8:
return 0.0
return float(np.dot(a_centered, b_centered) / denominator)
def cosine_similarity(a: np.ndarray, b: np.ndarray) -> float:
denominator = float(np.linalg.norm(a) * np.linalg.norm(b))
if denominator <= 1e-8:
return 0.0
return float(np.dot(a, b) / denominator)
def pitch_class_distance(left: int, right: int) -> int:
distance = abs((int(left) % 12) - (int(right) % 12))
return int(min(distance, 12 - distance))
def normalize_vector(values: np.ndarray) -> np.ndarray:
vector = np.asarray(values, dtype=np.float32)
total = float(vector.sum())
if total <= 0:
return np.zeros_like(vector)
return vector / total
def round_list(values: list[float], places: int = 4) -> list[float]:
return [round(float(value), places) for value in values]
def logistic(value: float) -> float:
return float(1.0 / (1.0 + np.exp(-float(value))))