למטה חמישה קבצים, כולם בפייתון עם numpy בלבד. הקוד לא הורץ ולא נבדק על קבצים אמיתיים. שים אותם בתיקייה אחת, לצד korg_decoder.py ו-pcm_decoder.py שכבר יש ל-Codex.

שני דברים שחשוב ש-Codex ידע:

באג בקוד הקודם. ב-region_for_key וב-region_ranges התייחסתי ל-original_key כגבול תחתון של אזור. זה לא נכון: original_key הוא ה-root של הדגימה, וגבול תחתון של אזור הוא TopKey של האזור הקודם ועוד 1. הפונקציה build_key_table בקובץ 3 מתקנת את זה. צריך להפסיק להשתמש בשתי הפונקציות הישנות.
מה שלא מכוסה כאן. מיפוי צלילי תופים ב-Pa600 (איזה צליל על איזה מקש בתוך drum kit) שמור כנראה ב-PCG, לא ב-KMP. libgig לא מפענח PCG, ואין לי מפרט מאומת שלו. לכן ה-resolver קורא את המפה האמיתית רק מ-KMP. עבור drum kit הוא משתמש ב-GM רק כרמז, ולא דורס איתו את המיפוי. פענוח PCG הוא עדיין משימת reverse-engineering בנפרד, עם אותה מתודולוגיית corpus.
קובץ 1: tools/audio_dsp.py
"""Shared DSP helpers (numpy only)."""
import wave
import numpy as np


def load_wav(path):
    """PCM WAV (8/16/24/32-bit int) -> (mono float32 [-1,1], sample_rate)."""
    with wave.open(path, "rb") as w:
        ch, sw, sr = w.getnchannels(), w.getsampwidth(), w.getframerate()
        raw = w.readframes(w.getnframes())
    if sw == 1:
        x = (np.frombuffer(raw, np.uint8).astype(np.float32) - 128.0) / 128.0
    elif sw == 2:
        x = np.frombuffer(raw, "<i2").astype(np.float32) / 32768.0
    elif sw == 3:
        b = np.frombuffer(raw, np.uint8).reshape(-1, 3).astype(np.int32)
        v = b[:, 0] | (b[:, 1] << 8) | (b[:, 2] << 16)
        v = np.where(v & 0x800000, v - 0x1000000, v)
        x = v.astype(np.float32) / 8388608.0
    elif sw == 4:
        x = np.frombuffer(raw, "<i4").astype(np.float32) / 2147483648.0
    else:
        raise ValueError(f"unsupported sample width {sw}")
    return x.reshape(-1, ch).mean(axis=1), sr


def resample_fft(x, sr_in, sr_out):
    """FFT resampling (fine for analysis; not for final audio rendering)."""
    if sr_in == sr_out or len(x) == 0:
        return np.asarray(x, np.float32)
    n_out = int(round(len(x) * sr_out / sr_in))
    X = np.fft.rfft(x)
    m = n_out // 2 + 1
    Y = np.zeros(m, dtype=complex)
    k = min(m, len(X))
    Y[:k] = X[:k]
    return (np.fft.irfft(Y, n_out) * (n_out / len(x))).astype(np.float32)


def stft_mag(x, n_fft=2048, hop=512, center=True):
    """Magnitude STFT, shape (bins, frames). center=True -> frame i is centred on i*hop."""
    x = np.asarray(x, np.float32)
    if center:
        x = np.pad(x, (n_fft // 2, n_fft // 2))
    if len(x) < n_fft:
        x = np.pad(x, (0, n_fft - len(x)))
    frames = np.lib.stride_tricks.sliding_window_view(x, n_fft)[::hop]
    n = len(frames)
    win = np.hanning(n_fft).astype(np.float32)
    out = np.empty((n_fft // 2 + 1, n), np.float32)
    for s in range(0, n, 1024):
        blk = frames[s:s + 1024] * win
        out[:, s:s + len(blk)] = np.abs(np.fft.rfft(blk, axis=1)).T
    return out


def frame_rms(x, win, hop):
    n = max(0, (len(x) - win) // hop) + 1
    c = np.concatenate([[0.0], np.cumsum(np.asarray(x, np.float64) ** 2)])
    s = np.arange(n) * hop
    e = np.minimum(s + win, len(x))
    return np.sqrt((c[e] - c[s]) / np.maximum(e - s, 1))
קובץ 2: tools/sample_analyzer.py
"""Sample understanding: features + fuzzy role classifier with confidence.

IMPORTANT (honest scope):
- Features are measured facts. The role classifier is PROBABILISTIC.
- Probabilities are UNCALIBRATED until you run fit_temperature() on a labeled corpus.
- Anything flagged needs_review must go to a human. Never write it silently into a style.
"""
import numpy as np
from audio_dsp import resample_fft, frame_rms

TARGET_SR = 44100
BAND_EDGES = (20, 150, 500, 2000, 6000, 20000)


# ------------------------------------------------------------------ features
def _count_bursts(env_db, floor=-20.0, dip_db=3.0):
    """Number of distinct envelope peaks (claps have several in the first ~80 ms)."""
    count, last, valley = 0, None, 0.0
    for i in range(1, len(env_db) - 1):
        v = env_db[i]
        if last is not None:
            valley = min(valley, v)
        if v >= env_db[i - 1] and v > env_db[i + 1] and v > floor:
            if last is None or v - valley >= dip_db:
                count += 1
                last, valley = v, v
    return count


def estimate_pitch(x, sr, start, fmin=30.0, fmax=2000.0):
    """Autocorrelation f0. Returns (f0_hz | None, clarity 0..1)."""
    start = int(min(start, max(0, len(x) - 4096)))
    seg = x[start:start + 4096].astype(np.float64)
    if len(seg) < 1024:
        return None, 0.0
    seg = (seg - seg.mean()) * np.hanning(len(seg))
    n = len(seg)
    F = np.fft.rfft(seg, 2 * n)
    r = np.fft.irfft(F * np.conj(F))[:n]
    r = r / (r[0] + 1e-12)
    lmin, lmax = int(sr / fmax), min(int(sr / fmin), n - 2)
    if lmax <= lmin + 2:
        return None, 0.0
    seg_r = r[lmin:lmax]
    loc = np.nonzero((seg_r[1:-1] >= seg_r[:-2]) & (seg_r[1:-1] > seg_r[2:]))[0] + 1
    if len(loc) == 0:
        return None, 0.0
    best = seg_r[loc].max()
    lag = lmin + int(loc[seg_r[loc] >= 0.9 * best][0])   # smallest lag ~ best -> avoids sub-octave errors
    a, b, c = r[lag - 1], r[lag], r[lag + 1]
    denom = a - 2 * b + c
    off = 0.5 * (a - c) / denom if abs(denom) > 1e-12 else 0.0
    return float(sr / (lag + off)), float(np.clip(b, 0, 1))


def extract_features(x, sr):
    x = np.asarray(x, np.float32)
    if x.ndim > 1:
        x = x.mean(axis=1)
    peak = float(np.max(np.abs(x))) if len(x) else 0.0
    if peak < 1e-5:
        return None
    x = x / peak
    if sr != TARGET_SR:
        x = resample_fft(x, sr, TARGET_SR)
        sr = TARGET_SR
    nz = np.nonzero(np.abs(x) > 10 ** (-50 / 20))[0]
    x = x[nz[0]:][:3 * sr]

    win, hop = int(0.006 * sr), int(0.002 * sr)
    env = frame_rms(x, win, hop)
    env_db = 20 * np.log10(env / (env.max() + 1e-12) + 1e-9)
    pk, hop_ms = int(np.argmax(env)), 1000.0 * hop / sr

    def fall_ms(db):
        idx = np.nonzero(env_db[pk:] < -db)[0]
        return float((idx[0] if len(idx) else len(env_db) - pk) * hop_ms)

    seg = x[:int(0.1 * sr)]
    if len(seg) > 8:
        seg = seg * np.hanning(len(seg))
    P = np.abs(np.fft.rfft(seg, 8192)) ** 2
    fr = np.fft.rfftfreq(8192, 1 / sr)
    m_all = (fr >= 20) & (fr <= 20000)
    tot = P[m_all].sum() + 1e-12
    centroid = float((fr[m_all] * P[m_all]).sum() / tot)
    cs = np.cumsum(P[m_all])
    rolloff = float(fr[m_all][min(np.searchsorted(cs, 0.85 * cs[-1]), len(cs) - 1)])
    m_fl = (fr >= 100) & (fr <= 16000)
    Pm = P[m_fl] + 1e-12
    flatness = float(np.exp(np.mean(np.log(Pm))) / np.mean(Pm))
    ratios = [float(P[(fr >= lo) & (fr < hi)].sum() / tot)
              for lo, hi in zip(BAND_EDGES[:-1], BAND_EDGES[1:])]

    f0, clarity = estimate_pitch(x, sr, pk * hop + int(0.02 * sr))
    if clarity < 0.4:
        f0 = None
    return {
        "duration_ms": 1000.0 * len(x) / sr,
        "attack_ms": pk * hop_ms,
        "t20_ms": fall_ms(20), "t40_ms": fall_ms(40),
        "bursts": _count_bursts(env_db[:80]),
        "centroid": centroid, "rolloff85": rolloff, "flatness": flatness,
        "low_ratio": ratios[0], "lowmid_ratio": ratios[1], "mid_ratio": ratios[2],
        "high_ratio": ratios[3], "air_ratio": ratios[4],
        "f0": f0, "clarity": clarity,
        "midi": (69 + 12 * float(np.log2(f0 / 440.0))) if f0 else None,
    }


# ---------------------------------------------------------------- classifier
def trap(x, a, b, c, d):
    if x < a or x > d:
        return 0.0
    if x < b:
        return (x - a) / (b - a) if b > a else 1.0
    if x <= c:
        return 1.0
    return (d - x) / (d - c) if d > c else 1.0


# feature, (a,b,c,d) trapezoid, weight
RULES = {
    "kick": [("centroid", (30, 60, 500, 1500), 2), ("low_ratio", (0.15, 0.35, 1, 1.01), 2),
             ("air_ratio", (-0.01, 0, 0.05, 0.2), 1), ("t20_ms", (15, 40, 400, 900), 1)],
    "snare": [("centroid", (900, 1600, 5000, 8000), 2), ("mid_ratio", (0.1, 0.2, 0.7, 0.95), 1),
              ("flatness", (0.005, 0.03, 1, 1.01), 1.5), ("t20_ms", (20, 50, 350, 700), 1),
              ("clarity", (-1, -1, 0.5, 0.8), 1)],
    "clap": [("centroid", (900, 1300, 3500, 6000), 1.5), ("bursts", (1.5, 2, 10, 12), 3),
             ("t20_ms", (30, 60, 400, 800), 1), ("flatness", (0.01, 0.04, 1, 1.01), 1)],
    "rim": [("t20_ms", (2, 5, 60, 150), 2.5), ("centroid", (1200, 2000, 6000, 9000), 1.5),
            ("clarity", (0.2, 0.4, 1, 1.01), 1), ("duration_ms", (5, 10, 200, 400), 1)],
    "hat_closed": [("centroid", (3500, 6000, 14000, 20000), 2), ("air_ratio", (0.15, 0.35, 1, 1.01), 2),
                   ("t20_ms", (3, 10, 100, 200), 2.5), ("flatness", (0.02, 0.08, 1, 1.01), 1)],
    "hat_open": [("centroid", (3500, 6000, 14000, 20000), 2), ("air_ratio", (0.15, 0.35, 1, 1.01), 2),
                 ("t20_ms", (120, 250, 1200, 2500), 2.5), ("flatness", (0.02, 0.08, 1, 1.01), 1)],
    "crash": [("centroid", (2500, 4000, 12000, 18000), 1.5), ("t40_ms", (600, 1200, 6000, 9000), 2.5),
              ("flatness", (0.01, 0.05, 1, 1.01), 1)],
    "ride": [("centroid", (2000, 3000, 9000, 14000), 1.5), ("t40_ms", (300, 600, 3000, 5000), 2),
             ("clarity", (0.1, 0.3, 1, 1.01), 1)],
    "tom": [("clarity", (0.4, 0.6, 1, 1.01), 2), ("f0", (60, 80, 350, 500), 2),
            ("t20_ms", (60, 120, 700, 1400), 1), ("centroid", (100, 200, 1500, 3000), 1)],
    "perc_tonal": [("clarity", (0.4, 0.6, 1, 1.01), 2), ("f0", (200, 300, 1200, 1800), 2),
                   ("t20_ms", (20, 50, 400, 900), 1)],
    "shaker": [("centroid", (3000, 5000, 12000, 18000), 1.5), ("attack_ms", (8, 20, 120, 250), 2),
               ("flatness", (0.05, 0.15, 1, 1.01), 1.5), ("t20_ms", (30, 60, 300, 600), 1)],
    "bass_tonal": [("f0", (25, 35, 200, 300), 2.5), ("clarity", (0.5, 0.7, 1, 1.01), 2),
                   ("t20_ms", (200, 400, 6000, 10000), 1)],
    "melodic_tonal": [("clarity", (0.5, 0.7, 1, 1.01), 2.5), ("f0", (30, 60, 2000, 2001), 1),
                      ("duration_ms", (250, 500, 10000, 20000), 1.5)],
}
DRUM_ROLES = {"kick", "snare", "clap", "rim", "hat_closed", "hat_open", "crash",
              "ride", "tom", "shaker", "perc_tonal"}


def score_classes(f):
    out = {}
    for cls, rules in RULES.items():
        num = den = 0.0
        for feat, (a, b, c, d), w in rules:
            v = f.get(feat)
            m = 0.15 if v is None else trap(v, a, b, c, d)   # missing evidence = mild penalty
            num += w * np.log(max(m, 0.05))
            den += w
        out[cls] = float(np.exp(num / den))
    out["fx"] = 0.12   # fallback
    return out


def _softmax_log(scores, temperature):
    v = np.log(np.array(list(scores.values()))) / temperature
    v -= v.max()
    p = np.exp(v)
    return p / p.sum()


def classify(f, temperature=0.4):
    s = score_classes(f)
    names = list(s)
    p = _softmax_log(s, temperature)
    order = np.argsort(-p)
    top, second = order[0], order[1]
    fit, margin = s[names[top]], float(p[top] - p[second])
    return {
        "role": names[top], "prob": float(p[top]), "margin": margin, "fit": fit,
        "ranking": [(names[i], float(p[i])) for i in order[:3]],
        "needs_review": bool(p[top] < 0.55 or margin < 0.2 or fit < 0.35),
    }


def fit_temperature(feature_list, labels, grid=None):
    """Calibrate probabilities on a labeled corpus (minimises NLL). Do this before trusting prob."""
    grid = np.linspace(0.1, 2.0, 39) if grid is None else grid
    all_scores = [score_classes(f) for f in feature_list]
    best_T, best_nll = None, 1e18
    for T in grid:
        nll = 0.0
        for s, y in zip(all_scores, labels):
            p = _softmax_log(s, T)
            nll -= np.log(p[list(s).index(y)] + 1e-9)
        if nll < best_nll:
            best_T, best_nll = float(T), nll
    return best_T, best_nll / max(len(labels), 1)


def analyze_sample(x, sr, temperature=0.4):
    f = extract_features(x, sr)
    if f is None:
        return {"features": None, "classification": {"role": "silent", "prob": 1.0,
                                                     "needs_review": True}}
    return {"features": f, "classification": classify(f, temperature)}
קובץ 3: tools/keymap_resolver.py
"""KMP + KSF -> per-key map (FACT) + per-sample role (PROBABILISTIC) + root-key sanity check.

Facts come from RLP1 (region ranges, root key). Roles come from audio analysis.
GM drum map is only a HINT and never overrides the file.
"""
import os
import numpy as np
from korg_decoder import parse_kmp, parse_ksf
from pcm_decoder import decode_pcm
from sample_analyzer import analyze_sample, DRUM_ROLES

NOTE_NAMES = ["C", "C#", "D", "D#", "E", "F", "F#", "G", "G#", "A", "A#", "B"]
GM_DRUM = {35: "kick", 36: "kick", 37: "rim", 38: "snare", 39: "clap", 40: "snare",
           41: "tom", 42: "hat_closed", 43: "tom", 44: "hat_closed", 45: "tom", 46: "hat_open",
           47: "tom", 48: "tom", 49: "crash", 50: "tom", 51: "ride", 52: "crash", 53: "ride",
           54: "shaker", 55: "crash", 56: "perc_tonal", 57: "crash", 59: "ride",
           60: "perc_tonal", 61: "perc_tonal", 62: "perc_tonal", 63: "perc_tonal",
           64: "perc_tonal", 65: "perc_tonal", 66: "perc_tonal", 69: "shaker", 70: "shaker"}


def note_name(n, octave_offset=0):
    """C4 = 60 by default. Verify Korg's display convention on the real instrument."""
    return f"{NOTE_NAMES[n % 12]}{n // 12 - 1 + octave_offset}"


def build_key_table(regions):
    """Per spec: region start = previous region TopKey + 1. Returns list[128] of region idx | None.
    Regions MUST be in file order (expected ascending TopKey)."""
    table, lo = [None] * 128, 0
    for idx, r in enumerate(regions):
        hi = min(r.top_key, 127)
        for k in range(lo, hi + 1):
            table[k] = idx
        lo = max(lo, hi + 1)
    return table


def find_ksf(directory, base):
    base = base.strip().upper()
    for fn in os.listdir(directory):
        stem, ext = os.path.splitext(fn)
        if ext.upper() == ".KSF" and stem.upper() == base:
            return os.path.join(directory, fn)
    return None


def resolve_kmp(kmp_path, ksf_dir, keep_audio=True):
    with open(kmp_path, "rb") as f:
        kmp = parse_kmp(f.read())
    warnings = []
    tops = [r.top_key for r in kmp.regions]
    if any(b < a for a, b in zip(tops, tops[1:])):
        warnings.append("TopKey not ascending in file order: verify RLP1 ordering against a real file")

    rows, lo = [], 0
    for i, r in enumerate(kmp.regions):
        hi = min(r.top_key, 127)
        row = {"index": i, "key_lo": lo, "key_hi": hi, "root": r.original_key, "tune": r.tune,
               "file": r.sample_file_name, "status": "ok", "shadowed": hi < lo}
        lo = max(lo, hi + 1)
        name = r.sample_file_name
        path = None if name.startswith(("SKIPPEDSAMPL", "INTERNAL")) else find_ksf(ksf_dir, name)
        if path is None:
            row["status"] = "special_or_missing"
            rows.append(row)
            continue
        with open(path, "rb") as f:
            ksf = parse_ksf(f.read())
        dec = decode_pcm(ksf)
        row["sample_rate"] = ksf.sample_rate
        if dec.is_compressed:
            row["status"] = "compressed_unsupported"     # never guess a proprietary codec
            rows.append(row)
            continue
        mono = dec.samples.mean(axis=1)
        ana = analyze_sample(mono, ksf.sample_rate)
        row.update(features=ana["features"], classification=ana["classification"])
        f = ana["features"]
        if f and f.get("midi") is not None and f["clarity"] >= 0.6:
            diff = f["midi"] - r.original_key
            octs = round(diff / 12)
            row["root_check"] = {
                "estimated_midi": round(f["midi"], 2), "diff_semitones": round(diff, 2),
                "ok": abs(diff) <= 0.6,
                "octave_error": bool(octs != 0 and abs(diff - 12 * octs) <= 0.6)}
        if keep_audio:
            row["_audio"] = (mono, ksf.sample_rate)
        rows.append(row)

    valid = [w for w in rows if w["key_hi"] >= w["key_lo"] and "classification" in w]
    widths = [w["key_hi"] - w["key_lo"] + 1 for w in valid]
    drum_frac = (np.mean([w["classification"]["role"] in DRUM_ROLES for w in valid])
                 if valid else 0.0)
    kind = "drum_kit_like" if widths and np.median(widths) <= 2 and drum_frac >= 0.6 else "multisample"
    if kind == "drum_kit_like":
        for w in valid:
            hint = GM_DRUM.get(w["key_lo"])
            w["gm_hint"] = hint
            w["gm_agrees"] = hint == w["classification"]["role"]
    return {"name": kmp.name16, "kind": kind, "regions": rows,
            "key_table": build_key_table(kmp.regions), "warnings": warnings}


def role_for_key(resolved, key):
    idx = resolved["key_table"][key]
    if idx is None:
        return None
    r = resolved["regions"][idx]
    c = r.get("classification")
    return {"region": idx, "file": r["file"], "role": c["role"] if c else None,
            "needs_review": c["needs_review"] if c else True}
קובץ 4: tools/music_analyzer.py
"""Music decoding: onsets, tempo, beats, meter/downbeat, template-NMF drum transcription
(using the KIT'S OWN samples as templates), then Canonical Pattern vs Groove separation.

Canonical pattern = the musical grid the part sits on.
Groove = how far each real hit deviates from it (kept separately, in ticks).
"""
import numpy as np
from audio_dsp import stft_mag, resample_fft, frame_rms

SR, N_FFT, HOP, PPQ = 44100, 2048, 256, 480
FMAX_BIN = int(16000 / (SR / N_FFT)) + 1
FPS = SR / HOP


def prepare(x, sr):
    x = np.asarray(x, np.float32)
    if x.ndim > 1:
        x = x.mean(axis=1)
    if sr != SR:
        x = resample_fft(x, sr, SR)
    m = float(np.max(np.abs(x))) if len(x) else 0.0
    return x / m if m > 0 else x


# ----------------------------------------------------------- onset detection
def onset_strength(mag, bands=((0, 150), (150, 2000), (2000, 20000))):
    freqs = np.fft.rfftfreq(N_FFT, 1 / SR)
    L = np.log1p(100.0 * mag / (mag.max() + 1e-9))
    d = np.pad(np.maximum(0, L[:, 1:] - L[:, :-1]), ((0, 0), (1, 0)))
    per = []
    for lo, hi in bands:
        e = d[(freqs >= lo) & (freqs < hi)].sum(axis=0)
        per.append(e / (np.percentile(e, 95) + 1e-9))
    return per, sum(per) / len(per)


def adaptive_threshold(env, fps=FPS, win_s=0.8, delta=0.5):
    w = max(3, int(win_s * fps))
    m = np.convolve(np.pad(env, (w // 2, w // 2), mode="edge"), np.ones(w) / w, "valid")[:len(env)]
    return m + delta * env.std()


def pick_peaks(env, min_dist, thresh):
    n = len(env)
    thr = np.broadcast_to(np.asarray(thresh, float), (n,))
    mask = (env[1:-1] >= env[:-2]) & (env[1:-1] > env[2:]) & (env[1:-1] > thr[1:-1])
    cand = sorted((np.nonzero(mask)[0] + 1).tolist(), key=lambda i: -env[i])
    used, taken = np.zeros(n, bool), []
    for i in cand:
        if not used[max(0, i - min_dist):i + min_dist + 1].any():
            taken.append(i)
            used[i] = True
    return np.array(sorted(taken), int)


def refine_times(x, times, back=0.03, fwd=0.01, win=64, hop=16):
    """Snap STFT-resolution onset times to the steepest RMS rise (~1 ms accuracy)."""
    env = frame_rms(x, win, hop)
    d = np.diff(env, prepend=env[0])
    out = []
    for t in times:
        c = int(t * SR / hop)
        lo, hi = max(0, c - int(back * SR / hop)), min(len(d), c + int(fwd * SR / hop) + 1)
        if hi <= lo:
            out.append(float(t))
            continue
        j = lo + int(np.argmax(d[lo:hi]))
        out.append((j * hop + win / 2) / SR)
    return np.array(out)


# --------------------------------------------------------------------- tempo
def estimate_tempo(env, fps=FPS, bpm_lo=55, bpm_hi=190, prior=120.0, sigma_oct=0.9, top=3):
    e = env - env.mean()
    n = len(e)
    F = np.fft.rfft(e, 2 * n)
    ac = np.fft.irfft(F * np.conj(F))[:n]
    ac = ac / (ac[0] + 1e-12)
    grid = np.arange(bpm_lo * 0.9, bpm_hi * 1.1, 0.25)
    lag = 60 * fps / grid
    idx = np.arange(n)
    A = lambda l: np.interp(l, idx, ac, right=0.0)
    S = A(lag) + 0.5 * A(2 * lag) + 0.25 * A(4 * lag)
    S = np.maximum(S, 0) * np.exp(-0.5 * (np.log2(grid / prior) / sigma_oct) ** 2)
    S[(grid < bpm_lo) | (grid > bpm_hi)] = 0
    loc = np.nonzero((S[1:-1] >= S[:-2]) & (S[1:-1] > S[2:]))[0] + 1
    loc = sorted(loc.tolist(), key=lambda i: -S[i])
    cands = []
    for i in loc:
        if all(abs(grid[i] / c["bpm"] - 1) > 0.04 for c in cands):
            cands.append({"bpm": float(grid[i]), "strength": float(S[i] / (S[loc[0]] + 1e-12))})
        if len(cands) >= top:
            break
    return cands   # cands[0] best; ambiguity is visible in the others (half/double time)


def track_beats(env, fps, bpm, tightness=100.0):
    """Ellis (2007) dynamic-programming beat tracker. Returns beat frame indices."""
    period = 60 * fps / bpm
    n = len(env)
    o = env / (env.std() + 1e-9)
    score, back = o.copy(), -np.ones(n, int)
    w0, w1 = int(round(period / 2)), int(round(period * 2))
    for t in range(n):
        hi, lo = t - w0, max(t - w1, 0)
        if hi < lo:
            continue
        prev = np.arange(lo, hi + 1)
        cand = score[prev] - tightness * np.log((t - prev) / period) ** 2
        j = int(np.argmax(cand))
        if cand[j] > 0:
            score[t] = o[t] + cand[j]
            back[t] = prev[j]
    loc = np.nonzero((score[1:-1] >= score[:-2]) & (score[1:-1] > score[2:]))[0] + 1
    if len(loc) == 0:
        return np.array([], int)
    keep = loc[score[loc] > 0.5 * np.median(score[loc])]
    t = int(keep[-1]) if len(keep) else int(loc[-1])
    beats = [t]
    while back[t] >= 0:
        t = int(back[t])
        beats.append(t)
    return np.array(beats[::-1], int)


def estimate_meter(beat_frames, env_low, env_all, candidates=(4, 3), prior=None):
    """Meter + downbeat phase from beat-synchronous low-band/overall accents. 4/4 vs 2/4 vs 6/8
    can be ambiguous: use confidence and fall back to human review when it is low."""
    prior = prior or {4: 1.1, 3: 1.0}
    nb = len(beat_frames)
    if nb < 8:
        return {"meter": 4, "downbeat_idx": 0, "confidence": 0.0}

    def local(env):
        return np.array([env[max(0, f - 2):f + 3].max() for f in beat_frames])

    s = 0.6 * local(env_low) / (env_low.std() + 1e-9) + 0.4 * local(env_all) / (env_all.std() + 1e-9)
    s = (s - s.mean()) / (s.std() + 1e-9)
    res = []
    for m in candidates:
        if nb < 2 * m:
            continue
        for ph in range(m):
            mask = np.zeros(nb, bool)
            mask[ph::m] = True
            res.append(((s[mask].mean() - s[~mask].mean()) * prior.get(m, 1.0), m, ph))
    if not res:
        return {"meter": 4, "downbeat_idx": 0, "confidence": 0.0}
    res.sort(reverse=True)
    second = res[1][0] if len(res) > 1 else 0.0
    conf = float(np.clip((res[0][0] - second) / (abs(res[0][0]) + 1e-9), 0, 1))
    return {"meter": res[0][1], "downbeat_idx": res[0][2], "confidence": conf}


# ------------------------------------------------- drum transcription (NMF)
def _v(mag):
    return np.power(mag / (mag.max() + 1e-9), 0.7)[:FMAX_BIN]


def attack_template(x, sr, attack_ms=30):
    x = prepare(x, sr)
    nz = np.nonzero(np.abs(x) > 10 ** (-45 / 20))[0]
    if len(nz) == 0:
        return None
    x = x[nz[0]:][:int(0.3 * SR)]
    V = _v(stft_mag(x, N_FFT, HOP, center=True))
    n = max(2, int(attack_ms / 1000 * SR / HOP) + N_FFT // (2 * HOP))
    t = V[:, :n].mean(axis=1)
    return t / (t.sum() + 1e-9)


def build_templates(kit):
    """kit: [{'role': str, 'x': mono float array, 'sr': int}] -> (roles, W[F, K])."""
    roles, cols = [], []
    for r in sorted({k["role"] for k in kit}):
        ts = [attack_template(k["x"], k["sr"]) for k in kit if k["role"] == r]
        ts = [t for t in ts if t is not None]
        if ts:
            t = np.mean(ts, axis=0)
            roles.append(r)
            cols.append(t / (t.sum() + 1e-9))
    return roles, np.stack(cols, axis=1).astype(np.float32)


def nmf_activations(V, W_fixed, n_free=4, iters=60, seed=0):
    """KL-NMF with FIXED kit templates + a few free components that absorb bass/keys/vocals."""
    rng = np.random.default_rng(seed)
    F, T = V.shape
    K = W_fixed.shape[1]
    Wf = rng.random((F, n_free)).astype(np.float32) + 0.1
    Wf /= Wf.sum(axis=0, keepdims=True)
    W = np.concatenate([W_fixed, Wf], axis=1)
    H = (W_fixed.T @ V) / (W_fixed.sum(axis=0)[:, None] + 1e-9)
    H = np.concatenate([H, rng.random((n_free, T)).astype(np.float32) + 0.1], axis=0) + 1e-6
    eps = 1e-9
    for _ in range(iters):
        R = V / (W @ H + eps)
        H *= (W.T @ R) / (W.sum(axis=0)[:, None] + eps)
        if n_free:
            R = V / (W @ H + eps)
            Wn = W[:, K:] * (R @ H[K:].T) / (H[K:].sum(axis=1)[None, :] + eps)
            W[:, K:] = Wn / (Wn.sum(axis=0, keepdims=True) + eps)
    return H[:K]


def transcribe_drums(x, roles, W, min_gap=0.03, rel_thresh=0.12, iters=60):
    V = _v(stft_mag(x, N_FFT, HOP, center=True)).astype(np.float32)
    H = nmf_activations(V, W, iters=iters)
    events, min_dist = [], max(1, int(min_gap * SR / HOP))
    for k, role in enumerate(roles):
        h = H[k]
        ref99 = np.percentile(h, 99.5)
        if ref99 <= 1e-9:
            continue
        pk = pick_peaks(h, min_dist, rel_thresh * ref99)
        if len(pk) == 0:
            continue
        times = refine_times(x, pk * HOP / SR)
        ref = np.percentile(h[pk], 95) + 1e-9
        for t, a in zip(times, h[pk]):
            events.append({"role": role, "time": float(t), "strength": float(a),
                           "velocity": int(np.clip(round(127 * (a / ref) ** 0.6), 1, 127))})
    return sorted(events, key=lambda e: e["time"])


# ---------------------------------------------- timing: canonical vs groove
def beat_position(t, beat_times):
    b = np.asarray(beat_times)
    i = int(np.clip(np.searchsorted(b, t) - 1, 0, len(b) - 2))
    return i + (t - b[i]) / (b[i + 1] - b[i])


def fit_grid(pos, divisions=(2, 3, 4, 6, 8, 12), n_ok=0.4, frac_ok=0.85):
    """Simplest grid (divisions per beat) explaining >=85% of hits. Normalised error 0=on grid, 1=midway."""
    best = None
    for d in divisions:
        n = np.abs(pos - np.round(pos * d) / d) * d * 2
        frac = float(np.mean(n <= n_ok))
        if best is None or frac > best[1]:
            best = (d, frac)
        if frac >= frac_ok:
            return d, frac
    return best   # nothing fit well: caller must lower confidence / ask for review


def quantize_events(events, beat_times, meter=4, downbeat_idx=0):
    if len(events) == 0 or len(beat_times) < 3:
        return [], {"division": None, "grid_fit": 0.0, "steps_per_bar": None}
    pos = np.array([beat_position(e["time"], beat_times) for e in events])
    d, fit = fit_grid(pos)
    q = np.round(pos * d) / d
    spb = meter * d
    out = []
    for e, p, qq in zip(events, pos, q):
        rel = int(round(qq * d)) - downbeat_idx * d
        err = abs(p - qq) * d * 2
        out.append({**e, "pos_beats": float(p), "bar": rel // spb, "step": rel % spb,
                    "tick": int(round(qq * PPQ)),
                    "groove_offset_ticks": int(round((p - qq) * PPQ)),
                    "grid_confidence": float(max(0.0, 1.0 - err))})
    return out, {"division": d, "grid_fit": fit, "steps_per_bar": spb}


def fold_pattern(qev, periods=(1, 2, 4)):
    """Fold bars into a repeating loop (majority vote). Returns canonical pattern + consistency."""
    bars = {}
    for e in qev:
        if e["bar"] >= 0:
            bars.setdefault(e["bar"], {})[(e["step"], e["role"])] = e["velocity"]
    keys = sorted(bars)
    if len(keys) < 2:
        return {"period_bars": 1, "consistency": 0.0, "pattern": []}
    results = []
    for p in periods:
        if len(keys) < 2 * p:
            continue
        groups = {g: [b for b in keys if b % p == g] for g in range(p)}
        canon, sims = {}, []
        for g, bl in groups.items():
            cnt = {}
            for b in bl:
                for k in bars[b]:
                    cnt[k] = cnt.get(k, 0) + 1
            canon[g] = {k for k, c in cnt.items() if c * 2 > len(bl)}
        for b in keys:
            a, c = set(bars[b]), canon[b % p]
            sims.append(len(a & c) / max(len(a | c), 1))
        results.append((p, float(np.mean(sims)), canon))
    if not results:
        return {"period_bars": 1, "consistency": 0.0, "pattern": []}
    top = max(r[1] for r in results)
    p, sim, canon = next(r for r in results if r[1] >= top - 0.03)   # smallest good period
    pattern = []
    for g in range(p):
        for (step, role) in sorted(canon[g]):
            vels = [bars[b][(step, role)] for b in keys if b % p == g and (step, role) in bars[b]]
            pattern.append({"bar_in_loop": g, "step": step, "role": role,
                            "velocity": int(np.median(vels))})
    return {"period_bars": p, "consistency": sim, "pattern": pattern}


# ------------------------------------------------------------------ pipeline
def analyze_song(x, sr, kit=None):
    x = prepare(x, sr)
    mag = stft_mag(x, N_FFT, HOP, center=True)
    bands, env = onset_strength(mag)
    tempo = estimate_tempo(env)
    beats_f = track_beats(env, FPS, tempo[0]["bpm"])
    beat_times = beats_f * HOP / SR
    bpm = float(60.0 / np.median(np.diff(beat_times))) if len(beat_times) > 3 else tempo[0]["bpm"]
    consistency = (float(np.mean(np.abs(np.diff(beat_times) / np.median(np.diff(beat_times)) - 1) < 0.08))
                   if len(beat_times) > 3 else 0.0)
    meter = estimate_meter(beats_f, bands[0], env)

    onsets = pick_peaks(env, max(1, int(0.03 * FPS)), adaptive_threshold(env))
    onset_times = refine_times(x, onsets * HOP / SR)
    coarse = [["low_hit", "mid_hit", "high_hit"][int(np.argmax([b[i] for b in bands]))] for i in onsets]

    result = {"tempo_candidates": tempo, "bpm": bpm, "beat_regularity": consistency,
              "beat_times": beat_times.tolist(), "meter": meter,
              "onsets": [{"time": float(t), "band": c} for t, c in zip(onset_times, coarse)]}
    if kit:
        roles, W = build_templates(kit)
        events = transcribe_drums(x, roles, W)
        q, info = quantize_events(events, beat_times, meter["meter"], meter["downbeat_idx"])
        result.update(drum_events=q, grid=info, loop=fold_pattern(q))
    return result


# ---------------------------------------------------------------- MIDI export
def _vlq(n):
    out = [n & 0x7F]
    n >>= 7
    while n:
        out.append((n & 0x7F) | 0x80)
        n >>= 7
    return bytes(reversed(out))


def write_midi(path, events, bpm, role_to_note, meter=4, use_groove=False, channel=9, note_len=60):
    ev = []
    for e in events:
        note = role_to_note.get(e["role"])
        if note is None:
            continue
        tick = max(0, e["tick"] + (e["groove_offset_ticks"] if use_groove else 0))
        ev.append((tick, 1, 0x90 | channel, note, e["velocity"]))
        ev.append((tick + note_len, 0, 0x80 | channel, note, 0))
    ev.sort(key=lambda z: (z[0], z[1]))            # note-off before note-on on the same tick
    trk = bytearray(b"\x00\xFF\x51\x03" + int(60_000_000 / bpm).to_bytes(3, "big"))
    trk += b"\x00\xFF\x58\x04" + bytes([meter, 2, 24, 8])
    last = 0
    for tick, _, st, n, v in ev:
        trk += _vlq(tick - last) + bytes([st, n, v])
        last = tick
    trk += b"\x00\xFF\x2F\x00"
    hdr = b"MThd" + (6).to_bytes(4, "big") + (0).to_bytes(2, "big") + (1).to_bytes(2, "big") + PPQ.to_bytes(2, "big")
    with open(path, "wb") as f:
        f.write(hdr + b"MTrk" + len(trk).to_bytes(4, "big") + bytes(trk))
קובץ 5: tools/m2s_pipeline.py
#!/usr/bin/env python3
"""CLI glue.
  python m2s_pipeline.py set  <file.kmp> <ksf_dir>
  python m2s_pipeline.py song <song.wav> <file.kmp> <ksf_dir> [out.mid]
"""
import sys
import json
from audio_dsp import load_wav
from keymap_resolver import resolve_kmp, note_name
from sample_analyzer import DRUM_ROLES
from music_analyzer import analyze_song, write_midi


def print_set(res):
    print(f"KMP '{res['name']}'  kind={res['kind']}")
    for w in res["warnings"]:
        print("  WARNING:", w)
    for r in res["regions"]:
        c = r.get("classification")
        line = (f"  [{note_name(r['key_lo']):>4}..{note_name(r['key_hi']):<4}] root={r['root']:3d} "
                f"{r['file']:<12} {r['status']}")
        if c:
            line += f"  role={c['role']} p={c['prob']:.2f}" + ("  REVIEW" if c["needs_review"] else "")
        if r.get("root_check") and not r["root_check"]["ok"]:
            line += f"  ROOT_MISMATCH({r['root_check']['diff_semitones']:+.2f}st)"
        if r.get("gm_agrees") is False:
            line += f"  gm_hint={r['gm_hint']}"
        print(line)


def kit_from_set(res, min_prob=0.5):
    kit, role_to_key = [], {}
    for r in res["regions"]:
        c = r.get("classification")
        if c and c["role"] in DRUM_ROLES and c["prob"] >= min_prob and "_audio" in r:
            x, sr = r["_audio"]
            kit.append({"role": c["role"], "x": x, "sr": sr})
            role_to_key.setdefault(c["role"], r["key_lo"])
    return kit, role_to_key


def main():
    mode = sys.argv[1]
    if mode == "set":
        print_set(resolve_kmp(sys.argv[2], sys.argv[3]))
    elif mode == "song":
        x, sr = load_wav(sys.argv[2])
        res = resolve_kmp(sys.argv[3], sys.argv[4])
        kit, role_to_key = kit_from_set(res)
        out = analyze_song(x, sr, kit)
        print(f"bpm={out['bpm']:.2f}  regularity={out['beat_regularity']:.2f}  "
              f"meter={out['meter']}  candidates={out['tempo_candidates']}")
        if "loop" in out:
            print(f"grid={out['grid']}  loop period={out['loop']['period_bars']} bars  "
                  f"consistency={out['loop']['consistency']:.2f}")
            print(json.dumps(out["loop"]["pattern"][:40], indent=1))
        if len(sys.argv) > 5:
            write_midi(sys.argv[5], out.get("drum_events", []), out["bpm"], role_to_key,
                       meter=out["meter"]["meter"])


if __name__ == "__main__":
    main()
מה זה נותן ל-Codex:

הבנת הדגימה. נמדדים attack, זמני דעיכה, מספר הפרצים (ל-clap), centroid, flatness, פיזור אנרגיה ב-5 תחומי תדר ו-f0 עם clarity. על בסיסם יש סיווג לתפקיד (kick, snare, hats, tom ועוד) עם הסתברות, margin ודגל needs_review. שדה שלא ניתן למדוד נחשב מעט לרעת הסיווג, ולא מומצא.
מיקום במקלדת. הטווחים וה-root נקראים מ-RLP1 כעובדה. בנוסף יש בדיקת root: pitch מוער מול OriginalKey, כולל זיהוי טעות אוקטבה. עבור drum kit, GM משמש רק כרמז.
הבנת המוזיקה. onsets רב-תחומיים מדויקים לכ-1ms, טמפו עם מועמדים חלופיים (חצי וכפול), beat tracking (DP של Ellis), מטר ו-downbeat עם confidence.
תמלול תופים איכותי. ה-NMF משתמש בדגימות ה-kit האמיתי כתבניות, והרכיבים החופשיים סופגים בס וקלידים. זה בדרך כלל מדויק יותר מ-ADT גנרי, אבל צריך למדוד את זה מול הקלטות אמיתיות. צליל קרוב (closed hat מול open hat) עלול להתבלבל.
תבנית קאנונית מול groove. נבחר הגריד הפשוט ביותר שמסביר לפחות 85% מהאירועים, ואז מפרידים בין המיקום הקאנוני להסטיית ה-groove בטיקים. הפאטרן מקופל ללולאה של 1, 2 או 4 תיבות, עם consistency. יש גם ייצוא MIDI.
מגבלות שכדאי לומר ל-Codex במפורש:

prob לא מכויל עד שמריצים fit_temperature על corpus מתויג.
דגימות דחוסות (compressed_unsupported) לא מנותחות, כי אין מפרט קודק.
בסיס ה-NMF מתאים לקבצים של דקות בודדות. לשירים ארוכים צריך לעבד בחלקים.
מטר של 4/4 מול 2/4 או 6/8 יכול להיות דו-משמעי. כש-meter.confidence נמוך צריך review אנושי.
הצעד הבא הוא להריץ python m2s_pipeline.py set על SET אמיתי, ולתייג ידנית 30-50 דגימות כדי לכייל את fit_temperature ולכוון את ה-RULES.

8 minutes ago



