From b1e2d532d8d0f3fcfd17f581e5fd3381047d4392 Mon Sep 17 00:00:00 2001 From: "prompt.ac/@jeffrey" Date: Sat, 5 Sep 2026 07:23:58 -0400 Subject: [PATCH] pop single-study-toolkit: read a finished single from the outside in MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit study/study.py measures a mastered track in four layers β€” master (LUFS/LRA/crest/stereo/tilt), structure (self-similarity -> section letters + phrase tier), arrangement (six-band energy, percussive share), harmony (key + dominant voice) β€” and study/compare.py lines several reports up. Registered as analysis.single-study / analysis.study-compare in the pop menu, with a critique-bench section in SCORE.md. study/out/ is gitignored; third-party audio never lands in the repo. Claude-Session: https://claude.ai/code/session_01LaRrW1RHanS8it7SzgrqeV --- pop/MENU-WISHLIST.md | 10 + pop/SCORE.md | 24 +++ pop/lib/menu.mjs | 10 + pop/study/.gitignore | 1 + pop/study/README.md | 52 +++++ pop/study/compare.py | 143 +++++++++++++ pop/study/study.py | 495 +++++++++++++++++++++++++++++++++++++++++++ 7 files changed, 735 insertions(+) create mode 100644 pop/study/.gitignore create mode 100644 pop/study/README.md create mode 100644 pop/study/compare.py create mode 100644 pop/study/study.py diff --git a/pop/MENU-WISHLIST.md b/pop/MENU-WISHLIST.md index e199f3023a..b906cc68fb 100644 --- a/pop/MENU-WISHLIST.md +++ b/pop/MENU-WISHLIST.md @@ -38,6 +38,16 @@ All from Abe's 2026-05-21 list. detector with hysteresis + retrigger lockout; beatbox a mic to fire drum samples. β†’ `lib/analysis.mjs:audioGate` +- 🟒 **single study** (`analysis.single-study`) β€” read a finished single + from the outside in: master (LUFS/LRA/crest/stereo), structure + (self-similarity β†’ section letters + phrase tier), arrangement + (six-band energy, percussive share), harmony (key, dominant voice). + Emits `report.json` + `REPORT.md` + four figures. + β†’ `study/study.py` (docs: `study/README.md`) +- 🟒 **study compare** (`analysis.study-compare`) β€” line up several + study reports: section timelines, band-balance heatmap, loudness + small multiples, `COMPARISON.md` stat table. + β†’ `study/compare.py` ### score β€” notation & sequencing - 🟒 **audio β†’ rhythm sequence** (`score.audio-to-rhythm`) β€” onset-detect diff --git a/pop/SCORE.md b/pop/SCORE.md index 196589a590..255213a2e1 100644 --- a/pop/SCORE.md +++ b/pop/SCORE.md @@ -191,6 +191,30 @@ JS twin across 76 rows, and both tracks re-render **byte-identical** WAVs. `acidtek.c` was left alone β€” its "off-beatness" is Barlow indispensability, not Toussaint's measure, and its comment now says so. +## Shared tooling β€” single study (critical listening) + +`pop/study/` is the single-study-toolkit: static analysis of a **finished** +single, read from the outside in the way a listener meets it β€” the mastered +object (LUFS, loudness range, crest, stereo image, spectral tilt), then its +shape in time (self-similarity β†’ section letters + a phrase tier), then who +is playing when (six-band energy, percussive share, onset density), then +what the notes are (key global + per-section, dominant voice). One command +per track, one to compare a shelf of them: + +```fish +cd pop +.venv/bin/python study/study.py track.mp3 --out study/out/slug --title T --artist A +.venv/bin/python study/compare.py study/out/*/report.json --out study/out/comparison +``` + +This is the mill's critique bench: run it on our own masters next to work we +admire and let the deltas argue. First use (oskie's *One Step* vs. four /pop +releases) is written up in `papers/study-one-step/`. Registered in the pop +menu as `analysis.single-study` / `analysis.study-compare`; honesty limits +(approximate LRA/true-peak, per-track section letters, streaming-rip air +band) live in `study/README.md`. Study inputs and outputs stay out of git β€” +`study/out/` is ignored, and third-party audio never gets committed. + ## Sample sources (commercial-safe) Tracks ship to DistroKid/Spotify, so any sourced audio MUST be CC0 / diff --git a/pop/lib/menu.mjs b/pop/lib/menu.mjs index 01e784f0fc..cb2c90429f 100644 --- a/pop/lib/menu.mjs +++ b/pop/lib/menu.mjs @@ -142,6 +142,16 @@ export const MENU = { params: ["threshold", "attackMs", "releaseMs", "minGapMs"], blurb: "amplitude onset trigger β€” beatbox a mic to fire samples", }, + "single-study": { + file: "study/study.py", + params: ["out", "title", "artist"], + blurb: "finished single, outside in β€” master/structure/arrangement/harmony report + figures", + }, + "study-compare": { + file: "study/compare.py", + params: ["out"], + blurb: "several study reports side by side β€” timelines, band balance, loudness", + }, }, vocal: { diff --git a/pop/study/.gitignore b/pop/study/.gitignore new file mode 100644 index 0000000000..89f9ac04aa --- /dev/null +++ b/pop/study/.gitignore @@ -0,0 +1 @@ +out/ diff --git a/pop/study/README.md b/pop/study/README.md new file mode 100644 index 0000000000..dfa1e26c26 --- /dev/null +++ b/pop/study/README.md @@ -0,0 +1,52 @@ +# single-study-toolkit + +Static analysis for finished singles β€” study a track **from the outside in**, +the way a listener meets it: the mastered object first, then its shape in +time, then who is playing when, then what the notes are. + +Registered in the pop menu (`lib/menu.mjs`) as `analysis.single-study` and +`analysis.study-compare`; the critique-bench posture lives in `SCORE.md` +under *Shared tooling β€” single study*. + +| layer | name | what it measures | +|---|---|---| +| L0 | master | LUFS, LRA, crest, true peak, stereo image, spectral tilt | +| L1 | structure | tempo, beat grid, self-similarity, section letters | +| L2 | arrangement | six-band energy over time, harmonic/percussive, onsets | +| L3 | harmony | chroma, global + per-section key, dominant-voice pitch | + +## use + +```fish +cd pop +.venv/bin/python study/study.py path/to/track.mp3 \ + --out study/out/track-slug --title "One Step" --artist oskie +``` + +Outputs land in `--out`: `report.json`, `REPORT.md`, and four figures +(`fig-structure`, `fig-ssm`, `fig-arrangement`, `fig-chroma`). + +Compare several studied tracks: + +```fish +.venv/bin/python study/compare.py study/out/*/report.json \ + --out study/out/comparison +``` + +That writes `COMPARISON.md` plus section-timeline, band-balance, and +loudness-small-multiple figures. + +## honesty notes + +- Loudness range and true peak are **approximations** (RMS-window LRA, + 4Γ— oversampled peak) β€” good for comparison, not for mastering QC. +- Section letters are repetition classes **within one track**; the same + letter on two different tracks means nothing. +- Key/melody estimates run on the harmonic component of the full mix; + treat them as evidence, not truth. +- A 128 kbps source rolls off β‰ˆ16 kHz β€” ignore the `air` band verdict + on streaming rips. + +Deps live in `pop/.venv`: librosa, scipy, soundfile, matplotlib, +pyloudnorm. First run of a study takes ~1–3 min per track (pyin is the +slow part). diff --git a/pop/study/compare.py b/pop/study/compare.py new file mode 100644 index 0000000000..0bb9229986 --- /dev/null +++ b/pop/study/compare.py @@ -0,0 +1,143 @@ +#!/usr/bin/env python3 +"""Compare several single-study reports side by side. + +Usage: + .venv/bin/python study/compare.py out/a/report.json out/b/report.json \ + --out study/out/comparison + +Writes COMPARISON.md plus three figures: section-strip timelines, +band-balance heatmap, loudness small multiples. +""" + +import argparse +import json +from pathlib import Path + +import numpy as np +import matplotlib + +matplotlib.use("Agg") +import matplotlib.pyplot as plt + +from study import (SURFACE, INK, INK2, GRID, SEQ_CMAP, BANDS, + sec_color) + + +def load(paths): + return [json.loads(Path(p).read_text()) for p in paths] + + +def fig_structures(out, reports): + fig, ax = plt.subplots(figsize=(9, 0.62 * len(reports) + 0.9)) + for i, r in enumerate(reports): + y = len(reports) - 1 - i + for sec in r["structure"]["sections"]: + ax.barh(y, sec["end_s"] - sec["start_s"], left=sec["start_s"], + height=0.52, color=sec_color(sec["label"]), + edgecolor=SURFACE, linewidth=1.2) + if sec["end_s"] - sec["start_s"] > 8: + ax.text((sec["start_s"] + sec["end_s"]) / 2, y, + sec["label"], ha="center", va="center", + fontsize=7, fontweight="bold", color=SURFACE) + ax.set_yticks(range(len(reports))) + ax.set_yticklabels([r["title"] for r in reversed(reports)], fontsize=8) + ax.set_xlabel("time (s)") + ax.set_title("section timelines (letters = repetition classes, " + "per track β€” colors do not match across rows)") + ax.grid(axis="x") + fig.tight_layout() + fig.savefig(out / "fig-compare-structure.png", dpi=180) + plt.close(fig) + + +def fig_balance(out, reports): + mat = np.array([[r["arrangement"]["band_balance_db"][b[0]] + for b in BANDS] for r in reports]) + fig, ax = plt.subplots(figsize=(4.8, 0.42 * len(reports) + 1.0)) + ax.imshow(np.clip(mat, -30, 0), cmap=SEQ_CMAP, aspect="auto", + vmin=-30, vmax=0) + for i in range(mat.shape[0]): + for j in range(mat.shape[1]): + v = mat[i, j] + ax.text(j, i, f"{v:.0f}", ha="center", va="center", fontsize=7, + color=SURFACE if v > -12 else INK2) + ax.set_xticks(range(len(BANDS))) + ax.set_xticklabels([b[0] for b in BANDS]) + ax.set_yticks(range(len(reports))) + ax.set_yticklabels([r["title"] for r in reports], fontsize=8) + ax.grid(False) + ax.set_title("band balance (dB rel. loudest band)") + fig.tight_layout() + fig.savefig(out / "fig-compare-balance.png", dpi=180) + plt.close(fig) + + +def fig_dynamics(out, reports, curves): + n = len(reports) + fig, axes = plt.subplots(n, 1, figsize=(5.6, 0.78 * n + 0.5), + sharex=True) + axes = np.atleast_1d(axes) + for ax, r, (t, db) in zip(axes, reports, curves): + ax.plot(t, db, color="#2a78d6", lw=1.4) + ax.set_ylim(-38, 0) + ax.set_ylabel(r["title"], rotation=0, ha="right", va="center", + fontsize=8, color=INK2) + ax.set_yticks([-30, -10]) + axes[-1].set_xlabel("time (s)") + axes[0].set_title("short-term RMS loudness (dBFS)") + fig.tight_layout() + fig.savefig(out / "fig-compare-dynamics.png", dpi=180) + plt.close(fig) + + +def write_markdown(out, reports): + rows = ["| track | dur | bpm | key | LUFS | LRAβ‰ˆ | crest | sections |" + " mean sec | onsets/s | perc share |", + "|---|---|---|---|---|---|---|---|---|---|---|"] + for r in reports: + m, s, a, h = (r["master"], r["structure"], + r["arrangement"], r["harmony"]) + secs = s["sections"] + mean_sec = np.mean([x["end_s"] - x["start_s"] for x in secs]) + rows.append( + f"| {r['title']} | {m['duration_s']:.0f}s | {s['tempo_bpm']} |" + f" {h['key']} | {m['integrated_lufs']} |" + f" {m['loudness_range_db_approx']} | {m['crest_factor_db']} |" + f" {s['n_sections']} | {mean_sec:.0f}s |" + f" {a['onsets_per_s_overall']} | {a['percussive_share']} |") + (out / "COMPARISON.md").write_text( + "# single-study comparison\n\n" + "\n".join(rows) + "\n") + + +def main(): + ap = argparse.ArgumentParser(description=__doc__) + ap.add_argument("reports", nargs="+") + ap.add_argument("--out", required=True) + args = ap.parse_args() + out = Path(args.out) + out.mkdir(parents=True, exist_ok=True) + reports = load(args.reports) + + # loudness curves are re-derived from each report's audio file so the + # comparison never needs the originals resampled together + import librosa + curves = [] + for r in reports: + y, sr = librosa.load(r["master"]["file"], sr=22050, mono=True) + win, hop = int(3 * sr), int(0.5 * sr) + t, db = [], [] + for start in range(0, max(1, len(y) - win), hop): + seg = y[start:start + win] + t.append((start + win / 2) / sr) + db.append(20 * np.log10(np.sqrt(np.mean(seg ** 2)) + 1e-12)) + curves.append((np.array(t), np.array(db))) + + fig_structures(out, reports) + fig_balance(out, reports) + fig_dynamics(out, reports, curves) + write_markdown(out, reports) + print(f"β†’ {out}") + + +if __name__ == "__main__": + main() diff --git a/pop/study/study.py b/pop/study/study.py new file mode 100644 index 0000000000..e54dd8ebf5 --- /dev/null +++ b/pop/study/study.py @@ -0,0 +1,495 @@ +#!/usr/bin/env python3 +"""single-study-toolkit β€” study a finished single from the outside in. + +Four layers, outermost first: + + L0 master the file as a mastered object: loudness, peaks, crest, + stereo image, spectral tilt + L1 structure tempo, beat grid, self-similarity, section boundaries + and letters (A/B/C… by repetition) + L2 arrangement six-band energy over time, harmonic/percussive split, + onset density β€” who is playing, when + L3 harmony chroma, global + per-section key, dominant-voice pitch + +Usage: + .venv/bin/python study/study.py AUDIO --out study/out/slug \ + [--title "One Step"] [--artist oskie] + +Writes report.json, REPORT.md and four figures into --out. +compare.py consumes multiple report.json files. +""" + +import argparse +import json +import sys +from pathlib import Path + +import numpy as np +import scipy.signal +import scipy.cluster.hierarchy as hier +import librosa +import soundfile as sf +import pyloudnorm + +import matplotlib + +matplotlib.use("Agg") +import matplotlib.pyplot as plt +from matplotlib.colors import LinearSegmentedColormap + +# ---------------------------------------------------------------- palette -- +# Reference dataviz palette (light mode) β€” categorical order is fixed. +SURFACE = "#fcfcfb" +INK = "#0b0b0b" +INK2 = "#52514e" +GRID = "#e5e4e0" +CATEGORICAL = ["#2a78d6", "#eb6834", "#1baf7a", "#eda100", + "#e87ba4", "#008300", "#4a3aa7", "#e34948"] +SEQ_CMAP = LinearSegmentedColormap.from_list( + "seq_blue", [SURFACE, "#9ec4ec", "#2a78d6", "#0f3f78"]) + +BANDS = [("sub", 20, 60), ("bass", 60, 250), ("lowmid", 250, 1000), + ("mid", 1000, 4000), ("high", 4000, 10000), ("air", 10000, 16000)] + +plt.rcParams.update({ + "figure.facecolor": SURFACE, "axes.facecolor": SURFACE, + "savefig.facecolor": SURFACE, "text.color": INK, + "axes.edgecolor": GRID, "axes.labelcolor": INK2, + "xtick.color": INK2, "ytick.color": INK2, + "axes.grid": True, "grid.color": GRID, "grid.linewidth": 0.6, + "axes.axisbelow": True, "font.size": 9, + "axes.titlesize": 10, "axes.titleweight": "bold", + "axes.spines.top": False, "axes.spines.right": False, +}) + + +# ---------------------------------------------------------------- L0 master -- +def layer_master(path, y_stereo, y, sr): + dur = len(y) / sr + meter = pyloudnorm.Meter(sr) + data = y_stereo.T if y_stereo.ndim == 2 else y_stereo + lufs = float(meter.integrated_loudness(data)) + + # short-term loudness proxy: 3 s RMS windows, 0.5 s hop (dBFS) + win, hop = int(3 * sr), int(0.5 * sr) + st_times, st_db = [], [] + for start in range(0, max(1, len(y) - win), hop): + seg = y[start:start + win] + rms = np.sqrt(np.mean(seg ** 2) + 1e-12) + st_times.append((start + win / 2) / sr) + st_db.append(20 * np.log10(rms + 1e-12)) + st_times, st_db = np.array(st_times), np.array(st_db) + active = st_db[st_db > st_db.max() - 40] # crude gate + lra = float(np.percentile(active, 95) - np.percentile(active, 10)) + + peak = float(np.max(np.abs(y_stereo))) + y4 = scipy.signal.resample_poly( + y_stereo, 4, 1, axis=-1 if y_stereo.ndim == 2 else 0) + true_peak = float(np.max(np.abs(y4))) + rms_all = float(np.sqrt(np.mean(y ** 2))) + crest = 20 * np.log10(peak / (rms_all + 1e-12)) + + stereo = None + if y_stereo.ndim == 2 and y_stereo.shape[0] == 2: + l, r = y_stereo + corr = float(np.corrcoef(l, r)[0, 1]) + mid, side = (l + r) / 2, (l - r) / 2 + side_db = 20 * np.log10( + (np.sqrt(np.mean(side ** 2)) + 1e-12) / + (np.sqrt(np.mean(mid ** 2)) + 1e-12)) + stereo = {"correlation": round(corr, 3), + "side_vs_mid_db": round(float(side_db), 1)} + + S = np.abs(librosa.stft(y, n_fft=4096)) + freqs = librosa.fft_frequencies(sr=sr, n_fft=4096) + centroid = librosa.feature.spectral_centroid(S=S, sr=sr)[0] + rolloff = librosa.feature.spectral_rolloff(S=S, sr=sr, roll_percent=0.95)[0] + mean_spec = np.mean(S, axis=1) + mask = (freqs > 100) & (freqs < 16000) + tilt = np.polyfit(np.log10(freqs[mask]), + 20 * np.log10(mean_spec[mask] + 1e-9), 1)[0] + + return { + "file": str(path), "duration_s": round(dur, 2), "sr": sr, + "channels": 1 if y_stereo.ndim == 1 else y_stereo.shape[0], + "integrated_lufs": round(lufs, 1), + "loudness_range_db_approx": round(lra, 1), + "sample_peak_dbfs": round(20 * np.log10(peak + 1e-12), 2), + "true_peak_dbtp_approx": round(20 * np.log10(true_peak + 1e-12), 2), + "crest_factor_db": round(float(crest), 1), + "stereo": stereo, + "spectral_centroid_hz_median": int(np.median(centroid)), + "rolloff95_hz_median": int(np.median(rolloff)), + "spectral_tilt_db_per_decade": round(float(tilt), 1), + }, (st_times, st_db) + + +# ------------------------------------------------------------- L1 structure -- +def foote_novelty(ssm, kernel=48): + k = kernel // 2 + g = scipy.signal.windows.gaussian(kernel, kernel / 4) + board = np.outer(g, g) + sign = np.ones((kernel, kernel)) + sign[:k, k:] = -1 + sign[k:, :k] = -1 + kern = board * sign + n = ssm.shape[0] + pad = np.pad(ssm, k, mode="edge") + nov = np.array([np.sum(pad[i:i + kernel, i:i + kernel] * kern) + for i in range(n)]) + nov -= nov.min() + return nov / (nov.max() + 1e-12) + + +def layer_structure(y, sr): + tempo, beats = librosa.beat.beat_track(y=y, sr=sr, trim=False) + tempo = float(np.atleast_1d(tempo)[0]) + beat_times = librosa.frames_to_time(beats, sr=sr) + + chroma = librosa.feature.chroma_cqt(y=y, sr=sr) + mfcc = librosa.feature.mfcc(y=y, sr=sr, n_mfcc=20) + rms = librosa.feature.rms(y=y) + feats = np.vstack([librosa.util.normalize(chroma, axis=0), + librosa.util.normalize(mfcc, axis=0), + librosa.util.normalize(rms, axis=0)]) + fb = librosa.util.sync(feats, beats, aggregate=np.median) + fb = librosa.util.normalize(fb, axis=0) + + # full cosine self-similarity β€” dense enough for checkerboard novelty + unit = fb / (np.linalg.norm(fb, axis=0, keepdims=True) + 1e-9) + ssm = unit.T @ unit + + nov = foote_novelty(ssm, kernel=min(48, max(8, ssm.shape[0] // 8))) + # drops and breaks announce themselves in energy before anything else + rms_b = librosa.util.sync(rms, beats, aggregate=np.median)[0] + d_rms = scipy.ndimage.gaussian_filter1d(np.abs(np.gradient(rms_b)), 2) + d_rms /= d_rms.max() + 1e-12 + nov = 0.6 * nov + 0.4 * d_rms[:len(nov)] + min_gap = max(4, int(8 * tempo / 60 / 2)) # β‰₯ ~8 s between boundaries + peaks, _ = scipy.signal.find_peaks( + nov, distance=min_gap, prominence=0.15) + bounds = [0] + [int(p) for p in peaks] + [len(beat_times) - 1] + bounds = sorted(set(bounds)) + segs = list(zip(bounds[:-1], bounds[1:])) + + # absorb slivers (< ~6 s) into whichever neighbor sounds more alike + def seg_mean(seg): + return fb[:, seg[0]:seg[1]].mean(axis=1) + + min_beats = int(6 * tempo / 60) + while len(segs) > 1: + lens = [b - a for a, b in segs] + i = int(np.argmin(lens)) + if lens[i] >= min_beats: + break + cands = [j for j in (i - 1, i + 1) if 0 <= j < len(segs)] + j = min(cands, key=lambda j: np.linalg.norm( + seg_mean(segs[i]) - seg_mean(segs[j]))) + a, b = min(i, j), max(i, j) + segs[a:b + 1] = [(segs[a][0], segs[b][1])] + + # letter segments by clustering their mean feature vectors + seg_means = np.array([seg_mean(s) for s in segs]) + if len(segs) > 1: + z = hier.linkage(seg_means, method="ward") + cut = 0.5 * z[:, 2].max() + raw = hier.fcluster(z, t=cut, criterion="distance") + else: + raw = [1] + letters, order = {}, [] + for c in raw: + if c not in letters: + letters[c] = chr(ord("A") + len(letters)) + order.append(letters[c]) + + # merge neighbors that ended up with the same letter + merged = [] + for seg, lab in zip(segs, order): + if merged and merged[-1][1] == lab: + merged[-1] = ((merged[-1][0][0], seg[1]), lab) + else: + merged.append((seg, lab)) + + sections = [] + for (a, b), lab in merged: + t0, t1 = float(beat_times[a]), float(beat_times[b]) + sections.append({"label": lab, "start_s": round(t0, 2), + "end_s": round(t1, 2), + "bars_approx": round((t1 - t0) * tempo / 60 / 4, 1)}) + bounds = [seg[0] for seg, _ in merged] + [merged[-1][0][1]] + + # phrase tier: the boundaries before letter-merging β€” the small waves + # (fills, drops, lifts) inside the macroform + phrases = [{"start_s": round(float(beat_times[a]), 2), + "end_s": round(float(beat_times[b]), 2)} for a, b in segs] + return { + "tempo_bpm": round(tempo, 1), + "n_beats": len(beat_times), + "n_sections": len(sections), + "sections": sections, + "n_phrases": len(phrases), + "median_phrase_s": round(float(np.median( + [p["end_s"] - p["start_s"] for p in phrases])), 1), + "phrases": phrases, + }, (beat_times, ssm, nov, bounds) + + +# ----------------------------------------------------------- L2 arrangement -- +def layer_arrangement(y, sr, sections): + S = np.abs(librosa.stft(y, n_fft=4096, hop_length=1024)) ** 2 + freqs = librosa.fft_frequencies(sr=sr, n_fft=4096) + times = librosa.times_like(S[0], sr=sr, hop_length=1024) + band_energy = np.array([ + S[(freqs >= lo) & (freqs < hi)].sum(axis=0) + for _, lo, hi in BANDS]) + band_db = 10 * np.log10(band_energy + 1e-12) + band_db -= band_db.max() + + H, P = librosa.decompose.hpss(librosa.stft(y)) + e_h, e_p = float(np.sum(np.abs(H) ** 2)), float(np.sum(np.abs(P) ** 2)) + onset_env = librosa.onset.onset_strength(y=y, sr=sr) + onsets = librosa.onset.onset_detect(y=y, sr=sr, units="time") + + profile = {} + total_db = 10 * np.log10(band_energy.sum(axis=1) + 1e-12) + for (name, _, _), db in zip(BANDS, total_db - total_db.max()): + profile[name] = round(float(db), 1) + + per_section = [] + for sec in sections: + m = (times >= sec["start_s"]) & (times < sec["end_s"]) + sec_prof = band_db[:, m].mean(axis=1) if m.any() else np.zeros(len(BANDS)) + dens = float(np.sum((onsets >= sec["start_s"]) & + (onsets < sec["end_s"])) / + max(0.1, sec["end_s"] - sec["start_s"])) + per_section.append({ + "label": sec["label"], "start_s": sec["start_s"], + "onsets_per_s": round(dens, 2), + "band_db": {n: round(float(v), 1) + for (n, _, _), v in zip(BANDS, sec_prof)}}) + return { + "band_balance_db": profile, + "percussive_share": round(e_p / (e_h + e_p), 3), + "onsets_per_s_overall": round(len(onsets) / (len(y) / sr), 2), + "per_section": per_section, + }, (times, band_db, onset_env) + + +# --------------------------------------------------------------- L3 harmony -- +KS_MAJOR = np.array([6.35, 2.23, 3.48, 2.33, 4.38, 4.09, + 2.52, 5.19, 2.39, 3.66, 2.29, 2.88]) +KS_MINOR = np.array([6.33, 2.68, 3.52, 5.38, 2.60, 3.53, + 2.54, 4.75, 3.98, 2.69, 3.34, 3.17]) +NOTES = ["C", "C#", "D", "D#", "E", "F", "F#", "G", "G#", "A", "A#", "B"] + + +def estimate_key(chroma_mean): + best = (-2, None) + for i in range(12): + rolled = np.roll(chroma_mean, -i) + for prof, mode in ((KS_MAJOR, "major"), (KS_MINOR, "minor")): + r = np.corrcoef(rolled, prof)[0, 1] + if r > best[0]: + best = (r, f"{NOTES[i]} {mode}") + return best[1], round(float(best[0]), 3) + + +def layer_harmony(y, sr, sections): + y_harm = librosa.effects.harmonic(y) + chroma = librosa.feature.chroma_cqt(y=y_harm, sr=sr) + times = librosa.times_like(chroma[0], sr=sr) + key, conf = estimate_key(chroma.mean(axis=1)) + + per_section = [] + for sec in sections: + m = (times >= sec["start_s"]) & (times < sec["end_s"]) + if m.any(): + k, c = estimate_key(chroma[:, m].mean(axis=1)) + per_section.append({"label": sec["label"], + "start_s": sec["start_s"], "key": k, + "confidence": c}) + + f0, voiced, _ = librosa.pyin( + y_harm, fmin=float(librosa.note_to_hz("C2")), + fmax=float(librosa.note_to_hz("C6")), sr=sr) + v = f0[voiced.astype(bool)] if voiced is not None else np.array([]) + melody = None + if v.size > 20: + melody = { + "voiced_fraction": round(float(np.mean(voiced)), 2), + "median_pitch": librosa.hz_to_note(float(np.median(v))), + "range_semitones": round(float( + 12 * np.log2(np.percentile(v, 95) / np.percentile(v, 5))), 1)} + return {"key": key, "key_confidence": conf, + "per_section_keys": per_section, "dominant_voice": melody}, \ + (chroma, times) + + +# ----------------------------------------------------------------- figures -- +def sec_color(label): + return CATEGORICAL[(ord(label) - ord("A")) % len(CATEGORICAL)] + + +def draw_sections(ax, sections, ymin, ymax, label_y=None): + for sec in sections: + ax.axvline(sec["start_s"], color=INK2, lw=0.7, alpha=0.6) + if label_y is not None and sec["end_s"] - sec["start_s"] > 5: + ax.text((sec["start_s"] + sec["end_s"]) / 2, label_y, + sec["label"], ha="center", va="center", + fontsize=8, fontweight="bold", + color=sec_color(sec["label"])) + + +def fig_structure(out, title, y, sr, st, sections, phrases=()): + st_times, st_db = st + t = np.arange(len(y)) / sr + step = max(1, len(y) // 4000) + fig, ax = plt.subplots(figsize=(9, 2.8)) + ax.fill_between(t[::step], y[::step], -y[::step], + color=GRID, lw=0, alpha=0.9) + for p in phrases: + ax.axvline(p["start_s"], ymin=0, ymax=0.05, color=INK2, lw=0.7) + ax2 = ax.twinx() # loudness overlays the waveform on its own scale + ax2.plot(st_times, st_db, color=CATEGORICAL[0], lw=2) + ax2.set_ylabel("short-term RMS (dBFS)", color=INK2) + ax2.grid(False) + ax2.spines["top"].set_visible(False) + ax.set_yticks([]) + draw_sections(ax, sections, -1, 1, label_y=0.88) + ax.set_ylim(-1, 1) + ax.set_xlim(0, t[-1]) + ax.set_xlabel("time (s)") + ax.set_title(f"{title} β€” waveform, loudness, sections") + fig.tight_layout() + fig.savefig(out / "fig-structure.png", dpi=180) + plt.close(fig) + + +def fig_ssm(out, title, ssm, beat_times, bounds): + fig, ax = plt.subplots(figsize=(4.6, 4.6)) + ax.imshow(ssm, origin="lower", cmap=SEQ_CMAP, aspect="equal", + extent=[beat_times[0], beat_times[-1], + beat_times[0], beat_times[-1]]) + for b in bounds[1:-1]: + ax.axvline(beat_times[b], color=INK, lw=0.6, alpha=0.5) + ax.axhline(beat_times[b], color=INK, lw=0.6, alpha=0.5) + ax.grid(False) + ax.set_xlabel("time (s)") + ax.set_ylabel("time (s)") + ax.set_title(f"{title} β€” self-similarity (beat-synced)") + fig.tight_layout() + fig.savefig(out / "fig-ssm.png", dpi=180) + plt.close(fig) + + +def fig_arrangement(out, title, times, band_db, sections): + fig, ax = plt.subplots(figsize=(9, 2.8)) + ax.imshow(np.clip(band_db, -50, 0), origin="lower", aspect="auto", + cmap=SEQ_CMAP, extent=[times[0], times[-1], 0, len(BANDS)], + vmin=-50, vmax=0) + ax.set_yticks([i + 0.5 for i in range(len(BANDS))]) + ax.set_yticklabels([b[0] for b in BANDS]) + draw_sections(ax, sections, 0, len(BANDS), label_y=len(BANDS) - 0.4) + ax.grid(False) + ax.set_xlabel("time (s)") + ax.set_title(f"{title} β€” band energy over time (dB rel. max)") + fig.tight_layout() + fig.savefig(out / "fig-arrangement.png", dpi=180) + plt.close(fig) + + +def fig_chroma(out, title, chroma, times, sections, keys): + fig, ax = plt.subplots(figsize=(9, 2.6)) + ax.imshow(chroma, origin="lower", aspect="auto", cmap=SEQ_CMAP, + extent=[times[0], times[-1], 0, 12]) + ax.set_yticks([i + 0.5 for i in range(12)]) + ax.set_yticklabels(NOTES, fontsize=7) + draw_sections(ax, sections, 0, 12) + for k in keys: + ax.text(k["start_s"] + 1, 11.3, k["key"], fontsize=7, + color=INK, fontweight="bold") + ax.grid(False) + ax.set_xlabel("time (s)") + ax.set_title(f"{title} β€” chroma + per-section key") + fig.tight_layout() + fig.savefig(out / "fig-chroma.png", dpi=180) + plt.close(fig) + + +# ------------------------------------------------------------------- report -- +def write_markdown(out, rpt): + m, s, a, h = (rpt["master"], rpt["structure"], + rpt["arrangement"], rpt["harmony"]) + lines = [f"# {rpt['title']} β€” {rpt['artist']}", ""] + lines += [ + f"- **duration** {m['duration_s']} s Β· **tempo** {s['tempo_bpm']} bpm" + f" Β· **key** {h['key']} ({h['key_confidence']})", + f"- **loudness** {m['integrated_lufs']} LUFS Β·" + f" LRAβ‰ˆ{m['loudness_range_db_approx']} dB Β·" + f" crest {m['crest_factor_db']} dB Β·" + f" true peakβ‰ˆ{m['true_peak_dbtp_approx']} dBTP", + f"- **spectrum** centroid {m['spectral_centroid_hz_median']} Hz Β·" + f" tilt {m['spectral_tilt_db_per_decade']} dB/decade", + f"- **rhythm** {a['onsets_per_s_overall']} onsets/s Β·" + f" percussive share {a['percussive_share']}", + "", "## sections", "", + "| # | label | start | end | barsβ‰ˆ | onsets/s |", + "|---|-------|-------|-----|-------|----------|"] + for i, (sec, per) in enumerate(zip(s["sections"], a["per_section"])): + lines.append( + f"| {i+1} | {sec['label']} | {sec['start_s']:.0f}s |" + f" {sec['end_s']:.0f}s | {sec['bars_approx']} |" + f" {per['onsets_per_s']} |") + lines += ["", "## band balance (dB rel. loudest band)", ""] + for k, v in a["band_balance_db"].items(): + lines.append(f"- {k}: {v}") + (out / "REPORT.md").write_text("\n".join(lines) + "\n") + + +def study(path, out, title, artist): + out.mkdir(parents=True, exist_ok=True) + y_stereo, sr = librosa.load(path, sr=None, mono=False) + y = librosa.to_mono(y_stereo) + + print("Β· L0 master") + master, st = layer_master(path, y_stereo, y, sr) + print("Β· L1 structure") + structure, (beat_times, ssm, nov, bounds) = layer_structure(y, sr) + print("Β· L2 arrangement") + arrangement, (bt, band_db, onset_env) = layer_arrangement( + y, sr, structure["sections"]) + print("Β· L3 harmony") + harmony, (chroma, ct) = layer_harmony(y, sr, structure["sections"]) + + rpt = {"title": title, "artist": artist, + "master": master, "structure": structure, + "arrangement": arrangement, "harmony": harmony} + (out / "report.json").write_text(json.dumps(rpt, indent=2)) + write_markdown(out, rpt) + + print("Β· figures") + label = f"{artist} β€” {title}" if artist else title + fig_structure(out, label, y, sr, st, structure["sections"], + structure["phrases"]) + fig_ssm(out, label, ssm, beat_times, bounds) + fig_arrangement(out, label, bt, band_db, structure["sections"]) + fig_chroma(out, label, chroma, ct, structure["sections"], + harmony["per_section_keys"]) + print(f"β†’ {out}") + return rpt + + +def main(): + ap = argparse.ArgumentParser(description=__doc__) + ap.add_argument("audio") + ap.add_argument("--out", required=True) + ap.add_argument("--title", default=None) + ap.add_argument("--artist", default="") + args = ap.parse_args() + path = Path(args.audio) + title = args.title or path.stem + study(path, Path(args.out), title, args.artist) + + +if __name__ == "__main__": + main() -- 2.51.2