From 21ef533e2da17469779f8ed5247aa35c0f7cb8d3 Mon Sep 17 00:00:00 2001 From: "prompt.ac/@jeffrey" Date: Mon, 17 Aug 2026 16:02:18 -0400 Subject: [PATCH] loner kickvox: draw her real waveforms into the timeline blocks, and the energy trim they exposed MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit @jeffrey, reading the study: "up comes too soon" · "curled is too short" · "check the length of the actual waveforms in the utterances, not just ur trim etc" · "please map / render those waveforms directly into the clips in the mp4 so i can see" · "and add live timecode". bin/timeline.py now draws each word's ACTUAL audio inside its block — per-column peak of the vox4/ lead render (the thing the study plays), normalized to the whole take, so dead air inside a slot is visible — plus a live timecode and bar·beat address in the corner, and the active word outlines in pink instead of filling so the waveform stays readable while it plays. Which immediately showed the trim was lying. Whisper's word boundaries are handoffs, not note ends: it gave "led" 0.99 s when she stops singing after ~0.55 and the rest is decay, and build_warp was time-stretching that silence across the slot along with the note — "led" sang to 78% of its block and then sat there. trim_units() pulls each unit's source span back to its real audio end (5 ms RMS against a −36 dB gate of the take's peak, +50 ms release margin) and the silence is DROPPED rather than warped, so only sung frames stretch. Guards: the last unit is untouched (the tail/release machinery owns it), trims under 80 ms aren't worth the surgery, no unit loses more than 65% of its span. On the whole line exactly two fire — led −390 ms · think −445 ms — and every word now sings to ≥92% of its slot (led 78% → 95%). Receipt: `trims` in vox4/.manifest.json. Bar 1 hand-pinned against that picture: CURLED alone fills it (cur 2 + led 2 — her own 0.74/0.99 s split says led ≥ cur), and UP, her 0.25 s tonic release, lands ON the bar-2 downbeat as an up-in pickup pair into "myself", which stays anchored at beat 9 so think/of/stone hold their tuned spots. timeline.py wants system python3 (PIL), not pop/.venv — noted in the README along with the whole study-reading recipe. --- pop/loner/README.md | 37 ++++++++++++++++++++ pop/loner/bin/halo3.py | 64 ++++++++++++++++++++++++++++++++--- pop/loner/bin/timeline.py | 46 +++++++++++++++++++++++-- pop/loner/c/loner-chart.h | 8 ++--- pop/loner/vox4/.chart.json | 10 +++--- pop/loner/vox4/.manifest.json | 13 +++++++ 6 files changed, 163 insertions(+), 15 deletions(-) diff --git a/pop/loner/README.md b/pop/loner/README.md index e4e09a7fdf..7eacd5d7ef 100644 --- a/pop/loner/README.md +++ b/pop/loner/README.md @@ -297,6 +297,43 @@ pop/loner/c/lonerremix # → out/loner-remix-v4-full.wav bash pop/loner/c/cut-v4.sh # → out/loner-remix-v4.mp3 (−14 LUFS) ``` +### The kick+vocals study, and reading it + +`MINIMAL=1 c/lonerremix` renders `out/loner-kickvox` — kick plus the +unbroken vocal, nothing else — and `bin/timeline.py` draws it: a +scrolling piano roll where every word is a block at its chart slot and +sung pitch, **with the word's real waveform drawn inside it** (per-column +peak of the `vox4/` lead render, normalized to the whole take), bars +numbered, the kick lane flashing, a live timecode and `bar · beat` +address in the corner. The active word outlines in pink rather than +filling, so the waveform stays readable while it plays. + +Drawing the audio into the blocks is what exposed **the energy trim**. +Whisper's word boundaries are handoffs, not note ends: it gave "led" +0.99 s when Camille stops singing after ~0.55 and the rest is decay, and +those dead frames were being time-stretched across the slot along with +the note — "led" sang to 78% of its block and then sat there. Each +unit's source span now ends where its audio actually ends (5 ms RMS +against a −36 dB gate of the take's peak, plus a 50 ms release margin), +and the silence is **dropped rather than warped**, so only sung frames +stretch. Guards: the last unit is untouched (the tail/release machinery +owns it), trims under 80 ms aren't worth the surgery, and no unit can +lose more than 65% of its span. On the whole line exactly two fire — +`led −390 ms · think −445 ms` — and every word now sings to ≥92% of its +slot. Receipt: `trims` in `vox4/.manifest.json`. + +Bar 1 is hand-pinned against that picture: **curled** alone fills it +(cur 2 + led 2 — her own 0.74/0.99 s split says led ≥ cur), and **up**, +her 0.25 s tonic release, lands *on* the bar-2 downbeat as an up-in +pickup pair into "myself". + +```bash +MINIMAL=1 pop/loner/c/lonerremix # → out/loner-kickvox-full.wav +python3 pop/loner/bin/timeline.py # → out/loner-kickvox-timeline.mp4 +``` + +`timeline.py` wants system `python3` (PIL), not `pop/.venv`. + ## Measured | check | v1 | v2 | v3 | diff --git a/pop/loner/bin/halo3.py b/pop/loner/bin/halo3.py index 2ae42c6503..be7d462b8a 100644 --- a/pop/loner/bin/halo3.py +++ b/pop/loner/bin/halo3.py @@ -164,9 +164,15 @@ CHART = { # POST-split indices. "splits": [5], # then: think = 2 · OF (the octave) = 4 · - # a stone = 4 - "durs": { 0: 4.0, 1: 1.5, 2: 2.0, 3: 0.5, - 4: 1.0, 5: 3.0, 6: 3.0, 7: 1.0, + # a stone = 4. "curled is too short · up + # comes too soon": CURLED alone fills bar 1 + # (cur 2 + led 2 — her own 37/50 split says + # led ≥ cur), and up — her 0.25 s tonic + # release — lands ON the bar-2 downbeat, an + # up-in pickup pair into myself, which + # stays anchored at beat 9. + "durs": { 0: 4.0, 1: 2.0, 2: 2.0, 3: 0.5, + 4: 0.5, 5: 3.0, 6: 3.0, 7: 1.0, 8: 2.0, 9: 4.0, 10: 0.5, 11: 3.5 } }, "w-sitting-curled": { "slice": "f-sitting-curled", "beats": 11.0 }, "w-i-think": { "slice": "f-i-think", "beats": 3.5 }, @@ -200,6 +206,52 @@ def derive_units(words, beats_total, stretch=None, durs=None): SLICES = json.load(open(os.path.join(LANE, "samples", ".manifest.json"))) +# ── the energy trim — only SUNG frames stretch ──────────────────────── +# @jeffrey, reading the waveforms drawn into the timeline: "check the +# length of the actual waveforms in the utterances, not just ur trim". +# Whisper's word boundaries are handoffs, not note ends: it gave "led" +# 0.99 s when she stops singing after ~0.5 and the rest is decay. Those +# dead frames were stretching across the slot with the note, so a word +# could fill 78% of its block and then sit there. Each unit's source +# span now ends where its audio actually ends (+ a release margin), and +# the silence is DROPPED rather than warped — the vowel takes the slot. +TRIM_GATE_DB = -36.0 # of the take's peak; keeps quiet fricatives +TRIM_MARGIN_S = 0.050 # let the release start before we cut +TRIM_MIN_S = 0.080 # below this, not worth the surgery +TRIM_KEEP = 0.35 # never leave a unit shorter than this share + + +def energy_end(x, fs, f0, f1, peak): + """Last frame in [f0,f1) whose 5 ms RMS clears the gate.""" + n = int(round(fs * FRAME_S)) + seg = x[f0 * n:f1 * n] + if len(seg) < n: + return f1 + m = len(seg) // n + rms = np.sqrt((seg[:m * n].reshape(m, n) ** 2).mean(axis=1)) + on = np.nonzero(rms > peak * 10.0 ** (TRIM_GATE_DB / 20.0))[0] + return f0 + (int(on[-1]) + 1 if len(on) else m) + + +def trim_units(x, fs, unit_src, names=None): + """Pull each unit's end back to its real audio end. Last unit keeps + its span (the tail/release machinery owns it). Returns (spans, log).""" + peak = np.max(np.abs(x)) or 1.0 + out, log = [], [] + for u, (s0, s1) in enumerate(unit_src): + if u == len(unit_src) - 1: + out.append((s0, s1)) + continue + e = energy_end(x, fs, s0, s1, peak) + int(round(TRIM_MARGIN_S / FRAME_S)) + e = max(s0 + int(round(TRIM_KEEP * (s1 - s0))), min(e, s1)) + cut = (s1 - e) * FRAME_S + if cut < TRIM_MIN_S: + e = s1 + elif names: + log.append(f"{names[u]} −{cut * 1000:.0f}ms") + out.append((s0, e)) + return out, log + def build_warp(a, unit_src, beats, dursb): """Frame index map with VOWEL-ON-THE-BEAT alignment. @@ -368,6 +420,8 @@ for name, ch in CHART.items(): s1 = int(round((words[i + 1]["start"] - t0_slice) / FRAME_S)) unit_src.append((max(0, s0), min(F, max(s0 + 1, s1)))) + unit_src, trims = trim_units(x, fs, unit_src, [w["t"] for w in words]) + idx, holds, fade, Z = build_warp(a, unit_src, [b for (b, d) in ch["units"]], [d for (b, d) in ch["units"]]) @@ -427,8 +481,10 @@ for name, ch in CHART.items(): chart_c.append((name, lead_in, beats_total, notes)) manifest[name] = dict(slice=slice_name, lead_in=round(lead_in, 3), beats=beats_total, renders=renders, - words=entry["words"]) + trims=trims, words=entry["words"]) print(f" {name:22s} {renders['lead']:5.2f}s lead·8ve×2·low3·low5 «{entry['words']}»") + if trims: + print(f" trimmed: {' · '.join(trims)}") json.dump(manifest, open(os.path.join(VOX4, ".manifest.json"), "w"), indent=1) diff --git a/pop/loner/bin/timeline.py b/pop/loner/bin/timeline.py index 3a77d67ebb..af1c669351 100644 --- a/pop/loner/bin/timeline.py +++ b/pop/loner/bin/timeline.py @@ -11,7 +11,7 @@ # pop/.venv/bin/python pop/loner/bin/timeline.py # → out/loner-kickvox-timeline.mp4 -import json, math, os, subprocess +import json, math, os, subprocess, wave import numpy as np from PIL import Image, ImageDraw, ImageFont @@ -50,6 +50,31 @@ FRAMES = int(math.ceil(dur * FPS)) F = lambda s: ImageFont.truetype("/System/Library/Fonts/Helvetica.ttc", s) f_title, f_bar, f_word, f_note, f_tiny = F(40), F(26), F(30), F(20), F(22) +M = lambda s: ImageFont.truetype("/System/Library/Fonts/Menlo.ttc", s) +f_tc, f_tc_small = M(38), M(22) + +# ── the actual sung waveform, so the eye can audit the trim ─────────── +# @jeffrey: "check the length of the actual waveforms in the utterances, +# not just ur trim etc — map / render those waveforms directly into the +# clips". The lead render (vox4/w-whole-line.wav) IS what the study +# plays; chart beat b lives at leadIn + b·SPB seconds in that file. +with wave.open(os.path.join(LANE, "vox4", "w-whole-line.wav"), "rb") as wf: + VFS = wf.getframerate() + VOX = np.frombuffer(wf.readframes(wf.getnframes()), + dtype=np.int16).astype(np.float64) / 32768.0 +VOX_PEAK = np.max(np.abs(VOX)) or 1.0 +LEAD_IN = chart["leadIn"] + +def vox_env(b0, b1, npx): + """Per-pixel-column peak of the lead render between chart beats.""" + s0 = int((LEAD_IN + b0 * SPB) * VFS) + s1 = int((LEAD_IN + b1 * SPB) * VFS) + seg = np.abs(VOX[max(0, s0):max(0, s1)]) + if len(seg) == 0 or npx <= 0: + return np.zeros(max(1, npx)) + edges = np.linspace(0, len(seg), npx + 1).astype(int) + return np.array([seg[a:b].max() if b > a else 0.0 + for a, b in zip(edges[:-1], edges[1:])]) / VOX_PEAK CHROM = ["A#", "B", "C", "C#", "D", "D#", "E", "F", "F#", "G", "G#", "A"] def note_name(st): @@ -109,6 +134,14 @@ for pb in PASSES: y0, y1 = y - 24, y + 24 d.rounded_rectangle([x0 + 2, y0, x1 - 3, y1], 10, fill=(255, 166, 202, 90), outline=INK, width=2) + # the real audio inside the clip: per-column peak of the lead + # render, normalized to the whole take — dead air is visible + env = vox_env(n["beat"], n["beat"] + n["dur"], max(1, x1 - x0 - 8)) + for j, a in enumerate(env): + ah = a * 20 + if ah >= 0.4: + xw = x0 + 4 + j + d.line([(xw, y - ah), (xw, y + ah)], fill=(26, 26, 34, 88), width=1) txt = word_text(n["t"]) tw = d.textlength(txt, font=f_word) wide = (x1 - x0) > tw + 20 @@ -154,13 +187,22 @@ for i in range(FRAMES): if t0 <= t < t1: lx0, lx1 = x0 - px, x1 - px if lx1 > 0 and lx0 < W: + # outline-only: the waveform inside stays scrutinizable d2.rounded_rectangle([lx0 + 2, y0, lx1 - 3, y1], 10, - fill=(255, 102, 168, 120), outline=PINK, width=3) + fill=(255, 102, 168, 40), outline=PINK, width=4) # kick flash kb = int(beat) if kb < KICK_BEATS and (beat - kb) < 0.22: x = X(kb) - px d2.rounded_rectangle([x + 4, KY0, x + PXB - 8, KY1], 6, fill=PINK) + # live timecode — song clock + musical address, updating every frame + tc = f"{int(t // 60):02d}:{t % 60:05.2f}" + bar_no = int(beat) // 4 - 2 + addr = (f"bar {bar_no} · beat {beat % 4 + 1:3.1f}" if beat >= 8 + else f"count-in · beat {beat % 8 + 1:3.1f}") + d2.text((W - 40 - d2.textlength(tc, font=f_tc), 26), tc, font=f_tc, fill=INK) + d2.text((W - 40 - d2.textlength(addr, font=f_tc_small), 78), addr, + font=f_tc_small, fill=BLUE) # playhead d2.line([(PLAYHEAD_X, ROLL_TOP - 44), (PLAYHEAD_X, KY1 + 8)], fill=PINK, width=4) d2.polygon([(PLAYHEAD_X - 10, ROLL_TOP - 56), (PLAYHEAD_X + 10, ROLL_TOP - 56), diff --git a/pop/loner/c/loner-chart.h b/pop/loner/c/loner-chart.h index 3ba2e62f39..cdf8c510cb 100644 --- a/pop/loner/c/loner-chart.h +++ b/pop/loner/c/loner-chart.h @@ -14,10 +14,10 @@ typedef struct { const char *name; double leadIn; double beats; static const ChartNote w_whole_line_notes[] = { { 0.00, 4.00, 7 }, - { 4.00, 1.50, 3 }, - { 5.50, 2.00, 2 }, - { 7.50, 0.50, 0 }, - { 8.00, 1.00, 0 }, + { 4.00, 2.00, 3 }, + { 6.00, 2.00, 2 }, + { 8.00, 0.50, 0 }, + { 8.50, 0.50, 0 }, { 9.00, 3.00, 5 }, { 12.00, 3.00, 5 }, { 15.00, 1.00, -2 }, diff --git a/pop/loner/vox4/.chart.json b/pop/loner/vox4/.chart.json index 85593df475..62b8425f3f 100644 --- a/pop/loner/vox4/.chart.json +++ b/pop/loner/vox4/.chart.json @@ -11,25 +11,25 @@ }, { "beat": 4.0, - "dur": 1.5, + "dur": 2.0, "st": 3, "t": "cur" }, { - "beat": 5.5, + "beat": 6.0, "dur": 2.0, "st": 2, "t": "led" }, { - "beat": 7.5, + "beat": 8.0, "dur": 0.5, "st": 0, "t": "up" }, { - "beat": 8.0, - "dur": 1.0, + "beat": 8.5, + "dur": 0.5, "st": 0, "t": "in" }, diff --git a/pop/loner/vox4/.manifest.json b/pop/loner/vox4/.manifest.json index 3efa80a1eb..898701471c 100644 --- a/pop/loner/vox4/.manifest.json +++ b/pop/loner/vox4/.manifest.json @@ -10,6 +10,10 @@ "low3": 27.99, "low5": 27.99 }, + "trims": [ + "led \u2212390ms", + "think \u2212445ms" + ], "words": "the whole lyric, one take" }, "w-sitting-curled": { @@ -23,6 +27,7 @@ "low3": 5.905, "low5": 5.905 }, + "trims": [], "words": "sitting curled up in myself" }, "w-i-think": { @@ -36,6 +41,7 @@ "low3": 2.225, "low5": 2.225 }, + "trims": [], "words": "i think" }, "w-of-a-stone": { @@ -49,6 +55,7 @@ "low3": 4.395, "low5": 4.395 }, + "trims": [], "words": "of a stone" }, "w-just-waiting": { @@ -62,6 +69,7 @@ "low3": 3.69, "low5": 3.69 }, + "trims": [], "words": "just waiting" }, "w-very-patiently": { @@ -75,6 +83,7 @@ "low3": 4.645, "low5": 4.645 }, + "trims": [], "words": "very patiently" }, "w-for-time-to-pass": { @@ -88,6 +97,7 @@ "low3": 6.835, "low5": 6.835 }, + "trims": [], "words": "for time to pass" }, "w-n-getting-curled": { @@ -101,6 +111,7 @@ "low3": 5.425, "low5": 5.425 }, + "trims": [], "words": "getting curled up in myself i think" }, "w-n-stone-waiting": { @@ -114,6 +125,7 @@ "low3": 8.4, "low5": 8.4 }, + "trims": [], "words": "of a stone just waiting very patiently" }, "w-n-for-time-to-pass": { @@ -127,6 +139,7 @@ "low3": 3.78, "low5": 3.78 }, + "trims": [], "words": "for time to pass" } } \ No newline at end of file -- 2.51.2