diff --git a/pop/loner/README.md b/pop/loner/README.md index e4e09a7fdf..7eacd5d7ef 100644 --- a/pop/loner/README.md +++ b/pop/loner/README.md @@ -297,6 +297,43 @@ pop/loner/c/lonerremix # → out/loner-remix-v4-full.wav bash pop/loner/c/cut-v4.sh # → out/loner-remix-v4.mp3 (−14 LUFS) ``` +### The kick+vocals study, and reading it + +`MINIMAL=1 c/lonerremix` renders `out/loner-kickvox` — kick plus the +unbroken vocal, nothing else — and `bin/timeline.py` draws it: a +scrolling piano roll where every word is a block at its chart slot and +sung pitch, **with the word's real waveform drawn inside it** (per-column +peak of the `vox4/` lead render, normalized to the whole take), bars +numbered, the kick lane flashing, a live timecode and `bar · beat` +address in the corner. The active word outlines in pink rather than +filling, so the waveform stays readable while it plays. + +Drawing the audio into the blocks is what exposed **the energy trim**. +Whisper's word boundaries are handoffs, not note ends: it gave "led" +0.99 s when Camille stops singing after ~0.55 and the rest is decay, and +those dead frames were being time-stretched across the slot along with +the note — "led" sang to 78% of its block and then sat there. Each +unit's source span now ends where its audio actually ends (5 ms RMS +against a −36 dB gate of the take's peak, plus a 50 ms release margin), +and the silence is **dropped rather than warped**, so only sung frames +stretch. Guards: the last unit is untouched (the tail/release machinery +owns it), trims under 80 ms aren't worth the surgery, and no unit can +lose more than 65% of its span. On the whole line exactly two fire — +`led −390 ms · think −445 ms` — and every word now sings to ≥92% of its +slot. Receipt: `trims` in `vox4/.manifest.json`. + +Bar 1 is hand-pinned against that picture: **curled** alone fills it +(cur 2 + led 2 — her own 0.74/0.99 s split says led ≥ cur), and **up**, +her 0.25 s tonic release, lands *on* the bar-2 downbeat as an up-in +pickup pair into "myself". + +```bash +MINIMAL=1 pop/loner/c/lonerremix # → out/loner-kickvox-full.wav +python3 pop/loner/bin/timeline.py # → out/loner-kickvox-timeline.mp4 +``` + +`timeline.py` wants system `python3` (PIL), not `pop/.venv`. + ## Measured | check | v1 | v2 | v3 | diff --git a/pop/loner/bin/halo3.py b/pop/loner/bin/halo3.py index 2ae42c6503..be7d462b8a 100644 --- a/pop/loner/bin/halo3.py +++ b/pop/loner/bin/halo3.py @@ -164,9 +164,15 @@ CHART = { # POST-split indices. "splits": [5], # then: think = 2 · OF (the octave) = 4 · - # a stone = 4 - "durs": { 0: 4.0, 1: 1.5, 2: 2.0, 3: 0.5, - 4: 1.0, 5: 3.0, 6: 3.0, 7: 1.0, + # a stone = 4. "curled is too short · up + # comes too soon": CURLED alone fills bar 1 + # (cur 2 + led 2 — her own 37/50 split says + # led ≥ cur), and up — her 0.25 s tonic + # release — lands ON the bar-2 downbeat, an + # up-in pickup pair into myself, which + # stays anchored at beat 9. + "durs": { 0: 4.0, 1: 2.0, 2: 2.0, 3: 0.5, + 4: 0.5, 5: 3.0, 6: 3.0, 7: 1.0, 8: 2.0, 9: 4.0, 10: 0.5, 11: 3.5 } }, "w-sitting-curled": { "slice": "f-sitting-curled", "beats": 11.0 }, "w-i-think": { "slice": "f-i-think", "beats": 3.5 }, @@ -200,6 +206,52 @@ def derive_units(words, beats_total, stretch=None, durs=None): SLICES = json.load(open(os.path.join(LANE, "samples", ".manifest.json"))) +# ── the energy trim — only SUNG frames stretch ──────────────────────── +# @jeffrey, reading the waveforms drawn into the timeline: "check the +# length of the actual waveforms in the utterances, not just ur trim". +# Whisper's word boundaries are handoffs, not note ends: it gave "led" +# 0.99 s when she stops singing after ~0.5 and the rest is decay. Those +# dead frames were stretching across the slot with the note, so a word +# could fill 78% of its block and then sit there. Each unit's source +# span now ends where its audio actually ends (+ a release margin), and +# the silence is DROPPED rather than warped — the vowel takes the slot. +TRIM_GATE_DB = -36.0 # of the take's peak; keeps quiet fricatives +TRIM_MARGIN_S = 0.050 # let the release start before we cut +TRIM_MIN_S = 0.080 # below this, not worth the surgery +TRIM_KEEP = 0.35 # never leave a unit shorter than this share + + +def energy_end(x, fs, f0, f1, peak): + """Last frame in [f0,f1) whose 5 ms RMS clears the gate.""" + n = int(round(fs * FRAME_S)) + seg = x[f0 * n:f1 * n] + if len(seg) < n: + return f1 + m = len(seg) // n + rms = np.sqrt((seg[:m * n].reshape(m, n) ** 2).mean(axis=1)) + on = np.nonzero(rms > peak * 10.0 ** (TRIM_GATE_DB / 20.0))[0] + return f0 + (int(on[-1]) + 1 if len(on) else m) + + +def trim_units(x, fs, unit_src, names=None): + """Pull each unit's end back to its real audio end. Last unit keeps + its span (the tail/release machinery owns it). Returns (spans, log).""" + peak = np.max(np.abs(x)) or 1.0 + out, log = [], [] + for u, (s0, s1) in enumerate(unit_src): + if u == len(unit_src) - 1: + out.append((s0, s1)) + continue + e = energy_end(x, fs, s0, s1, peak) + int(round(TRIM_MARGIN_S / FRAME_S)) + e = max(s0 + int(round(TRIM_KEEP * (s1 - s0))), min(e, s1)) + cut = (s1 - e) * FRAME_S + if cut < TRIM_MIN_S: + e = s1 + elif names: + log.append(f"{names[u]} −{cut * 1000:.0f}ms") + out.append((s0, e)) + return out, log + def build_warp(a, unit_src, beats, dursb): """Frame index map with VOWEL-ON-THE-BEAT alignment. @@ -368,6 +420,8 @@ for name, ch in CHART.items(): s1 = int(round((words[i + 1]["start"] - t0_slice) / FRAME_S)) unit_src.append((max(0, s0), min(F, max(s0 + 1, s1)))) + unit_src, trims = trim_units(x, fs, unit_src, [w["t"] for w in words]) + idx, holds, fade, Z = build_warp(a, unit_src, [b for (b, d) in ch["units"]], [d for (b, d) in ch["units"]]) @@ -427,8 +481,10 @@ for name, ch in CHART.items(): chart_c.append((name, lead_in, beats_total, notes)) manifest[name] = dict(slice=slice_name, lead_in=round(lead_in, 3), beats=beats_total, renders=renders, - words=entry["words"]) + trims=trims, words=entry["words"]) print(f" {name:22s} {renders['lead']:5.2f}s lead·8ve×2·low3·low5 «{entry['words']}»") + if trims: + print(f" trimmed: {' · '.join(trims)}") json.dump(manifest, open(os.path.join(VOX4, ".manifest.json"), "w"), indent=1) diff --git a/pop/loner/bin/timeline.py b/pop/loner/bin/timeline.py index 3a77d67ebb..af1c669351 100644 --- a/pop/loner/bin/timeline.py +++ b/pop/loner/bin/timeline.py @@ -11,7 +11,7 @@ # pop/.venv/bin/python pop/loner/bin/timeline.py # → out/loner-kickvox-timeline.mp4 -import json, math, os, subprocess +import json, math, os, subprocess, wave import numpy as np from PIL import Image, ImageDraw, ImageFont @@ -50,6 +50,31 @@ FRAMES = int(math.ceil(dur * FPS)) F = lambda s: ImageFont.truetype("/System/Library/Fonts/Helvetica.ttc", s) f_title, f_bar, f_word, f_note, f_tiny = F(40), F(26), F(30), F(20), F(22) +M = lambda s: ImageFont.truetype("/System/Library/Fonts/Menlo.ttc", s) +f_tc, f_tc_small = M(38), M(22) + +# ── the actual sung waveform, so the eye can audit the trim ─────────── +# @jeffrey: "check the length of the actual waveforms in the utterances, +# not just ur trim etc — map / render those waveforms directly into the +# clips". The lead render (vox4/w-whole-line.wav) IS what the study +# plays; chart beat b lives at leadIn + b·SPB seconds in that file. +with wave.open(os.path.join(LANE, "vox4", "w-whole-line.wav"), "rb") as wf: + VFS = wf.getframerate() + VOX = np.frombuffer(wf.readframes(wf.getnframes()), + dtype=np.int16).astype(np.float64) / 32768.0 +VOX_PEAK = np.max(np.abs(VOX)) or 1.0 +LEAD_IN = chart["leadIn"] + +def vox_env(b0, b1, npx): + """Per-pixel-column peak of the lead render between chart beats.""" + s0 = int((LEAD_IN + b0 * SPB) * VFS) + s1 = int((LEAD_IN + b1 * SPB) * VFS) + seg = np.abs(VOX[max(0, s0):max(0, s1)]) + if len(seg) == 0 or npx <= 0: + return np.zeros(max(1, npx)) + edges = np.linspace(0, len(seg), npx + 1).astype(int) + return np.array([seg[a:b].max() if b > a else 0.0 + for a, b in zip(edges[:-1], edges[1:])]) / VOX_PEAK CHROM = ["A#", "B", "C", "C#", "D", "D#", "E", "F", "F#", "G", "G#", "A"] def note_name(st): @@ -109,6 +134,14 @@ for pb in PASSES: y0, y1 = y - 24, y + 24 d.rounded_rectangle([x0 + 2, y0, x1 - 3, y1], 10, fill=(255, 166, 202, 90), outline=INK, width=2) + # the real audio inside the clip: per-column peak of the lead + # render, normalized to the whole take — dead air is visible + env = vox_env(n["beat"], n["beat"] + n["dur"], max(1, x1 - x0 - 8)) + for j, a in enumerate(env): + ah = a * 20 + if ah >= 0.4: + xw = x0 + 4 + j + d.line([(xw, y - ah), (xw, y + ah)], fill=(26, 26, 34, 88), width=1) txt = word_text(n["t"]) tw = d.textlength(txt, font=f_word) wide = (x1 - x0) > tw + 20 @@ -154,13 +187,22 @@ for i in range(FRAMES): if t0 <= t < t1: lx0, lx1 = x0 - px, x1 - px if lx1 > 0 and lx0 < W: + # outline-only: the waveform inside stays scrutinizable d2.rounded_rectangle([lx0 + 2, y0, lx1 - 3, y1], 10, - fill=(255, 102, 168, 120), outline=PINK, width=3) + fill=(255, 102, 168, 40), outline=PINK, width=4) # kick flash kb = int(beat) if kb < KICK_BEATS and (beat - kb) < 0.22: x = X(kb) - px d2.rounded_rectangle([x + 4, KY0, x + PXB - 8, KY1], 6, fill=PINK) + # live timecode — song clock + musical address, updating every frame + tc = f"{int(t // 60):02d}:{t % 60:05.2f}" + bar_no = int(beat) // 4 - 2 + addr = (f"bar {bar_no} · beat {beat % 4 + 1:3.1f}" if beat >= 8 + else f"count-in · beat {beat % 8 + 1:3.1f}") + d2.text((W - 40 - d2.textlength(tc, font=f_tc), 26), tc, font=f_tc, fill=INK) + d2.text((W - 40 - d2.textlength(addr, font=f_tc_small), 78), addr, + font=f_tc_small, fill=BLUE) # playhead d2.line([(PLAYHEAD_X, ROLL_TOP - 44), (PLAYHEAD_X, KY1 + 8)], fill=PINK, width=4) d2.polygon([(PLAYHEAD_X - 10, ROLL_TOP - 56), (PLAYHEAD_X + 10, ROLL_TOP - 56), diff --git a/pop/loner/c/loner-chart.h b/pop/loner/c/loner-chart.h index 3ba2e62f39..cdf8c510cb 100644 --- a/pop/loner/c/loner-chart.h +++ b/pop/loner/c/loner-chart.h @@ -14,10 +14,10 @@ typedef struct { const char *name; double leadIn; double beats; static const ChartNote w_whole_line_notes[] = { { 0.00, 4.00, 7 }, - { 4.00, 1.50, 3 }, - { 5.50, 2.00, 2 }, - { 7.50, 0.50, 0 }, - { 8.00, 1.00, 0 }, + { 4.00, 2.00, 3 }, + { 6.00, 2.00, 2 }, + { 8.00, 0.50, 0 }, + { 8.50, 0.50, 0 }, { 9.00, 3.00, 5 }, { 12.00, 3.00, 5 }, { 15.00, 1.00, -2 }, diff --git a/pop/loner/vox4/.chart.json b/pop/loner/vox4/.chart.json index 85593df475..62b8425f3f 100644 --- a/pop/loner/vox4/.chart.json +++ b/pop/loner/vox4/.chart.json @@ -11,25 +11,25 @@ }, { "beat": 4.0, - "dur": 1.5, + "dur": 2.0, "st": 3, "t": "cur" }, { - "beat": 5.5, + "beat": 6.0, "dur": 2.0, "st": 2, "t": "led" }, { - "beat": 7.5, + "beat": 8.0, "dur": 0.5, "st": 0, "t": "up" }, { - "beat": 8.0, - "dur": 1.0, + "beat": 8.5, + "dur": 0.5, "st": 0, "t": "in" }, diff --git a/pop/loner/vox4/.manifest.json b/pop/loner/vox4/.manifest.json index 3efa80a1eb..898701471c 100644 --- a/pop/loner/vox4/.manifest.json +++ b/pop/loner/vox4/.manifest.json @@ -10,6 +10,10 @@ "low3": 27.99, "low5": 27.99 }, + "trims": [ + "led \u2212390ms", + "think \u2212445ms" + ], "words": "the whole lyric, one take" }, "w-sitting-curled": { @@ -23,6 +27,7 @@ "low3": 5.905, "low5": 5.905 }, + "trims": [], "words": "sitting curled up in myself" }, "w-i-think": { @@ -36,6 +41,7 @@ "low3": 2.225, "low5": 2.225 }, + "trims": [], "words": "i think" }, "w-of-a-stone": { @@ -49,6 +55,7 @@ "low3": 4.395, "low5": 4.395 }, + "trims": [], "words": "of a stone" }, "w-just-waiting": { @@ -62,6 +69,7 @@ "low3": 3.69, "low5": 3.69 }, + "trims": [], "words": "just waiting" }, "w-very-patiently": { @@ -75,6 +83,7 @@ "low3": 4.645, "low5": 4.645 }, + "trims": [], "words": "very patiently" }, "w-for-time-to-pass": { @@ -88,6 +97,7 @@ "low3": 6.835, "low5": 6.835 }, + "trims": [], "words": "for time to pass" }, "w-n-getting-curled": { @@ -101,6 +111,7 @@ "low3": 5.425, "low5": 5.425 }, + "trims": [], "words": "getting curled up in myself i think" }, "w-n-stone-waiting": { @@ -114,6 +125,7 @@ "low3": 8.4, "low5": 8.4 }, + "trims": [], "words": "of a stone just waiting very patiently" }, "w-n-for-time-to-pass": { @@ -127,6 +139,7 @@ "low3": 3.78, "low5": 3.78 }, + "trims": [], "words": "for time to pass" } } \ No newline at end of file