From 4575d2628d45ab97e57c1c81e5857fcb2f387f63 Mon Sep 17 00:00:00 2001 From: prompt.ac/@jeffrey Date: Mon, 04 May 2026 10:04:29 +0000 Subject: [PATCH] pop/big-pictures: per-syllable tiktok pipeline + folk bell accompaniment initial check-in of the big-pictures pop song pipeline. score-driven storyboard (.np → per-syllable slides), pitchsnap WORLD vocal correction, FLUX word-image gen + OCR validation, render_frames.py held-center keyframes (slides arrive td after audio onset, hold through full sustain), finalize.mjs cover art with non-overlapping bands. mary scored across all four traditional verses + chick-chick- boom outro hook for the folk_backing gunshot SFX trigger. Co-Authored-By: Claude Opus 4.7 (1M context) --- pop/.gitignore | 6 ++++++ pop/AUTOTUNE-ALGORITHMS.md | 178 ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ pop/JEFFREY-VOICE.md | 173 +++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ pop/RESEARCH-DIRECTION.md | 78 ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ pop/SCORE.md | 49 +++++++++++++++++++++++++++++++++++++++++++++++++ pop/SPEECH-TO-SINGING-V2.md | 172 ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ pop/SPEECH-TO-SINGING.md | 102 ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ pop/VOICE.md | 45 +++++++++++++++++++++++++++++++++++++++++++++ pop/big-pictures/README.md | 76 ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ pop/big-pictures/ac-da.txt | 29 +++++++++++++++++++++++++++++ pop/big-pictures/ac.txt | 31 +++++++++++++++++++++++++++++++ pop/big-pictures/amazing.np | 18 ++++++++++++++++++ pop/big-pictures/amazing.txt | 5 +++++ pop/big-pictures/elephant.np | 9 +++++++++ pop/big-pictures/elephant.txt | 2 ++ pop/big-pictures/mary.np | 33 +++++++++++++++++++++++++++++++++ pop/big-pictures/mary.txt | 19 +++++++++++++++++++ pop/big-pictures/plork.np | 43 +++++++++++++++++++++++++++++++++++++++++++ pop/big-pictures/plork.txt | 50 ++++++++++++++++++++++++++++++++++++++++++++++++++ pop/big-pictures/row.np | 10 ++++++++++ pop/big-pictures/row.txt | 5 +++++ pop/big-pictures/sentence.np | 6 ++++++ pop/big-pictures/sentence.txt | 2 ++ pop/big-pictures/uncomfortable.np | 6 ++++++ pop/big-pictures/uncomfortable.txt | 2 ++ pop/bin/align.mjs | 79 +++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ pop/bin/finalize.mjs | 456 ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ pop/bin/musicxml_to_np.py | 313 +++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ pop/bin/pitchcheck.mjs | 264 ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ pop/bin/pitchsnap.mjs | 1017 +++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ pop/bin/pitchsnap_world.py | 276 ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ pop/bin/pitchwords.mjs | 199 +++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ pop/bin/refine_onsets.py | 92 ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ pop/bin/render_frames.py | 890 ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ pop/bin/repair_letter.py | 152 ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ pop/bin/say.mjs | 193 +++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ pop/bin/storyboard.mjs | 226 ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ pop/bin/tiktok.mjs | 637 +++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ pop/bin/timefit.mjs | 133 +++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ pop/bin/validate_word.py | 313 +++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ pop/references/README.md | 18 ++++++++++++++++++ 41 file(s) changed, 6407 insertion(s)(+), 0 deletion(s)(-) diff --git a/pop/.gitignore b/pop/.gitignore new file mode 100644 --- /dev/null +++ b/pop/.gitignore @@ -0,0 +1,6 @@ +# reference corpus stays vault-only — never commit third-party lyrics +references/* +!references/README.md + +# track outputs +big-pictures/out/ diff --git a/pop/AUTOTUNE-ALGORITHMS.md b/pop/AUTOTUNE-ALGORITHMS.md new file mode 100644 --- /dev/null +++ b/pop/AUTOTUNE-ALGORITHMS.md @@ -0,0 +1,178 @@ +# autotune algorithms — non-neural offline pitch correction + +scope: the actual internals of pitch-correction, with code we can lift. neural systems (rvc, diffsinger, etc.) are the other agent's beat. this is dsp, c, and python — the stuff that runs on jeffrey's 8gb macbook without a gpu and finishes before lunch. + +the problem with `pop/bin/pitchsnap.mjs` today: it slices a vocal by word and runs `rubberband -p N` per segment. that's a *static* shift — every sample inside the word slides up the same n semitones. the source's natural pitch contour rides along, which is the opposite of autotune. autotune *clamps* — it replaces f0 with a target curve. + +## 1. what autotune actually does — five sentences + +step 1: estimate f0 (pitch) frame-by-frame on the input, ~5 ms hop. step 2: for each frame, snap the detected midi note to the nearest member of an allowed scale (with hysteresis so a wobble around the boundary doesn't flip notes every frame). step 3: compute a per-frame correction ratio `target_f0 / source_f0`. step 4: shift the audio by that ratio while preserving formants — so the timbre stays human, the pitch becomes mechanical. step 5: stitch frames back together with phase coherence (or, in psola/world world, resynth from decomposed parameters). + +that's it. the magic is entirely in steps 1, 4, and 5 being good. + +## 2. pitch detection — yin / pyin + +naive autocorrelation (what `pitchsnap.mjs` uses today) finds octave errors because the autocorrelation peak at lag `t` is matched at `2t`, `3t`, `4t`, etc. any noise pushes the search to the wrong harmonic. yin (de cheveigné & kawahara, 2002) fixes this with the *cumulative mean normalized difference function* — the curve starts at 1 and only dips below when a true period is found, with the lower-lag (higher pitch) match preferred. + +```python +# from patriceguyot/Yin (mit license) — compact form +def differenceFunction(x, w, tau_max): + # d(tau) = sum_i (x[i] - x[i+tau])^2, computed via fft for speed + x_cumsum = np.concatenate(([0.], (x*x).cumsum())) + fc = np.fft.rfft(x, size_pad) + conv = np.fft.irfft(fc * fc.conjugate())[:tau_max] + return x_cumsum[w:w-tau_max:-1] + x_cumsum[w] - x_cumsum[:tau_max] - 2*conv + +def cmndf(df, N): + # normalize so curve starts at 1, drops < 1 at true period + return np.insert(df[1:] * range(1, N) / np.cumsum(df[1:]), 0, 1) + +def getPitch(cmdf, tau_min, tau_max, harmo_th=0.1): + tau = tau_min + while tau < tau_max: + if cmdf[tau] < harmo_th: + while tau+1 < tau_max and cmdf[tau+1] < cmdf[tau]: + tau += 1 # walk down to local minimum + return tau + tau += 1 + return 0 # no pitch found (unvoiced) +``` + +source: `patriceguyot/Yin/yin.py`, also vendored into `nvidia/mellotron`. pyin (mauch & dixon, 2014) extends yin with a probabilistic hmm over candidate periods — much more robust on noisy/breathy vocals. `librosa.yin` and `librosa.pyin` ship both. + +## 3. pitch shifting — psola vs phase vocoder vs world resynth + +three families, each with a tradeoff: + +**td-psola** (charpentier & stella, 1986). estimate f0, find pitch-period peaks, extract two-period hann-windowed grains, re-space the grains at the *target* period, overlap-add. key property: formants are intrinsic to the grain shape, so they survive untouched — no separate formant flag needed. weak on polyphonic / unvoiced material. clean reference: `sannawag/TD-PSOLA`: + +```python +def shift_pitch(signal, fs, f_ratio): + peaks = find_peaks(signal, fs) # autocorr-based pitch-mark detection + return psola(signal, peaks, f_ratio) # window+respace+overlap-add +``` + +**phase vocoder** (flanagan & golden, 1966; dolson 1986). stft → for each bin, estimate true frequency from the phase derivative across hops → resynthesize at scaled frame rate → resample to original duration. this is what librosa's `effects.pitch_shift` does (time-stretch via phase vocoder, then resample). also the engine inside rubber band's `R2` mode. weakness: "phasiness" — bins drift apart, transients smear. fixes (laroche & dolson 1999, "phase locking") are what rubber band's `R3 (--finer)` mode adds. + +**world resynth** (morise et al., 2016). decompose speech into three independent streams: f0, smoothed spectral envelope (cheaptrick), aperiodicity (d4c). modify any of them. resynth. this is the cleanest "replace the f0 curve" pipeline that exists outside the neural world — see section 4. + +| method | formants | best on | failure mode | +|---|---|---|---| +| td-psola | preserved (intrinsic) | clean monophonic vocals | breathy / unvoiced | +| phase vocoder | requires extra filter | broadband / polyphonic | phasiness, transient smear | +| world resynth | preserved (envelope split out) | speech / vocals | not free of artifacts on heavy distortion | + +## 4. world / pyworld — the f0-replace pipeline + +this is the one. `pyworld` is the python wrapper around morise's c library. install: `pip install pyworld`. has wheels for arm64 macos. the canonical autotune flow is six lines: + +```python +import pyworld as pw + +# 1. decompose +_f0, t = pw.dio(x, fs) # raw f0 candidates (or pw.harvest for accuracy) +f0 = pw.stonemask(x, _f0, t, fs) # refine f0 to sub-frame precision +sp = pw.cheaptrick(x, f0, t, fs) # smoothed spectral envelope (formants) +ap = pw.d4c(x, f0, t, fs) # aperiodicity (breath, fricatives) + +# 2. modify f0 — this is where autotune happens +f0_corrected = quantize_to_scale(f0, scale_midi_notes) # see section 5 + +# 3. resynthesize +y = pw.synthesize(f0_corrected, sp, ap, fs) +``` + +note what's not here: there is no pitch-shift step. you literally write the new f0 curve into the synthesizer. `sp` (formants) and `ap` (breath texture) are unchanged, so the speaker's identity survives perfectly. this is the api surface we want. + +`harvest` is slower but recovers from the octave errors that bite `dio` on jeffrey's takes. for vocals always prefer `harvest` + `stonemask`. + +## 5. scale quantization with hysteresis + +given an instantaneous f0 in hz and a target scale (set of midi numbers), the snap is straightforward — hz → midi via `69 + 12*log2(f/440)`, then nearest-member lookup. the trick is hysteresis: if you're sitting between two scale degrees and the f0 wobbles ±10 cents, naive nearest-snap will *flip notes every frame*, producing a digital trill. fix: + +```python +def quantize_to_scale(f0, scale_midi, prev_target=None, + hysteresis_cents=30, retain=1.0): + out = np.zeros_like(f0) + cur = prev_target + for i, f in enumerate(f0): + if f <= 0: # unvoiced — keep silent + out[i] = 0; continue + midi = 69 + 12*np.log2(f/440) + # candidates sorted by distance to current frame + cands = sorted(scale_midi, key=lambda n: abs(n - midi)) + nearest = cands[0] + if cur is not None and nearest != cur: + # only switch if we've moved > hysteresis cents past the boundary + if abs(midi - nearest)*100 + hysteresis_cents > abs(midi - cur)*100: + nearest = cur + cur = nearest + # `retain` interpolates between source pitch (0) and full snap (1) + corrected_midi = midi + retain * (nearest - midi) + out[i] = 440 * 2**((corrected_midi - 69)/12) + return out +``` + +`retain=1.0` is full hard auto-tune (the t-pain sound). `retain=0.6` is melodyne-style — keeps human bend. `hysteresis_cents=30` to `50` kills the trill without slowing legit pitch transitions. + +## 6. shortlist — what we can plug in tonight + +| lib | language | offline? | install | what we get | +|---|---|---|---|---| +| **pyworld** | python | yes | `pip install pyworld` | f0 extract + replace + resynth, formants intrinsic | +| psola | python | yes | `pip install psola` | td-psola via parselmouth/praat — needs target-pitch numpy array | +| librosa | python | yes | `pip install librosa` | `yin`, `pyin`, `effects.pitch_shift` (phase vocoder) | +| crepe | python | gpu-light | `pip install crepe` | neural pitch detector — slower but cleanest f0 | +| autotalent | c (ladspa) | yes | source build | reference c implementation, gpl2 | +| rubberband | c++ | yes | `brew install rubberband` (already installed) | phase vocoder pitch shift, no f0-replace | +| sannawag/TD-PSOLA | python | yes | `git clone` | minimal readable td-psola reference | + +all of the python options install on apple silicon python 3.14 with native wheels in 2026. pyworld's only build dep is numpy + cython. + +## 7. integration sketch — pitchsnap.mjs → world + +the right move is to delegate the per-word pitch step to a small python helper that does world-style f0 replacement. `pitchsnap.mjs` keeps its responsibilities (whisper alignment, score parsing, grid snap, slice extraction) and calls the helper for the actual correction: + +```python +# pop/bin/pitchsnap_world.py — called per-word from pitchsnap.mjs +import sys, numpy as np, pyworld as pw, soundfile as sf + +in_wav, out_wav, target_midi, retain = sys.argv[1:5] +target_midi = float(target_midi); retain = float(retain) + +x, fs = sf.read(in_wav, dtype="float64") +if x.ndim > 1: x = x.mean(axis=1) + +f0_raw, t = pw.harvest(x, fs, f0_floor=80, f0_ceil=600) +f0 = pw.stonemask(x, f0_raw, t, fs) +sp = pw.cheaptrick(x, f0, t, fs) +ap = pw.d4c(x, f0, t, fs) + +# replace f0 with target — hold the bend with `retain` +target_hz = 440 * 2**((target_midi - 69)/12) +voiced = f0 > 0 +f0_new = np.where(voiced, np.exp((1-retain)*np.log(np.maximum(f0,1e-6)) + + retain*np.log(target_hz)), 0) + +y = pw.synthesize(f0_new, sp, ap, fs) +sf.write(out_wav, y.astype(np.float32), fs) +``` + +then in `pitchsnap.mjs`, replace the rubberband call inside the per-word loop with: + +```javascript +spawnSync("python3", [ + resolve(import.meta.dirname, "pitchsnap_world.py"), + sliceWav, shiftedWav, String(targetMidi), String(retain) +], { stdio: "inherit" }); +``` + +next phase: instead of a single `target_midi` per word, write a per-frame target curve that interpolates between syllable notes — that's the syllable-glide we already plan in `pitchsnap.mjs` curve mode, but executed by world instead of cross-fading two rubberband renders. + +--- + +**recommendation: integrate pyworld first.** it's the only option in this list that lets us *replace* the f0 curve instead of shifting it, formants stay intact for free, and the install is one line. + +``` +pip install pyworld soundfile numpy +``` diff --git a/pop/JEFFREY-VOICE.md b/pop/JEFFREY-VOICE.md new file mode 100644 --- /dev/null +++ b/pop/JEFFREY-VOICE.md @@ -0,0 +1,173 @@ +# jeffrey voice — for writing rap lyrics + +a style guide for `pop/big-pictures/`. read this before drafting a lyric. all quotes below are real, pulled from the AC repo — papers, recap narrations, code comments. the voice you're chasing already exists; the job is to compress it into bars without flattening it. + +## 0. the posture in one line + +> "the music is real and the math just flipped" — `pop/big-pictures/plork.txt` + +quiet conviction, present tense, plain words. nobody is being convinced. somebody is being shown. + +## 1. recurring couplet patterns (lift these) + +### a. "X is the Y, Y is the Z" — chained identity + +the load-bearing one. jeffrey reaches for it whenever a system collapses into itself. + +- "every piece a url / every url a score" (`ac.txt`) +- "the address is the score" (`ac.txt`) +- "the goal was the room. the room was the soul" (`plork.txt`) +- "the gear was the gate, the gate was the price tag" (`plork.txt`) +- "the deployment mechanism is a file, not a supply chain" (`plork.tex` §10) + +lands when the chain is a real collapse — A actually equals B, not "is sort of like". don't fake it. + +### b. "every X a Y" — universalizer + +- "every piece a url" (`ac.txt`) +- "every URL is the program" (paraphrased through whole `papers/SCORE.md`) +- "every orchestra instrument has a name. now every laptop does too." (`plork.tex` §4) + +works in hooks. one stress per side, no filler. + +### c. "no X, no Y, no Z" — the negation triplet + +- "no investors, no playbook, no series-a scheme" (`ac.txt`) +- "one-person stack, no runway, no dream" (`ac.txt`) +- "no userspace, no desktop environment, no package manager, no shell" (`plork.tex` §4) +- "no install, no app store, no auth at the door" (general AC posture, paraphrasable) + +three is the number. four is academic. two is incomplete. + +### d. "this is not a metaphor. it is a Y." — the slam-pivot + +the canonical jeffrey line: + +> "PLOrk'ing the planet is not a metaphor. It is a logistics problem. And the logistics just got a hundred times cheaper." (`plork.tex` §13) + +already lifted into bar form: + +- "the music is real — the logistics got cheap" (hook of `plork.txt`) +- "plork the planet — not a theory anymore, a tool" (`plork.txt`) + +use this when the listener might be reaching for the metaphor read. block it. name what it actually is. + +### e. "i'm not building a product, i'm building a Y." — the disavowal + +- "i'm not pitching a product, i'm holding a stand" (`ac.txt`) +- "i'm not asking permission — i'm just lifting it" (`plork.txt`) +- "i'm not bitter — i'm honest, i keep score on the track" (`plork.txt`) + +the second clause has to do real work. "not X, just Y" only lands if Y is more specific than X, not more abstract. + +### f. "X was the Y" — reframing what something already was + +- "the laptop orchestra was a beautiful idea that reached almost no one" (`plork.tex` §1) +- "this was not a barrier for Princeton. it was a barrier for everyone else." (`plork.tex` §1) +- "the surplus laptop is not a degraded general-purpose computer. it is an unrealized dedicated device." (`plork.tex` §5) +- "the whistlegraph is a score that teaches you how to play it" (`whistlegraph.tex` abstract) + +structurally a setup-and-flip. the rap version puts the flip on the rhyme. + +## 2. vocabulary jeffrey reaches for + +real words from the corpus, with locations. lift these. they carry his fingerprint. + +| word / phrase | example | +|---|---| +| **score** | "the address is the score" / "a score that teaches you how to play it" (`whistlegraph.tex`) | +| **address** | "the address is the score" (`ac.txt`); "color address" (`recap/audience/jeffrey-24h-2026-04-30.mjs`) | +| **piece** | "every piece a url" / "the piece was the wave" (`ac.txt`) | +| **real / it's real** | "the music is real" (used three times in `plork.txt`); "real, deadpan, very 'him'" (recap) | +| **the whole X** | "the whole move" / "the whole stack" / "the whole runtime" (`ac.txt`); "the whole app does less work when hidden" (`jeffrey-73h`) | +| **load-bearing** | "load-bearing infrastructure" (`jeffrey-24h-2026-05-01`); from `pop/VOICE.md`: "earned and load-bearing" | +| **honest** | "i'm not bitter — i'm honest" (`plork.txt`); "be honest about what doesn't work" (`papers/VOICE.md`) | +| **lands / it lands** | "fragments when they land" (`pop/VOICE.md`); "end with something that lands" (`papers/VOICE.md`) | +| **shipping / shipped** | "menuband shipped six releases in a row" (`jeffrey-73h`) | +| **logistics** | "it is a logistics problem. and the logistics just got a hundred times cheaper" (`plork.tex`) | +| **the gate / gated** | "the gear was the gate, the gate was the price tag" (`plork.txt`); "gated by enrollment" (`plork.tex`) | +| **kernel / flash a kernel** | "i flash a kernel from a usb and call it a school" (`plork.txt`) | +| **stack** | "one-person stack, no runway, no dream" (`ac.txt`); "the menubar audio stack" (`jeffrey-73h`) | +| **barely / under / sublinear** | "scales sublinearly with complexity" (`plork.tex`); "boot in under 8 seconds" | +| **the room** | "the goal was the room. the room was the soul" (`plork.txt`) | +| **tool / convivial** | "not a theory anymore, a tool" (`plork.txt`); "tools for conviviality" (`plork.tex` §8) | +| **planet / planetary** | "plork the planet" / "planetary laptop orchestra" (`plork.tex` everywhere) | +| **moat** | "build the research moat" (`papers/RESEARCH-DIRECTION.md`) | +| **fork / forkable** | "fork the whole runtime — same kid, same shape" (`ac.txt`); "URL-safe, and forkable" (`plork.tex`) | +| **embarrassingly** | "embarrassingly abundant" (`plork.tex` conclusion) | + +avoid technical synonyms. when jeffrey means kernel, he says kernel. when he means url, he says url. don't translate `mjs` to "module" or `usb` to "drive". the specific word is the texture. + +## 3. punctuation + rhythm tics + +- **em-dash** for the swerve. comma is too soft, parens are academic, semicolons are forbidden. + - "the music is real — the logistics got cheap" + - "i'm not bitter — i'm honest, i keep score on the track" +- **period for the slam.** short final sentence after a longer setup. the bar that ends a verse is almost always shorter than the bar before it. + - "It is a logistics problem. And the logistics just got a hundred times cheaper." +- **lowercase as default.** capitalize only where convention demands it (handles, proper nouns the listener actually knows). brand names like AC and PLOrk capitalize in papers but stay lowercase in lyric drafts. +- **fragments after a longer setup.** "fair. but the full stack was never the goal." — the one-word concession lands harder than the explanation. +- **never a question mark in a hook.** statements, not pitches. (the verse can ask a question; the hook closes one.) + +## 4. what jeffrey AVOIDS (cut these from drafts) + +from `pop/VOICE.md` and `papers/VOICE.md` directly + the corpus pattern: + +- **academic hedging**: "we propose", "the authors", "in this paper we", "a novel approach to", "results suggest potential viability" +- **defensive citation stacking** — cite when it matters, not to prove you read things +- **ChatGPT prose tells**: "moreover", "furthermore", "it's worth noting", parallelism for parallelism's sake, three-item lists where two would do, smoothing the rough edge +- **rap cliche**: money, cars, club, ice, bottles, "ayy", "skrt", "let's get it", "you already know" — `pop/VOICE.md` is explicit: "AC has nothing to flex about and that's the point" +- **corporate vocab**: "ecosystem", "stakeholders", "leverage", "synergy", "vision statement" +- **generic emotion**: "tired" is not a feeling — `pop/VOICE.md`: "tired of explaining kidlisp at every dinner" is +- **explaining the vision before singing it.** the lyric should *be* the vision, not narrate it (`pop/VOICE.md`) +- **smoothing the catch.** if there's a catch, name it. plork.txt verse 2 opens with "look — i know what they'll say. it sounds like a hack" — that's the move. + +## 5. lyric-specific patterns (the rap layer) + +these don't appear in the prose corpus because they're rap-only. they extend the voice into bars without breaking it. + +- **internal rhyme over end rhyme.** the rhyme should fall mid-bar at least as often as it falls on the four. (`pop/VOICE.md`) + - working: "two-forty million laptops, **windows ten retired** / flash a kernel — every one of them just got **hired**" — internal "retired/hired" plus the period after "kernel" +- **the hook is the vision compressed to one repeatable line.** if you can't say what the song's about in the hook's first line, the hook's not done. + - "the music is real — the logistics got cheap" (`plork.txt`) + - "every piece a url / every url a score" (`ac.txt`) +- **second verse can be the limitations section.** confess the catch, then keep moving. exactly mirrors the papers' Limitations sections. + - `plork.txt` v2: "look — i know what they'll say. it sounds like a hack / fifty-dollar laptops can't run the full stack / fair. but the full stack was never the goal" +- **technical names land plainly, no ceremony.** kidlisp, notepat, fedac, plork, mjs, hp-gl, ywft, rs-232 — drop them in unglossed. the listener catches up. don't say "my custom lisp called kidlisp"; just say kidlisp. +- **conviction quiet but absolute.** never "i think the music is real". just "the music is real". hedging is the audible giveaway that the line isn't from him. +- **ad-libs are minimal.** one or two per verse, low in the mix. "look —" works. "yeah yeah" doesn't. +- **the t-shirt rule (recap-derived).** every chapter image has a one-word command on the shirt that resolves at the AC prompt. lyric equivalent: every verse has at least one bar that names a real piece you can type and run. (notepat, kidlisp, $roz, plork…) + +## 6. examples gallery — works / doesn't + +pulled straight from `pop/big-pictures/*.txt` to mark what's already landing and what's still drifting toward generic. + +**works:** + +- `plork.txt` hook: "the music is real — the logistics got cheap / landfill's full of orchestras dreaming in their sleep" — em-dash slam, then a specific image (landfill, orchestras, sleeping). no abstraction. +- `plork.txt`: "i'm not bitter — i'm honest, i keep score on the track" — disavowal pattern + the double meaning on "score" is exactly the kind of pun jeffrey actually makes (score = music, score = tally). +- `plork.txt`: "the gear was the gate, the gate was the price tag" — chained-identity (pattern 1a) compressed into one bar. perfect. +- `plork.txt`: "look — i know what they'll say. it sounds like a hack / fifty-dollar laptops can't run the full stack / fair. but the full stack was never the goal / the goal was the room. the room was the soul" — the limitations-section move done as bars. the "fair." is the Adornoesque period-slam. +- `ac.txt` hook: "every piece a url / every url a score / push enter — you're inside / aesthetic dot computer" — pattern 1b twice, then an instruction, then the address. four bars, four moves, no waste. +- `ac.txt`: "i wanted a place my friends could just paint / where the paint was the piece, the piece was the wave" — pattern 1a, but starting from "i wanted" gives it the personal ground. this is the emo-rap-honesty register `pop/VOICE.md` asks for. + +**doesn't (yet):** + +- `ac.txt`: "i'm not pitching a product, i'm holding a stand" — pattern 1e is right but "holding a stand" is vaguer than what jeffrey would actually say. compare to "i'm not asking permission — i'm just lifting it" from `plork.txt`, which has a real verb. "stand" reads like the line was reaching for a rhyme with "hand". the second clause has to specify, not abstract. +- `ac.txt`: "the music is real and it works in your hand" — "the music is real" is load-bearing jeffrey, but "works in your hand" is a marketing flourish. cf. `plork.txt`'s "the music is real and the math just flipped" — same opener, then a specific claim. swap the second clause for something only AC can say. +- `plork.txt`: "i watched plork in a basement on a livestream / fifteen laptops in a hall doing rituals like a dream" — "rituals like a dream" is poetic-rap default. the rest of the song avoids that register; this couplet drifts. the fix is the specific number-noun ("fifteen laptops in a hall") without the "like a dream" softener. + +## 7. checklist before you ship a draft + +- [ ] does the hook pass the one-line-vision test? if you read just line one, do you know the song? +- [ ] is there at least one chained-identity bar (pattern 1a)? +- [ ] is there at least one period-slam at a verse end? +- [ ] does verse 2 own a limitation honestly? +- [ ] are technical names (kidlisp, notepat, fedac…) landed without gloss? +- [ ] every "the music is real / X is the Y" — does X actually equal Y, or did you reach? +- [ ] zero "ayy / skrt / let's get it / yeah yeah". zero "ecosystem / leverage / stakeholders". zero "moreover". +- [ ] read it aloud lowercase. does it sound like the recap narrations or like a poster? + +--- + +*maintained by @jeffrey — update when the voice evolves. all examples above are real, pulled from the AC repo as of 2026-05-03.* diff --git a/pop/RESEARCH-DIRECTION.md b/pop/RESEARCH-DIRECTION.md new file mode 100644 --- /dev/null +++ b/pop/RESEARCH-DIRECTION.md @@ -0,0 +1,78 @@ +# Research Direction · AC Pop + +**Last updated**: 2026-05-03 +**Author**: @jeffrey + +--- + +## Posture + +**bottom-up + compositional.** suno / udio are product-in, top-down, overused, and not compositional — you give them a prompt and get a finished song you didn't compose. that is not the AC posture. tracks here are built from AC's own instruments (the same notepat / chord / sinebells / beat primitives the recap waltz bed already uses), bar by bar. the composition is the artifact. + +## Vocal Architecture (decided 2026-05-03) + +big-pictures vocal = **jeffrey-pvc via ElevenLabs**. same `provider: "jeffrey", voice: "neutral:0"` already used by the 24h recap pipeline (`/api/say` proxy in `system/netlify/functions/`, called from `recap/bin/tts.mjs`). hooks and verses both. it's literally jeffrey's cloned voice — the most authentic option and already wired up. + +**AC-native vocal was attempted and dropped on 2026-05-03.** built a 3-formant synth (`recap/bin/vocal.mjs`) and a `.np` score reader, rendered the plork hook over the trap bed. result: it sounded like tones, not voice. real vocoder/talkbox character would need glottal pulse, F4/F5, pitch jitter, breath, consonant articulation — a full research lane. not blocking track 1 on it. `vocal.mjs` stays in the repo as experimental research; may find use as a *melodic instrument* layer (formant-shaped lead) rather than a vocal. + +if/when AC-native vocal resumes, the right starting point is **Pink Trombone** (Neil Thapen, 2017) — a browser-based physical model of the vocal tract that produces articulate speech-like sounds in pure JS. open-source, copy-pasteable. ports cleanly to a node renderer that consumes the `.np` score directly. see `reference_pink_trombone.md` in memory. + +after the ElevenLabs stem returns, we run **WhisperX forced alignment** on it (the same dependency the recap pipeline already uses for subtitle timing — see `feedback_recap_subtitle_timing_drift.md`) to get per-word timestamps. those word boundaries: + +1. **confirm** every word lands on or near a 16th-note grid line of the trap bed +2. drive the **bar-snap** pass that nudges drift to the nearest beat (per `feedback_recap_musical_snapping.md`, ~200ms tolerance) +3. become the score for any visual layer (subtitle overlays, notepat-style scrolling lyric track, AC piece tied to the track) +4. become the **edit unit** for vocal post-production (next section) + +the **`.np` score** for each track stays useful: it pairs the lyrics with a pitch contour, which is the right input to a notepat-style scrolling lyric visual layer and to any future kidlisp-driven music piece. + +## Vocal Post-Production (per-word) + +once a vocal track has word boundaries (WhisperX for ElevenLabs verses, or directly from the `.np` score for `vocal.mjs` hooks), each word is an editable segment. the post-prod stage applies per-word edits driven by a recipe file (`pop/big-pictures/.edit.json`): + +- **pitch** — shift up/down semitones, autotune to the scale of the bed (C minor pentatonic for trap) +- **elongate** — time-stretch to fit a target duration (rubberband / atempo, formant-preserving) +- **effect** — reverb / delay / distortion / vocoder / formant-shift / saturation, per-word +- **harmonize** — duplicate stem, pitch-shift each copy (+3rd, +5th, octave), mix back +- **realign** — move the word's start to match a target beat / 16th-note slot + +the **aggression** is a per-edit (or per-track) knob: `gentle` (≤50ms nudge), `firm` (≤200ms), `aggressive` (snap to nearest beat regardless of distance), `off` (leave timestamps as-is). aggression applies independently to each operation — you can pitch-correct gently and realign aggressively, or vice versa. + +post-prod is **generic across vocal sources** — works the same on ElevenLabs verse stems and `vocal.mjs` hook output. the `.edit.json` recipe is paired with the lyrics/score. + +implementation hooks: ffmpeg filtergraph (`asetrate`, `atempo`, `aecho`, `aphaser`, `chorus`), rubberband for pitch+time independence, sox for finer effects. + +## Current Goals + +1. **Land the first big-pictures track** — one AC vision, ninety seconds, mixed and listenable end-to-end. proves the pipeline before scaling +2. **Route lyrics through jeffrey-pvc TTS** — feed plork lyrics into `/api/say` with the existing voice config (`provider: "jeffrey", voice: "neutral:0"`); cache the stem; mix over the trap bed +3. **Build a curated reference corpus** — ~10–20 emo rap tracks' lyrics in the vault, used as in-context style examples for the lyric generator (not training data) +4. **Write the lyric generator** — paper / vision → 16-bar verse + 4-line hook in jeffrey-pvc voice with emo-rap overlay +5. **WhisperX align + bar-snap** — run forced alignment on the jeffrey-pvc stem; emit per-word timestamps; nudge drift to bar grid +6. **Build the per-word post-prod stage** — `pop/bin/vocal-post.mjs`: takes a vocal track + word timestamps + `.edit.json` recipe → applies pitch / elongate / effect / harmonize / realign, with per-edit aggression +7. **Tune the AC-native trap bed** — extend `recap/bin/trap.mjs` (now landed) with: dedicated 808 sub voice, swing toggle, fill on bar 16, optional pan stage + +## Open Questions + +- copyright posture for reference lyrics: in-context style examples only, vault-only, never committed. enough? +- elevenlabs cadence control: how tightly can we pin the vocal to the bar grid? rap needs the stem to land on the beat, not float. WhisperX + snap should fix most drift; test before committing the lane +- jeffrey-pvc rap performance: the 24h-recap voice is calm + descriptive. does it deliver rap cadence at all, or do we need a separate voice variant ("jeffrey-pvc:rap")? test on the plork hook first +- "big pictures" the show vs. the format: is each track its own episode, or are they collected into albums? defer until track 3 + +## First Track Plan — `plork` + +source: `papers/arxiv-plork/plork.tex`. core hook line already in the paper: *the music is real, the logistics just got cheaper.* + +``` +1. lyrics → pop/big-pictures/plork.txt [done — v1 draft] +2. score → pop/big-pictures/plork.np [done — hook + verse 1 + outro; visual layer / future] +3. trap bed → recap/out/trap.mp3 [done — 16 bars, 140 BPM] +4. vocal stem → /api/say with jeffrey-pvc [next: feed plork.txt as narration body] +5. align → WhisperX → per-word timestamps +6. snap → nudge words to nearest bar-grid 16th-note +7. post-prod → vocal-post.mjs --edit plork.edit.json + (pitch / elongate / effect / harmonize / realign, aggression-tunable) +8. mix → pop/big-pictures/out/plork.mp3 (~1:30) +``` + +If the jeffrey-pvc TTS doesn't carry rap cadence — try a different ElevenLabs voice variant or work the lyric line-breaks / punctuation to coax better delivery before resorting to a different provider. diff --git a/pop/SCORE.md b/pop/SCORE.md new file mode 100644 --- /dev/null +++ b/pop/SCORE.md @@ -0,0 +1,49 @@ +# Score for Pop + +## Mill Mission + +`pop` is the research home for music that comes out of Aesthetic Computer — songs, instrumentals, and the writing about them. The papers platter pushes AC's thinking out as text. `pop` does the same job in the form of tracks: short, finished pieces of music that can leave the building and be heard. + +The mill is not a label. It is a research lane. Every track here exists because it was the most honest way to compress an idea — a feature, a vision, a paper, a moment in the project. If a thread can survive being written as a song, it was real. If it can survive being compressed to ninety seconds, it was essential. + +## What This Is + +Pop tracks are one output of the AC research platter. They share source material with `papers/` — the same threads, the same readings, the same code — but render as audio. The platter feeds both. Some threads become papers, some become tracks, some become both. + +## Posture + +**bottom-up + compositional.** tracks here are composed from AC's own instruments — the notepat sample bank, sinebells, chord, beat — the same primitives the recap waltz bed already uses. no suno-style end-to-end song generation; that's product-in, top-down, and not compositional. AI vocal (ElevenLabs) is the one exception, since vocal is performance on top of the composition, not the compositional substrate. + +## Process + +``` +platter (raw material: notes, code, conversations, papers) + → thread (a vision worth singing) + → draft lyrics (in jeffrey-pvc voice + per-genre voice) + → vocal + beat (per-lane pipeline) + → mix (~1:30 mp3, audio-only) +``` + +Audio-only by default. No video, no chrome. If a track later becomes a video lane, that's a recap-side concern, not a `pop` concern. + +## Swimlanes + +### 1. big pictures (`big-pictures/`) + +Hip hop / trap dance versions of jeffrey's AC visions. Roughly **1:30 per track**. Lyrics rapped over a 4/4 trap bed (808s, triplet hats), one track per "big picture" — a single AC vision pulled from the papers platter (laptop orchestras, kidlisp, native OS, identity, latency, etc). + +Voice posture: emo-rap honesty. Conviction quiet but absolute. No flexing, no industry posture. Internal rhyme over end rhyme. The vision is the hook. + +See [`big-pictures/README.md`](big-pictures/README.md) for the format spec. + +### 2. (open) + +More lanes will land here as they prove themselves. Candidates: kidlisp-as-instrument tracks, AC-native ensemble cuts, voice-memo-grade demo lane. None of them have earned a swimlane yet — they need a real track first. + +## References + +Third-party lyrics (emo rap reference corpus, etc) live in the vault. They are not committed to this repo. See [`references/README.md`](references/README.md). + +--- + +*maintained by @jeffrey* diff --git a/pop/SPEECH-TO-SINGING-V2.md b/pop/SPEECH-TO-SINGING-V2.md new file mode 100644 --- /dev/null +++ b/pop/SPEECH-TO-SINGING-V2.md @@ -0,0 +1,172 @@ +# speech-to-singing — v2 follow-up + +v1 picked the saitou recipe and we built it. world (`pyworld`) lands the median pitch on target to ~2¢. but the holes are now audible — word-start stutters, held-note ringing, harsh sibilants, and outlier octave errors from naive autocorrelation in `pitchcheck.mjs`. v2 goes shopping for fixes to *those specific symptoms*, not for another full pipeline. + +three lanes: + +— **stay inside world, fix it.** parameter passes, second-pass correction, voicing-edge ramps. +— **swap parts of world out.** psola via parselmouth, sinusoidal modeling, neural pitch detectors. +— **bypass world.** rvc / so-vits-svc / nsf as a "real singer" lane. bigger payoff, bigger install. + +## 1. better pitch detection — the quick outlier kill + +our `pitchcheck.mjs` autocorrelation is the *measurement* tool, not the renderer — but its octave errors inflate the mean drift report and make us chase ghosts. world internally uses `harvest` already, which is good. the question is: when *we* measure the post-render f0 to verify, what do we use. + +| tool | install | gpu | what we get | +|---|---|---|---| +| **librosa.pyin** | already installed in the env | no | viterbi-smoothed yin, monophonic-only — perfect for our use. one function call: `librosa.pyin(y, sr, fmin=70, fmax=600)`. drop-in replacement for `detectPitch()`. | +| **torchcrepe** | `pip install torchcrepe` (~80 mb plus torch) | optional | pretrained cnn, sub-cent accuracy on clean voice. has a `viterbi=True` flag for smoothing. cpu-runnable, ~2x real-time on m-series. | +| **rmvpe** | `pip install rmvpe` (~120 mb model) | optional | best-on-singing benchmark (87.2% rpa on mir-1k vs crepe 85.3%). robust to noise. used internally by rvc/applio. | +| **fcpe / torchfcpe** | `pip install torchfcpe` (~30 mb) | optional | 2026 model. ~5× faster than rmvpe, ~77× faster than crepe. cpu-realtime. accuracy on par with rmvpe on clean monophonic. | +| **penn** | `pip install penn` (~50 mb) | optional | morrison's cross-domain neural pitch + periodicity, 11× rt on cpu, returns confidence — useful for voicing detection. | + +**verdict.** `librosa.pyin` is the immediate win — already on disk, two-line swap inside `pitchcheck.mjs`, removes the 170¢ mean-drift outliers in one move. for a future "second-pass autotune" (see §3), `torchfcpe` is the right pick — small install, cpu-friendly, sings well. + +integration sketch (pitchcheck.mjs): + +```python +# pop/bin/pitchcheck_pyin.py — child process called from pitchcheck.mjs +import sys, librosa, soundfile as sf, numpy as np +y, sr = sf.read(sys.argv[1]) +if y.ndim > 1: y = y.mean(axis=1) +f0, voiced, conf = librosa.pyin(y.astype(np.float32), sr=sr, + fmin=70, fmax=600, frame_length=2048) +# emit timestamp, hz, confidence csv +``` + +## 2. fix the world stutter — voicing-transition pops + +the audible "skip" at word starts is the world synth flipping unvoiced→voiced abruptly. our current script removed the `voicing-ramp` because multiplying f0 by 0..1 forced the first 40ms to be unvoiced. correct fix is the opposite — *interpolate f0 across unvoiced regions before synthesis*, then synthesise, then mute the unvoiced portions in the time domain after. + +this is the standard f0-contour smoother trick (zeehio / edinburgh speech tools, harvest paper §4): + +```python +# fill unvoiced gaps in f0 with linear interpolation +voiced_idx = np.where(f0 > 0)[0] +if len(voiced_idx) >= 2: + f0_filled = np.interp(np.arange(len(f0)), voiced_idx, f0[voiced_idx]) +else: + f0_filled = f0 +# build unvoiced mask in time-domain *after* synth +unvoiced_mask_audio = np.repeat(f0 == 0, samples_per_frame) +y = pw.synthesize(f0_filled, sp, ap, fs) # smooth pitch through gaps +# crossfade-mute the unvoiced regions (5 ms ramps at boundaries) +y *= smooth_mute(unvoiced_mask_audio, ramp_ms=5) +``` + +this gives world a continuous f0 curve to work against (no abrupt 0→target jumps), then re-imposes the voiced/unvoiced structure as a smoothed amplitude mask. costs ~10 lines of python; addresses the *single largest* perceptual artifact we currently hit. + +related fix: **bump the cheaptrick fft size**. default is auto-computed from `f0_floor=70`. raising `f0_floor` to 90 hz (jeffrey's voice never goes below 90) shrinks the analysis window, which reduces the "echo / ringing" on held vowels — that ring is cheaptrick smearing low-frequency formant estimates over too long a window. + +```python +fft_size = pw.get_cheaptrick_fft_size(fs, f0_floor=90.0) +sp = pw.cheaptrick(x, f0, t, fs, fft_size=fft_size, f0_floor=90.0) +``` + +## 3. second-pass correction — melodyne-lite + +even with world's f0-replace, we still see drift on outliers. add a dirt-cheap *second pass*: + +1. measure actual f0 of world output (with `torchfcpe` or `librosa.pyin`). +2. compute residual cents-error per frame: `err = 1200 * log2(measured / target)`. +3. if median |err| over a sustained note > 15 cents, build a per-frame correction ratio and re-render with world (only that note). otherwise pass through. + +this is what auto-tune internally does after its first snap pass. one extra world round-trip per problem note, ~50 lines, no new deps. works because the *first* pass got us to within ~2¢ median; the second pass surgically fixes the tail. + +## 4. psola alternative — when world rings, try grains + +td-psola (`psola` on pypi, wraps parselmouth → praat) does pitch-shift in the time domain by repositioning glottal-pulse-aligned grains. **no spectral-envelope estimation step**, so no cheaptrick ringing. the cost: psola is monophonic-only and breathy/unvoiced material can break it. + +```python +import psola +# target_pitch: numpy array same length as audio, hz per sample +y = psola.vocode(audio, sample_rate, target_pitch=target_pitch_curve, + fmin=70, fmax=600) +``` + +| | world | td-psola | +|---|---|---| +| held-note ringing | yes (cheaptrick smearing) | no (no spectral step) | +| sibilant clipping | yes (envelope smoothes 's') | yes (psola breaks on unvoiced) | +| word-start stutter | yes (v/uv transition) | less (grains overlap-add naturally) | +| install | `pip install pyworld` | `pip install psola praat-parselmouth` | +| cpu speed | fast | slow (~3× slower) | + +**recommended use:** dual-render — psola on sustained vowels, world on consonants — composited per-phoneme. that's a half-day build but addresses two of our four symptoms. + +## 5. sinusoidal modeling — the slow lane + +`sms-tools` (mtg/upf, xavier serra), `loris` / `loristrck`, and `simpl` decompose audio into time-varying sinusoidal partials + a noise residual. you can shift each partial independently, which gives surgical control over harmonics vs sibilance vs breath. + +— `pip install sms-tools` (apple silicon wheels exist as of 2026). +— per-partial pitch shifting: pure tones move, residual noise stays. +— **fixes sibilants directly**: classify partials by frequency band, leave 5–10 khz partials unshifted, shift only the harmonic stack below 4 khz. + +cost: orders of magnitude slower than world, more knobs, more code (~300 lines). flag this as a research-track experiment, not a production swap. + +## 6. consonant-region artifact suppression — the cleanup pass + +even with everything above, world will still occasionally muck up "s", "k", "t" because it tries to apply f0 to noisy regions. the cheap fix is a *post-vocoder* spectral repair pass: + +— **de-esser**: dynamic compressor on the 5–8 khz band, sidechain-keyed by an envelope of that same band. classic broadcast trick. `scipy.signal.iirfilter` + a one-pole detector + per-sample gain in ~30 lines. or use the `pedalboard` library (spotify, `pip install pedalboard`) which has `Compressor` + `HighpassFilter` building blocks ready. +— **transient gate**: at frame boundaries that fall inside unvoiced regions, replace world's output with the *original* recording's audio in that region, crossfaded ±5 ms. world handles vowels; original audio handles fricatives and stops. this is the "world for pitch, source for noise" composite. + +```python +# composite: keep world output on voiced frames, original audio on unvoiced +voiced_audio = upsample(voiced_mask, fs) # 0..1 per sample, 5 ms ramps +y_final = voiced_audio * y_world + (1 - voiced_audio) * x_original +``` + +this is the single biggest "sounds clean" win for the same effort budget as §2. probably do both. + +## 7. neural lane — rvc / so-vits-svc / nsf + +if §1–§6 still fall short, the field's actual answer in 2026 is voice conversion: train a model on jeffrey's voice, feed it a sung melody (synthesised or hummed), get sung jeffrey out. + +| project | status | install | model size | cpu? | what it gives us | +|---|---|---|---|---|---| +| **rvc / applio** | active, large community | `git clone IAHispano/Applio` + cli | ~500 mb base + ~50 mb per voice | yes (~1× rt on m-series for non-real-time) | trained voice clone, melody-driven. uses rmvpe internally. | +| **so-vits-svc-fork** (voicepaw) | actively maintained | `pip install so-vits-svc-fork` | ~200 mb base + ~150 mb per voice | yes (~3× rt cpu) | similar to rvc, slightly older arch. | +| **diff-svc / diffsinger** | research-grade | `git clone prophesier/diff-svc` | ~800 mb | gpu strongly preferred | diffusion, highest quality, slow. | +| **nnsvs + pc-nsf-hifigan** | active, score-driven | `pip install nnsvs` | ~400 mb | yes — nsf models specifically beat hifi-gan on cpu | feeds *score + lyrics*, not a guide vocal. complementary lane. | + +**critical caveat for our pipeline.** rvc and so-vits-svc need a *sung* input. they convert sung-anyone → sung-jeffrey. they don't solve speech→sung. so the pipeline becomes: + +``` +speech → world (saitou f0-replace) → rough sung → rvc (jeffrey clone) → polished sung +``` + +rvc as a *cleanup pass* downstream of our world output is the highest-ceiling option in this doc. it would mask all of §2/§4/§6 — rvc's neural decoder doesn't care if our world stage stutters, it resamples everything through the trained voice. estimated install + train: 1 day for setup, 4–8 hours of jeffrey audio collection, ~2 hours of training on cpu (overnight) or 30 min on a rented gpu. + +— **diff-pitcher** (jhu-lcap, waspaa 2023, `haidog-yaqub/DiffPitcher` on github) is the closest thing to "pitch-correct neurally without a trained singer". diffusion-based correction that takes out-of-tune audio + target midi, returns in-tune audio with timbre preserved. no voice training step. ~600 mb model, cpu-runnable but slow (~5× rt). worth a one-day evaluation if §1–§6 leaves residual artifacts on jeffrey's takes specifically. + +## 8. nsf vocoders — better than world without leaving the dsp lane + +if we want to *replace* world entirely with something newer but stay non-neural-conversion, **pc-nsf-hifigan** (pitch-controllable neural source-filter, used by nnsvs as of 2025) takes `(f0, mel)` and produces a waveform. quality on singing beats world by a wide margin in the published evals (yamagishi lab samples, source-filter hifi-gan paper 2022). cpu-runnable, ~10× faster than world per second of audio. + +integration cost is the install (`pip install nnsvs` brings ~400 mb of torch + checkpoints) and re-fitting our pipeline to emit mel-spectrograms instead of cheaptrick envelopes. about a 2-day rewrite. defer until §1–§6 are exhausted. + +## the new symptom-to-fix table + +| symptom | root cause | new fix proposed | section | +|---|---|---|---| +| word-start stutter | abrupt 0→target f0 at voicing transitions | interpolate f0 across unvoiced gaps + post-synth amplitude mask | §2 | +| held-note ringing | cheaptrick analysis window too long | raise `f0_floor` to 90 hz, recompute fft_size | §2 | +| harsh sibilants | world tries to apply f0 to fricatives | composite — world on voiced, source audio on unvoiced | §6 | +| residual drift on outliers | one-pass correction insufficient | melodyne-lite second-pass over world output, measured by torchfcpe | §3 | +| octave errors in `pitchcheck.mjs` report | naive autocorrelation | swap to `librosa.pyin` (already installed) | §1 | +| occasional vowel mush | cheaptrick envelope over-smooths | dual-render with td-psola on sustained vowels | §4 | + +## shortlist — three things to try next, ranked + +**1. f0-gap interpolation + voiced/unvoiced source compositing.** *highest payoff, ~3 hour build, no new deps.* sections 2 and 6 combined. this kills the two worst symptoms (stutter at word starts, harsh sibilants) inside `pitchsnap_world.py` with ~30 lines of numpy. `librosa.pyin` swap in `pitchcheck.mjs` rolls in for free. zero new install. + +**2. cheaptrick fft_size tuning + melodyne-lite second pass.** *medium payoff, ~half-day build, adds torchfcpe.* sections 2 (the `f0_floor=90` line) and 3. addresses held-note ringing and tail outliers. `pip install torchfcpe` is the only new dep — small, cpu-friendly, no gpu. + +**3. rvc/applio as a downstream cleanup pass.** *highest ceiling, ~2 day build, large install.* the "real" answer if signal-processing exhausts itself. world output → applio cli → polished sung jeffrey. requires collecting clean jeffrey audio (probably already in the recap corpus) and one overnight training run. defer until (1) and (2) ship and we have a tagged "world-only" baseline to a/b against. + +## strongest single new finding + +**the word-start stutter is fixable inside world without changing the vocoder.** v1 assumed any pop at voicing transitions was a fundamental world limitation; the literature is clear it's a configuration problem. the fix is the standard "interpolate f0 through unvoiced gaps, synthesise on the smoothed curve, then re-impose the voiced mask in the time domain with crossfaded ramps" — used by every production pitch-correction pipeline, missing from our current script because we tried (and removed) the wrong version of it (multiplying f0 by 0→1 envelope, which forced unvoiced and made the problem worse). doing it the documented way (interpolate the contour, mute the *output* not the f0) is ~30 lines and should clean up the stutter completely. + +**recommended immediate action:** patch `pitchsnap_world.py` to interpolate f0 across unvoiced frames before synthesis, mute the unvoiced output via a 5 ms-ramped amplitude mask, and bump `f0_floor` to 90 hz in cheaptrick. that single commit should hit two of the four current symptoms. then `librosa.pyin` swap in `pitchcheck.mjs` to retire the autocorrelation outliers from the drift report. day one. diff --git a/pop/SPEECH-TO-SINGING.md b/pop/SPEECH-TO-SINGING.md new file mode 100644 --- /dev/null +++ b/pop/SPEECH-TO-SINGING.md @@ -0,0 +1,102 @@ +# speech-to-singing — survey + +a map of how the field gets sung output from spoken input, and which pieces drop into our existing `pitchsnap.mjs` lane. + +scope: jeffrey-pvc tts → sung melody. neural svc / svs lanes are noted but flagged as a separate research track (gpu, trained models, weeks of setup). + +## the field + +three eras, all still in active use: + +**signal-processing era (2007–2014).** saitou et al.'s "speech-to-singing synthesis system" (waspaa 2007 / 2009) is the canonical paper. you read the lyrics out loud, hand it a score, and the system replaces the f0 contour with the score's contour, lengthens phonemes to fit notes, and reshapes the spectrum to add the singer's formant. it runs on the **straight** vocoder (kawahara) — analyse → modify (f0, duration, spectrum) → resynthesise. saitou explicitly lists the three things missing from speech: pitched melody, sustained vowels, and a singer's-formant ring. that decomposition is still the right mental model. + +**statistical-parametric era (2010–2018).** **sinsy** (nagoya institute of technology, hmm-based, bsd-licensed, on sourceforge / github) takes musicxml in and produces sung audio. it learns from a corpus what real singers do with a score. spiritually similar but you don't get to inject your own speech timbre — it's full synthesis from a model, not transformation of a recording. + +**neural era (2019–now).** parekh et al. (icassp 2020, "speech-to-singing conversion in an encoder-decoder framework") was the first end-to-end learned s2s — input is speech spectrogram + target melody, output is sung spectrogram. on the synthesis side, **diffsinger** (aaai 2022, diffusion + hifigan) and **nnsvs** (r9y9, 2022, the open-source successor to neutrino) take a score and a trained voice and produce a sung waveform. on the conversion side, **so-vits-svc** and **rvc** turn a guide vocal (sung melody) into the target singer's voice — they need a sung melody as input, so they don't solve our problem on their own; they'd sit *downstream* of a working s2s pipeline. + +## the techniques + +each technique listed with what it actually does to the signal. the problem isn't that we don't have enough techniques — it's that pitchsnap currently uses one (rubberband segment-shifting) and skips the rest. + +— **f0 substitution.** decompose audio with a vocoder that gives you `(f0, spectral_envelope, aperiodicity)` as separate streams. throw the source f0 away, write a new f0 contour from the score, resynthesise. spectral envelope unchanged → vowel identity preserved → it still sounds like jeffrey, just with sung pitch. this is the saitou recipe and also what world / pyworld / praat do natively. our current rubberband path *shifts* the existing f0 by an interval, which is why "spoken prosody fights melody" — the prosody is *in* the f0 we're shifting. + +— **phoneme-aware time-stretch (vowel sustain).** detect vowel regions, stretch them; leave consonants alone. straight + saitou's "duration control model" does this. world's `synthesizeRequiem` lets you pass a per-frame time-warp. praat's manipulation object exposes a `durationtier` you can set per phoneme. without this step, "real" stretched 4× sounds like "rrrrreal" — every phoneme drags. with it, only the `ee` drags and you get "reeeeeal". + +— **phoneme alignment.** **montreal forced aligner** (mfa, kaldi-based, free, has english pretrained models) gives you per-phoneme timestamps from audio + transcript. **charsiu** (lingjzhu, wav2vec2-based, lighter than mfa) does the same with a smaller install and decent quality. either gives consonant/vowel boundaries — required for vowel-sustain to know which spans to stretch. + +— **f0 detection.** our current `pitchsnap.mjs` uses naive autocorrelation in `detectPitch()` (lines 150–182), which the comments admit picks 2× / ½× on individual words. **pyin** (mauch & dixon 2014, in librosa as `librosa.pyin`) adds viterbi smoothing and is the standard for monophonic speech/singing. **crepe** (kim et al. 2018) is a cnn, more accurate, needs tensorflow. **dio + stonemask** in world is fast and clean for voice. swapping autocorrelation → pyin or world's dio is a one-day fix that removes octave errors immediately. + +— **f0 contour shaping.** real sung notes don't step. saitou models four sub-effects: + - *overshoot* (~50 cents past the target on attack, decays to centre over ~200 ms) + - *vibrato* (4.5–6.5 hz, 50–120 cents peak-to-peak, fades in after sustain begins) + - *preparation* (slight dip below the target before a rising interval) + - *fine fluctuation* (sub-cent jitter for naturalness) + add these on top of a stepped target curve and a flat synthetic line starts to read as performed. + +— **singer's formant.** sundberg 1974 — trained singers cluster f3/f4/f5 around 2.5–3 khz for "ring" that cuts through an orchestra. in spectral-envelope terms: boost a ~500 hz-wide band centred near 2.8 khz by ~6–10 db. on a vocoder that exposes the spectral envelope (world, straight, praat) this is one filter operation per frame. we can fake it on raw audio with a parametric eq at +6 db / 2.8 khz / q=4, but it's cleaner inside a vocoder where the boost rides the formant rather than ringing. + +— **autotune (correction style).** the antares trick: run pyin → snap each frame's f0 to the nearest scale degree → resynthesise via a phase vocoder or psola. with a low retune-time you get the stepped t-pain effect; with higher retune-time you get gentle correction. the *aggressive* variant is identical to "f0 substitution from score" except the target curve comes from snapping rather than from a score. + +— **psola / phase vocoder.** **td-psola** (moulines & charpentier 1990, in praat) modifies pitch and duration in the time domain by repositioning glottal-pulse-aligned grains; preserves formants by construction. phase vocoder (frequency-domain) is what rubberband and librosa use under the hood. for singing, psola tends to sound more natural on small shifts, phase vocoder handles bigger transformations better but smears transients. + +## the tools + +| tool | language | install | gpu? | what it gives us | +|---|---|---|---|---| +| **rubberband** (cli, current) | c++ | already installed | no | pitch + time stretch, phase vocoder. has `--formant`, `--pitchmap`, `--smoothing` flags we're under-using. | +| **world / pyworld** | c++ / python | `pip install pyworld` (~1 mb) | no | analyse → `(f0, sp, ap)` → modify f0 directly → resynthesise. cleanest path to f0 substitution. dio for f0, cheaptrick for envelope, d4c for aperiodicity. | +| **praat** (cli + scripts) | c | `brew install praat` | no | manipulation object with pitchtier + durationtier, td-psola underneath. fully scriptable. heavier orchestration than pyworld but doesn't need python. | +| **librosa** | python | `pip install librosa` (~50 mb) | no | `librosa.pyin` for f0 detection, `pitch_shift` / `time_stretch` (lower fidelity than rubberband). good as a measurement tool, not a render tool. | +| **crepe** | python (tf) | `pip install crepe` (~500 mb tf) | optional | best-in-class f0 detection. heavy install for one feature. | +| **mfa** | python (kaldi) | `conda install -c conda-forge montreal-forced-aligner` (~2 gb incl. acoustic models) | no | per-phoneme alignment from audio + transcript. heavy. | +| **charsiu** | python (torch) | `pip install transformers + ckpt` (~500 mb) | optional | per-phoneme alignment, lighter than mfa, comparable accuracy. | +| **sinsy** | c++ | source build | no | full hmm-based singer from musicxml; outputs sung audio, not a transformation tool — different lane. | +| **nnsvs / diffsinger / so-vits-svc / rvc** | python (torch) | gpu strongly recommended; multi-gb models | yes | parallel research lane. produce excellent results but require trained voices and gpus. flagged out-of-scope-for-now. | + +ffmpeg's bundled rubberband filter is not available in our build — confirmed earlier; we shell out to the cli. + +## what fixes our specific symptoms + +| symptom we hit | root cause | technique that fixes it | tool | +|---|---|---|---| +| "spoken prosody fights melody" | rubberband shifts existing f0 by interval; original speech contour rides on top of every shift | f0 substitution — replace, don't shift | pyworld (dio + stonemask + cheaptrick + synthesize) | +| "vowels don't sustain" | word-level slices stretched uniformly drag consonants too; word-final vowels are too short to ring | phoneme alignment + selective vowel stretch | charsiu (or mfa) → per-phoneme durationtier in praat / world | +| "sounds tonal not sung" | flat stepped pitch, no overshoot / vibrato / fine fluctuation; missing singer's formant ring | saitou f0-contour shaping + spectral-envelope boost at 2.5–3 khz | pyworld envelope edit + post-eq, or praat scripted | +| "octave errors on shifts" | autocorrelation in `detectPitch()` picks 2× / ½× on some words | viterbi-smoothed f0 | librosa.pyin or world.dio | +| "consonants get cut at word edges" | whisper word boundaries land mid-consonant | phoneme alignment, slice on phoneme edges | charsiu | +| "segment boundaries click" | fixed 20 ms crossfade in segmented rendering doesn't align with glottal pulses | psola — grains aligned to f0 periods | praat td-psola, or world resynth (no clicks by construction) | +| "formant character lost on big shifts" | rubberband `--formant` not currently passed in `pitchsnap.mjs` | enable formant preservation | rubberband `--formant` flag (one-line change) | + +## shortlist + +ranked by effort vs. payoff. opinionated. + +**1. swap the rendering core to world (pyworld) for f0 substitution.** *highest payoff, ~2-day build.* this is the saitou recipe and addresses three of our top symptoms in one move: + - rubberband segment-shift → world `(f0, sp, ap)` decompose, write target f0 from score, resynth. + - kills "spoken prosody fights melody" (because we replace, not shift). + - kills octave errors (dio is reliable on voice). + - removes click artefacts at segment boundaries (no segmentation needed). + - keeps jeffrey's timbre intact (spectral envelope unchanged). + - new piece: a python child process called from `pitchsnap.mjs` that takes the slice + target f0 contour json and returns a wav. ~150 lines of python, ~1 mb pip install, no gpu. + - immediately enables vibrato / overshoot / singer's-formant boost as f0/envelope edits in the same script. + +**2. add phoneme alignment via charsiu, switch from word-snap to phoneme-snap.** *medium payoff, ~1-day build, depends on (1) for full effect.* gets us: + - clean consonant/vowel boundaries, no chopped tails. + - vowel-sustain becomes a real operation: identify the vowel span, stretch only that. + - lyric-carrying consonants stay short and crisp; the *note* lives on the vowel. + - charsiu over mfa for install size — 500 mb vs. 2 gb, comparable quality on english. + +**3. saitou f0 sub-effects: vibrato + overshoot + fine fluctuation.** *lowest effort once (1) lands, big perceptual win.* once we own the f0 contour, layer: + - 5.5 hz sine, 70 cents peak-to-peak, fading in after the first 150 ms of sustain → vibrato + - 50 cents above target on attack, exponential decay over 200 ms → overshoot + - low-pass-filtered noise at ±10 cents → fine fluctuation + - all parameters in a recipe file, per-track tunable. + this is ~50 lines on top of the world renderer and is the single biggest "sounds sung" lever once the substrate is right. + +**deferred (parallel lane).** so-vits-svc / rvc / diffsinger / nnsvs as a separate research track. they produce the best results in the field today, but they need trained models on jeffrey's voice (so-vits-svc / rvc) or are full synthesizers from score (diffsinger / nnsvs) and skip our "preserve the speech recording" posture. start with (1)–(3); evaluate the neural lane only after we know what the signal-processing path actually sounds like on our material. + +## the strongest pattern + +every working s2s system in the literature — saitou 2007, sinsy 2010, vocalistener 2009, diffsinger 2022, parekh 2020 — separates f0 from spectral envelope and treats them as independent edit streams. our current pipeline doesn't. we shift one composite signal in semitones, which means jeffrey's spoken prosody is still riding on top of every "snapped" note. the single biggest upgrade is moving to a vocoder that exposes those streams separately so we can replace f0 wholesale instead of shifting it. + +**top recommendation for the immediate next pitchsnap upgrade:** add a `--engine world` option to `pitchsnap.mjs` that shells out to a small `pitchsnap-world.py` for the per-word render. python takes the wav slice + target midi(s), runs `pyworld.dio` → `pyworld.stonemask` → `pyworld.cheaptrick` → `pyworld.d4c`, replaces f0 with the target curve (with a 200 ms attack overshoot and a 5.5 hz vibrato fading in after the first sustained quarter-second), and resynths with `pyworld.synthesize`. keeps the rest of the snap / scale-walk / score-mapping logic identical. one new dependency, one new flag, no neural models. that's the saitou pipeline and it should turn "tonal speech" into "sung jeffrey" in a single weekend. diff --git a/pop/VOICE.md b/pop/VOICE.md new file mode 100644 --- /dev/null +++ b/pop/VOICE.md @@ -0,0 +1,45 @@ +# Voice Guide for AC Pop + +pop tracks should sound like @jeffrey writing songs, not like a brand writing copy. lyrics are not advertising. they are the most compressed form a vision can take before it stops being music. + +## the basics + +- first person is fine. "i built this" not "the artist built" +- lowercase default in lyric drafts. capitalization in mix metadata only when convention demands it +- em dashes for asides — not semicolons, not parentheses +- short lines. fragments when they land. no filler syllables to pad the bar +- internal rhyme over end rhyme. the rhyme should fall mid-bar at least as often as it falls on the four +- state the hard thing plainly. don't soften with hedging language +- conviction is quiet but absolute — "the music is real" not "i'm feeling kinda confident about this" +- humor is allowed. the line that makes you laugh on the first listen is doing work +- be honest about what doesn't work. the second verse can be the limitations section + +## what to avoid + +- flexing posture. money / cars / club bars. AC has nothing to flex about and that's the point +- industry filler — "ayy", "skrt", "let's get it" — unless they're earned and load-bearing +- explaining the vision before singing it. the lyric should *be* the vision, not narrate it +- writing differently for different platforms. one lyric file, one mix, no platform variants +- corporate vocabulary. "ecosystem", "stakeholders", "leverage" — drop them +- generic emotion. pick the specific feeling. "tired" is not a feeling. "tired of explaining kidlisp at every dinner" is + +## per-genre overlays + +each lane has its own voice on top of this base. + +### big pictures (emo rap / trap) + +- emo rap honesty — vulnerability without self-pity. confess the thing, then keep moving +- triplet flow allowed but not default. use it when the line needs urgency +- ad-libs: minimal. one or two per verse, low in the mix, never "yeah yeah yeah" filler +- the hook is the vision compressed to one line — repeatable, undeniable, true +- 16 bars per verse, 4-line hook, ~1:30 total. roughly: hook → verse → hook → verse → hook → outro +- reference voice: see vault. ingestion is in-context style, not training corpus + +## examples of the voice working + +placeholders. real lines land here when the first track is mixed. + +--- + +*maintained by @jeffrey — update this when the voice evolves* diff --git a/pop/big-pictures/README.md b/pop/big-pictures/README.md new file mode 100644 --- /dev/null +++ b/pop/big-pictures/README.md @@ -0,0 +1,76 @@ +# big pictures + +audio-only hip hop / trap versions of jeffrey's AC visions. one vision per track. ~1:30 each. + +## format spec + +- **length**: 90 seconds, ±10s +- **structure**: hook (4 bars) → verse (16 bars) → hook → verse (16 bars) → hook → outro +- **tempo**: ~140 BPM, 4/4 +- **bed**: trap — 808 sub, triplet hats, sparse snare on 3, room for the vocal +- **vocal**: rapped, not sung. emo-rap honesty (see `../VOICE.md`) +- **output**: single mp3 per track in `out/.mp3`. no video. + +## source → track + +each track corresponds to one paper or one vision from the platter. the lyric is the compression of that paper into the form a song can carry. the hook is the vision in one line. + +``` +papers/arxiv-/.tex + → pop/big-pictures/.txt (plain lyrics) + → pop/big-pictures/.np (notepat score: NOTE:syllable per syllable) + → pop/big-pictures/out/.mp3 (mix) +``` + +the `.np` (notepat) file is the score in the same notation as the folk-songs paper (`papers/arxiv-folk-songs/folk-songs.tex` §3). every syllable carries a pitch — making the lyric playable on notepat in song mode and renderable through `recap/bin/vocal.mjs` (formant synth) or any other pitch-driven voice. the file is its own URL when fed to `notepat.com?song=...`. + +## lyric file format + +plain text. no metadata header. blocks separated by blank lines, labeled in lowercase: + +``` +hook +<4 lines> + +verse 1 +<16 lines> + +hook + +verse 2 +<16 lines> + +hook + +outro +<2-4 lines> +``` + +## pipeline (planned) + +`recap/bin/big-pictures.mjs` — mirrors the recap cli pattern. cached per step so reruns cost nothing. + +``` +read paper + → draft lyrics (jeffrey-pvc voice + emo-rap overlay) + → write notepat score (.np) — every syllable carries pitch (visual / kidlisp future) + → AC-native trap bed (recap/bin/trap.mjs) + → vocal stem: /api/say with jeffrey-pvc (provider:"jeffrey", voice:"neutral:0") + → WhisperX forced alignment — per-word timestamps + → snap drift to bar grid — ±200ms tolerance, 16th-note quantization + → vocal-post per-word edits — pitch / elongate / effect / harmonize, aggression-tunable + → mix (bed + vocal) + → mp3 +``` + +## vocal source + +big-pictures uses **jeffrey-pvc via ElevenLabs** (the same voice the 24h recap pipeline already uses). hooks and verses both. it's literally jeffrey's cloned voice and is already wired up through `/api/say`. + +an AC-native formant-synth vocal was attempted and dropped on 2026-05-03 — it produced melodic tones but not voice; getting real vocoder/talkbox character would need glottal pulse + F4/F5 + pitch jitter + consonants, a full research lane. `recap/bin/vocal.mjs` remains in the repo as experimental research; may resurface as a *melodic instrument* layer (formant-shaped lead in the bed) rather than a vocal. + +**bottom-up posture preserved at the composition layer.** the bed is composed bar-by-bar from AC instruments (`trap.mjs` over `percussion.mjs`); the score is hand-written in `.np` notation. ElevenLabs is the *performance* on top of that composition. + +## tracks + +none yet. first candidate: `plork` (laptop orchestras, planetary scale). diff --git a/pop/big-pictures/ac-da.txt b/pop/big-pictures/ac-da.txt new file mode 100644 --- /dev/null +++ b/pop/big-pictures/ac-da.txt @@ -0,0 +1,29 @@ +hook +hvert stykke er en url +hver url er et nodeark +tryk enter — så er du inde +æstetisk dot computer + +verse 1 +kidlisp i browseren +notepat på qwerty +samme samples, native og statisk +dual-channel, under femti millisek +hvert stykke — én mjs-fil +adressen er kilden + +hook + +verse 2 +en mand alene, ingen runway +jeg ville bare mine venner kunne male +kildekoden er på github +cachen holder altid +jeg bygger ikke et produkt +jeg bygger et lille sted + +hook + +outro +tryk enter +du er inde i det diff --git a/pop/big-pictures/ac.txt b/pop/big-pictures/ac.txt new file mode 100644 --- /dev/null +++ b/pop/big-pictures/ac.txt @@ -0,0 +1,31 @@ +hook +every piece a url +every url a score +push enter — you're inside +aesthetic dot computer + +verse 1 +kidlisp evaluates parens live in the browser +notepat — qwerty, chromatic, two octaves over +sample bank's the same on native and the static +half-meg of api, every event with a purpose +mjs is the unit, the address is the score +i write a piece, i save it — that's the whole move + +hook + +verse 2 +one-person stack, no runway, no dream +no investors, no playbook, no series-a scheme +i wanted a place my friends could just paint +where the paint was the piece, the piece was the wave +the source is on github, the cache stays the cache +fork the whole runtime — same kid, same shape +i'm not pitching a product, i'm holding a stand +the music is real and it works in your hand + +hook + +outro +push enter +you're inside it diff --git a/pop/big-pictures/amazing.np b/pop/big-pictures/amazing.np new file mode 100644 --- /dev/null +++ b/pop/big-pictures/amazing.np @@ -0,0 +1,18 @@ +# Amazing Grace — full verse 1. "New Britain" tune, William Walker 1835. +# notation: NOTE:syllable*beats +# Phrase 1+2 melody from papers/arxiv-folk-songs/folk-songs.tex:181 +# (the canonical pentatonic transcription). +# +# key: G major pentatonic. octave 3 — sits in jeffrey-pvc's natural +# baritone. Climax at D4 on "blind" (line 4) is the highest note. +# Beat values follow the dotted-half / quarter pattern of 3/4 hymn +# delivery. Each line ends with an extended hold (5 beats) for the +# phrase-end breath. +# +# Use --beat-mode --bpm 70. + +verse 1 +D3:a-*1 G3:-ma-*3 B3:-zing*1 G3:grace*3 B3:how*1 A3:sweet*3 G3:the*1 B3:sound*5 +D3:that*1 E3:saved*3 D3:a*1 B3:wretch*3 G3:like*1 B3:me*5 +D3:i*1 G3:once*3 D4:was*1 D4:lost*3 B3:but*1 D4:now*3 A3:am*1 G3:found*5 +B3:was*1 D4:blind*3 D4:but*1 B3:now*3 A3:i*1 G3:see*5 diff --git a/pop/big-pictures/amazing.txt b/pop/big-pictures/amazing.txt new file mode 100644 --- /dev/null +++ b/pop/big-pictures/amazing.txt @@ -0,0 +1,5 @@ +verse 1 +amazing grace how sweet the sound +that saved a wretch like me +i once was lost but now am found +was blind but now i see diff --git a/pop/big-pictures/elephant.np b/pop/big-pictures/elephant.np new file mode 100644 --- /dev/null +++ b/pop/big-pictures/elephant.np @@ -0,0 +1,9 @@ +# elephant × 4 — C minor i–VI–iv–V arpeggio progression +# 3 syllables per word: el-e-phant +# rep 1: i (Cm) → C3 Eb3 G3 +# rep 2: VI (Ab) → Ab3 C4 Eb3 +# rep 3: iv (Fm) → F3 Ab3 C4 +# rep 4: V (G) → G3 Bb3 F3 (resolves back to root next bar) + +verse 1 +C3:el- Eb3:-e- G3:-phant Ab3:el- C4:-e- Eb3:-phant F3:el- Ab3:-e- C4:-phant G3:el- Bb3:-e- F3:-phant diff --git a/pop/big-pictures/elephant.txt b/pop/big-pictures/elephant.txt new file mode 100644 --- /dev/null +++ b/pop/big-pictures/elephant.txt @@ -0,0 +1,2 @@ +verse 1 +elephant elephant elephant elephant diff --git a/pop/big-pictures/mary.np b/pop/big-pictures/mary.np new file mode 100644 --- /dev/null +++ b/pop/big-pictures/mary.np @@ -0,0 +1,33 @@ +# Mary Had a Little Lamb — notepat score, all 4 traditional verses +# plus the gen-alpha "chick chick BOOM no more lamb" outro. +# notation: NOTE:syllable, hyphens for syllable continuation +# (folk-songs paper §3, papers/arxiv-folk-songs/folk-songs.tex:152) +# +# key: C major. classic kindergarten melody. octave 4 — middle C and +# above. WORLD engine keeps voice timbre intact even with the +12 to +# +16 semitone shift from jeffrey-pvc's natural baritone. Same E-D-C-D- +# E-E-E phrase repeats across all four verses. +# +# The closing "chick chick BOOM" verse triggers a gunshot SFX on the +# BOOM syllable (folk_backing.py overlays a synthesized PISTOL preset +# gunshot wherever the syllable text matches /^boom$/i). + +verse 1 +E4:mar- D4:-y C4:had D4:a E4:lit- E4:-tle E4:lamb +D4:lit- D4:-tle E4:lamb D4:lit- D4:-tle E4:lamb +E4:mar- D4:-y C4:had D4:a E4:lit- E4:-tle E4:lamb +E4:its D4:fleece D4:was E4:white D4:as C4:snow +E4:ev- D4:-ery- C4:-where D4:that E4:mar- E4:-y E4:went +D4:mar- D4:-y E4:went D4:mar- D4:-y E4:went +E4:ev- D4:-ery- C4:-where D4:that E4:mar- E4:-y E4:went +E4:the D4:lamb D4:was E4:sure D4:to C4:go +E4:fol- D4:-lowed C4:her D4:to E4:school E4:one E4:day +D4:school D4:one E4:day D4:school D4:one E4:day +E4:fol- D4:-lowed C4:her D4:to E4:school E4:one E4:day +E4:which D4:was D4:a- E4:-gainst D4:the C4:rules +E4:made D4:the C4:chil- D4:-dren E4:laugh E4:and E4:play +D4:laugh D4:and E4:play D4:laugh D4:and E4:play +E4:made D4:the C4:chil- D4:-dren E4:laugh E4:and E4:play +E4:to D4:see D4:a E4:lamb D4:at C4:school +E4:mar- D4:-y C4:had D4:a E4:lit- E4:-tle E4:lamb +D4:chick D4:chick E4:boom D4:no D4:more E4:lamb*2 diff --git a/pop/big-pictures/mary.txt b/pop/big-pictures/mary.txt new file mode 100644 --- /dev/null +++ b/pop/big-pictures/mary.txt @@ -0,0 +1,19 @@ +verse 1 +mary had a little lamb +little lamb, little lamb +mary had a little lamb +its fleece was white as snow +everywhere that mary went +mary went, mary went +everywhere that mary went +the lamb was sure to go +followed her to school one day +school one day, school one day +followed her to school one day +which was against the rules +made the children laugh and play +laugh and play, laugh and play +made the children laugh and play +to see a lamb at school +mary had a little lamb +chick chick boom, no more lamb diff --git a/pop/big-pictures/plork.np b/pop/big-pictures/plork.np new file mode 100644 --- /dev/null +++ b/pop/big-pictures/plork.np @@ -0,0 +1,43 @@ +# big pictures: plork — notepat score +# notation: NOTE:syllable, hyphens for syllable continuation +# (folk-songs paper §3, papers/arxiv-folk-songs/folk-songs.tex:152) +# +# key: C minor (pentatonic-leaning: C Eb F G Bb) +# fits trap progression i VI iv V → Cm Ab Fm G +# rap contour: verses sit close to root, peaks on key words; hook is +# the most melodic line and lands on tonic. + +hook +Eb:the C:mu- Eb:-sic G:is G:real Eb:the C:lo- Eb:-gis- F:-tics G:got C:cheap +Eb:land- Eb:-fill's Eb:full Eb:of C:or- Eb:-ches- F:-tras G:dream- F:-ing Eb:in Eb:their C:sleep +G:two- G:-for- G:-ty G:mil- G:-lion Bb:lap- G:-tops, F:win- F:-dows F:ten Eb:re- C:-tired +Eb:flash Eb:a F:ker- Eb:-nel G:ev- G:-ery G:one Eb:of Eb:them F:just G:got C:hired + +verse 1 +C:i C:watched C:plork C:in C:a Eb:base- C:-ment C:on C:a Eb:live- C:-stream +C:fif- C:-teen C:lap- C:-tops Eb:in Eb:a F:hall Eb:do- Eb:-ing Eb:rit- C:-u- C:-als C:like C:a G:dream +Eb:two Eb:thou- C:-sand C:six F:prince- Eb:-ton C:had C:the Eb:keys C:to C:the C:hall +Eb:twen- C:-ty Eb:twen- C:-ty F:half Eb:those Eb:or- C:-ches- C:-tras Eb:don't Eb:an- Eb:-swer C:the C:call +C:co- C:-vid Eb:took Eb:the F:rooms F:and Eb:no- C:-bo- C:-dy Eb:built C:them C:back +C:i'm C:not Eb:bit- C:-ter C:i'm Eb:hon- C:-est C:i Eb:keep G:score Eb:on C:the C:track +C:the Eb:gear C:was C:the F:gate C:the F:gate C:was C:the G:price C:tag +C:fif- C:-teen C:hun- C:-dred C:a Eb:seat C:that's C:not C:a Eb:school C:that's C:a G:flag +Eb:mean- C:-while Eb:dell Eb:un- C:-loads F:pal- Eb:-lets C:at C:the Eb:auc- C:-tion C:door +C:fif- C:-ty Eb:bucks C:du- C:-al Eb:core F:screen Eb:scratched C:on C:the C:floor +C:i Eb:flash C:a Eb:ker- C:-nel C:from C:a C:u- C:-s- C:-b Eb:and Eb:call C:it C:a G:school +G:plork F:the Eb:plan- C:-et Eb:not C:a Eb:the- C:-o- C:-ry Eb:an- C:-y- C:-more C:a C:tool +Eb:ge C:wang Eb:made C:chuck Eb:pa- C:-pert Eb:made C:the F:tur- Eb:-tle C:and C:the C:kid +Eb:il- C:-lich Eb:said Eb:de- C:-school F:the Eb:build- C:-ing C:was C:the C:lid +C:i'm C:not Eb:ask- C:-ing F:per- Eb:-mis- C:-sion C:i'm Eb:just G:lift- F:-ing C:it +Eb:the C:mu- Eb:-sic G:is G:real Eb:and Eb:the F:math G:just C:flipped + +hook + +# verse 2 — TBD (plain lyrics in plork.txt; pitch contour pending) + +hook + +outro +Eb:plug Eb:it C:in Eb:lis- C:-ten G:close +Eb:the C:mu- Eb:-sic G:is C:real +Eb:the C:lo- Eb:-gis- F:-tics Eb:the C:lo- Eb:-gis- F:-tics G:got C:cheap diff --git a/pop/big-pictures/plork.txt b/pop/big-pictures/plork.txt new file mode 100644 --- /dev/null +++ b/pop/big-pictures/plork.txt @@ -0,0 +1,50 @@ +hook +the music is real — the logistics got cheap +landfill's full of orchestras dreaming in their sleep +two-forty million laptops, windows ten retired +flash a kernel — every one of them just got hired + +verse 1 +i watched plork in a basement on a livestream +fifteen laptops in a hall doing rituals like a dream +two thousand six — princeton had the keys to the hall +twenty twenty — half those orchestras don't answer the call +covid took the rooms and nobody built them back +i'm not bitter — i'm honest, i keep score on the track +the gear was the gate, the gate was the price tag +fifteen hundred a seat — that's not a school, that's a flag +meanwhile dell unloads pallets at the auction door +fifty bucks, dual core, screen scratched on the floor +i flash a kernel from a usb and call it a school +plork the planet — not a theory anymore, a tool +ge wang made chuck, papert made the turtle and the kid +illich said deschool — the building was the lid +i'm not asking permission — i'm just lifting it +the music is real and the math just flipped + +hook + +verse 2 +look — i know what they'll say. it sounds like a hack +fifty-dollar laptops can't run the full stack +fair. but the full stack was never the goal +the goal was the room. the room was the soul +we'll lose the latency wars to the fiber-pumped bay +pay attention to mesh networks — they're closer to the day +i can't promise the speakers won't pop in the rain +i can't promise the kernel won't panic on the train +but princeton couldn't promise that either — they just had a wing +small classroom, big idea — same hands, same swing +i'm building from the bottom, you'll hear the plywood +that's not the cheap part — that's the part that's good +e-waste is a deadline. october was the gun +microsoft retired ten — the count's begun +two-forty million machines — pick the ones you want to use +or every single one of them is going down with the refuse + +hook + +outro +plug it in — listen close +the music is real +the logistics — the logistics got cheap diff --git a/pop/big-pictures/row.np b/pop/big-pictures/row.np new file mode 100644 --- /dev/null +++ b/pop/big-pictures/row.np @@ -0,0 +1,10 @@ +# Row Row Row Your Boat — notepat score +# notation: NOTE:syllable, hyphens for syllable continuation +# canonical c major round melody. octave 4 (middle C up to C5 on the +# "merrily" peak). + +verse 1 +C4:row C4:row C4:row D4:your E4:boat*3 +E4:gen- D4:-tly E4:down F4:the G4:stream*3 +C5:mer- C5:-ri- C5:-ly G4:mer- G4:-ri- G4:-ly E4:mer- E4:-ri- E4:-ly C4:mer- C4:-ri- C4:-ly +G4:life F4:is E4:but D4:a C4:dream*4 diff --git a/pop/big-pictures/row.txt b/pop/big-pictures/row.txt new file mode 100644 --- /dev/null +++ b/pop/big-pictures/row.txt @@ -0,0 +1,5 @@ +verse 1 +row row row your boat +gently down the stream +merrily merrily merrily merrily +life is but a dream diff --git a/pop/big-pictures/sentence.np b/pop/big-pictures/sentence.np new file mode 100644 --- /dev/null +++ b/pop/big-pictures/sentence.np @@ -0,0 +1,6 @@ +# the music is real — short hook phrase, c minor pentatonic +# rising contour landing on G3 for "real" (the emphasized word) +# 5 syllables: the / mu- / -sic / is / real + +verse 1 +C3:the Eb3:mu- Eb3:-sic F3:is G3:real diff --git a/pop/big-pictures/sentence.txt b/pop/big-pictures/sentence.txt new file mode 100644 --- /dev/null +++ b/pop/big-pictures/sentence.txt @@ -0,0 +1,2 @@ +verse 1 +the music is real diff --git a/pop/big-pictures/uncomfortable.np b/pop/big-pictures/uncomfortable.np new file mode 100644 --- /dev/null +++ b/pop/big-pictures/uncomfortable.np @@ -0,0 +1,6 @@ +# uncomfortable — single-word melody substrate. +# 5 syllables: un-com-fort-a-ble +# descending arc, c minor: G3 → F3 → Eb3 → D3 → C3. + +verse 1 +G3:un- F3:-com- Eb3:-fort- D3:-a- C3:-ble diff --git a/pop/big-pictures/uncomfortable.txt b/pop/big-pictures/uncomfortable.txt new file mode 100644 --- /dev/null +++ b/pop/big-pictures/uncomfortable.txt @@ -0,0 +1,2 @@ +verse 1 +uncomfortable diff --git a/pop/bin/align.mjs b/pop/bin/align.mjs new file mode 100644 --- /dev/null +++ b/pop/bin/align.mjs @@ -0,0 +1,79 @@ +#!/usr/bin/env node +// align.mjs — run whisper-cli on a vocal stem, emit per-word timestamps. +// +// Mirrors `recap/bin/transcribe.mjs` but takes the stem path as an +// argument and writes `-words.json` next to the source. Uses the +// same whisper.cpp model already in recap/models/. Caches by source +// content hash; --force to bypass. +// +// Output format (matches recap pipeline): [{text, fromMs, toMs}, ...] +// +// Usage: +// node bin/align.mjs ../big-pictures/out/plork-hook-vocal.mp3 +// node bin/align.mjs --force + +import { execFileSync } from "node:child_process"; +import { readFileSync, writeFileSync, existsSync, unlinkSync } from "node:fs"; +import { resolve, dirname, basename } from "node:path"; +import { fileURLToPath } from "node:url"; +import { createHash } from "node:crypto"; + +const HERE = dirname(fileURLToPath(import.meta.url)); +const ROOT = resolve(HERE, ".."); +const REPO = resolve(ROOT, ".."); +const MODEL = `${REPO}/recap/models/ggml-base.en.bin`; + +const argv = process.argv.slice(2); +const force = argv.includes("--force"); +const stemPath = resolve(process.cwd(), argv.find((a) => !a.startsWith("--")) || ""); + +if (!stemPath || !existsSync(stemPath)) { + console.error("usage: node bin/align.mjs [--force]"); + process.exit(1); +} +if (!existsSync(MODEL)) { + console.error(`✗ missing whisper model: ${MODEL}\n download: curl -L -o ${MODEL} https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-base.en.bin`); + process.exit(1); +} + +const inputHash = createHash("sha256") + .update(readFileSync(stemPath)) + .digest("hex").slice(0, 16); + +const wordsPath = stemPath.replace(/\.mp3$/, "-words.json"); +const hashFile = `${wordsPath}.hash`; + +if (!force && existsSync(wordsPath) && existsSync(hashFile)) { + const cached = readFileSync(hashFile, "utf8").trim(); + if (cached === inputHash) { + const words = JSON.parse(readFileSync(wordsPath, "utf8")); + const last = words[words.length - 1]; + console.log(`✓ ${wordsPath} cached · ${words.length} words · hash ${inputHash} — skipping whisper`); + if (last) console.log(` audio ends at ${(last.toMs / 1000).toFixed(2)}s`); + process.exit(0); + } +} + +console.log(`→ whisper-cli · ${stemPath}`); +const stemDir = dirname(stemPath); +const stemBase = basename(stemPath).replace(/\.mp3$/, ""); +const tmpJson = `${stemDir}/${stemBase}.json`; + +execFileSync( + "whisper-cli", + ["-m", MODEL, "-f", stemPath, "-ojf", "-of", `${stemDir}/${stemBase}`, "--max-len", "1", "-ml", "1", "-sow"], + { stdio: ["ignore", "ignore", "inherit"] }, +); + +const raw = JSON.parse(readFileSync(tmpJson, "utf8")); +const words = raw.transcription + .map((s) => ({ text: s.text.trim(), fromMs: s.offsets.from, toMs: s.offsets.to })) + .filter((w) => w.text.length > 0); + +writeFileSync(wordsPath, JSON.stringify(words, null, 2)); +writeFileSync(hashFile, inputHash + "\n"); +try { unlinkSync(tmpJson); } catch {} + +const last = words[words.length - 1]; +console.log(`✓ ${wordsPath} · ${words.length} words · ${(last.toMs / 1000).toFixed(2)}s · hash ${inputHash}`); +console.log(` first 6: ${words.slice(0, 6).map(w => `${w.text}@${(w.fromMs/1000).toFixed(2)}s`).join(" ")}`); diff --git a/pop/bin/finalize.mjs b/pop/bin/finalize.mjs new file mode 100644 --- /dev/null +++ b/pop/bin/finalize.mjs @@ -0,0 +1,456 @@ +#!/usr/bin/env node +// finalize.mjs — emit a finalized mp3 with ID3v2 metadata + embedded +// auto-generated, timestamped cover art. Last leg of the +// pop/big-pictures lane: takes a rendered track (output of any of the +// align/pitchsnap/timefit pipeline) and produces a streaming-ready file. +// +// Cover generation runs entirely offline — no APIs, no canvas +// dependencies. PNG is hand-rolled with zlib + a 6x10 bitmap font +// extracted from fedac/native/src/font-6x10.h (public-domain X11 +// terminal font), scaled up to draw big titles at 3000×3000. +// +// Usage: +// node bin/finalize.mjs --in out/ac-mix.mp3 --slug ac --title "aesthetic 24" +// node bin/finalize.mjs --in out/mary-tuned.mp3 --slug mary +// node bin/finalize.mjs --in out/plork-chorus.mp3 --slug plork --cover my-cover.png +// node bin/finalize.mjs --in out/ac-mix.mp3 --slug ac --out custom-final.mp3 --force + +import { spawnSync } from "node:child_process"; +import { writeFileSync, readFileSync, mkdirSync, existsSync, statSync } from "node:fs"; +import { resolve, dirname, basename } from "node:path"; +import { fileURLToPath } from "node:url"; +import { createHash } from "node:crypto"; +import { homedir } from "node:os"; +import { deflateSync } from "node:zlib"; + +const HERE = dirname(fileURLToPath(import.meta.url)); +const ROOT = resolve(HERE, ".."); + +// ── argv parsing (matches say.mjs / timefit.mjs idiom) ───────────────── +function parseArgs(argv) { + const flags = {}; + const positional = []; + for (let i = 0; i < argv.length; i++) { + const a = argv[i]; + if (a.startsWith("--")) { + const k = a.slice(2); + const next = argv[i + 1]; + if (next !== undefined && !next.startsWith("--")) { flags[k] = next; i++; } + else flags[k] = true; + } else positional.push(a); + } + return { flags, positional }; +} + +function expandHome(p) { + if (!p || typeof p !== "string") return p; + if (p === "~") return homedir(); + if (p.startsWith("~/")) return resolve(homedir(), p.slice(2)); + return p; +} + +const { flags } = parseArgs(process.argv.slice(2)); + +if (!flags.in || !flags.slug) { + console.error("usage: node bin/finalize.mjs --in --slug [--title \"...\"] [--cover ] [--out ] [--force]"); + process.exit(1); +} + +const IN_PATH = resolve(process.cwd(), expandHome(flags.in)); +if (!existsSync(IN_PATH)) { + console.error(`✗ input mp3 not found: ${IN_PATH}`); + process.exit(1); +} + +const SLUG = String(flags.slug).trim().toLowerCase(); +const TITLE = flags.title && flags.title !== true + ? String(flags.title) + : SLUG.replace(/[-_]+/g, " "); +const FORCE = flags.force === true; +const OUT_PATH = expandHome(flags.out) + ? resolve(process.cwd(), expandHome(flags.out)) + : `${ROOT}/big-pictures/out/${SLUG}-final.mp3`; +const COVER_OVERRIDE = flags.cover && flags.cover !== true + ? resolve(process.cwd(), expandHome(flags.cover)) + : null; + +mkdirSync(dirname(OUT_PATH), { recursive: true }); + +// ── Probe input duration (mirror of timefit.mjs) ─────────────────────── +const probe = spawnSync( + "ffprobe", + ["-v", "error", "-show_entries", "format=duration", + "-of", "default=noprint_wrappers=1:nokey=1", IN_PATH], + { encoding: "utf8" }, +); +const inputDur = Number(probe.stdout.trim()); +if (!(inputDur > 0)) { + console.error(`✗ ffprobe could not read duration of ${IN_PATH}`); + process.exit(1); +} + +// ── Build metadata ───────────────────────────────────────────────────── +const now = new Date(); +const pad = (n) => String(n).padStart(2, "0"); +const isoDate = `${now.getUTCFullYear()}-${pad(now.getUTCMonth() + 1)}-${pad(now.getUTCDate())}`; +const isoYear = String(now.getUTCFullYear()); +const isoStamp = now.toISOString(); +const localStampShort = `${isoDate} ${pad(now.getUTCHours())}:${pad(now.getUTCMinutes())} UTC`; + +const ARTIST = "@jeffrey"; +const ALBUM = "big pictures"; +const GENRE = "emo trap"; + +// Lyric file (sibling .txt) → USLT +const lyricPath = `${ROOT}/big-pictures/${SLUG}.txt`; +let lyricsText = null; +if (existsSync(lyricPath)) { + lyricsText = readFileSync(lyricPath, "utf8").trim(); +} + +const meta = { + title: TITLE, + artist: ARTIST, + album_artist: ARTIST, + album: ALBUM, + date: isoDate, + year: isoYear, + genre: GENRE, + comment: `rendered ${isoStamp} · source: ${basename(IN_PATH)}`, +}; + +// ── Cache key (idempotent on input contents + key metadata) ──────────── +const inputBuf = readFileSync(IN_PATH); +const cacheKey = createHash("sha256").update(inputBuf).update(JSON.stringify({ + slug: SLUG, title: TITLE, lyricsText, coverOverride: COVER_OVERRIDE, +})).digest("hex").slice(0, 16); +const hashFile = `${OUT_PATH}.hash`; + +if (!FORCE && existsSync(OUT_PATH) && existsSync(hashFile)) { + const cached = readFileSync(hashFile, "utf8").trim(); + if (cached === cacheKey) { + const size = (statSync(OUT_PATH).size / 1024).toFixed(0); + console.log(`✓ ${OUT_PATH} cached (${size} KB · hash ${cacheKey}) — skipping finalize`); + process.exit(0); + } +} + +// ── 6×10 bitmap font (public-domain X11 fixed, mined from fedac) ─────── +// Each glyph is 10 bytes; high 6 bits of each byte are the row pixels +// (MSB = leftmost column). Glyph 0 = ASCII 32 (space). 95 glyphs total. +const FONT_W = 6, FONT_H = 10; +const FONT_B64 = + "AAAAAAAAAAAAAAAgICAgIAAgAAAAUFBQAAAAAAAAAFBQ+FD4UFAAAAAgcKBwKHAgAAAASKhQIFCo" + + "kAAAAECgoECokGgAAAAgICAAAAAAAAAAECBAQEAgEAAAAEAgEBAQIEAAAAAAiFD4UIgAAAAAACAg" + + "+CAgAAAAAAAAAAAAMCBAAAAAAAD4AAAAAAAAAAAAAAAgcCAAAAgIECBAgIAAAAAgUIiIiFAgAAAA" + + "IGCgICAg+AAAAHCICDBAgPgAAAD4CBAwCIhwAAAAEDBQkPgQEAAAAPiAsMgIiHAAAAAwQICwyIhw" + + "AAAA+AgQECBAQAAAAHCIiHCIiHAAAABwiJhoCBBgAAAAACBwIAAgcCAAAAAgcCAAMCBAAAAIECBA" + + "IBAIAAAAAAD4APgAAAAAAEAgEAgQIEAAAABwiBAgIAAgAAAAcIiYqLCAcAAAACBQiIj4iIgAAADw" + + "SEhwSEjwAAAAcIiAgICIcAAAAPBISEhISPAAAAD4gIDwgID4AAAA+ICA8ICAgAAAAHCIgICYiHAA" + + "AACIiIj4iIiIAAAAcCAgICAgcAAAADgQEBAQkGAAAACIkKDAoJCIAAAAgICAgICA+AAAAIiI2KiI" + + "iIgAAACIiMiomIiIAAAAcIiIiIiIcAAAAPCIiPCAgIAAAABwiIiIiKhwCAAA8IiI8KCQiAAAAHCI" + + "gHAIiHAAAAD4ICAgICAgAAAAiIiIiIiIcAAAAIiIiFBQUCAAAACIiIioqNiIAAAAiIhQIFCIiAAA" + + "AIiIUCAgICAAAAD4CBAgQID4AAAAcEBAQEBAcAAAAICAQCAQCAgAAABwEBAQEBBwAAAAIFCIAAAA" + + "AAAAAAAAAAAAAAD4ACAQAAAAAAAAAAAAAABwCHiIeAAAAICAsMiIyLAAAAAAAHCIgIhwAAAACAho" + + "mIiYaAAAAAAAcIj4gHAAAAAwSEDwQEBAAAAAAAB4iIh4CIhwAICAsMiIiIgAAAAgAGAgICBwAAAA" + + "CAAYCAgISEgwAICAiJDgkIgAAABgICAgICBwAAAAAADQqKioiAAAAAAAsMiIiIgAAAAAAHCIiIhw" + + "AAAAAACwyIjIsICAAAAAaJiImGgICAAAALDIgICAAAAAAABwgHAI8AAAAEBA8EBASDAAAAAAAIiI" + + "iJhoAAAAAACIiFBQIAAAAAAAiIioqFAAAAAAAIhQIFCIAAAAAACIiJhoCIhwAAAA+BAgQPgAAAAY" + + "IBBgECAYAAAAICAgICAgIAAAAGAQIBggEGAAAABIqJAAAAAAAAA="; +const FONT = Buffer.from(FONT_B64, "base64"); + +// pixel sample at (x, y) in glyph space [0..6) × [0..10). +function glyphPixel(ch, x, y) { + const code = ch.charCodeAt(0); + if (code < 32 || code > 126) return false; + const idx = code - 32; + const row = FONT[idx * FONT_H + y]; + if (row === undefined) return false; + // High 6 bits hold the columns; bit 7 = leftmost. + return (row & (0x80 >> x)) !== 0; +} + +// Measure text width at scale (no kerning, fixed-pitch font). +function textWidth(str, scale) { + return str.length * FONT_W * scale; +} + +// ── Pure-Node 8-bit RGB framebuffer ──────────────────────────────────── +function makeFB(w, h, bg) { + const buf = Buffer.alloc(w * h * 3); + for (let i = 0; i < buf.length; i += 3) { + buf[i] = bg[0]; buf[i + 1] = bg[1]; buf[i + 2] = bg[2]; + } + return { w, h, buf }; +} + +function setPx(fb, x, y, rgb) { + if (x < 0 || y < 0 || x >= fb.w || y >= fb.h) return; + const i = (y * fb.w + x) * 3; + fb.buf[i] = rgb[0]; fb.buf[i + 1] = rgb[1]; fb.buf[i + 2] = rgb[2]; +} + +function fillRect(fb, x0, y0, w, h, rgb) { + for (let y = y0; y < y0 + h; y++) { + for (let x = x0; x < x0 + w; x++) setPx(fb, x, y, rgb); + } +} + +function drawText(fb, str, x0, y0, scale, rgb) { + for (let i = 0; i < str.length; i++) { + const ch = str[i]; + const gx0 = x0 + i * FONT_W * scale; + for (let gy = 0; gy < FONT_H; gy++) { + for (let gx = 0; gx < FONT_W; gx++) { + if (!glyphPixel(ch, gx, gy)) continue; + // Stamp scale×scale square per pixel. + fillRect(fb, gx0 + gx * scale, y0 + gy * scale, scale, scale, rgb); + } + } + } +} + +// ── Hand-rolled PNG writer (deflate, RGB8, no filtering) ─────────────── +function crc32(buf) { + let c, table = crc32.table; + if (!table) { + table = new Uint32Array(256); + for (let n = 0; n < 256; n++) { + c = n; + for (let k = 0; k < 8; k++) c = (c & 1) ? (0xEDB88320 ^ (c >>> 1)) : (c >>> 1); + table[n] = c >>> 0; + } + crc32.table = table; + } + c = 0xFFFFFFFF; + for (let i = 0; i < buf.length; i++) c = table[(c ^ buf[i]) & 0xFF] ^ (c >>> 8); + return (c ^ 0xFFFFFFFF) >>> 0; +} + +function chunk(type, data) { + const len = Buffer.alloc(4); len.writeUInt32BE(data.length, 0); + const tbuf = Buffer.from(type, "ascii"); + const crcBuf = Buffer.alloc(4); + crcBuf.writeUInt32BE(crc32(Buffer.concat([tbuf, data])), 0); + return Buffer.concat([len, tbuf, data, crcBuf]); +} + +function writePNG(fb) { + const sig = Buffer.from([0x89, 0x50, 0x4E, 0x47, 0x0D, 0x0A, 0x1A, 0x0A]); + // IHDR + const ihdr = Buffer.alloc(13); + ihdr.writeUInt32BE(fb.w, 0); + ihdr.writeUInt32BE(fb.h, 4); + ihdr[8] = 8; // bit depth + ihdr[9] = 2; // color type: RGB + ihdr[10] = 0; // compression + ihdr[11] = 0; // filter + ihdr[12] = 0; // interlace + // IDAT — prepend 0x00 filter byte to each scanline + const stride = fb.w * 3; + const filtered = Buffer.alloc(fb.h * (stride + 1)); + for (let y = 0; y < fb.h; y++) { + filtered[y * (stride + 1)] = 0; + fb.buf.copy(filtered, y * (stride + 1) + 1, y * stride, (y + 1) * stride); + } + const idat = deflateSync(filtered, { level: 6 }); + return Buffer.concat([sig, chunk("IHDR", ihdr), chunk("IDAT", idat), chunk("IEND", Buffer.alloc(0))]); +} + +// ── Slug-deterministic palette (HSL → RGB, AC-saturated) ─────────────── +function hslToRgb(h, s, l) { + // h ∈ [0..360), s ∈ [0..1], l ∈ [0..1] + const c = (1 - Math.abs(2 * l - 1)) * s; + const hp = h / 60; + const x = c * (1 - Math.abs((hp % 2) - 1)); + let r1 = 0, g1 = 0, b1 = 0; + if (hp < 1) { r1 = c; g1 = x; } + else if (hp < 2) { r1 = x; g1 = c; } + else if (hp < 3) { g1 = c; b1 = x; } + else if (hp < 4) { g1 = x; b1 = c; } + else if (hp < 5) { r1 = x; b1 = c; } + else { r1 = c; b1 = x; } + const m = l - c / 2; + return [Math.round((r1 + m) * 255), Math.round((g1 + m) * 255), Math.round((b1 + m) * 255)]; +} + +function paletteFromSlug(slug) { + const h = createHash("sha256").update(slug).digest(); + const hue = (h[0] / 256) * 360; + // AC palette: rich, saturated, fairly dark backgrounds with a complementary accent + const bg = hslToRgb(hue, 0.72, 0.16); + const accent = hslToRgb((hue + 36) % 360, 0.85, 0.58); + // Bright cream type for max contrast against the dark bg + const fg = hslToRgb(hue, 0.10, 0.96); + const dim = hslToRgb(hue, 0.40, 0.62); + return { bg, fg, accent, dim }; +} + +// ── Compose the cover (3000×3000) ────────────────────────────────────── +function buildCover(slug, title, stampShort) { + const W = 3000, H = 3000; + const pal = paletteFromSlug(slug); + const fb = makeFB(W, H, pal.bg); + + // Top decoration bars — geometric, hand-drawn-feeling. + const h = createHash("sha256").update(slug).digest(); + const barCount = 5 + (h[1] % 5); + for (let i = 0; i < barCount; i++) { + fillRect(fb, 200, 90 + i * 22, W - 400, 4, pal.dim); + } + + // ── Vertical bands (no overlap) ───────────────────────────────────── + // 90..200 top bars + // 280..1080 title (scale ≤80) + // 1180..1480 waveform + // 1560..1780 stamp + // 1860..2230 album mark "big pictures" + // 2310..2870 handle "@jeffrey" + // 2900..2980 bottom bars + + // Title — large, occupies most of the top band. + const titleStr = String(title || slug).toLowerCase(); + let titleScale = Math.floor((W - 400) / (titleStr.length * FONT_W)); + titleScale = Math.min(titleScale, 80); + titleScale = Math.max(titleScale, 28); + const titleW = textWidth(titleStr, titleScale); + const titleX = Math.round((W - titleW) / 2); + const titleH = FONT_H * titleScale; + const titleY = 280 + Math.round((800 - titleH) / 2); + const shadowOff = Math.max(4, Math.round(titleScale * 0.18)); + drawText(fb, titleStr, titleX + shadowOff, titleY + shadowOff, titleScale, pal.accent); + drawText(fb, titleStr, titleX, titleY, titleScale, pal.fg); + + // Waveform glyph + const wvBars = 32; + const wvY = 1330; + const wvW = Math.round(W * 0.66); + const wvX = Math.round((W - wvW) / 2); + const barWidth = Math.floor(wvW / wvBars) - 4; + for (let i = 0; i < wvBars; i++) { + const seed = h[i % h.length]; + const amp = 30 + ((seed * (i + 1)) % 130); + const x = wvX + i * Math.floor(wvW / wvBars); + fillRect(fb, x, wvY - amp, barWidth, amp * 2, pal.accent); + } + + // Timestamp — small, centered. + const stampScale = 22; + const stampW = textWidth(stampShort, stampScale); + drawText(fb, stampShort, Math.round((W - stampW) / 2), 1560, stampScale, pal.dim); + + // Album mark "big pictures" — auto-fit width. + const mark = "big pictures"; + let markScale = Math.floor((W - 300) / (mark.length * FONT_W)); + markScale = Math.min(markScale, 38); + markScale = Math.max(markScale, 24); + const markW = textWidth(mark, markScale); + const markH = FONT_H * markScale; + const markY = 1860 + Math.round((370 - markH) / 2); + drawText(fb, mark, Math.round((W - markW) / 2), markY, markScale, pal.fg); + + // Handle "@jeffrey" — biggest text on the cover, max-fit to width. + const handle = "@jeffrey"; + let handleScale = Math.floor((W - 240) / (handle.length * FONT_W)); + handleScale = Math.min(handleScale, 70); + handleScale = Math.max(handleScale, 28); + const handleW = textWidth(handle, handleScale); + const handleX = Math.round((W - handleW) / 2); + const handleH = FONT_H * handleScale; + const handleY = 2310 + Math.round((560 - handleH) / 2); + const handleShadow = Math.max(6, Math.round(handleScale * 0.14)); + drawText(fb, handle, handleX + handleShadow, handleY + handleShadow, handleScale, pal.accent); + drawText(fb, handle, handleX, handleY, handleScale, pal.fg); + + // Bottom decoration bars + for (let i = 0; i < barCount; i++) { + fillRect(fb, 200, H - 90 - i * 14, W - 400, 4, pal.dim); + } + + return writePNG(fb); +} + +// ── Resolve cover path ───────────────────────────────────────────────── +let coverPath; +if (COVER_OVERRIDE) { + if (!existsSync(COVER_OVERRIDE)) { + console.error(`✗ --cover not found: ${COVER_OVERRIDE}`); + process.exit(1); + } + coverPath = COVER_OVERRIDE; + console.log(`→ using provided cover: ${coverPath}`); +} else { + coverPath = `${ROOT}/big-pictures/out/${SLUG}-cover.png`; + console.log(`→ generating cover: ${coverPath} (3000×3000, ${localStampShort})`); + const png = buildCover(SLUG, TITLE, localStampShort); + writeFileSync(coverPath, png); + console.log(` wrote ${(png.length / 1024).toFixed(0)} KB`); +} + +// ── ffmpeg: mux audio + cover, attach ID3v2 ──────────────────────────── +// Notes: +// * `-map 0:a -map 1` keeps audio + cover. +// * `-c copy` preserves the source mp3 bitstream (no re-encode). +// * `-disposition:v attached_pic` sets the APIC role. +// * `-id3v2_version 3` keeps ID3v2.3 (best player support). +// * `-write_id3v1 1` is a courtesy for legacy players. +// * `-metadata:s:v` sets the per-stream cover description. +const args = [ + "-hide_banner", "-y", "-loglevel", "error", + "-i", IN_PATH, + "-i", coverPath, + "-map", "0:a", + "-map", "1", + "-c", "copy", + "-id3v2_version", "3", + "-write_id3v1", "1", + "-disposition:v", "attached_pic", + "-metadata:s:v", "title=Album cover", + "-metadata:s:v", "comment=Cover (front)", +]; + +for (const [k, v] of Object.entries(meta)) { + args.push("-metadata", `${k}=${v}`); +} +if (lyricsText) { + args.push("-metadata", `lyrics-eng=${lyricsText}`); +} + +args.push(OUT_PATH); + +console.log(`→ ffmpeg mux · in=${basename(IN_PATH)} cover=${basename(coverPath)} → ${basename(OUT_PATH)}`); +const ff = spawnSync("ffmpeg", args, { stdio: "inherit" }); +if (ff.status !== 0) { + console.error("✗ ffmpeg failed"); + process.exit(1); +} + +// ── Verify output ────────────────────────────────────────────────────── +const verify = spawnSync( + "ffprobe", + ["-v", "error", "-show_entries", + "format=duration:format_tags=title,artist,album,album_artist,date,genre,comment,lyrics-eng:stream=codec_type,codec_name,disposition:stream_tags=title,comment", + "-of", "default=noprint_wrappers=1", OUT_PATH], + { encoding: "utf8" }, +); + +const outDur = (() => { + const m = (verify.stdout || "").match(/^duration=(.+)$/m); + return m ? Number(m[1]) : null; +})(); + +const hasCover = (verify.stdout || "").includes("codec_type=video"); +const drift = outDur !== null ? outDur - inputDur : null; + +writeFileSync(hashFile, cacheKey + "\n"); +const outSize = (statSync(OUT_PATH).size / 1024).toFixed(0); +console.log(`✓ ${OUT_PATH} (${outSize} KB · hash ${cacheKey})`); +console.log(` duration ${outDur?.toFixed(3) ?? "?"}s` + + (drift !== null ? ` (drift ${drift >= 0 ? "+" : ""}${drift.toFixed(3)}s)` : "") + + ` · cover ${hasCover ? "embedded" : "MISSING"}` + + ` · lyrics ${lyricsText ? "yes" : "no"}`); +if (!hasCover) { + console.error("✗ cover stream not detected in output — verify ffprobe report:"); + console.error(verify.stdout); + process.exit(1); +} diff --git a/pop/bin/musicxml_to_np.py b/pop/bin/musicxml_to_np.py new file mode 100644 --- /dev/null +++ b/pop/bin/musicxml_to_np.py @@ -0,0 +1,313 @@ +#!/usr/bin/env python3 +""" +musicxml_to_np.py — convert a MusicXML lead sheet (melody + lyrics) to +AC's `.np` score format used by pitchsnap.mjs. + +The .np format pairs one note with one syllable per token: + NOTE:syllable*beats + + - NOTE — scientific pitch notation, e.g. D3, G#3 (flats normalized to + enharmonic sharps so the AC parser doesn't have to know "Bb3") + - syllable — lowercase. MusicXML markers map to AC dashes: + single → "grace" + begin → "a-" + middle → "-ma-" + end → "-zing" + - beats — duration in beats (MusicXML / ). + Rounded to int when whole, otherwise 2-decimal. + +Tied notes are merged into one logical note (sum of durations, lyric +from the first). Chord tones (non-melody pitches) and rests are +skipped — this is a melody-extraction pass for the AC vocal lane, not +a full MusicXML round-trip. + +Usage: + python bin/musicxml_to_np.py input.musicxml output.np \\ + [--bpm 70] [--key "G major"] [--title "Amazing Grace"] + +Designed to be run from sources like Hymnary, Mutopia (after +`lilypond --output=musicxml`), or any MuseScore export. +""" +import argparse +import sys +import xml.etree.ElementTree as ET +from pathlib import Path +from fractions import Fraction + + +def strip_ns(tag): + """Strip XML namespace if present (MusicXML files in the wild are + inconsistent about whether they declare a namespace).""" + return tag.split("}", 1)[-1] if "}" in tag else tag + + +def find(elem, name): + """Find a direct child by local name, ignoring namespaces.""" + if elem is None: + return None + for child in elem: + if strip_ns(child.tag) == name: + return child + return None + + +def findall(elem, name): + if elem is None: + return [] + return [c for c in elem if strip_ns(c.tag) == name] + + +# Flats → enharmonic sharps (AC pitch parser sticks to sharps) +FLAT_TO_SHARP = {"D": "C#", "E": "D#", "G": "F#", "A": "G#", "B": "A#"} + + +def pitch_to_np(step, alter, octave): + name = step.upper() + if alter == 1: + name += "#" + elif alter == 2: + # Double sharp — rare. Roll forward one whole step. + roll = {"C": "D", "D": "E", "F": "G", "G": "A", "A": "B"} + name = roll.get(name, name + "##") + elif alter == -1: + eq = FLAT_TO_SHARP.get(step.upper()) + if eq is not None: + name = eq + elif step.upper() == "C": + # Cb → B (same pitch, octave - 1) + return f"B{octave - 1}" + elif step.upper() == "F": + # Fb → E + name = "E" + elif alter == -2: + # Double flat — also rare; roll back one whole step + roll = {"E": "D", "B": "A", "A": "G", "G": "F", "D": "C"} + name = roll.get(name, name + "bb") + return f"{name}{octave}" + + +def syllabify(text, syllabic): + text = (text or "").lower().strip() + # Strip punctuation that would confuse the AC parser + text = text.replace("_", "").replace(",", "").replace(".", "").replace("?", "").replace("!", "") + if not text: + return "_" + if syllabic == "begin": + return text + "-" + if syllabic == "middle": + return "-" + text + "-" + if syllabic == "end": + return "-" + text + return text # single (or default) + + +def beats_str(beats): + if beats == int(beats): + return str(int(beats)) + # 2-dec, but trim trailing zeros (1.50 → 1.5) + s = f"{beats:.2f}".rstrip("0").rstrip(".") + return s or "0" + + +class TiedAccumulator: + """Collects duration across until we + see . The first note in the chain owns the lyric.""" + + def __init__(self): + self.active = False + self.pitch = None + self.duration = 0 + self.divisions = 1 + self.lyric = None + self.syllabic = None + + def reset(self): + self.__init__() + + +def extract_melody(part): + """Walk one , yielding (np_pitch, syllable, beats) tuples in + order. Handles tied notes, chord tones, rests, key/voice changes.""" + tokens = [] + line_breaks = [] # measure index of each line break candidate + + divisions = 1 + tied = TiedAccumulator() + main_voice = None + + measures = findall(part, "measure") + for m_idx, measure in enumerate(measures): + attrs = find(measure, "attributes") + if attrs is not None: + d = find(attrs, "divisions") + if d is not None and d.text: + divisions = int(d.text) + + for note in findall(measure, "note"): + voice_el = find(note, "voice") + voice = voice_el.text if voice_el is not None else "1" + if main_voice is None: + main_voice = voice + if voice != main_voice: + continue + + # Chord tone (non-first pitch in a chord) — skip; we only + # transcribe the topmost melody. + if find(note, "chord") is not None: + continue + + duration_el = find(note, "duration") + if duration_el is None or not duration_el.text: + continue + duration = int(duration_el.text) + + # Rest: flush any tied note, then skip + if find(note, "rest") is not None: + if tied.active: + tokens.append(_emit_tied(tied)) + tied.reset() + continue + + pitch = find(note, "pitch") + if pitch is None: + continue + + step = (find(pitch, "step").text or "C").upper() + octave = int((find(pitch, "octave").text or "4")) + alter_el = find(pitch, "alter") + alter = int(alter_el.text) if alter_el is not None and alter_el.text else 0 + np_pitch = pitch_to_np(step, alter, octave) + + # Lyric (only from notes that *start* a syllable) + lyric = None + syllabic = "single" + for ly in findall(note, "lyric"): + txt = find(ly, "text") + if txt is not None and txt.text: + lyric = txt.text + syl = find(ly, "syllabic") + syllabic = syl.text if syl is not None and syl.text else "single" + break + + # Tie handling + ties = findall(note, "tie") + tie_types = {t.attrib.get("type") for t in ties} + + if "start" in tie_types and "stop" not in tie_types: + # Start of a tied chain + if tied.active: + tokens.append(_emit_tied(tied)) + tied.active = True + tied.pitch = np_pitch + tied.duration = duration + tied.divisions = divisions + tied.lyric = lyric + tied.syllabic = syllabic + elif tie_types == {"start", "stop"} or tie_types == {"stop", "start"}: + # Continuation in the middle of a chain + if tied.active and tied.pitch == np_pitch: + tied.duration += duration + else: + # Stray; treat as new + if tied.active: + tokens.append(_emit_tied(tied)) + tied.reset() + tokens.append((np_pitch, syllabify(lyric, syllabic), + duration / divisions)) + elif "stop" in tie_types: + # End of a tied chain + if tied.active and tied.pitch == np_pitch: + tied.duration += duration + tokens.append(_emit_tied(tied)) + else: + tokens.append((np_pitch, syllabify(lyric, syllabic), + duration / divisions)) + tied.reset() + else: + # Plain note + if tied.active: + tokens.append(_emit_tied(tied)) + tied.reset() + tokens.append((np_pitch, syllabify(lyric, syllabic), + duration / divisions)) + + line_breaks.append(len(tokens)) + + if tied.active: + tokens.append(_emit_tied(tied)) + + return tokens, line_breaks + + +def _emit_tied(t): + return (t.pitch, syllabify(t.lyric, t.syllabic), t.duration / t.divisions) + + +def find_part(root): + """Return the first element. Score may have + or at the root, with namespaces, etc.""" + for elem in root.iter(): + if strip_ns(elem.tag) == "part": + return elem + return None + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("input", help="MusicXML file") + ap.add_argument("output", help="Output .np file") + ap.add_argument("--bpm", type=int, help="Tempo in BPM (written as a comment)") + ap.add_argument("--key", help="Key name (written as a comment)") + ap.add_argument("--title", help="Title (written as a comment)") + ap.add_argument("--verse", default="verse 1", help='Verse heading (default: "verse 1")') + ap.add_argument("--line-every", type=int, default=8, + help="Wrap output to a new line every N notes (default: 8)") + args = ap.parse_args() + + tree = ET.parse(args.input) + root = tree.getroot() + part = find_part(root) + if part is None: + print("✗ no in MusicXML", file=sys.stderr) + sys.exit(1) + + tokens, line_breaks = extract_melody(part) + if not tokens: + print("✗ no melody notes extracted", file=sys.stderr) + sys.exit(1) + + # Build output + lines = [] + if args.title: + lines.append(f"# {args.title}") + if args.key: + lines.append(f"# key: {args.key}") + if args.bpm: + lines.append(f"# Use --beat-mode --bpm {args.bpm}.") + if lines: + lines.append("") # blank separator + + lines.append(args.verse) + + # Wrap line every N notes (simple heuristic; user re-flows by hand) + cur = [] + for i, (pitch, syl, beats) in enumerate(tokens): + cur.append(f"{pitch}:{syl}*{beats_str(beats)}") + if len(cur) >= args.line_every: + lines.append(" ".join(cur)) + cur = [] + if cur: + lines.append(" ".join(cur)) + + Path(args.output).write_text("\n".join(lines) + "\n") + print(f"✓ {args.output}") + print(f" {len(tokens)} notes · {len(line_breaks)} measures · " + f"≈{sum(t[2] for t in tokens):.1f} beats") + # Show first line preview + first_line = next((l for l in lines if l and not l.startswith("#") and l != args.verse), "") + if first_line: + print(f" first line: {first_line[:120]}{'…' if len(first_line) > 120 else ''}") + + +if __name__ == "__main__": + main() diff --git a/pop/bin/pitchcheck.mjs b/pop/bin/pitchcheck.mjs new file mode 100644 --- /dev/null +++ b/pop/bin/pitchcheck.mjs @@ -0,0 +1,264 @@ +#!/usr/bin/env node +// pitchcheck.mjs — measure the actual fundamental of each word in a +// rendered vocal stem and compare to what pitchsnap *intended* to +// shift each word to. Reads the `.events.json` emitted by pitchsnap +// (avoids re-aligning the rendered output, which whisper degrades on +// heavily-shifted audio). +// +// Pitch detection: autocorrelation over a central window of each +// word's slice, restricted to voice range [80 Hz, 600 Hz]. Parabolic +// interpolation around the peak for sub-sample precision. Skip +// silence by RMS gate. +// +// Output: per-word table of expected vs measured + cents drift, and a +// summary of mean / median absolute drift. ±50¢ = quarter-tone, ±25¢ +// = "in tune." +// +// Usage: +// node bin/pitchcheck.mjs --vocal big-pictures/out/mary-sung.mp3 +// (auto-finds mary-sung.events.json next to the mp3) + +import { spawnSync } from "node:child_process"; +import { existsSync, readFileSync, mkdirSync, rmSync } from "node:fs"; +import { resolve, dirname, basename } from "node:path"; + +function parseArgs(argv) { + const flags = {}; + for (let i = 0; i < argv.length; i++) { + const a = argv[i]; + if (!a.startsWith("--")) continue; + const k = a.slice(2); + const next = argv[i + 1]; + if (next !== undefined && !next.startsWith("--")) { flags[k] = next; i++; } + else flags[k] = true; + } + return flags; +} + +const flags = parseArgs(process.argv.slice(2)); +const vocalPath = resolve(process.cwd(), flags.vocal || ""); +if (!existsSync(vocalPath)) { + console.error("usage: --vocal "); + process.exit(1); +} +const eventsPath = resolve(process.cwd(), + flags.events || vocalPath.replace(/\.mp3$/, ".events.json")); +if (!existsSync(eventsPath)) { + console.error(`✗ events file not found: ${eventsPath}\n rerun pitchsnap.mjs to generate it.`); + process.exit(1); +} +const SAMPLE_RATE = 48_000; +const F_MIN = Number(flags["f-min"]) || 80; +const F_MAX = Number(flags["f-max"]) || 600; + +// ── helpers ─────────────────────────────────────────────────────────── +function freqToMidi(f) { return 69 + 12 * Math.log2(f / 440); } +function midiToName(midi) { + const names = ["C","C#","D","Eb","E","F","F#","G","G#","A","Bb","B"]; + const r = Math.round(midi); + return `${names[((r % 12) + 12) % 12]}${Math.floor(r / 12) - 1}`; +} + +function readWav(path) { + const buf = readFileSync(path); + let i = 12; + while (i < buf.length - 8) { + const id = buf.toString("ascii", i, i + 4); + const size = buf.readUInt32LE(i + 4); + if (id === "data") { + i += 8; + const samples = new Float32Array(size / 2); + for (let j = 0; j < samples.length; j++) { + samples[j] = buf.readInt16LE(i + j * 2) / 32768; + } + return samples; + } + i += 8 + size; + } + throw new Error(`no data chunk in ${path}`); +} + +// Autocorrelation pitch detection. Naive but works for clean voice. +// Skip first/last 20% of samples (attack/release transients). +function detectPitch(samples, sr, fmin, fmax) { + if (samples.length < sr * 0.05) return null; // < 50ms — too short + const start = Math.floor(samples.length * 0.2); + const end = Math.floor(samples.length * 0.8); + const win = samples.slice(start, end); + + // RMS gate — skip silence + let rms = 0; + for (let i = 0; i < win.length; i++) rms += win[i] * win[i]; + rms = Math.sqrt(rms / win.length); + if (rms < 0.005) return null; + + const lagMin = Math.floor(sr / fmax); + const lagMax = Math.min(Math.floor(sr / fmin), Math.floor(win.length / 2)); + + let bestLag = lagMin; + let bestScore = -Infinity; + for (let lag = lagMin; lag <= lagMax; lag++) { + let sum = 0; + let n = win.length - lag; + for (let i = 0; i < n; i++) sum += win[i] * win[i + lag]; + sum /= n; + if (sum > bestScore) { bestScore = sum; bestLag = lag; } + } + + // Parabolic interpolation around peak for sub-sample precision + let lagF = bestLag; + if (bestLag > lagMin && bestLag < lagMax) { + const acAt = (k) => { + let s = 0; + const n = win.length - k; + for (let i = 0; i < n; i++) s += win[i] * win[i + k]; + return s / n; + }; + const a = acAt(bestLag - 1); + const b = acAt(bestLag); + const c = acAt(bestLag + 1); + const denom = a - 2 * b + c; + if (Math.abs(denom) > 1e-9) lagF = bestLag - 0.5 * (c - a) / denom; + } + + return sr / lagF; +} + +// ── main ────────────────────────────────────────────────────────────── +const events = JSON.parse(readFileSync(eventsPath, "utf8")); + +const tmpDir = `${dirname(vocalPath)}/.pitchcheck-tmp`; +rmSync(tmpDir, { recursive: true, force: true }); +mkdirSync(tmpDir, { recursive: true }); + +console.log( + `→ pitchcheck · ${events.events.length} events against ${basename(eventsPath)}\n` + + ` vocal=${basename(vocalPath)} · stretch=${events.stretch}× curve=${events.curve}\n` +); +console.log(` ${"i".padStart(3)} ${"word".padEnd(12)} ${"expected".padEnd(14)} ${"measured".padEnd(20)} drift`); +console.log(` ${"─".repeat(60)}`); + +let drifts = []; +let confidentCount = 0; + +for (const ev of events.events) { + const startSec = ev.snappedStart; + const endSec = startSec + ev.durSec; + + const sliceWav = `${tmpDir}/w${ev.i.toString().padStart(3,"0")}.wav`; + spawnSync("ffmpeg", + ["-hide_banner","-y","-loglevel","error", + "-ss",startSec.toFixed(4),"-to",endSec.toFixed(4), + "-i",vocalPath, + "-c:a","pcm_s16le","-ar",String(SAMPLE_RATE),"-ac","1",sliceWav], + { stdio: ["ignore","ignore","ignore"] }); + if (!existsSync(sliceWav)) continue; + + const samples = readWav(sliceWav); + const f0 = detectPitch(samples, SAMPLE_RATE, F_MIN, F_MAX); + + if (f0 === null) { + console.log(` ${ev.i.toString().padStart(3)} ${ev.text.padEnd(12)} ${ev.targetNote.padEnd(14)} ${"(silence)".padEnd(20)}`); + continue; + } + + const measuredMidi = freqToMidi(f0); + const measuredName = midiToName(measuredMidi); + const driftCents = (measuredMidi - ev.targetMidi) * 100; + drifts.push(driftCents); + confidentCount++; + + const driftStr = `${driftCents >= 0 ? "+" : ""}${driftCents.toFixed(0)}¢`; + const measuredStr = `${measuredName} (${f0.toFixed(1)}Hz)`; + console.log( + ` ${ev.i.toString().padStart(3)} ${ev.text.padEnd(12)} ${ev.targetNote.padEnd(14)} ${measuredStr.padEnd(20)} ${driftStr}` + ); +} + +rmSync(tmpDir, { recursive: true, force: true }); + +if (confidentCount === 0) { + console.log("\n no confident measurements — too much silence or noise"); + process.exit(0); +} + +drifts.sort((a, b) => Math.abs(a) - Math.abs(b)); +const median = Math.abs(drifts[Math.floor(drifts.length / 2)]); +const mean = drifts.reduce((a, b) => a + Math.abs(b), 0) / drifts.length; +const max = Math.max(...drifts.map(Math.abs)); + +console.log(`\n summary · ${confidentCount}/${events.events.length} measured`); +console.log(` median |drift| = ${median.toFixed(0)}¢`); +console.log(` mean |drift| = ${mean.toFixed(0)}¢`); +console.log(` max |drift| = ${max.toFixed(0)}¢`); +console.log(`\n reference: ±50¢ = within a quarter-tone, ±25¢ = "in tune"`); + +// ── Stutter detection ────────────────────────────────────────────────── +// Look for amplitude dips (RMS drops > 70% within 50ms then recovers +// within 80ms) and f0 jumps (frame-to-frame f0 ratio > 1.5 = >7 +// semitones in 5ms). Both indicate WORLD phase resets / vocal skips. +{ + const fullSliceWav = `/tmp/pitchcheck-stutter-${Date.now()}.wav`; + spawnSync("ffmpeg", + ["-hide_banner", "-y", "-loglevel", "error", + "-i", vocalPath, + "-c:a", "pcm_s16le", "-ar", String(SAMPLE_RATE), "-ac", "1", fullSliceWav], + { stdio: ["ignore", "ignore", "ignore"] }); + if (existsSync(fullSliceWav)) { + const samples = readWav(fullSliceWav); + const hop = Math.floor(0.010 * SAMPLE_RATE); + const nF = Math.floor(samples.length / hop); + const rms = new Float32Array(nF); + for (let f = 0; f < nF; f++) { + let r = 0; + for (let j = 0; j < hop; j++) { + const v = samples[f * hop + j]; + r += v * v; + } + rms[f] = Math.sqrt(r / hop); + } + // Smooth RMS with a 3-frame rolling mean for stability + const sm = new Float32Array(nF); + for (let f = 0; f < nF; f++) { + let s = 0, c = 0; + for (let k = -1; k <= 1; k++) { + if (f + k >= 0 && f + k < nF) { s += rms[f + k]; c++; } + } + sm[f] = s / c; + } + const peak = sm.reduce((m, v) => v > m ? v : m, 0); + // Stutter = dip below 30% of peak that's surrounded by content > 60% + const dipThr = peak * 0.30; + const surroundThr = peak * 0.60; + const stutters = []; + for (let f = 5; f < nF - 5; f++) { + if (sm[f] < dipThr) { + // Check if surrounded by content + let preMax = 0, postMax = 0; + for (let k = 1; k <= 5; k++) { + if (sm[f - k] > preMax) preMax = sm[f - k]; + if (sm[f + k] > postMax) postMax = sm[f + k]; + } + if (preMax > surroundThr && postMax > surroundThr) { + // It's a dip — check if it's a real stutter (recovers within 80ms) + let recoveredBy = 8; + for (let k = 1; k <= 8; k++) { + if (f + k < nF && sm[f + k] > surroundThr) { recoveredBy = k; break; } + } + stutters.push({ time: f * 0.010, dipDepth: 1 - sm[f] / preMax, recoveryFrames: recoveredBy }); + // skip ahead past this dip + f += recoveredBy; + } + } + } + if (stutters.length === 0) { + console.log(`\n stutters: none detected ✓`); + } else { + console.log(`\n stutters: ${stutters.length} amplitude dip${stutters.length === 1 ? "" : "s"} flagged`); + for (const s of stutters.slice(0, 12)) { + console.log(` ${s.time.toFixed(2)}s depth ${(s.dipDepth * 100).toFixed(0)}% recover ${s.recoveryFrames * 10}ms`); + } + if (stutters.length > 12) console.log(` ... and ${stutters.length - 12} more`); + } + } +} diff --git a/pop/bin/pitchsnap.mjs b/pop/bin/pitchsnap.mjs new file mode 100644 --- /dev/null +++ b/pop/bin/pitchsnap.mjs @@ -0,0 +1,1017 @@ +#!/usr/bin/env node +// pitchsnap.mjs — aggressive per-word post-prod, no elongation. +// +// For each whisper-aligned word: +// 1. Snap its START to the nearest 16th-note slot at the target BPM +// (no time-stretch — word duration stays natural) +// 2. Pitch-shift to the target note from the .np score (proportional +// word→syllable mapping, formant-preserving via rubberband) +// 3. Place the pitched slice into a fresh buffer at the snapped start +// +// Inter-word gaps end up shifted slightly (sometimes longer, sometimes +// shorter) — that's the snap. Words themselves keep natural speech +// rate, so jeffrey-pvc doesn't sound rushed; only their *placement* +// quantizes to the grid. +// +// Stretch ("lazy" mode): pass `--stretch FACTOR` to time-stretch every +// word by FACTOR using rubberband (formant + pitch preserving), then +// re-snap the stretched starts to the grid. 1.0 = natural, 1.5 = lazy +// (50% longer per word), 2.0 = drone. Total track duration grows. +// +// Usage: +// node bin/pitchsnap.mjs --vocal big-pictures/out/ac-vocal.mp3 \ +// --score big-pictures/plork.np --section hook \ +// --bpm 140 --grid 16 --ref-note C3 \ +// --stretch 1.4 \ +// --out big-pictures/out/ac-snapped-pitched.mp3 + +import { spawnSync } from "node:child_process"; +import { existsSync, mkdirSync, readFileSync, writeFileSync, rmSync } from "node:fs"; +import { resolve, dirname, basename } from "node:path"; +import { fileURLToPath } from "node:url"; + +const HERE = dirname(fileURLToPath(import.meta.url)); +const POP_ROOT = resolve(HERE, ".."); + +function parseArgs(argv) { + const flags = {}; + for (let i = 0; i < argv.length; i++) { + const a = argv[i]; + if (!a.startsWith("--")) continue; + const k = a.slice(2); + const next = argv[i + 1]; + if (next !== undefined && !next.startsWith("--")) { flags[k] = next; i++; } + else flags[k] = true; + } + return flags; +} + +const flags = parseArgs(process.argv.slice(2)); + +const vocalPath = resolve(process.cwd(), flags.vocal || ""); +if (!existsSync(vocalPath)) { + console.error("usage: --vocal --score [--section hook] [--bpm 140] [--grid 16] [--ref-note C3] [--out path.mp3]"); + process.exit(1); +} +const wordsPath = resolve(process.cwd(), flags.words || vocalPath.replace(/\.mp3$/, "-words.json")); +if (!existsSync(wordsPath)) { + console.error(`✗ words.json not found at ${wordsPath}. run bin/align.mjs first.`); + process.exit(1); +} +const scorePath = resolve(process.cwd(), flags.score || ""); +if (!existsSync(scorePath)) { + console.error(`✗ --score file required (path to .np)`); + process.exit(1); +} +const SECTION = (flags.section || "hook").toLowerCase(); +const BPM = Number(flags.bpm) || 140; +const GRID = Number(flags.grid) || 16; // 16 = sixteenth notes per bar +const REF_NOTE = flags["ref-note"] || "C3"; +const STRETCH = Number(flags.stretch) || 1.0; // 1.0 = natural, >1 = lazier +const CURVE = flags.curve || "flat"; // "flat" | "linear" | "bezier" +// Engine: "rubberband" (default — segmented per-syllable rubberband +// shifts, preserves source prosody shifted up/down) or "world" (calls +// pitchsnap_world.py for WORLD-vocoder f0 replacement, fully clamps +// pitch to target with vocal timbre intact). World requires the .venv +// at pop/.venv with pyworld + soundfile installed. +const ENGINE = flags.engine || "rubberband"; +const RETAIN = Number(flags.retain ?? 1.0); // world only: 0 = source, 1 = clamp +const VIBRATO_HZ = Number(flags["vibrato-hz"] ?? 0); +const VIBRATO_CENTS = Number(flags["vibrato-cents"] ?? 0); +// Transpose every target note by N semitones at runtime (no need to +// rewrite the .np). Useful when the score is in a register too high +// for the source voice (jeffrey-pvc baritone ≈ C3, so kid-songs at +// C4-E4 sound chipmunky — try --transpose -12 to drop them an octave). +const TRANSPOSE = Number(flags.transpose ?? 0); +// Beat mode: interpret syllable `*weight` as BEATS at the given BPM, +// not as relative multipliers. Each word's stretch becomes +// (sum(beats) * 60/BPM) / naturalWordDuration. Required for real +// song timing — speech speeds rarely match the song's meter. +const BEAT_MODE = flags["beat-mode"] === true; +// Detect syllable boundaries within each word via librosa onset +// detection (in pitchsnap_world.py). For multi-syllable words this +// snaps the per-syllable pitch targets to natural energy peaks in +// the audio rather than weighted-proportional splits. +const DETECT_BOUNDARIES = flags["detect-boundaries"] === true; +// Scale walk: comma-separated notes (e.g. "C3,D3,Eb3,F3,G3,Ab3,Bb3,C4") +// that override the .np score's syllable pitches. The full scale is +// laid across each word's duration as evenly-spaced pitchmap +// waypoints — useful for single-word melody experiments where you +// want a slur through more notes than the word has syllables. +const SCALE_WALK = flags["scale-walk"] + ? String(flags["scale-walk"]).split(",").map((s) => s.trim()).filter(Boolean) + : null; +// autotune: off | "global" (median source pitch — uniform shifts, robust) +// | "word" (per-word source — accurate but octave-error-prone). +// Default "global" because autocorrelation f0-detection occasionally +// finds 2× or ½× on individual words, causing audible octave jumps. +const AUTOTUNE = flags.autotune === true ? "global" + : flags.autotune === false ? "off" + : (flags.autotune || "off"); +const SAMPLE_RATE = 48_000; +const OUT_PATH = flags.out + ? resolve(process.cwd(), flags.out) + : vocalPath.replace(/\.mp3$/, "-snapped-pitched.mp3"); + +// ── helpers ─────────────────────────────────────────────────────────── +const NOTE_TO_SEMI = { c: 0, d: 2, e: 4, f: 5, g: 7, a: 9, b: 11 }; +function noteToMidi(p) { + const m = p.trim().toLowerCase().match(/^([a-g])([#b]?)(-?\d+)$/); + if (!m) throw new Error(`bad note: ${p}`); + let semi = NOTE_TO_SEMI[m[1]]; + if (m[2] === "#") semi += 1; + if (m[2] === "b") semi -= 1; + const oct = parseInt(m[3], 10); + return 12 * (oct + 1) + semi; +} + +function parseNp(text) { + const sections = {}; + let current = null; + for (const raw of text.split("\n")) { + const line = raw.trim(); + if (!line || line.startsWith("#")) continue; + if (!line.includes(":") && /^[a-z][a-z0-9 ]*$/.test(line)) { + current = line.toLowerCase(); + if (!sections[current]) sections[current] = []; + continue; + } + if (!current) { current = "default"; sections[current] = []; } + const tokens = line.split(/\s+/).filter(Boolean); + for (const tok of tokens) { + // Note: letter + optional sharp/flat + optional octave digit + ":" + syllable + // Token grammar: :[*] + // Weight is a relative duration multiplier (default 1). E.g. + // `G3:-ma-*2` makes "ma" twice as long as a default syllable + // when distributing time across the parent word. + const m = tok.match(/^([A-Ga-g][#b]?\d?):(.+?)(?:\*([\d.]+))?$/); + if (!m) continue; + const note = m[1].charAt(0).toUpperCase() + m[1].slice(1); + const weight = m[3] ? Number(m[3]) : 1; + sections[current].push({ pitch: note, syl: m[2], weight }); + } + } + return sections; +} + +function probeDuration(p) { + const r = spawnSync( + "ffprobe", + ["-v", "error", "-show_entries", "format=duration", + "-of", "default=noprint_wrappers=1:nokey=1", p], + { encoding: "utf8" } + ); + return Number(r.stdout.trim()); +} + +// Heuristic syllable count for an English word — counts vowel groups, +// drops trailing silent 'e'. Good enough for lyric-to-score alignment; +// failure modes (e.g. "fire" → 1 vs. true 2) drift by ~1 syllable per +// word but the proportional mapping recovers across line lengths. +function syllableCount(word) { + const w = word.toLowerCase().replace(/[^a-z]/g, ""); + if (!w) return 1; + const groups = w.match(/[aeiouy]+/g) || []; + let count = groups.length; + // Silent 'ed' suffix in past-tense verbs (saved, called, loved). + // Exception: "ed" preceded by t/d IS pronounced (wanted, rated). + if (w.endsWith("ed") && w.length > 2 && count > 1) { + const beforeEd = w.charAt(w.length - 3); + if (beforeEd !== "t" && beforeEd !== "d") count--; + } + // Silent trailing 'e' (rake, love, since, more) — but not after the + // ed-rule has already fired. + else if (w.endsWith("e") && count > 1) count--; + // 'le' creates an extra syllable when preceded by a consonant + // (apple, little, bottle). After silent-e adjustment. + if (w.endsWith("le") && w.length > 2 && !"aeiouy".includes(w.charAt(w.length - 3))) count++; + return Math.max(1, count); +} + +// Autocorrelation pitch detection — same algorithm as pitchcheck.mjs. +// Restricted to voice range, uses central window, parabolic interp. +function detectPitch(samples, sr, fmin = 80, fmax = 600) { + if (samples.length < sr * 0.05) return null; + const start = Math.floor(samples.length * 0.2); + const end = Math.floor(samples.length * 0.8); + const win = samples.slice(start, end); + let rms = 0; + for (let i = 0; i < win.length; i++) rms += win[i] * win[i]; + rms = Math.sqrt(rms / win.length); + if (rms < 0.005) return null; + const lagMin = Math.floor(sr / fmax); + const lagMax = Math.min(Math.floor(sr / fmin), Math.floor(win.length / 2)); + let bestLag = lagMin, bestScore = -Infinity; + for (let lag = lagMin; lag <= lagMax; lag++) { + let sum = 0; + const n = win.length - lag; + for (let i = 0; i < n; i++) sum += win[i] * win[i + lag]; + sum /= n; + if (sum > bestScore) { bestScore = sum; bestLag = lag; } + } + let lagF = bestLag; + if (bestLag > lagMin && bestLag < lagMax) { + const acAt = (k) => { + let s = 0; + const n = win.length - k; + for (let i = 0; i < n; i++) s += win[i] * win[i + k]; + return s / n; + }; + const a = acAt(bestLag - 1), b = acAt(bestLag), c = acAt(bestLag + 1); + const denom = a - 2 * b + c; + if (Math.abs(denom) > 1e-9) lagF = bestLag - 0.5 * (c - a) / denom; + } + return sr / lagF; +} +function freqToMidi(f) { return 69 + 12 * Math.log2(f / 440); } + +function readWav(path) { + // Read a 16-bit PCM mono wav written by ffmpeg into a Float32Array + // of normalized samples. Skips RIFF/fmt headers via the data chunk. + const buf = readFileSync(path); + // Find 'data' chunk + let i = 12; + while (i < buf.length - 8) { + const id = buf.toString("ascii", i, i + 4); + const size = buf.readUInt32LE(i + 4); + if (id === "data") { + i += 8; + const samples = new Float32Array(size / 2); + for (let j = 0; j < samples.length; j++) { + samples[j] = buf.readInt16LE(i + j * 2) / 32768; + } + return samples; + } + i += 8 + size; + } + throw new Error(`no data chunk in ${path}`); +} + +// ── load inputs ─────────────────────────────────────────────────────── +const words = JSON.parse(readFileSync(wordsPath, "utf8")); +const score = parseNp(readFileSync(scorePath, "utf8")); +const syllables = score[SECTION]; +if (!syllables || !syllables.length) { + console.error(`✗ section '${SECTION}' empty in ${scorePath}`); + process.exit(1); +} + +const refMidi = noteToMidi(REF_NOTE); +const beatSec = 60 / BPM; +const stepSec = beatSec * 4 / GRID; // 16th = beatSec / 4 + +const totalNaturalDur = probeDuration(vocalPath); + +const tmpDir = `${dirname(OUT_PATH)}/.${basename(OUT_PATH).replace(/\..*$/, "")}-ps-tmp`; +rmSync(tmpDir, { recursive: true, force: true }); +mkdirSync(tmpDir, { recursive: true }); + +// ── Pre-pass: measure source pitch across all words for "global" autotune ── +// Slice every word once with ffmpeg, run autocorrelation, take median. +// One pre-pass means the per-word loop below uses a stable reference. +let globalSourceMidi = null; +if (AUTOTUNE === "global") { + const detected = []; + for (let i = 0; i < words.length; i++) { + const w = words[i]; + const startSec = w.fromMs / 1000; + const endSec = i < words.length - 1 ? words[i + 1].fromMs / 1000 : totalNaturalDur; + if (startSec >= totalNaturalDur - 0.005) continue; + const safeEnd = Math.min(endSec, totalNaturalDur); + const probeWav = `${tmpDir}/probe${i.toString().padStart(3, "0")}.wav`; + spawnSync( + "ffmpeg", + ["-hide_banner", "-y", "-loglevel", "error", + "-ss", startSec.toFixed(4), "-to", safeEnd.toFixed(4), + "-i", vocalPath, + "-c:a", "pcm_s16le", "-ar", String(SAMPLE_RATE), "-ac", "1", probeWav], + { stdio: ["ignore", "ignore", "ignore"] } + ); + if (!existsSync(probeWav)) continue; + const samples = readWav(probeWav); + const f = detectPitch(samples, SAMPLE_RATE, 80, 280); + if (f !== null) detected.push(freqToMidi(f)); + } + if (detected.length) { + detected.sort((a, b) => a - b); + globalSourceMidi = detected[Math.floor(detected.length / 2)]; + console.log(` global source pitch: median MIDI ${globalSourceMidi.toFixed(2)} from ${detected.length} words`); + } else { + console.warn(" ! global autotune: no pitch detections — falling back to ref-relative shifts"); + } +} + +console.log( + `→ pitchsnap · ${words.length} whisper words → ${syllables.length} score syllables\n` + + ` bpm=${BPM} grid=1/${GRID}-note (${(stepSec * 1000).toFixed(1)}ms) ref=${REF_NOTE} ` + + `stretch=${STRETCH.toFixed(2)}× curve=${CURVE} autotune=${AUTOTUNE}` +); + +// ── per-word: extract slice → pitch shift → record snapped start ───── +const slices = []; // { snappedStart, samples, text } +let maxEndSec = 0; +let sylCursor = 0; // syllable-aware mapping: advance by each word's syllable count + +// Beat-mode pre-pass: walk the score syllable cursor exactly the same +// way the per-word loop will, so we can produce a beats-cumulative +// start time per word. Without this, words begin at speech-time and +// individually-stretched durations cause overlap. +let beatStarts = null; +if (BEAT_MODE) { + beatStarts = []; + let beats = 0; + let cur = 0; + const beatSec = 60 / BPM; + for (let i = 0; i < words.length; i++) { + beatStarts.push(beats * beatSec); + const ws = syllableCount(words[i].text); + let wordBeats = 0; + for (let k = cur; k < cur + ws && k < syllables.length; k++) { + const sw = syllables[k] && typeof syllables[k].weight === "number" + ? syllables[k].weight : 1; + wordBeats += sw; + } + beats += wordBeats; + cur += ws; + } + console.log(` beat-mode timeline: ${beats} beats total = ${(beats * beatSec).toFixed(2)}s @ ${BPM} BPM`); +} + +for (let i = 0; i < words.length; i++) { + const w = words[i]; + const naturalStart = w.fromMs / 1000; + const naturalEnd = i < words.length - 1 ? words[i + 1].fromMs / 1000 : totalNaturalDur; + + // Snap target: in BEAT_MODE, place each word at its cumulative beat + // position from the score. Otherwise preserve the speech timeline + // (scaled by global STRETCH and snapped to grid). + const snappedStart = beatStarts + ? beatStarts[i] + : Math.round((naturalStart * STRETCH) / stepSec) * stepSec; + + // Syllable-aware mapping: each word claims `syllableCount(word)` + // entries from the score, starting at sylCursor. Use the FIRST + // syllable's pitch as the word's main target; the LAST syllable's + // pitch as the next-target for curve glide. + const wordSyls = syllableCount(w.text); + const startSylIdx = Math.min(syllables.length - 1, sylCursor); + const endSylIdx = Math.min(syllables.length - 1, sylCursor + wordSyls - 1); + sylCursor += wordSyls; + + const syl = syllables[startSylIdx]; + const noteStr = /\d/.test(syl.pitch) ? syl.pitch : syl.pitch + "3"; + const targetMidi = noteToMidi(noteStr) + TRANSPOSE; + let semitones = targetMidi - refMidi; + + // Next-target for curve mode: end syllable of this word (intra-word + // glide for multi-syllable words) or the first syllable of the next word. + let nextSemitones = semitones; + if (endSylIdx > startSylIdx) { + // Multi-syllable word — glide to its own last syllable + const lastSyl = syllables[endSylIdx]; + const lastNoteStr = /\d/.test(lastSyl.pitch) ? lastSyl.pitch : lastSyl.pitch + "3"; + nextSemitones = noteToMidi(lastNoteStr) - refMidi; + } else if (sylCursor < syllables.length) { + // Single-syllable word — glide toward next word's first syllable + const nextSyl = syllables[Math.min(syllables.length - 1, sylCursor)]; + const nextNoteStr = /\d/.test(nextSyl.pitch) ? nextSyl.pitch : nextSyl.pitch + "3"; + nextSemitones = noteToMidi(nextNoteStr) - refMidi; + } + + // Guard: skip words whose start has gone past the natural duration. + if (naturalStart >= totalNaturalDur - 0.005) { + console.warn(` ! word ${i} (${w.text}) starts past audio end — skipping`); + continue; + } + // Extend each word's slice end by 60ms so trailing consonants / + // releases don't get cut. The next word will overlap slightly on + // playback (handled by additive mixing) — better than chopped tails. + const safeEnd = Math.min(naturalEnd + 0.06, totalNaturalDur); + const sliceWav = `${tmpDir}/w${i.toString().padStart(3, "0")}.wav`; + spawnSync( + "ffmpeg", + ["-hide_banner", "-y", "-loglevel", "error", + "-ss", naturalStart.toFixed(4), "-to", safeEnd.toFixed(4), + "-i", vocalPath, + "-c:a", "pcm_s16le", "-ar", String(SAMPLE_RATE), "-ac", "1", sliceWav], + { stdio: ["ignore", "ignore", "inherit"] } + ); + if (!existsSync(sliceWav)) { + console.warn(` ! word ${i} (${w.text}) ffmpeg slice failed — skipping`); + continue; + } + + // Trim leading/trailing silence within the slice so WORLD analyses + // only the actual word content. Whisper word boundaries can land + // mid-vowel of the previous word or mid-consonant of the current + // one; trimming on RMS-envelope at 5% of peak finds the real + // start/end. Adds 8ms of padding either side so we don't cut into + // attack transients. + { + const buf = readWav(sliceWav); + const hop = Math.floor(0.010 * SAMPLE_RATE); + const nFrames = Math.max(1, Math.floor(buf.length / hop)); + const env = new Float32Array(nFrames); + let peakE = 0; + for (let f = 0; f < nFrames; f++) { + let r = 0; + const a = f * hop; + const b = Math.min(buf.length, a + hop); + for (let j = a; j < b; j++) r += buf[j] * buf[j]; + env[f] = Math.sqrt(r / (b - a)); + if (env[f] > peakE) peakE = env[f]; + } + if (peakE > 0.005) { + const thr = peakE * 0.05; + let s = 0; while (s < nFrames && env[s] < thr) s++; + let e = nFrames - 1; while (e > s && env[e] < thr) e--; + // ATTACK DETECTION (only for short slot allocations). + // For words with allocated weight ≥ 3 beats (sustained notes + // like 'found', 'see' on *5), keep the natural ramp-in intact + // because rubberband needs that material to stretch into the + // long sustain. Aggressively trimming a slow-attack 5-beat note + // produces a clipped sustain — the word ends early in its slot + // and the listener hears silence. Short words (1-2 beats) get + // the full attack-detection trim so their attack lands on beat. + const wordBeats = syllables + .slice(startSylIdx, endSylIdx + 1) + .reduce((sum, s2) => sum + (s2.weight || 1), 0); + if (wordBeats <= 2) { + const lookaheadFrames = Math.min(25, e - s); // 250ms @ 10ms hop + let maxRise = 0; + let attackFrame = s; + for (let af = s + 1; af < s + lookaheadFrames; af++) { + const rise = env[af] - env[Math.max(s, af - 3)]; + if (rise > maxRise) { + maxRise = rise; + attackFrame = af; + } + } + if (attackFrame - s > 1) { + const preRollFrames = Math.max(0, Math.floor(0.015 / 0.010)); + s = Math.max(s, attackFrame - preRollFrames); + } + } + const pad = Math.floor(0.008 * SAMPLE_RATE); + const startSamp = Math.max(0, s * hop - pad); + const endSamp = Math.min(buf.length, (e + 1) * hop + pad); + if (endSamp > startSamp + Math.floor(0.030 * SAMPLE_RATE)) { + // Write trimmed back to sliceWav so downstream steps see the cleaned cut. + const trimmed = buf.slice(startSamp, endSamp); + const sampleBytes = Buffer.alloc(trimmed.length * 2); + for (let k = 0; k < trimmed.length; k++) { + const v = Math.max(-1, Math.min(1, trimmed[k])); + sampleBytes.writeInt16LE(Math.floor(v * 32767), k * 2); + } + // Minimal RIFF header for 48kHz mono 16-bit PCM + const dataLen = sampleBytes.length; + const header = Buffer.alloc(44); + header.write("RIFF", 0); header.writeUInt32LE(36 + dataLen, 4); + header.write("WAVE", 8); header.write("fmt ", 12); + header.writeUInt32LE(16, 16); header.writeUInt16LE(1, 20); + header.writeUInt16LE(1, 22); header.writeUInt32LE(SAMPLE_RATE, 24); + header.writeUInt32LE(SAMPLE_RATE * 2, 28); header.writeUInt16LE(2, 32); + header.writeUInt16LE(16, 34); header.write("data", 36); + header.writeUInt32LE(dataLen, 40); + writeFileSync(sliceWav, Buffer.concat([header, sampleBytes])); + } + } + } + + // Autotune: shift to land ON the target note. + // - global: shift by (target − globalSourceMidi). Uniform across + // all words, robust against per-word f0-detection octave errors. + // - word: shift by (target − measured per-word source). Most + // accurate when detection is clean, but susceptible to octave + // errors that produce audible jumps. + let sourceMidi = null; + if (AUTOTUNE === "global" && globalSourceMidi !== null) { + sourceMidi = globalSourceMidi; + semitones = targetMidi - sourceMidi; + } else if (AUTOTUNE === "word") { + const probeSamples = readWav(sliceWav); + const sourceF = detectPitch(probeSamples, SAMPLE_RATE, 80, 280); + if (sourceF !== null) { + sourceMidi = freqToMidi(sourceF); + semitones = targetMidi - sourceMidi; + } + // If pitch detection failed, fall back to ref-based shift. + } + + // Waypoints describe the pitch curve through this word; consumed by + // both the rubberband segmented rendering and the sine + tick overlay below. + let waypoints = []; + + // Segmented rendering — chops the source slice into N equal pieces + // and pitch-shifts each to its own target. Step pitch changes that + // are audibly distinct per segment. Bypasses rubberband's --pitchmap + // which produced ambiguous timing on this version. + // + // Notes for the segments come from one of: + // 1. SCALE_WALK (CLI override, walks notes regardless of syllables) + // 2. .np per-syllable pitches (multi-syllable word mapped to its + // syllable count's worth of score notes) + let segNotes = null; + let segWeights = null; + if (SCALE_WALK && SCALE_WALK.length >= 2) { + segNotes = SCALE_WALK.map((s) => /\d/.test(s) ? s : s + "3"); + segWeights = segNotes.map(() => 1); + } else if (CURVE !== "flat" && (endSylIdx - startSylIdx + 1) >= 1) { + // Populate for ALL words (1+ syllables) so single-syllable words + // also flow through the WORLD / beat-mode path. Previously gated + // on >= 2, which silently skipped half the lyric. + segNotes = []; + segWeights = []; + for (let k = startSylIdx; k <= endSylIdx; k++) { + const sylk = syllables[k]; + const baseNote = /\d/.test(sylk.pitch) ? sylk.pitch : sylk.pitch + "3"; + if (TRANSPOSE !== 0) { + const m = noteToMidi(baseNote) + TRANSPOSE; + const names = ["C","C#","D","Eb","E","F","F#","G","G#","A","Bb","B"]; + const oct = Math.floor(m / 12) - 1; + const idx = ((m % 12) + 12) % 12; + segNotes.push(`${names[idx]}${oct}`); + } else { + segNotes.push(baseNote); + } + segWeights.push(typeof sylk.weight === "number" ? sylk.weight : 1); + } + } + + // ── WORLD engine: replace f0 wholesale, no segmenting ──────────── + if (segNotes && segNotes.length >= 1 && ENGINE === "world") { + // Compute per-word stretch. + let perWordStretch = STRETCH; + if (BEAT_MODE && segWeights) { + const naturalWordDur = readWav(sliceWav).length / SAMPLE_RATE; + const beatSec = 60 / BPM; + const targetDur = segWeights.reduce((a, b) => a + b, 0) * beatSec; + if (naturalWordDur > 0.01) { + // Cap raised from 8× to 20× — ElevenLabs utterances of short + // words like "see" are ~250-350ms naturally and need to fill + // 4-5 beat sustain slots. With the old 8× ceiling, sustained + // notes ended early and the listener heard silence after. + perWordStretch = Math.max(0.5, Math.min(20.0, targetDur / naturalWordDur)); + } + console.log(` beat-mode · '${w.text}' ${segWeights.join("+")}b → ${targetDur.toFixed(2)}s @ ${perWordStretch.toFixed(2)}× stretch (nat=${naturalWordDur.toFixed(2)}s)`); + } + + // Pre-stretch with rubberband (formant-preserving) so the stretched + // wav is what WORLD analyses. This delivers per-word stretch on the + // world engine path (WORLD itself doesn't stretch). + let worldInputWav = sliceWav; + if (Math.abs(perWordStretch - 1.0) >= 0.001) { + worldInputWav = `${tmpDir}/w${i.toString().padStart(3,"0")}-stretched.wav`; + const rb = spawnSync( + "rubberband", + ["-t", String(perWordStretch), sliceWav, worldInputWav], + { stdio: ["ignore", "ignore", "ignore"] } + ); + if (rb.status !== 0 || !existsSync(worldInputWav)) worldInputWav = sliceWav; + } + const pieceWavWorld = `${tmpDir}/w${i.toString().padStart(3,"0")}-world.wav`; + const venvPython = resolve(POP_ROOT, ".venv/bin/python"); + const helperPath = resolve(POP_ROOT, "bin/pitchsnap_world.py"); + const args = [ + helperPath, worldInputWav, pieceWavWorld, + "--notes", segNotes.join(","), + "--retain", String(RETAIN), + ]; + if (segWeights && segWeights.some((w) => w !== 1)) { + args.push("--weights", segWeights.join(",")); + } + if (DETECT_BOUNDARIES) args.push("--detect-boundaries"); + if (VIBRATO_HZ > 0) { + args.push("--vibrato-hz", String(VIBRATO_HZ)); + args.push("--vibrato-cents", String(VIBRATO_CENTS)); + } + const r = spawnSync(venvPython, args, { stdio: ["ignore", "inherit", "inherit"] }); + if (r.status !== 0 || !existsSync(pieceWavWorld)) { + console.warn(` ! word ${i} world engine failed — falling back to rubberband`); + } else { + // Build waypoints (for sine overlay + events trace) + const refForShift = (AUTOTUNE === "global" && globalSourceMidi !== null) + ? globalSourceMidi : (sourceMidi !== null ? sourceMidi : refMidi); + const wlen = readWav(pieceWavWorld).length; + for (let k = 0; k < segNotes.length; k++) { + const midiK = noteToMidi(segNotes[k]); + waypoints.push({ + sample: Math.floor((k / segNotes.length) * wlen), + midi: midiK, + semi: midiK - refForShift, + }); + } + let samples = readWav(pieceWavWorld); + // Hard-trim each word to its beat allocation (BEAT_MODE) so + // stretched WORLD output doesn't bleed into the next word's slot + // and create overlap/echo. 30ms cosine fade-out at the trim point + // smooths the cut. Without this, words that ended up slightly + // longer than their beat target leaked tonal tail into following + // words → audible echo on the single voice. + if (BEAT_MODE && segWeights) { + const beatSec = 60 / BPM; + const allocatedSec = segWeights.reduce((a, b) => a + b, 0) * beatSec; + const allocatedSamples = Math.floor(allocatedSec * SAMPLE_RATE); + if (samples.length > allocatedSamples) { + const trimmed = samples.slice(0, allocatedSamples); + const fadeS = Math.min(Math.floor(0.030 * SAMPLE_RATE), Math.floor(trimmed.length / 8)); + for (let k = 0; k < fadeS; k++) { + const j = trimmed.length - fadeS + k; + const env = 0.5 + 0.5 * Math.cos((Math.PI * k) / fadeS); + trimmed[j] *= env; + } + samples = trimmed; + } + } + slices.push({ + snappedStart, samples, text: w.text, naturalStart, + semitones, noteStr, targetMidi, sourceMidi, waypoints, + }); + const endSec = snappedStart + samples.length / SAMPLE_RATE; + if (endSec > maxEndSec) maxEndSec = endSec; + continue; + } + } + + if (segNotes && segNotes.length >= 2) { + const naturalSamples = readWav(sliceWav).length; + const refForShift = (AUTOTUNE === "global" && globalSourceMidi !== null) + ? globalSourceMidi + : (sourceMidi !== null ? sourceMidi : refMidi); + + const N = segNotes.length; + // Weighted segment boundaries — per-syllable `*weight` from the .np + // score lets us hold longer notes (e.g. "a-MAAA-zing"). + const weights = (segWeights && segWeights.length === N) ? segWeights : segNotes.map(() => 1); + const wTotal = weights.reduce((a, b) => a + b, 0) || N; + const cumW = [0]; + for (let k = 0; k < N; k++) cumW.push(cumW[k] + weights[k] / wTotal); + + const segPieces = []; + for (let k = 0; k < N; k++) { + const noteAtK = segNotes[k]; + const midiAtK = noteToMidi(noteAtK); + const semiAtK = midiAtK - refForShift; + waypoints.push({ + sample: Math.floor(cumW[k] * naturalSamples * STRETCH), + midi: midiAtK, + semi: semiAtK, + }); + + // Slice source segment k by its weighted boundaries. + const segStartSec = cumW[k] * (naturalSamples / SAMPLE_RATE); + const segEndSec = cumW[k + 1] * (naturalSamples / SAMPLE_RATE); + const segSrcWav = `${tmpDir}/w${i.toString().padStart(3,"0")}-seg${k}.wav`; + spawnSync( + "ffmpeg", + ["-hide_banner", "-y", "-loglevel", "error", + "-ss", segStartSec.toFixed(4), "-to", segEndSec.toFixed(4), + "-i", sliceWav, + "-c:a", "pcm_s16le", "-ar", String(SAMPLE_RATE), "-ac", "1", segSrcWav], + { stdio: ["ignore", "ignore", "ignore"] } + ); + if (!existsSync(segSrcWav)) continue; + + // Pitch-shift + stretch this segment. + const segOutWav = `${tmpDir}/w${i.toString().padStart(3,"0")}-seg${k}-p.wav`; + const rb = spawnSync( + "rubberband", + ["-p", String(semiAtK), "-t", String(STRETCH), segSrcWav, segOutWav], + { stdio: ["ignore", "ignore", "ignore"] } + ); + if (rb.status === 0 && existsSync(segOutWav)) segPieces.push(segOutWav); + else segPieces.push(segSrcWav); + } + + // Concat segments with overlap-add crossfade (20ms) so segment + // boundaries don't click. Read each piece as samples, mix into a + // running buffer with the tail of segment k overlapping the head + // of segment k+1. + const xfade = Math.floor(0.020 * SAMPLE_RATE); + const segBufs = segPieces.map((p) => readWav(p)); + let totalLen = 0; + for (const b of segBufs) totalLen += b.length; + totalLen -= xfade * Math.max(0, segBufs.length - 1); + const walked = new Float32Array(Math.max(0, totalLen) + xfade); + let cursor = 0; + for (let k = 0; k < segBufs.length; k++) { + const b = segBufs[k]; + const startK = cursor; + for (let j = 0; j < b.length; j++) { + let env = 1; + if (k > 0 && j < xfade) env = j / xfade; + if (k < segBufs.length - 1 && j >= b.length - xfade) { + env = Math.min(env, (b.length - j) / xfade); + } + const dst = startK + j; + if (dst >= 0 && dst < walked.length) walked[dst] += b[j] * env; + } + cursor += b.length - xfade; + } + // Trim trailing zeros if we overcounted. + let endIdx = walked.length; + while (endIdx > 0 && Math.abs(walked[endIdx - 1]) < 1e-9) endIdx--; + const samples = walked.slice(0, endIdx); + slices.push({ + snappedStart, samples, text: w.text, naturalStart, + semitones, noteStr, targetMidi, sourceMidi, waypoints, + }); + const endSec = snappedStart + samples.length / SAMPLE_RATE; + if (endSec > maxEndSec) maxEndSec = endSec; + continue; // skip the rest of the per-word loop body for SCALE_WALK + } + + let pieceWav = sliceWav; + const needsPitch = Math.abs(semitones) >= 0.01 || (CURVE !== "flat" && Math.abs(nextSemitones - semitones) >= 0.01); + const needsStretch = Math.abs(STRETCH - 1.0) >= 0.001; + if (needsPitch || needsStretch) { + const target = `${tmpDir}/w${i.toString().padStart(3, "0")}-p.wav`; + const rbArgs = []; + + // (waypoints declared at the per-word scope below) + if (CURVE !== "flat") { + const sliceSamples = readWav(sliceWav).length; + const stretchedSamples = Math.floor(sliceSamples * STRETCH); + + const refForShift = (AUTOTUNE === "global" && globalSourceMidi !== null) + ? globalSourceMidi + : (sourceMidi !== null ? sourceMidi : refMidi); + + if (SCALE_WALK && SCALE_WALK.length >= 2) { + for (let k = 0; k < SCALE_WALK.length; k++) { + const noteAtK = /\d/.test(SCALE_WALK[k]) ? SCALE_WALK[k] : SCALE_WALK[k] + "3"; + const midiAtK = noteToMidi(noteAtK); + const sample = Math.floor((k / (SCALE_WALK.length - 1)) * stretchedSamples); + waypoints.push({ sample, midi: midiAtK, semi: midiAtK - refForShift }); + } + } else if (CURVE === "linear" && (endSylIdx - startSylIdx + 1) >= 2) { + const sylsInWord = endSylIdx - startSylIdx + 1; + for (let k = 0; k < sylsInWord; k++) { + const sylAtK = syllables[startSylIdx + k]; + const noteAtK = /\d/.test(sylAtK.pitch) ? sylAtK.pitch : sylAtK.pitch + "3"; + const midiAtK = noteToMidi(noteAtK); + const sample = Math.floor((k / (sylsInWord - 1)) * stretchedSamples); + waypoints.push({ sample, midi: midiAtK, semi: midiAtK - refForShift }); + } + } else if (CURVE === "linear") { + waypoints.push({ sample: 0, midi: targetMidi, semi: semitones }); + const nextMidi = (sylCursor < syllables.length) + ? noteToMidi(/\d/.test(syllables[Math.min(syllables.length - 1, sylCursor)].pitch) + ? syllables[Math.min(syllables.length - 1, sylCursor)].pitch + : syllables[Math.min(syllables.length - 1, sylCursor)].pitch + "3") + : targetMidi; + waypoints.push({ sample: stretchedSamples, midi: nextMidi, semi: nextSemitones }); + } else if (CURVE === "bezier") { + const fifthOffset = 7; + const midSemi = semitones + Math.sign(nextSemitones - semitones || 1) * (fifthOffset / 4); + waypoints.push({ sample: 0, midi: targetMidi, semi: semitones }); + waypoints.push({ sample: Math.floor(stretchedSamples / 2), midi: targetMidi + midSemi - semitones, semi: midSemi }); + waypoints.push({ sample: stretchedSamples, midi: targetMidi + nextSemitones - semitones, semi: nextSemitones }); + } + + const pitchmap = `${tmpDir}/w${i.toString().padStart(3, "0")}.pmap`; + writeFileSync(pitchmap, waypoints.map((w) => `${w.sample} ${w.semi.toFixed(3)}`).join("\n") + "\n"); + rbArgs.push("--pitchmap", pitchmap); + rbArgs.push("-t", String(STRETCH)); + } else { + if (needsPitch) rbArgs.push("-p", String(semitones)); + if (needsStretch) rbArgs.push("-t", String(STRETCH)); + } + + rbArgs.push(sliceWav, target); + const r = spawnSync("rubberband", rbArgs, { stdio: ["ignore", "ignore", "ignore"] }); + if (r.status === 0 && existsSync(target)) { + pieceWav = target; + } else { + // rubberband can fail on very short slices — fall back to natural. + console.warn(` ! word ${i} (${w.text}) rubberband fell back to natural slice`); + } + } + + const samples = readWav(pieceWav); + slices.push({ + snappedStart, samples, text: w.text, naturalStart, + semitones, noteStr, targetMidi, sourceMidi, + waypoints, // for sine overlay + events trace + }); + const endSec = snappedStart + samples.length / SAMPLE_RATE; + if (endSec > maxEndSec) maxEndSec = endSec; +} + +// ── assemble buffer ─────────────────────────────────────────────────── +const totalSamples = Math.ceil((maxEndSec + 0.5) * SAMPLE_RATE); +const out = new Float32Array(totalSamples); + +// Word-boundary fade — 5ms cosine in/out per word slice. Just enough +// to prevent splice clicks; longer fade-ins were softening attacks of +// already-trimmed words, defeating the perceptual-onset alignment in +// the per-word silence trim above. The LAST word doesn't fade out so +// the song resolves naturally on its final note. +const wordFadeS = Math.floor(0.005 * SAMPLE_RATE); +for (let sIdx = 0; sIdx < slices.length; sIdx++) { + const s = slices[sIdx]; + const isLast = sIdx === slices.length - 1; + const startIdx = Math.floor(s.snappedStart * SAMPLE_RATE); + const len = s.samples.length; + const fadeIn = Math.min(wordFadeS, Math.floor(len / 8)); + const fadeOut = isLast ? 0 : Math.min(wordFadeS, Math.floor(len / 8)); + const sustainEnd = len - fadeOut; + for (let i = 0; i < len; i++) { + const dst = startIdx + i; + if (dst < 0 || dst >= out.length) continue; + let env = 1; + if (i < fadeIn) env = 0.5 - 0.5 * Math.cos((Math.PI * i) / fadeIn); + else if (fadeOut > 0 && i >= sustainEnd) env = 0.5 - 0.5 * Math.cos((Math.PI * (len - i)) / fadeOut); + out[dst] += s.samples[i] * env; + } +} + +// Normalize to ~ -3 dBFS peak. +let peak = 0; +for (let i = 0; i < out.length; i++) { + const a = Math.abs(out[i]); + if (a > peak) peak = a; +} +if (peak > 0) { + const norm = 0.85 / peak; + for (let i = 0; i < out.length; i++) out[i] *= norm; +} + +// Write float32 raw + ffmpeg → mp3. +const rawPath = `${tmpDir}/out.f32.raw`; +const buf = Buffer.alloc(out.length * 4); +for (let i = 0; i < out.length; i++) buf.writeFloatLE(out[i], i * 4); +writeFileSync(rawPath, buf); + +const ff = spawnSync( + "ffmpeg", + ["-hide_banner", "-y", "-loglevel", "error", + "-f", "f32le", "-ar", String(SAMPLE_RATE), "-ac", "1", + "-i", rawPath, + "-c:a", "libmp3lame", "-q:a", "3", OUT_PATH], + { stdio: "inherit" } +); +if (ff.status !== 0) { + console.error("✗ ffmpeg encode failed"); + process.exit(1); +} + +// ── summary ─────────────────────────────────────────────────────────── +console.log(`\n ${"i".padStart(3)} ${"word".padEnd(14)} ${"natural→snapped".padEnd(20)} pitch`); +console.log(` ${"─".repeat(55)}`); +for (let i = 0; i < slices.length; i++) { + const s = slices[i]; + const arrow = s.semitones >= 0 ? "+" : ""; + const drift = s.snappedStart - s.naturalStart; + const driftMs = Math.round(drift * 1000); + const driftStr = `${s.naturalStart.toFixed(2)}→${s.snappedStart.toFixed(2)}s (${driftMs >= 0 ? "+" : ""}${driftMs}ms)`; + console.log( + ` ${i.toString().padStart(3)} ${s.text.padEnd(14)} ${driftStr.padEnd(20)} ${s.noteStr} ${arrow}${s.semitones} st` + ); +} + +// ── Optional: render a sine reference under the vocal ───────────────── +// Pure sine at each event's target pitch for the event's duration, +// gated by a gentle envelope. Mixed at -12 dB by default. Useful for +// auditing whether pitch shifts landed — you should hear the vocal +// "tracking" the sine. Set --sine-overlay 0 to mute (still emits the +// file), or --sine-overlay to override (e.g. 0.3 = -10 dB). +const SINE_OVERLAY = flags["sine-overlay"] !== undefined; +if (SINE_OVERLAY) { + const sineGain = flags["sine-overlay"] === true ? 0.25 : Number(flags["sine-overlay"]); + const sineBuf = new Float32Array(totalSamples); + const twoPiOverSr = (2 * Math.PI) / SAMPLE_RATE; + const ATTACK_SEC = 0.015; + const RELEASE_SEC = 0.060; + const sineXfade = Math.floor(0.030 * SAMPLE_RATE); // 30ms + for (const s of slices) { + const startIdx = Math.floor(s.snappedStart * SAMPLE_RATE); + const len = s.samples.length; + + const wps = (s.waypoints && s.waypoints.length >= 2) + ? s.waypoints.map((w) => ({ pos: Math.min(len, w.sample), midi: w.midi })) + : [{ pos: 0, midi: s.targetMidi }, { pos: len, midi: s.targetMidi }]; + + // Sine reference — phase-continuous, with linear pitch crossfade + // in a 30ms window centered on each waypoint boundary. The pitch + // glides smoothly between adjacent held notes so there are no + // glitches/clicks. Outer attack/release envelope applies only at + // the very start/end of the whole word. + let phase = 0; + let segIdx = 0; + const wordAtt = Math.floor(ATTACK_SEC * SAMPLE_RATE); + const wordRel = Math.floor(RELEASE_SEC * SAMPLE_RATE); + const wordSustainEnd = len - wordRel; + for (let i = 0; i < len; i++) { + const dst = startIdx + i; + if (dst < 0 || dst >= sineBuf.length) continue; + while (segIdx + 1 < wps.length && i >= wps[segIdx + 1].pos) segIdx++; + const cur = wps[segIdx]; + // Determine pitch with optional crossfade near the next waypoint. + let midiNow = cur.midi; + const nxt = wps[segIdx + 1]; + if (nxt) { + const distToNext = nxt.pos - i; + if (distToNext < sineXfade) { + const t = 1 - distToNext / sineXfade; + midiNow = cur.midi * (1 - t) + nxt.midi * t; + } + } + const freq = 440 * Math.pow(2, (midiNow - 69) / 12); + phase += twoPiOverSr * freq; + + let env; + if (i < wordAtt) env = i / wordAtt; + else if (i >= wordSustainEnd) env = Math.max(0, (len - i) / wordRel); + else env = 1; + sineBuf[dst] += Math.sin(phase) * env * sineGain; + } + + // Ticks at each waypoint position so the listener can HEAR the + // pitch grid changing. Loud and unambiguous — 1 kHz tone, 50 ms, + // sharp exponential decay. Amplitude clipped to 0.95 so it always + // pokes through the mix. + const tickLen = Math.floor(0.050 * SAMPLE_RATE); + for (const wp of wps) { + const tickStart = startIdx + wp.pos; + for (let j = 0; j < tickLen; j++) { + const dst = tickStart + j; + if (dst < 0 || dst >= sineBuf.length) continue; + const env = Math.exp(-j / (tickLen * 0.20)); + const tone = Math.sin(2 * Math.PI * 1000 * j / SAMPLE_RATE); + sineBuf[dst] += tone * env * 0.95; + } + } + } + + // Diagnostic: write the sine + tick layer alone so it can be + // auditioned without the vocal masking it. + const refOut = OUT_PATH.replace(/\.mp3$/, "-ref.mp3"); + const refRaw = `${tmpDir}/ref.f32.raw`; + let refPeak = 0; + for (let i = 0; i < sineBuf.length; i++) { + const a = Math.abs(sineBuf[i]); + if (a > refPeak) refPeak = a; + } + const refScale = refPeak > 0 ? 0.9 / refPeak : 1; + const refBuf = Buffer.alloc(sineBuf.length * 4); + for (let i = 0; i < sineBuf.length; i++) refBuf.writeFloatLE(sineBuf[i] * refScale, i * 4); + writeFileSync(refRaw, refBuf); + spawnSync( + "ffmpeg", + ["-hide_banner", "-y", "-loglevel", "error", + "-f", "f32le", "-ar", String(SAMPLE_RATE), "-ac", "1", + "-i", refRaw, + "-c:a", "libmp3lame", "-q:a", "3", refOut], + { stdio: "inherit" } + ); + console.log(` sine+tick reference: ${refOut}`); + // Mix sineBuf into out (additively). + for (let i = 0; i < out.length && i < sineBuf.length; i++) out[i] += sineBuf[i]; + // Re-normalize because we just added energy. + let p2 = 0; + for (let i = 0; i < out.length; i++) { + const a = Math.abs(out[i]); + if (a > p2) p2 = a; + } + if (p2 > 0) { + const norm = 0.85 / p2; + for (let i = 0; i < out.length; i++) out[i] *= norm; + } + console.log(` sine overlay: enabled at gain ${sineGain.toFixed(2)}`); +} + +// Emit an events file alongside the mp3 so pitchcheck.mjs and other +// downstream tools can compare measured pitch vs intended target +// without re-aligning the rendered output (whisper degrades on +// heavily-shifted audio). +const eventsPath = OUT_PATH.replace(/\.mp3$/, ".events.json"); +const eventsObj = { + source: vocalPath, + score: scorePath, + section: SECTION, + bpm: BPM, + grid: GRID, + refNote: REF_NOTE, + refMidi, + stretch: STRETCH, + curve: CURVE, + totalDur: maxEndSec, + events: slices.map((s, i) => ({ + i, + text: s.text, + naturalStart: s.naturalStart, + snappedStart: s.snappedStart, + durSec: s.samples.length / SAMPLE_RATE, + targetNote: s.noteStr, + targetMidi: s.targetMidi, + targetFreq: 440 * Math.pow(2, (s.targetMidi - 69) / 12), + sourceMidi: s.sourceMidi, + semitones: s.semitones, + })), +}; +writeFileSync(eventsPath, JSON.stringify(eventsObj, null, 2)); + +rmSync(tmpDir, { recursive: true, force: true }); +console.log(`\n✓ ${OUT_PATH} · ${maxEndSec.toFixed(2)}s`); +console.log(` events: ${eventsPath}`); diff --git a/pop/bin/pitchsnap_world.py b/pop/bin/pitchsnap_world.py new file mode 100644 --- /dev/null +++ b/pop/bin/pitchsnap_world.py @@ -0,0 +1,276 @@ +#!/usr/bin/env python3 +""" +pitchsnap_world.py — WORLD-vocoder-based pitch correction helper for +pitchsnap.mjs. Replaces the f0 curve of a vocal slice with a target +melody (one or more notes laid across the duration) while preserving +the spectral envelope and aperiodicity — i.e. jeffrey's voice +character is unchanged but the pitch lands on the score. + +Pipeline (Saitou 2007 / Morise et al. 2016): + audio + → harvest (f0 candidates) + → stonemask (f0 refinement) + → cheaptrick (spectral envelope = formants = voice identity) + → d4c (aperiodicity = breath / consonants) + → REPLACE f0 with target curve + → synthesize + +Multi-syllable: --notes "C3,D3,Eb3,F3,G3" lays N evenly-spaced +target pitches across the audio. Each segment holds steady; transitions +crossfade in log-pitch space (one frame). + +Usage: + pitchsnap_world.py \ + --notes "C3,Eb3,G3" [--retain 1.0] [--vibrato-hz 5.5] \ + [--vibrato-cents 0] [--xfade-ms 30] +""" +import argparse +import sys +import numpy as np +import soundfile as sf +import pyworld as pw + +try: + import librosa + HAS_LIBROSA = True +except ImportError: + HAS_LIBROSA = False + +NOTE_SEMI = {"c": 0, "d": 2, "e": 4, "f": 5, "g": 7, "a": 9, "b": 11} + +def note_to_midi(s: str) -> int: + s = s.strip().lower() + semi = NOTE_SEMI[s[0]] + i = 1 + if i < len(s) and s[i] in "#b": + semi += 1 if s[i] == "#" else -1 + i += 1 + octave = int(s[i:]) + return 12 * (octave + 1) + semi + +def midi_to_hz(m: float) -> float: + return 440.0 * (2.0 ** ((m - 69.0) / 12.0)) + +def main() -> int: + p = argparse.ArgumentParser() + p.add_argument("in_wav") + p.add_argument("out_wav") + p.add_argument("--notes", required=True, + help='Comma-separated notes laid evenly across the audio, e.g. "C3,Eb3,G3"') + p.add_argument("--weights", default=None, + help='Comma-separated relative duration weights per note (default 1 each)') + p.add_argument("--note-starts", default=None, + help='Comma-separated absolute start times (seconds) per note. ' + 'When provided, target curve is anchored at these times ' + '(useful for whole-audio continuous resynth).') + p.add_argument("--retain", type=float, default=1.0, + help="0 = source f0 unchanged, 1 = full clamp to target (default)") + p.add_argument("--vibrato-hz", type=float, default=0.0, + help="Vibrato frequency in Hz (0 = off)") + p.add_argument("--vibrato-cents", type=float, default=0.0, + help="Vibrato depth in cents (peak-to-peak / 2)") + p.add_argument("--vibrato-onset-ms", type=float, default=200.0, + help="Delay before vibrato fades in (ms)") + p.add_argument("--xfade-ms", type=float, default=80.0, + help="Crossfade between adjacent target notes (ms). Larger " + "values smooth pitch transitions but blur the attack.") + p.add_argument("--voicing-ramp-ms", type=float, default=40.0, + help="Ramp f0 in/out over this window at voiced/unvoiced " + "boundaries (word starts/ends). Smooths the WORLD " + "synth pop where pitch goes 0→target abruptly.") + p.add_argument("--detect-boundaries", action="store_true", + help="Use librosa onset detection to find natural syllable " + "boundaries in the audio, rather than weight-proportional " + "splits. Only applies when there are >= 2 notes.") + args = p.parse_args() + + notes = [n.strip() for n in args.notes.split(",") if n.strip()] + if not notes: + print("✗ --notes required", file=sys.stderr) + return 1 + target_midis = np.array([note_to_midi(n) for n in notes], dtype=np.float64) + target_hzs = np.array([midi_to_hz(m) for m in target_midis]) + + if args.weights: + weights = np.array([float(w) for w in args.weights.split(",")], dtype=np.float64) + if len(weights) != len(notes): + print(f"✗ --weights length {len(weights)} != notes length {len(notes)}", file=sys.stderr) + return 1 + else: + weights = np.ones(len(notes), dtype=np.float64) + weights = weights / weights.sum() + cum_w = np.concatenate([[0.0], np.cumsum(weights)]) + + x, fs = sf.read(args.in_wav, dtype="float64") + if x.ndim > 1: + x = x.mean(axis=1) + + # WORLD decompose. f0_floor=90 (jeffrey-pvc never goes below 90 Hz) + # tightens cheaptrick's analysis window, which kills the held-note + # "ring/echo" artifact (low-floor → wide window → formant smearing). + f0_floor = 90.0 + f0_raw, t = pw.harvest(x, fs, f0_floor=f0_floor, f0_ceil=600.0, frame_period=5.0) + f0 = pw.stonemask(x, f0_raw, t, fs) + fft_size = pw.get_cheaptrick_fft_size(fs, f0_floor=f0_floor) + sp = pw.cheaptrick(x, f0, t, fs, fft_size=fft_size, f0_floor=f0_floor) + ap = pw.d4c(x, f0, t, fs, fft_size=fft_size) + + # Build per-frame target f0 curve. N notes laid across the time axis; + # each note holds its pitch; transitions crossfade in log space over + # `--xfade-ms` ms. + n_frames = len(t) + frame_period_ms = (t[1] - t[0]) * 1000.0 if n_frames > 1 else 5.0 + xfade_frames = max(1, int(args.xfade_ms / frame_period_ms)) + + # Weighted segment boundaries in frame indices. + seg_starts = (cum_w * n_frames).astype(np.int64) + + # Absolute start-time override — pin each note to a specific time + # in the audio (seconds). Useful for whole-audio resynth where note + # boundaries should align to whisper word timestamps, not be + # distributed evenly across the file. + if args.note_starts: + starts_sec = [float(s) for s in args.note_starts.split(",") if s.strip()] + if len(starts_sec) != len(notes): + print(f"✗ --note-starts length {len(starts_sec)} != notes length {len(notes)}", file=sys.stderr) + return 1 + seg_starts = np.array( + [int(round(s / (frame_period_ms / 1000.0))) for s in starts_sec] + [n_frames], + dtype=np.int64, + ) + seg_starts = np.clip(seg_starts, 0, n_frames) + + # Optionally replace with librosa-detected syllable boundaries — + # finds onset frames in the audio and snaps the segment starts to + # the nearest detected onset. Falls back to weighted splits if + # detection finds the wrong number of onsets. + if args.detect_boundaries and HAS_LIBROSA and len(notes) >= 2: + # librosa needs float32 mono; use the WORLD-analyzed signal. + try: + onset_env = librosa.onset.onset_strength(y=x.astype(np.float32), sr=fs, hop_length=256) + onset_frames = librosa.onset.onset_detect( + onset_envelope=onset_env, sr=fs, hop_length=256, + backtrack=True, + ) + # Convert from librosa hop frames to WORLD frames + librosa_to_world = (256 / fs) / (frame_period_ms / 1000.0) + world_onset_frames = (onset_frames * librosa_to_world).astype(np.int64) + # Always anchor first segment at 0; need (len(notes)-1) onset hits + need = len(notes) - 1 + if len(world_onset_frames) >= need: + # Pick the `need` largest-strength onsets + strengths = onset_env[onset_frames] + top_idx = np.argsort(strengths)[-need:] + top_onsets = np.sort(world_onset_frames[top_idx]) + detected = np.concatenate([[0], top_onsets, [n_frames]]) + seg_starts = detected.astype(np.int64) + print(f" librosa onsets: {len(onset_frames)} found, used {need} → {top_onsets.tolist()}") + else: + print(f" librosa onsets: only {len(onset_frames)} found, need {need} — using weighted splits") + except Exception as e: + print(f" librosa onset detection failed: {e} — using weighted splits") + + target_log = np.zeros(n_frames) + for i in range(n_frames): + # Which segment is this frame in? + seg = int(np.searchsorted(seg_starts[1:], i, side="right")) + seg = min(seg, len(notes) - 1) + center_log = np.log(target_hzs[seg]) + # Crossfade with the next segment if we're near its boundary. + if seg + 1 < len(notes): + next_start = seg_starts[seg + 1] + dist_to_next = next_start - i + if dist_to_next < xfade_frames: + tval = 1.0 - (dist_to_next / xfade_frames) + center_log = (1 - tval) * center_log + tval * np.log(target_hzs[seg + 1]) + target_log[i] = center_log + + target_curve = np.exp(target_log) + + # Vibrato — sine LFO on top of target, fading in after onset. + if args.vibrato_hz > 0 and args.vibrato_cents > 0: + time_sec = np.arange(n_frames) * (frame_period_ms / 1000.0) + onset_sec = args.vibrato_onset_ms / 1000.0 + fade = np.clip((time_sec - onset_sec) / max(0.05, onset_sec), 0.0, 1.0) + depth_ratio = (args.vibrato_cents / 100.0) / 12.0 # cents → semitones → log2 + lfo = np.sin(2 * np.pi * args.vibrato_hz * time_sec) * depth_ratio * fade + target_curve = target_curve * (2.0 ** lfo) + + # Build the f0 curve passed to WORLD synth. Two key moves vs naive: + # + # 1. INTERPOLATE through unvoiced gaps before synthesis. WORLD pops + # on f0=0→target jumps; feeding it a continuous curve eliminates + # those phase-resets. Voiced/unvoiced structure gets re-imposed + # in the time domain after synthesis (step 2). + # 2. MUTE unvoiced regions post-synth with 5ms crossfade ramps — + # keeps consonants/sibilants from being colored by tone. + voiced = f0 > 0 + + # Build the per-frame target the synth will see (interpolated curve). + if args.retain >= 0.999: + # Hard clamp to target on voiced frames; interpolate target + # through unvoiced gaps so the synth has continuous pitch. + f0_synth = target_curve.copy() + else: + log_src = np.log(np.maximum(f0, 1e-6)) + log_tgt = np.log(target_curve) + log_blend = (1.0 - args.retain) * log_src + args.retain * log_tgt + # Interpolate the source side across unvoiced frames so the blend + # has continuous data; otherwise unvoiced frames blend toward 0. + if voiced.sum() >= 2: + voiced_idx = np.where(voiced)[0] + log_src_interp = np.interp(np.arange(len(f0)), voiced_idx, log_src[voiced_idx]) + log_blend_smooth = (1.0 - args.retain) * log_src_interp + args.retain * log_tgt + f0_synth = np.exp(log_blend_smooth) + else: + f0_synth = np.exp(log_blend) + + f0_new = f0_synth # name kept for downstream code consistency + + y = pw.synthesize(f0_new, sp, ap, fs, frame_period=5.0) + + # Re-impose voiced/unvoiced structure as a time-domain amplitude + # mask. Unvoiced regions get muted with 5ms crossfades at every + # transition so consonants ("s", "k", "th") aren't colored by + # whatever tone WORLD synthesized through the gap. This is the + # "WORLD for pitch, original passes through for noise" trick. + samples_per_frame = int(round(fs * 5.0 / 1000.0)) + voiced_audio_mask = np.repeat(voiced.astype(np.float64), samples_per_frame) + if len(voiced_audio_mask) < len(y): + voiced_audio_mask = np.pad(voiced_audio_mask, (0, len(y) - len(voiced_audio_mask)), + mode="edge") + voiced_audio_mask = voiced_audio_mask[:len(y)] + + # Smooth the 0/1 mask with 5ms cosine crossfades at edges. + ramp = int(0.005 * fs) + if ramp > 1: + edges = np.diff(voiced_audio_mask.astype(np.int8)) + for idx in np.where(edges == 1)[0]: + for k in range(ramp): + pos = idx + 1 + k + if pos < len(voiced_audio_mask): + voiced_audio_mask[pos] *= 0.5 - 0.5 * np.cos(np.pi * (k + 1) / ramp) + for idx in np.where(edges == -1)[0]: + for k in range(ramp): + pos = idx - k + if pos >= 0: + voiced_audio_mask[pos] *= 0.5 - 0.5 * np.cos(np.pi * (k + 1) / ramp) + + # Composite: WORLD audio in voiced regions, original audio in + # unvoiced regions. Preserves natural sibilants and stops while + # keeping the pitch-corrected vowels. mask is 1 in voiced, 0 in + # unvoiced (with 5ms ramps at boundaries). + n = min(len(y), len(x), len(voiced_audio_mask)) + y_composite = np.zeros(n) + y_composite = voiced_audio_mask[:n] * y[:n] + (1.0 - voiced_audio_mask[:n]) * x[:n] + + sf.write(args.out_wav, y_composite.astype(np.float32), fs) + + # Print a one-line summary so pitchsnap.mjs can show it. + voiced_pct = 100.0 * np.mean(voiced) + print(f" world · {len(notes)} notes · {n_frames} frames · {voiced_pct:.0f}% voiced " + f"· retain={args.retain} · vibrato={args.vibrato_hz}Hz/{args.vibrato_cents}¢") + return 0 + +if __name__ == "__main__": + sys.exit(main()) diff --git a/pop/bin/pitchwords.mjs b/pop/bin/pitchwords.mjs new file mode 100644 --- /dev/null +++ b/pop/bin/pitchwords.mjs @@ -0,0 +1,199 @@ +#!/usr/bin/env node +// pitchwords.mjs — first stage of vocal-post: pitch each word in a +// vocal stem to its target note from the .np score. +// +// Slice plan: for each whisper word at [fromMs, toMs], grab the audio +// from fromMs to the NEXT word's fromMs (or end of file for the last +// word). This keeps the inter-word silence with each word so the +// concatenated output preserves rhythm. +// +// Pitch plan: walk whisper words in order; map each word to a syllable +// in the .np section at the proportional index (whisper words are +// usually fewer than .np syllables — multi-syllable words pick one +// syllable's note). Pitch-shift the slice by (target − ref) semitones +// using the `rubberband` CLI (formant-preserving). +// +// Aggressive by default — full target-note pitch shift, no smoothing. +// Future: gentle / firm / off knobs per the post-prod memory. +// +// Usage: +// node bin/pitchwords.mjs --vocal big-pictures/out/plork-hook-vocal.mp3 \ +// --score big-pictures/plork.np --section hook \ +// --ref-note C3 \ +// --out big-pictures/out/plork-hook-pitched.mp3 + +import { spawnSync } from "node:child_process"; +import { existsSync, mkdirSync, readFileSync, writeFileSync, rmSync } from "node:fs"; +import { resolve, dirname, basename } from "node:path"; + +// ── arg parse ───────────────────────────────────────────────────────── +function parseArgs(argv) { + const flags = {}; + for (let i = 0; i < argv.length; i++) { + const a = argv[i]; + if (!a.startsWith("--")) continue; + const k = a.slice(2); + const next = argv[i + 1]; + if (next !== undefined && !next.startsWith("--")) { flags[k] = next; i++; } + else flags[k] = true; + } + return flags; +} + +const flags = parseArgs(process.argv.slice(2)); + +const vocalPath = resolve(process.cwd(), flags.vocal || ""); +if (!existsSync(vocalPath)) { + console.error("usage: --vocal --score [--section hook] [--ref-note C3] [--out path.mp3]"); + process.exit(1); +} +const wordsPath = resolve(process.cwd(), flags.words || vocalPath.replace(/\.mp3$/, "-words.json")); +if (!existsSync(wordsPath)) { + console.error(`✗ words.json not found at ${wordsPath}. run bin/align.mjs first.`); + process.exit(1); +} +const scorePath = resolve(process.cwd(), flags.score || ""); +if (!existsSync(scorePath)) { + console.error(`✗ --score file required (path to .np)`); + process.exit(1); +} +const SECTION = (flags.section || "hook").toLowerCase(); +const REF_NOTE = flags["ref-note"] || "C3"; +const OUT_PATH = flags.out + ? resolve(process.cwd(), flags.out) + : vocalPath.replace(/\.mp3$/, "-pitched.mp3"); + +// ── helpers ─────────────────────────────────────────────────────────── +const NOTE_TO_SEMI = { c: 0, d: 2, e: 4, f: 5, g: 7, a: 9, b: 11 }; +function noteToMidi(p) { + const m = p.trim().toLowerCase().match(/^([a-g])([#b]?)(-?\d+)$/); + if (!m) throw new Error(`bad note: ${p}`); + let semi = NOTE_TO_SEMI[m[1]]; + if (m[2] === "#") semi += 1; + if (m[2] === "b") semi -= 1; + const oct = parseInt(m[3], 10); + return 12 * (oct + 1) + semi; +} + +// Same parser shape as recap/bin/vocal.mjs's parseNp — flatten to a +// list of {pitch, syl} per section. +function parseNp(text) { + const sections = {}; + let current = null; + for (const raw of text.split("\n")) { + const line = raw.trim(); + if (!line || line.startsWith("#")) continue; + if (!line.includes(":") && /^[a-z][a-z0-9 ]*$/.test(line)) { + current = line.toLowerCase(); + if (!sections[current]) sections[current] = []; + continue; + } + if (!current) { current = "default"; sections[current] = []; } + const tokens = line.split(/\s+/).filter(Boolean); + for (const tok of tokens) { + const m = tok.match(/^([A-Ga-g][#b]?):(.+)$/); + if (!m) continue; + const note = m[1].charAt(0).toUpperCase() + m[1].slice(1); + sections[current].push({ pitch: note, syl: m[2] }); + } + } + return sections; +} + +function probeDuration(p) { + const r = spawnSync( + "ffprobe", + ["-v", "error", "-show_entries", "format=duration", + "-of", "default=noprint_wrappers=1:nokey=1", p], + { encoding: "utf8" } + ); + return Number(r.stdout.trim()); +} + +// ── main ────────────────────────────────────────────────────────────── +const words = JSON.parse(readFileSync(wordsPath, "utf8")); +const score = parseNp(readFileSync(scorePath, "utf8")); +const syllables = score[SECTION]; +if (!syllables || !syllables.length) { + console.error(`✗ section '${SECTION}' empty in ${scorePath}`); + process.exit(1); +} + +const refMidi = noteToMidi(REF_NOTE); +const totalDur = probeDuration(vocalPath); + +const tmpDir = `${dirname(OUT_PATH)}/.${basename(OUT_PATH).replace(/\..*$/, "")}-pw-tmp`; +rmSync(tmpDir, { recursive: true, force: true }); +mkdirSync(tmpDir, { recursive: true }); + +console.log(`→ pitchwords · ${words.length} whisper words → ${syllables.length} score syllables · ref=${REF_NOTE}`); + +const slicePaths = []; +for (let i = 0; i < words.length; i++) { + const w = words[i]; + const startSec = w.fromMs / 1000; + const endSec = i < words.length - 1 ? words[i + 1].fromMs / 1000 : totalDur; + + // Proportional mapping word i → syllable + const sylIdx = Math.min(syllables.length - 1, + Math.floor(i * syllables.length / words.length)); + const syl = syllables[sylIdx]; + const noteStr = /\d/.test(syl.pitch) ? syl.pitch : syl.pitch + "3"; + const targetMidi = noteToMidi(noteStr); + const semitones = targetMidi - refMidi; + + const sliceWav = `${tmpDir}/w${i.toString().padStart(3, "0")}.wav`; + const cut = spawnSync( + "ffmpeg", + ["-hide_banner", "-y", "-loglevel", "error", + "-ss", startSec.toFixed(4), "-to", endSec.toFixed(4), + "-i", vocalPath, + "-c:a", "pcm_s16le", "-ar", "48000", "-ac", "1", sliceWav], + { stdio: ["ignore", "ignore", "inherit"] } + ); + if (cut.status !== 0) { + console.error(`✗ ffmpeg slice failed at word ${i}`); + process.exit(1); + } + + let pieceWav = sliceWav; + if (Math.abs(semitones) >= 0.01) { + pieceWav = `${tmpDir}/w${i.toString().padStart(3, "0")}-p.wav`; + const rb = spawnSync( + "rubberband", + ["-p", String(semitones), sliceWav, pieceWav], + { stdio: ["ignore", "ignore", "ignore"] } + ); + if (rb.status !== 0) { + console.error(`✗ rubberband failed at word ${i} (Δ${semitones} st)`); + process.exit(1); + } + } + slicePaths.push(pieceWav); + + const arrow = semitones >= 0 ? "+" : ""; + console.log( + ` ${i.toString().padStart(2, "0")} ${w.text.padEnd(14)} ` + + `${startSec.toFixed(2)}–${endSec.toFixed(2)}s → ${syl.syl.padEnd(8)} ${noteStr} ` + + `(${arrow}${semitones} st)` + ); +} + +// ── concat ───────────────────────────────────────────────────────────── +const concatList = `${tmpDir}/concat.txt`; +writeFileSync(concatList, slicePaths.map(p => `file '${p}'`).join("\n") + "\n"); + +const concat = spawnSync( + "ffmpeg", + ["-hide_banner", "-y", "-loglevel", "error", + "-f", "concat", "-safe", "0", "-i", concatList, + "-c:a", "libmp3lame", "-q:a", "3", OUT_PATH], + { stdio: "inherit" } +); +if (concat.status !== 0) { + console.error("✗ ffmpeg concat failed"); + process.exit(1); +} + +rmSync(tmpDir, { recursive: true, force: true }); +console.log(`✓ ${OUT_PATH}`); diff --git a/pop/bin/refine_onsets.py b/pop/bin/refine_onsets.py new file mode 100644 --- /dev/null +++ b/pop/bin/refine_onsets.py @@ -0,0 +1,92 @@ +#!/usr/bin/env python3 +""" +refine_onsets.py — VALIDATOR (not modifier). + +The score (.np file → events.json with `snappedStart` beat positions) +is the authoritative source of timing in this pipeline. The visual +storyboard rigidly follows the score; any drift between score and +audio is a *pitchsnap* problem to fix at the audio generation layer, +not papered over visually. + +This script measures the gap between score-defined slide.start times +and detected audio onsets, prints a report of mismatches, and exits. +It does NOT modify the storyboard. Use the report to identify which +words pitchsnap rendered late/early — that's where the audio pipeline +needs slot-rigid trimming/padding. + +Usage: + refine_onsets.py +""" +import json +import sys +from pathlib import Path + +import numpy as np +import librosa + +SEARCH_WIN_S = 0.85 # how far from snappedStart to look for an onset +MIN_FIX_MS = 25 # don't override if onset is within this of original + # (avoids jitter from low-confidence detections) + + +def main(): + if len(sys.argv) < 3: + print("usage: refine_onsets.py