From 4fdfbc4f224693b79444f12a8611657a0d38a274 Mon Sep 17 00:00:00 2001 From: "prompt.ac/@jeffrey" Date: Mon, 18 May 2026 20:08:22 -0400 Subject: [PATCH] pop: vocal/melody pipeline tooling + big-pictures tracks Per-word/per-line align, MFA + whisper realign, MIDI/onset->.np converters, score pitch/render, best-of-takes; chillwave local say/sing; big-pictures frere/twinkle/small-world; /api/say update; pipeline audit report. Co-Authored-By: Claude Opus 4.7 (1M context) --- pop/VOICE.md | 14 + pop/big-pictures/amazing.np | 38 ++- pop/big-pictures/amazing.txt | 36 +++ pop/big-pictures/cli.mjs | 467 +++++++++++++++++++++++++++++++ pop/big-pictures/frere.np | 16 ++ pop/big-pictures/frere.txt | 9 + pop/big-pictures/recap-test.np | 56 ++++ pop/big-pictures/small-world.np | 27 ++ pop/big-pictures/small-world.txt | 8 + pop/big-pictures/twinkle.np | 22 ++ pop/big-pictures/twinkle.txt | 9 + pop/bin/align-words.mjs | 122 ++++++++ pop/bin/best-of-takes.mjs | 229 +++++++++++++++ pop/bin/detect_onsets.py | 58 ++++ pop/bin/melody-bells.mjs | 275 ++++++++++++++++++ pop/bin/mfa-align.mjs | 230 +++++++++++++++ pop/bin/midi_to_np.py | 145 ++++++++++ pop/bin/os_to_np.py | 152 ++++++++++ pop/bin/perline.mjs | 232 +++++++++++++++ pop/bin/perword.mjs | 269 ++++++++++++++++++ pop/bin/realign-from-whisper.mjs | 116 ++++++++ pop/bin/realign-storyboard.mjs | 140 +++++++++ pop/bin/refine_words.py | 102 +++++++ pop/bin/render_frames.py | 129 ++++++--- pop/bin/say.mjs | 83 +++++- pop/bin/score-pitch.mjs | 207 ++++++++++++++ pop/bin/score-render.mjs | 299 ++++++++++++++++++++ pop/bin/storyboard.mjs | 194 ++++++++++--- pop/bin/timeline.py | 90 ++++-- pop/bin/track-poster.py | 441 +++++++++++++++++++++++++++++ pop/bin/validate_word.py | 36 +++ pop/chillwave/bin/gen-illy.mjs | 15 +- pop/chillwave/bin/render.mjs | 281 +++++++++++++++++-- pop/chillwave/bin/say-local.mjs | 114 ++++++++ pop/chillwave/bin/sing.mjs | 123 ++++++++ pop/chillwave/undabeach.illy.txt | 6 +- pop/chillwave/undabeach.vocal.np | 22 ++ pop/dance/bin/cover.mjs | 10 +- reports/pop-pipeline-audit.md | 173 ++++++++++++ system/netlify/functions/say.js | 99 +++++-- 40 files changed, 4946 insertions(+), 148 deletions(-) create mode 100644 pop/big-pictures/cli.mjs create mode 100644 pop/big-pictures/frere.np create mode 100644 pop/big-pictures/frere.txt create mode 100644 pop/big-pictures/recap-test.np create mode 100644 pop/big-pictures/small-world.np create mode 100644 pop/big-pictures/small-world.txt create mode 100644 pop/big-pictures/twinkle.np create mode 100644 pop/big-pictures/twinkle.txt create mode 100644 pop/bin/align-words.mjs create mode 100644 pop/bin/best-of-takes.mjs create mode 100644 pop/bin/detect_onsets.py create mode 100644 pop/bin/melody-bells.mjs create mode 100644 pop/bin/mfa-align.mjs create mode 100644 pop/bin/midi_to_np.py create mode 100644 pop/bin/os_to_np.py create mode 100644 pop/bin/perline.mjs create mode 100644 pop/bin/perword.mjs create mode 100644 pop/bin/realign-from-whisper.mjs create mode 100644 pop/bin/realign-storyboard.mjs create mode 100644 pop/bin/refine_words.py create mode 100644 pop/bin/score-pitch.mjs create mode 100644 pop/bin/score-render.mjs create mode 100644 pop/bin/track-poster.py create mode 100644 pop/chillwave/bin/say-local.mjs create mode 100644 pop/chillwave/bin/sing.mjs create mode 100644 pop/chillwave/undabeach.vocal.np create mode 100644 reports/pop-pipeline-audit.md diff --git a/pop/VOICE.md b/pop/VOICE.md index 160f78c84..c2cae0c73 100644 --- a/pop/VOICE.md +++ b/pop/VOICE.md @@ -36,6 +36,20 @@ each lane has its own voice on top of this base. - 16 bars per verse, 4-line hook, ~1:30 total. roughly: hook → verse → hook → verse → hook → outro - reference voice: see vault. ingestion is in-context style, not training corpus +### jungle (ragga toast) + +- this lane's voice is **fía's**, not jeffrey's. she is the MC. the + base guide above still holds (lowercase drafts, no filler, specific + feeling) but the words are hers to write and rewrite +- spanish by default. code-switch only when it lands +- toast craft, not rap craft: short lines that *jump over* the drums. + call-and-response. a hook is one phrase you can shout back +- ride the half-time pocket — land weight on the slow snare, not the + fast hats. leave space; the break is busy enough +- the `.txt` next to each `.np` is a scratch cue scaffold, not a lyric. + it exists to be thrown away +- sunlit, not menace — golden block-party energy, playful, warm + ## examples of the voice working placeholders. real lines land here when the first track is mixed. diff --git a/pop/big-pictures/amazing.np b/pop/big-pictures/amazing.np index 09ee61e25..8765028de 100644 --- a/pop/big-pictures/amazing.np +++ b/pop/big-pictures/amazing.np @@ -1,4 +1,4 @@ -# Amazing Grace — full verse 1. "New Britain" tune, William Walker 1835. +# Amazing Grace — full 7 verses. "New Britain" tune, William Walker 1835. # notation: NOTE:syllable*beats # Phrase 1+2 melody from papers/arxiv-folk-songs/folk-songs.tex:181 # (the canonical pentatonic transcription). @@ -16,3 +16,39 @@ D3:a-*1 G3:-ma-*3 B3:-zing*1 G3:grace*3 B3:how*1 A3:sweet*3 G3:the*1 B3:sound*5 D3:that*1 E3:saved*3 D3:a*1 B3:wretch*3 G3:like*1 B3:me*5 D3:i*1 G3:once*3 D4:was*1 D4:lost*3 B3:but*1 D4:now*3 A3:am*1 G3:found*5 B3:was*1 D4:blind*3 D4:but*1 B3:now*3 A3:i*1 G3:see*5 + +verse 2 +D3:twas*1 G3:grace*3 B3:that*1 G3:taught*3 B3:my*1 A3:heart*3 G3:to*1 B3:fear*5 +D3:and*1 E3:grace*3 D3:my*1 B3:fears*3 G3:re-*1 B3:-lieved*5 +D3:how*1 G3:pre-*3 D4:-cious*1 D4:did*3 B3:that*1 D4:grace*3 A3:ap-*1 G3:-pear*5 +B3:the*1 D4:hour*3 D4:i*1 B3:first*3 A3:be-*1 G3:-lieved*5 + +verse 3 +D3:through*1 G3:ma-*3 B3:-ny*1 G3:dan-*3 B3:-gers*1 A3:toils*3 G3:and*1 B3:snares*5 +D3:i*1 E3:have*3 D3:al-*1 B3:-rea-*3 G3:-dy*1 B3:come*5 +D3:tis*1 G3:grace*3 D4:hath*1 D4:brought*3 B3:me*1 D4:safe*3 A3:thus*1 G3:far*5 +B3:and*1 D4:grace*3 D4:will*1 B3:lead*3 A3:me*1 G3:home*5 + +verse 4 +D3:the*1 G3:lord*3 B3:has*1 G3:pro-*3 B3:-mised*1 A3:good*3 G3:to*1 B3:me*5 +D3:his*1 E3:word*3 D3:my*1 B3:hope*3 G3:se-*1 B3:-cures*5 +D3:he*1 G3:will*3 D4:my*1 D4:shield*3 B3:and*1 D4:por-*3 A3:-tion*1 G3:be*5 +B3:as*1 D4:long*3 D4:as*1 B3:life*3 A3:en-*1 G3:-dures*5 + +verse 5 +D3:yea*1 G3:when*3 B3:this*1 G3:flesh*3 B3:and*1 A3:heart*3 G3:shall*1 B3:fail*5 +D3:and*1 E3:mor-*3 D3:-tal*1 B3:life*3 G3:shall*1 B3:cease*5 +D3:i*1 G3:shall*3 D4:pos-*1 D4:-sess*3 B3:with-*1 D4:-in*3 A3:the*1 G3:veil*5 +B3:a*1 D4:life*3 D4:of*1 B3:joy*3 A3:and*1 G3:peace*5 + +verse 6 +D3:the*1 G3:earth*3 B3:shall*1 G3:soon*3 B3:dis-*1 A3:-solve*3 G3:like*1 B3:snow*5 +D3:the*1 E3:sun*3 D3:for-*1 B3:-bear*3 G3:to*1 B3:shine*5 +D3:but*1 G3:god*3 D4:who*1 D4:called*3 B3:me*1 D4:here*3 A3:be-*1 G3:-low*5 +B3:shall*1 D4:be*3 D4:for-*1 B3:-ev-*3 A3:-er*1 G3:mine*5 + +verse 7 +D3:when*1 G3:weve*3 B3:been*1 G3:there*3 B3:ten*1 A3:thou-*3 G3:-sand*1 B3:years*5 +D3:bright*1 E3:shi-*3 D3:-ning*1 B3:as*3 G3:the*1 B3:sun*5 +D3:weve*1 G3:no*3 D4:less*1 D4:days*3 B3:to*1 D4:sing*3 A3:gods*1 G3:praise*5 +B3:than*1 D4:when*3 D4:wed*1 B3:first*3 A3:be-*1 G3:-gun*5 diff --git a/pop/big-pictures/amazing.txt b/pop/big-pictures/amazing.txt index 06793d34f..809e56338 100644 --- a/pop/big-pictures/amazing.txt +++ b/pop/big-pictures/amazing.txt @@ -3,3 +3,39 @@ amazing grace how sweet the sound that saved a wretch like me i once was lost but now am found was blind but now i see + +verse 2 +twas grace that taught my heart to fear +and grace my fears relieved +how precious did that grace appear +the hour i first believed + +verse 3 +through many dangers toils and snares +i have already come +tis grace hath brought me safe thus far +and grace will lead me home + +verse 4 +the lord has promised good to me +his word my hope secures +he will my shield and portion be +as long as life endures + +verse 5 +yea when this flesh and heart shall fail +and mortal life shall cease +i shall possess within the veil +a life of joy and peace + +verse 6 +the earth shall soon dissolve like snow +the sun forbear to shine +but god who called me here below +shall be forever mine + +verse 7 +when weve been there ten thousand years +bright shining as the sun +weve no less days to sing gods praise +than when wed first begun diff --git a/pop/big-pictures/cli.mjs b/pop/big-pictures/cli.mjs new file mode 100644 index 000000000..b8068e729 --- /dev/null +++ b/pop/big-pictures/cli.mjs @@ -0,0 +1,467 @@ +#!/usr/bin/env node +// big-pictures/cli.mjs — single-command pipeline for vocal tracks. +// +// Each step is content-hash cached by its underlying script (say.mjs, +// align.mjs, score-pitch.mjs, etc.); rerunning costs $0 unless inputs +// change. The CLI just wires them together with sane defaults. +// +// Pipeline: +// 1. say.mjs .txt → -vocal.mp3 +// 2. align.mjs -vocal.mp3 → -vocal-words.json +// 3. score-pitch.mjs WORLD f0 replacement → -pitched.mp3 +// + -pitched-alignment.json +// 4. score-stretch.mjs per-word rubberband → -pitched-stretched.mp3 +// (skipped with --no-stretch) +// 5. melody-bells.mjs pad/bell accompaniment → -pad.mp3 +// 6. waltz.mjs sinebells harmonic bed → -bed.mp3 +// 7. ffmpeg amix vocal-forward 3-layer → -mix.mp3 +// 8. stamp concat ac signoff (chipmunk + +// 4-bit crush) → -stamped.mp3 +// (skipped with --no-stamp) +// 9. finalize.mjs ID3 + cover + lyrics → -final.mp3 +// 10. cp → ~/Desktop/.mp3 +// +// Usage: +// node cli.mjs +// node cli.mjs all build every .np in this dir +// node cli.mjs --status show cache state +// node cli.mjs --bpm 80 --transpose 0 --voice pad +// node cli.mjs --no-stretch natural prosody, no hymn pacing +// node cli.mjs --no-stamp skip the ac signoff +// node cli.mjs --force bust caches end to end +// +// Per-song defaults live in DEFAULTS below. Anything not listed gets +// the universal fallback (bpm 80, transpose 0, pad voice, etc.). + +import { spawnSync } from "node:child_process"; +import { + readFileSync, writeFileSync, copyFileSync, + existsSync, statSync, mkdirSync, readdirSync, +} from "node:fs"; +import { resolve, dirname } from "node:path"; +import { fileURLToPath } from "node:url"; +import { homedir } from "node:os"; + +const HERE = dirname(fileURLToPath(import.meta.url)); +const POP = resolve(HERE, ".."); +const REPO = resolve(POP, ".."); +const RECAP = resolve(REPO, "recap"); +const OUT = `${HERE}/out`; +mkdirSync(OUT, { recursive: true }); + +const argv = process.argv.slice(2); +const flags = {}; +const positional = []; +for (let i = 0; i < argv.length; i++) { + const a = argv[i]; + if (a.startsWith("--")) { + const k = a.slice(2); + const next = argv[i + 1]; + if (next !== undefined && !next.startsWith("--")) { flags[k] = next; i++; } + else flags[k] = true; + } else positional.push(a); +} + +const SLUG = positional[0]; +if (!SLUG || SLUG === "--help" || SLUG === "-h") { + console.log(`big-pictures cli — vocal track pipeline + +usage: + node cli.mjs + node cli.mjs all every .np in this dir + node cli.mjs --status cache state + node cli.mjs --bpm 80 --transpose 0 --voice pad + node cli.mjs --no-stretch natural-paced (no hymn rubberband) + node cli.mjs --no-stamp skip the ac signoff + node cli.mjs --force bust caches end to end + +flags: + --bpm N tempo for stretch + bells + bed + --transpose N semitone shift on the vocal (sign matters) + --transpose-bed N semitone shift on the bed (key root, e.g. 7 = G) + --max-stretch X rubberband ceiling (default 6.0; held notes hit this) + --onset-shift-ms N shift cuts earlier so vowels land on beats (default 60) + --voice pad|bell accompaniment timbre (default pad) + --scale major|minor bed scale + --progression "0,3,0,4" chord-degree cycle for the bed +`); + process.exit(SLUG === "--help" || SLUG === "-h" ? 0 : 1); +} + +// ── per-song defaults ──────────────────────────────────────────────── +// Flags override these. Any slug not listed gets DEFAULT_FALLBACK. +const DEFAULT_FALLBACK = { + bpm: 80, transpose: 0, maxStretch: 6.0, voice: "pad", + scale: "major", transposeBed: 0, progression: "0,3,4,0", + title: null, +}; +const DEFAULTS = { + amazing: { bpm: 70, transpose: 0, maxStretch: 8.0, voice: "pad", + scale: "major", transposeBed: 7, progression: "0,3,0,4", + title: "amazing grace" }, + twinkle: { bpm: 80, transpose: 0, maxStretch: 6.0, voice: "pad", + scale: "major", transposeBed: 0, progression: "0,3,4,0", + title: "twinkle twinkle little star" }, + mary: { bpm: 100, transpose: 0, maxStretch: 5.0, voice: "pad", + scale: "major", transposeBed: 0, progression: "0,3,4,0", + title: "mary had a little lamb" }, + row: { bpm: 90, transpose: -5, maxStretch: 6.0, voice: "pad", + scale: "major", transposeBed: 0, progression: "0,3,4,0", + title: "row row row your boat" }, + elephant: { bpm: 100, transpose: 0, maxStretch: 5.0, voice: "pad", + scale: "minor", transposeBed: 0, progression: "0,5,3,4", + title: "elephant" }, + plork: { bpm: 90, transpose: 0, maxStretch: 5.0, voice: "bell", + scale: "minor", transposeBed: 0, progression: "0,5,3,4", + title: "plork" }, + sentence: { bpm: 90, transpose: 0, maxStretch: 5.0, voice: "pad", + scale: "minor", transposeBed: 0, progression: "0,5,3,4", + title: "the music is real" }, + "small-world": { bpm: 110, transpose: -5, maxStretch: 5.0, voice: "pad", + scale: "major", transposeBed: 5, progression: "0,3,4,0", + title: "it's a small world" }, + frere: { bpm: 100, transpose: -5, maxStretch: 5.0, voice: "pad", + scale: "major", transposeBed: 0, progression: "0,4,0,4", + title: "frère jacques" }, +}; + +// ── 'all' subcommand ───────────────────────────────────────────────── +if (SLUG === "all") { + const slugs = readdirSync(HERE) + .filter((f) => f.endsWith(".np")) + .map((f) => f.replace(/\.np$/, "")) + .sort(); + console.log(`▸ building ${slugs.length} song(s): ${slugs.join(", ")}`); + for (const s of slugs) { + console.log(`\n━━━━━━━━━━━━━━━━━━━━ ${s} ━━━━━━━━━━━━━━━━━━━━`); + const r = spawnSync(process.execPath, [process.argv[1], s, ...argv.slice(1).filter((a) => a !== "all")], + { stdio: "inherit", cwd: HERE }); + if (r.status !== 0) console.error(`✗ ${s} failed (continuing)`); + } + process.exit(0); +} + +// ── resolve config for this slug ───────────────────────────────────── +const D = { ...DEFAULT_FALLBACK, ...(DEFAULTS[SLUG] || {}) }; +const BPM = Number(flags.bpm ?? D.bpm); +const TRANSPOSE = Number(flags.transpose ?? D.transpose); +const MAX_STRETCH = Number(flags["max-stretch"] ?? D.maxStretch); +const ONSET_SHIFT = Number(flags["onset-shift-ms"] ?? 60); +const VOICE = flags.voice || D.voice; +const SCALE = flags.scale || D.scale; +const TRANSPOSE_BED = Number(flags["transpose-bed"] ?? D.transposeBed); +const PROGRESSION = flags.progression || D.progression; +const STRETCH = !flags["no-stretch"]; +const STAMP = !flags["no-stamp"]; +const FORCE = flags.force === true; +const TITLE = flags.title || D.title || SLUG.replace(/[-_]+/g, " "); + +const TXT = `${HERE}/${SLUG}.txt`; +const NP = `${HERE}/${SLUG}.np`; +if (!existsSync(TXT)) { console.error(`✗ missing ${TXT}`); process.exit(1); } +if (!existsSync(NP)) { console.error(`✗ missing ${NP}`); process.exit(1); } + +// Output paths (all underneath OUT) +const VOCAL = `${OUT}/${SLUG}-vocal.mp3`; +const WORDS = `${OUT}/${SLUG}-vocal-words.json`; +const PITCHED = `${OUT}/${SLUG}-pitched.mp3`; +const STRETCHED = `${OUT}/${SLUG}-pitched-stretched.mp3`; +const PAD = `${OUT}/${SLUG}-pad.mp3`; +const BED = `${OUT}/${SLUG}-bed.mp3`; +const MIX = `${OUT}/${SLUG}-mix.mp3`; +const STAMPED = `${OUT}/${SLUG}-stamped.mp3`; +const FINAL = `${OUT}/${SLUG}-final.mp3`; +const STAMP_VOCAL = `${OUT}/ac-stamp-vocal.mp3`; +const STAMP_FX = `${OUT}/ac-stamp-crunched.mp3`; +const DESK = `${homedir()}/Desktop/${SLUG}.mp3`; + +// ── --status: show cache state and bail ────────────────────────────── +if (flags.status) { + const tally = (label, p) => { + if (existsSync(p)) { + const s = statSync(p); + const kb = (s.size / 1024).toFixed(0); + const age = ((Date.now() - s.mtimeMs) / 1000 / 60).toFixed(1); + console.log(` ✓ ${label.padEnd(28)} ${kb.padStart(7)} KB · ${age}m old`); + } else { + console.log(` · ${label.padEnd(28)} (not built)`); + } + }; + console.log(`▸ cache for '${SLUG}'`); + tally("vocal (TTS)", VOCAL); + tally("words.json (whisper)", WORDS); + tally("pitched (WORLD)", PITCHED); + tally("pitched-stretched", STRETCHED); + tally("pad/bells", PAD); + tally("bed (waltz)", BED); + tally("mix", MIX); + tally("stamped", STAMPED); + tally("final", FINAL); + tally("Desktop copy", DESK); + process.exit(0); +} + +// ── helpers ────────────────────────────────────────────────────────── +function step(name, fn) { + const t0 = Date.now(); + process.stdout.write(`▸ ${name}\n`); + fn(); + const ms = Date.now() - t0; + console.log(` ${(ms / 1000).toFixed(1)}s`); +} + +function run(cmd, args, cwd = POP) { + const r = spawnSync(cmd, args, { cwd, stdio: ["ignore", "inherit", "inherit"] }); + if (r.status !== 0) { + console.error(`✗ ${cmd} ${args.join(" ")} failed (exit ${r.status})`); + process.exit(1); + } +} + +// Pull every NOTE: token's midi from the .np for range diagnostics. +function scoreMidis() { + const NOTE_BASE = { C:0,"C#":1,DB:1,D:2,"D#":3,EB:3,E:4,F:5, + "F#":6,GB:6,G:7,"G#":8,AB:8,A:9,"A#":10,BB:10,B:11 }; + const lines = readFileSync(NP, "utf8").split("\n"); + const out = []; + for (const raw of lines) { + const l = raw.trim(); + if (!l || l.startsWith("#")) continue; + if (/^(verse|hook|bridge|chorus|outro|intro)( \d+)?$/i.test(l)) continue; + for (const tok of l.split(/\s+/)) { + const m = tok.match(/^([A-Ga-g][#b]?)-?(\d):/); + if (!m) continue; + const name = m[1].toUpperCase(); + const oct = parseInt(m[2], 10); + out.push(12 * (oct + 1) + NOTE_BASE[name]); + } + } + return out; +} + +// Warn when the post-transpose target range falls below jeffrey's +// sweet spot. WORLD f0 replacement only sounds like SINGING when the +// target sits above natural speech (~midi 56-58 / 175 Hz). When the +// target median lands in his speaking baritone, listeners hear his +// prosody, not the melody. This was the mary -7 bug. +function pitchSanityCheck() { + const midis = scoreMidis(); + if (!midis.length) return; + const sorted = [...midis].sort((a, b) => a - b); + const median = sorted[Math.floor(sorted.length / 2)] + TRANSPOSE; + const climax = sorted[sorted.length - 1] + TRANSPOSE; + const SPEECH_TOP = 58; // ~Bb3, top of jeffrey's natural baritone + const ICONIC_LOW = 65; // E4 — below this and pitching feels subtle + const ICONIC_TOP = 76; // E5 — above this and source struggles + const noteName = (m) => { + const oct = Math.floor(m / 12) - 1; + const n = ["C","C#","D","D#","E","F","F#","G","G#","A","A#","B"][m % 12]; + return `${n}${oct}`; + }; + if (median <= SPEECH_TOP) { + console.log(`⚠ post-transpose median ${noteName(median)} (midi ${median}) is in jeffrey's speech range`); + console.log(` WORLD's pitch shift will be subtle / inaudible. Try --transpose ${ICONIC_LOW - sorted[Math.floor(sorted.length/2)]} or higher.`); + } else if (climax > ICONIC_TOP) { + console.log(`⚠ post-transpose climax ${noteName(climax)} (midi ${climax}) above E5 — formant artifacts likely`); + console.log(` Consider --transpose ${ICONIC_TOP - sorted[sorted.length - 1]} or lower.`); + } +} + +// Sum beats × inter-verse gaps from the .np to predict total duration. +// Used to size the bed and trim the mix. +function computeDuration() { + const lines = readFileSync(NP, "utf8").split("\n"); + let beats = 0; + let sectionCount = 0; + for (const raw of lines) { + const l = raw.trim(); + if (!l || l.startsWith("#")) continue; + if (/^(verse|hook|bridge|chorus|outro|intro)( \d+)?$/i.test(l)) { + sectionCount++; continue; + } + for (const tok of l.split(/\s+/)) { + const m = tok.match(/^[A-Ga-g][#b]?-?\d:[^*]+(?:\*(\d+(?:\.\d+)?))?$/); + if (!m) continue; + beats += Number(m[1] ?? 1); + } + } + if (sectionCount > 1) beats += 2 * (sectionCount - 1); // inter-section rest + return Math.ceil((beats * 60) / BPM); +} + +// ── pipeline ───────────────────────────────────────────────────────── +const t0 = Date.now(); +console.log(`━━━ ${SLUG} ━━━ bpm ${BPM} · transpose ${TRANSPOSE >= 0 ? "+" : ""}${TRANSPOSE}st · ${VOICE} · ${STRETCH ? "stretched" : "natural"}${STAMP ? " · stamp" : ""}\n`); +pitchSanityCheck(); + +// ElevenLabs `/with-timestamps` returns exact per-character alignment +// for the TTS it just synthesized — no recognition step, no drift. +// Falls back to whisper if the alignment sidecar doesn't appear (e.g. +// the server endpoint isn't patched). +const VOCAL_ALIGN = `${OUT}/${SLUG}-vocal.mp3.alignment.json`; + +step("1 · say (jeffrey-pvc TTS)", () => { + // Try /with-timestamps first; if the server isn't patched (returns + // audio/mpeg instead of JSON), fall back to plain TTS + whisper. + const baseArgs = ["bin/say.mjs", `big-pictures/${SLUG}.txt`, + "--stability", "0.6", "--similarity", "0.9", + "--out", `big-pictures/out/${SLUG}-vocal.mp3`]; + const argsWithTs = [...baseArgs, "--timestamps"]; + if (FORCE) { baseArgs.push("--force"); argsWithTs.push("--force"); } + const r = spawnSync("node", argsWithTs, { cwd: POP, stdio: ["ignore", "inherit", "pipe"] }); + if (r.status === 0) return; + // /with-timestamps unsupported — retry plain. Existing cached vocal + // (without alignment sidecar) will be reused via content hash. + console.log(" ↪ falling back to plain TTS + whisper alignment"); + run("node", baseArgs); +}); + +step("2 · align (ElevenLabs timestamps → words.json)", () => { + if (existsSync(VOCAL_ALIGN)) { + // Convert ElevenLabs alignment → words.json schema that score-pitch + // and melody-bells already expect ([{text, fromMs, toMs}]). + const a = JSON.parse(readFileSync(VOCAL_ALIGN, "utf8")); + if (Array.isArray(a.words) && a.words.length) { + writeFileSync(WORDS, JSON.stringify(a.words, null, 0)); + console.log(` ✓ ${a.words.length} words from ElevenLabs (exact)`); + return; + } + console.log(` ⚠ alignment sidecar empty; falling back to whisper`); + } + const args = ["bin/align.mjs", `big-pictures/out/${SLUG}-vocal.mp3`]; + if (FORCE) args.push("--force"); + run("node", args); +}); + +// Step 2.5 — librosa onset refinement. Whisper word boundaries can +// drift 50-150ms from the actual consonant attack. librosa's +// onset_detect (with backtrack) finds energy onsets to ~30ms. We snap +// each whisper word's fromMs to the nearest onset within ±200ms, +// keeping whisper's word IDENTITY but acoustically-precise TIMING. +// +// Skipped automatically when the ElevenLabs timestamps path was used — +// those are already exact and don't need refinement. +const REFINED_WORDS = `${OUT}/${SLUG}-vocal-words-refined.json`; +step("2.5 · refine word boundaries (librosa onset detection)", () => { + if (existsSync(VOCAL_ALIGN)) { + // ElevenLabs timestamps are ground truth — just copy. + copyFileSync(WORDS, REFINED_WORDS); + console.log(" ↪ skipped (ElevenLabs timestamps already exact)"); + return; + } + run(`${POP}/.venv/bin/python`, [`${POP}/bin/refine_words.py`, + `big-pictures/out/${SLUG}-vocal.mp3`, + `big-pictures/out/${SLUG}-vocal-words.json`, + `big-pictures/out/${SLUG}-vocal-words-refined.json`]); +}); + +step(`3 · score-pitch (WORLD · transpose ${TRANSPOSE >= 0 ? "+" : ""}${TRANSPOSE}st)`, () => { + // Prefer the librosa-refined boundaries when available; the file + // schema is identical so score-pitch reads it transparently. + const wordsArg = existsSync(REFINED_WORDS) + ? `big-pictures/out/${SLUG}-vocal-words-refined.json` + : `big-pictures/out/${SLUG}-vocal-words.json`; + run("node", ["bin/score-pitch.mjs", "--slug", SLUG, "--section", "all", + "--transpose", String(TRANSPOSE), + "--vocal", `big-pictures/out/${SLUG}-vocal.mp3`, + "--words", wordsArg, + "--out", `big-pictures/out/${SLUG}-pitched.mp3`]); +}); + +let VOCAL_TRACK = PITCHED; +if (STRETCH) { + step(`4 · score-stretch (rubberband · ${BPM}bpm · max ${MAX_STRETCH}× · +${ONSET_SHIFT}ms onset)`, () => { + run("node", ["bin/score-stretch.mjs", "--slug", SLUG, "--section", "all", + "--bpm", String(BPM), + "--max-stretch", String(MAX_STRETCH), + "--onset-shift-ms", String(ONSET_SHIFT), + "--in", `big-pictures/out/${SLUG}-pitched.mp3`, + "--alignment", `big-pictures/out/${SLUG}-pitched-alignment.json`, + "--out", `big-pictures/out/${SLUG}-pitched-stretched.mp3`]); + }); + VOCAL_TRACK = STRETCHED; +} else { + console.log(`▸ 4 · score-stretch (skipped — natural pace)\n`); +} + +step(`5 · melody-bells (${VOICE} · ${BPM}bpm)`, () => { + run("node", ["bin/melody-bells.mjs", "--slug", SLUG, "--section", "all", + "--voice", VOICE, "--transpose", "0", "--gain", "0.45", + "--bpm", String(BPM), "--inter-verse-beats", "2", + "--out", `big-pictures/out/${SLUG}-pad.mp3`]); +}); + +step(`6 · waltz bed (sinebells · ${SCALE} · ${PROGRESSION})`, () => { + const dur = STRETCH ? computeDuration() + 2 : 0; + // For natural-paced cuts, size the bed to the WORDS sidecar duration. + let bedDur = dur; + if (!STRETCH && existsSync(WORDS)) { + const w = JSON.parse(readFileSync(WORDS, "utf8")); + if (w.length) bedDur = Math.ceil(w[w.length - 1].toMs / 1000) + 2; + } + run("node", ["bin/waltz.mjs", "general", + "--voice", "sinebells", + "--bpm", String(BPM), + "--scale", SCALE, + "--transpose", String(TRANSPOSE_BED), + "--progression", PROGRESSION, + "--density", "0.35", "--gain", "0.18", + "--duration", String(bedDur), + "--seed", `${SLUG}-bp`, + "--out", BED], RECAP); +}); + +step("7 · amix (vocal forward · vox 2.5 · pad 0.55 · bed 0.20)", () => { + // Trim the bed to match the vocal length so it doesn't overhang. + let trimDur; + if (STRETCH) trimDur = computeDuration() + 1; + else { + const w = JSON.parse(readFileSync(WORDS, "utf8")); + trimDur = Math.ceil(w[w.length - 1].toMs / 1000) + 1; + } + const filter = `[0:a]volume=2.5[v];[1:a]volume=0.55[m];[2:a]volume=0.20,atrim=duration=${trimDur}[b];[v][m][b]amix=inputs=3:duration=longest:dropout_transition=0:weights='1 1 1':normalize=0,atrim=duration=${trimDur + 1}`; + run("ffmpeg", ["-y", "-loglevel", "error", + "-i", VOCAL_TRACK, "-i", PAD, "-i", BED, + "-filter_complex", filter, + "-c:a", "libmp3lame", "-q:a", "2", MIX]); +}); + +let TO_FINAL = MIX; +if (STAMP) { + step("8 · stamp (ac signoff · chipmunk + 4-bit crush)", () => { + if (!existsSync(STAMP_VOCAL)) { + const tmpTxt = `/tmp/ac-stamp-${process.pid}.txt`; + writeFileSync(tmpTxt, "aesthetic computer.\n"); + run("node", ["bin/say.mjs", tmpTxt, + "--stability", "0.7", "--similarity", "0.9", "--speed", "0.9", + "--out", `big-pictures/out/ac-stamp-vocal.mp3`]); + } + if (!existsSync(STAMP_FX) || statSync(STAMP_FX).mtimeMs < statSync(STAMP_VOCAL).mtimeMs) { + run("ffmpeg", ["-y", "-loglevel", "error", "-i", STAMP_VOCAL, + "-af", "asetrate=66150,aresample=44100,acrusher=bits=4:mode=lin:aa=1,aformat=sample_rates=22050,aresample=44100,volume=1.6", + STAMP_FX]); + } + const filter = "[0:a]apad=pad_dur=0.5[song];[1:a]volume=1.0[stamp];[song][stamp]concat=n=2:v=0:a=1"; + run("ffmpeg", ["-y", "-loglevel", "error", + "-i", MIX, "-i", STAMP_FX, + "-filter_complex", filter, + "-c:a", "libmp3lame", "-q:a", "2", STAMPED]); + }); + TO_FINAL = STAMPED; +} else { + console.log("▸ 8 · stamp (skipped)\n"); +} + +step("9 · finalize (ID3 + cover + lyrics)", () => { + run("node", ["bin/finalize.mjs", + "--in", TO_FINAL.replace(POP + "/", ""), + "--slug", SLUG, + "--title", TITLE, + "--out", FINAL.replace(POP + "/", ""), + "--force"]); +}); + +copyFileSync(FINAL, DESK); +const dt = ((Date.now() - t0) / 1000).toFixed(1); +const sz = (statSync(DESK).size / 1024 / 1024).toFixed(2); +console.log(`\n✓ ~/Desktop/${SLUG}.mp3 ${sz} MB · pipeline ${dt}s`); diff --git a/pop/big-pictures/frere.np b/pop/big-pictures/frere.np new file mode 100644 index 000000000..511c11786 --- /dev/null +++ b/pop/big-pictures/frere.np @@ -0,0 +1,16 @@ +# Frère Jacques / Are You Sleeping — French traditional round. +# Key: C major. Canonical four-phrase shape, each repeated: +# 1. "are you sleeping" C-D-E-C +# 2. "brother john" E-F-G +# 3. "morning bells are ringing" G-A-G-F-E-C +# 4. "ding ding dong" C-G-C (octave dip) + +verse 1 +C4:are*1 D4:you*1 E4:sleep-*1 C4:-ing*2 +C4:are*1 D4:you*1 E4:sleep-*1 C4:-ing*2 +E4:bro-*1 F4:-ther*1 G4:john*2 +E4:bro-*1 F4:-ther*1 G4:john*2 +G4:morn-*1 A4:-ing*1 G4:bells*1 F4:are*1 E4:ring-*1 C4:-ing*1 +G4:morn-*1 A4:-ing*1 G4:bells*1 F4:are*1 E4:ring-*1 C4:-ing*1 +C4:ding*1 G3:ding*1 C4:dong*2 +C4:ding*1 G3:ding*1 C4:dong*2 diff --git a/pop/big-pictures/frere.txt b/pop/big-pictures/frere.txt new file mode 100644 index 000000000..3d8b065e3 --- /dev/null +++ b/pop/big-pictures/frere.txt @@ -0,0 +1,9 @@ +verse 1 +are you sleeping +are you sleeping +brother john +brother john +morning bells are ringing +morning bells are ringing +ding ding dong +ding ding dong diff --git a/pop/big-pictures/recap-test.np b/pop/big-pictures/recap-test.np new file mode 100644 index 000000000..024bc682f --- /dev/null +++ b/pop/big-pictures/recap-test.np @@ -0,0 +1,56 @@ +# auto-generated rich-melody test .np +# scale=G2 B2 D3 G3 A3 B3 D4 E4 G4 +# arc: cosine half, apex around 55-69%, alternating low/high per phrase +# sustains: hard *2 soft *1 + +verse +G2:Hey*1 B2:ev-*1 D3:-er-*1 A3:-yb-*1 B2:-od-*1 G2:-y*2 +D3:Her-*1 D3:-e's*1 G3:the*1 B3:last*1 B3:24*1 D4:hours*1 E4:at*1 G4:aesth-*1 E4:-et-*1 D4:-ic.c-*1 A3:-omp-*1 A3:-ut-*1 D3:-er*2 +G2:Five*1 B2:small*1 D3:things*1 A3:land-*1 G2:-ed*2 +D3:Let*1 G3:me*1 B3:walk*1 E4:you*1 A3:through*1 D3:them*2 +G3:First*1 +D3:men-*1 A3:-u*1 D3:band*2 +D3:Men-*1 D3:-u*1 G3:band*1 B3:is*1 B3:the*1 D4:men-*1 D4:-u*1 D4:bar*1 A3:instr-*1 D3:-um-*1 D3:-ent*1 +D3:a*1 D3:tin-*1 G3:-y*1 B3:pian-*1 A3:-o*1 B3:that*1 D4:liv-*1 G4:-es*1 E4:in*1 E4:your*1 B3:Mac*1 B3:men-*1 D3:-u*1 D3:bar*1 +D3:so*1 D3:you*1 D3:can*1 A3:fing-*1 G3:-er*1 A3:tap*1 A3:or*1 D4:type*1 D4:play*1 D4:mus-*1 E4:-ic*1 G4:with-*1 G4:-out*1 G4:leav-*1 G4:-ing*1 G4:what-*1 D4:-ev-*1 B3:-er*1 G3:else*1 G3:you're*1 D3:doing*2 +D3:Tod-*1 G3:-ay*1 A3:it*1 E4:grew*1 D4:a*1 B3:live*1 G3:sheet*1 G3:card*2 +G2:Hold*1 G3:a*1 G2:chord*1 +D3:and*1 D3:Var-*1 D3:-ov-*1 A3:-io*1 G3:engr-*1 A3:-av-*1 A3:-es*1 D4:it*1 D4:as*1 D4:real*1 E4:staff*1 G4:not-*1 G4:-at-*1 G4:-ion*1 G4:right*1 G4:ins-*1 D4:-ide*1 B3:the*1 G3:pop-*1 G3:-ov-*1 D3:-er*2 +D3:That*1 D3:means*1 G3:mus-*1 D4:-ic-*1 D4:-ians*1 D4:can*1 B3:read*1 B3:al-*1 G3:-ong*1 D3:now*1 +D3:not*1 A3:just*1 B3:cod-*1 G3:-ers*2 +G2:Same*1 B2:instr-*1 G3:-um-*1 B2:-ent*1 +D3:small-*1 A3:-er*1 B3:pol-*1 G3:-ish*1 +D3:the*1 D3:pop-*1 G3:-ov-*1 B3:-er*1 A3:glass*1 B3:now*1 D4:tints*1 G4:to*1 E4:the*1 E4:act-*1 B3:-ive*1 B3:voice*1 D3:col-*1 D3:-or*1 +D3:so*1 D3:the*1 D3:chrome*1 G3:its-*1 G3:-elf*1 G3:tells*1 G3:you*1 B3:which*1 A3:voice*1 B3:is*1 B3:in*1 D4:front*1 D4:bef-*1 D4:-ore*1 E4:you've*1 G4:ev-*1 E4:-en*1 G4:look-*1 G4:-ed*1 G4:at*1 G4:the*1 G4:lab-*1 E4:-el*1 G4:for*1 D4:fast*1 A3:switch-*1 A3:-ing*1 A3:that*1 D3:matt-*1 D3:-ers*2 +D3:All*1 D3:of*1 G3:that*1 B3:pol-*1 B3:-ish*1 D4:went*1 D4:int-*1 D4:-o*1 A3:a*1 D3:rel-*1 D3:-ease*2 +D3:Men-*1 G3:-u*1 A3:band*1 D4:0.09.3*1 D4:shipp-*1 B3:-ed*1 G3:tod-*1 G3:-ay*1 +G2:sign-*1 G2:-ed*1 +D3:not-*1 A3:-ar-*1 B3:-iz-*1 G3:-ed*1 +G2:stapl-*1 G2:-ed*2 +D3:If*1 G3:you've*1 A3:got*1 D4:it*1 D4:inst-*1 A3:-all-*1 D3:-ed*1 +D3:it'll*1 D3:aut-*1 A3:-o*1 D4:upd-*1 D4:-ate*1 D4:ov-*1 A3:-er*1 A3:the*1 D3:night*2 +D3:Zoom-*1 D3:-ing*1 G3:way*1 D4:out*1 D4:to*1 D4:the*1 B3:writ-*1 A3:-ing*1 D3:bench*1 +D3:the*1 D3:coll-*1 D3:-ard*1 A3:ccat*1 G3:tech*1 A3:dir-*1 B3:-ect-*1 D4:-or*1 D4:doss-*1 E4:-ier*1 E4:pick-*1 G4:-ed*1 G4:up*1 G4:its*1 G4:ill-*1 G4:-ustr-*1 B3:-at-*1 A3:-ed*1 D3:head-*1 G3:-er*1 +D3:Sage*1 D3:Jens-*1 G3:-en*1 A3:got*1 A3:wir-*1 B3:-ed*1 D4:int-*1 G4:-o*1 E4:the*1 E4:proj-*1 D4:-ect-*1 D4:-ion*1 A3:mapp-*1 G3:-ing*1 D3:lin-*1 G3:-eage*1 +D3:and*1 D3:a*1 D3:fact*1 A3:check*1 A3:pass-*1 B3:-ed*1 B3:surf-*1 E4:-ace*1 E4:the*1 E4:giff-*1 G4:-y*1 G4:and*1 G4:link-*1 E4:-ed*1 D4:by*1 B3:air*1 G3:cred-*1 D3:-ent-*1 D3:-ials*1 +D3:and*1 D3:refr-*1 G3:-am-*1 A3:-ed*1 A3:cadd-*1 B3:-est*1 B3:plus*1 E4:smk*1 E4:as*1 E4:coll-*1 E4:-ect-*1 E4:-ion*1 A3:plac-*1 G3:-em-*1 D3:-ents*2 +D3:Tech*1 D3:dir-*1 G3:-ect-*1 B3:-or*1 B3:res-*1 D4:-id-*1 D4:-enc-*1 G4:-ies*1 E4:are*1 D4:how*1 D4:the*1 B3:pract-*1 G3:-ice*1 D3:scal-*1 D3:-es*2 +D3:Stud-*1 D3:-ents*1 G3:inh-*1 B3:-er-*1 B3:-it*1 D4:the*1 E4:toolk-*1 G4:-it*1 D4:and*1 A3:ext-*1 G3:-end*1 G3:it*2 +G2:On*1 B2:the*1 D3:prod-*1 A3:-uct-*1 D3:-ion*1 G2:side*1 +D3:the*1 D3:mark-*1 D3:-et-*1 A3:-ing*1 G3:fold-*1 A3:-er*1 B3:fin-*1 E4:-all-*1 D4:-y*1 E4:grad-*1 G4:-uat-*1 G4:-ed*1 G4:out*1 G4:of*1 E4:the*1 G4:deskt-*1 D4:-op*1 B3:and*1 A3:int-*1 A3:-o*1 G3:the*1 D3:rep-*1 D3:-o*2 +D3:Gen*1 D3:prom-*1 G3:-o*1 B3:as*1 A3:the*1 B3:centr-*1 D4:-al*1 G4:im-*1 E4:-ag-*1 E4:-e-g-*1 B3:-en*1 B3:entr-*1 D3:-y*1 D3:point*1 +D3:a*1 D3:shar-*1 G3:-ed*1 B3:id-*1 A3:-ent-*1 B3:-it-*1 D4:-y*1 E4:ref's*1 B3:libr-*1 G3:-ar-*1 D3:-y*1 +G2:a*1 B2:script*1 G3:that*1 G3:pulls*1 G2:real*1 +D3:act-*1 G3:-ive*1 B3:capt-*1 B3:-ur-*1 D3:-es*1 +G2:and*1 G2:the*1 B2:first*1 G3:two*1 G3:seed*1 D3:camp-*1 G2:-aigns*2 +D3:That's*1 D3:the*1 G3:front*1 A3:door*1 A3:for*1 B3:ev-*1 D4:-er-*1 G4:-yone*1 E4:who*1 G4:hasn't*1 G4:typ-*1 G4:-ed*1 E4:a*1 D4:comm-*1 B3:-and*1 B3:at*1 G3:the*1 D3:prompt*1 D3:yet*2 +G2:And*1 G3:to*1 G2:close*1 +D3:a*1 D3:paint-*1 D3:-ing*1 A3:hang-*1 A3:-ing*1 A3:in*1 B3:the*1 E4:Ven-*1 D4:-ice*1 E4:Fam-*1 E4:-il-*1 G4:-y*1 D4:Clin-*1 A3:-ic*1 G3:auct-*1 G3:-ion*1 +G2:its*1 B2:QR*1 D3:code*1 A3:us-*1 G3:-ed*1 B2:to*1 G2:404*2 +D3:It*1 D3:now*1 G3:res-*1 B3:-olv-*1 B3:-es*1 D4:to*1 D4:a*1 D4:real*1 G3:AC*1 D3:piece*1 +D3:the*1 D3:draw-*1 G3:-ing*1 A3:breath-*1 A3:-es*1 B3:on*1 D4:a*1 G4:slow*1 E4:cycl-*1 E4:-ing*1 B3:purple*1 A3:backgr-*1 D3:-ound*1 +D3:and*1 D3:a*1 A3:tap*1 D4:op-*1 D4:-ens*1 D4:the*1 A3:auct-*1 A3:-ion*1 D3:lot*2 +G2:One*1 B2:tin-*1 D3:-y*1 A3:art-*1 D3:-if-*1 G2:-act*1 +D3:the*1 D3:whole*1 G3:shape*1 B3:of*1 A3:the*1 B3:proj-*1 D4:-ect*1 E4:in*1 B3:min-*1 G3:-iat-*1 D3:-ure*2 +G2:That's*1 G3:the*1 G2:24*2 +D3:Thanks*1 A3:for*1 B3:watch-*1 G3:-ing*2 diff --git a/pop/big-pictures/small-world.np b/pop/big-pictures/small-world.np new file mode 100644 index 000000000..2a70d6c41 --- /dev/null +++ b/pop/big-pictures/small-world.np @@ -0,0 +1,27 @@ +# It's a Small World — Sherman Brothers, Disney 1964. +# Melody sourced from Online Sequencer #509647 (instrument 14, "main +# singing voice starting on C5"). Key: F major. +# +# CHORUS PHRASE (sol-re-do-FA-fa-fa-mi shape, climax sustained on FA): +# "It's-a small WORLD af -ter all" +# C5 G4 F4 Bb4(2) Bb4 Bb4 A4 +# sol re do fa fa fa mi +# +# TAG: descending stepwise to tonic. +# "It's-a-small-small-world" +# G4 F4 E4 D4 F4 +# +# VERSE: derived from the same shape, opening lower for contrast and +# climbing into the chorus phrase. +# +# Range: D4-C5. With CLI transpose -5 → A3-G4 (jeffrey-pvc tenor). + +verse 1 +F4:its*1 F4:a*1 G4:world*1 G4:of*1 A4:laugh-*1 A4:-ter*2 G4:a*1 A4:world*1 G4:of*1 F4:tears*3 +F4:its*1 F4:a*1 G4:world*1 G4:of*1 A4:hopes*1 A4:and*1 G4:a*1 A4:world*1 G4:of*1 F4:fears*3 +G4:theres*1 A4:so*1 A#4:much*1 A#4:that*1 A#4:we*1 A4:share*2 G4:that*1 A4:its*1 A#4:time*1 A4:were*1 G4:a-*1 F4:-ware*3 +C5:its*1 G4:a*1 F4:small*1 A#4:world*2 A#4:af-*1 A#4:-ter*1 A4:all*3 + +chorus +C5:its*1 G4:a*1 F4:small*1 A#4:world*2 A#4:af-*1 A#4:-ter*1 A4:all*3 +G4:its*1 F4:a*1 E4:small*1 D4:small*1 F4:world*4 diff --git a/pop/big-pictures/small-world.txt b/pop/big-pictures/small-world.txt new file mode 100644 index 000000000..1dcf907e5 --- /dev/null +++ b/pop/big-pictures/small-world.txt @@ -0,0 +1,8 @@ +verse 1 +its a world of laughter a world of tears +its a world of hopes and a world of fears +theres so much that we share that its time were aware +its a small world after all + +chorus +its a small small world diff --git a/pop/big-pictures/twinkle.np b/pop/big-pictures/twinkle.np new file mode 100644 index 000000000..6beb1c384 --- /dev/null +++ b/pop/big-pictures/twinkle.np @@ -0,0 +1,22 @@ +# Twinkle Twinkle Little Star — canonical nursery-rhyme melody. +# Mozart variations / Bourée / "Ah! vous dirai-je, maman" +# Key: C major. Range: C4–A4. 4/4, last note of each phrase doubled. +# +# Words: twinkle twinkle little star +# Syllables: twin-kle twin-kle li-ttle star +# Notes: C4 C4 G4 G4 A4 A4 G4*2 +# +# notation: NOTE:syllable*beats +# Multi-syllable words use leading hyphen on continuation syllables. +# +# bpm: 80, 4/4. Each syllable = 1 beat, phrase ends = 2 beats. + +verse 1 +C4:twin-*1 C4:-kle*1 G4:twin-*1 G4:-kle*1 A4:li-*1 A4:-ttle*1 G4:star*2 +F4:how*1 F4:i*1 E4:won-*1 E4:-der*1 D4:what*1 D4:you*1 C4:are*2 +G4:up*1 G4:a-*1 F4:-bove*1 F4:the*1 E4:world*1 E4:so*1 D4:high*2 +G4:like*1 G4:a*1 F4:dia-*1 F4:-mond*1 E4:in*1 E4:the*1 D4:sky*2 + +verse 2 +C4:twin-*1 C4:-kle*1 G4:twin-*1 G4:-kle*1 A4:li-*1 A4:-ttle*1 G4:star*2 +F4:how*1 F4:i*1 E4:won-*1 E4:-der*1 D4:what*1 D4:you*1 C4:are*2 diff --git a/pop/big-pictures/twinkle.txt b/pop/big-pictures/twinkle.txt new file mode 100644 index 000000000..8a19be0b4 --- /dev/null +++ b/pop/big-pictures/twinkle.txt @@ -0,0 +1,9 @@ +verse 1 +twinkle twinkle little star +how i wonder what you are +up above the world so high +like a diamond in the sky + +verse 2 +twinkle twinkle little star +how i wonder what you are diff --git a/pop/bin/align-words.mjs b/pop/bin/align-words.mjs new file mode 100644 index 000000000..83dc20f3d --- /dev/null +++ b/pop/bin/align-words.mjs @@ -0,0 +1,122 @@ +// align-words.mjs — reconcile whisper word timestamps against the +// canonical score word list. +// +// Whisper occasionally splits/merges/substitutes words ("twas" → "to as", +// "tis" → "to his", "am" → "I'm") which cascades the entire downstream +// pitch alignment. This module walks both sequences and produces, for +// each score word, a (fromMs, toMs) window from the source audio. +// +// Strategy: greedy walk with three operations: +// match — current pair are the same word (token-equal or fuzzy) +// merge — score word == whisper[j] + whisper[j+1] concatenated +// (covers "twas" = "to" + "as") +// skip-w — whisper has an inserted/spurious word; skip it +// (covers whisper hearing "to his" where score expects "tis") +// +// Falls back to taking the current whisper time as a best-effort guess +// when no rule fires, so we never lose sync entirely. +// +// export: alignWords(scoreWords: string[], whisperWords: {text,fromMs,toMs}[]) +// → { fromMs, toMs }[] of length scoreWords.length + +const norm = (s) => (s || "").toLowerCase().replace(/[^a-z']/g, ""); + +// Levenshtein with early reject. Returns true iff edit distance > max. +function editsExceed(a, b, max) { + if (Math.abs(a.length - b.length) > max) return true; + const m = a.length, n = b.length; + let prev = new Array(n + 1); + let curr = new Array(n + 1); + for (let j = 0; j <= n; j++) prev[j] = j; + for (let i = 1; i <= m; i++) { + curr[0] = i; + let row_min = i; + for (let j = 1; j <= n; j++) { + curr[j] = a[i - 1] === b[j - 1] + ? prev[j - 1] + : 1 + Math.min(prev[j - 1], prev[j], curr[j - 1]); + if (curr[j] < row_min) row_min = curr[j]; + } + if (row_min > max) return true; + [prev, curr] = [curr, prev]; + } + return prev[n] > max; +} + +function fuzzyMatch(a, b) { + if (!a || !b) return false; + if (a === b) return true; + const ax = a.replace(/'/g, ""), bx = b.replace(/'/g, ""); + if (ax === bx) return true; + // Short words (≤ 4 chars): allow 1 edit. Longer: allow up to 2. + const maxLen = Math.max(ax.length, bx.length); + const tol = maxLen <= 4 ? 1 : 2; + if (!editsExceed(ax, bx, tol)) return true; + if (a.length >= 3 && b.length >= 3 && (a.startsWith(b) || b.startsWith(a))) return true; + return false; +} + +// Merge tolerance — score word vs concatenated whisper pair. More +// permissive (covers "twas" ≈ "toas", "tis" ≈ "tohis"). +function fuzzyMerge(scoreW, mergedW) { + if (scoreW === mergedW) return true; + const a = scoreW.replace(/'/g, ""), b = mergedW.replace(/'/g, ""); + if (a === b) return true; + // Allow edits up to half the merged length (e.g. tohis→tis = 2 edits, length 5) + const tol = Math.min(3, Math.ceil(Math.max(a.length, b.length) / 2)); + return !editsExceed(a, b, tol); +} + +export function alignWords(scoreWords, whisperWords) { + const out = new Array(scoreWords.length); + let i = 0, j = 0; + let lastEnd = 0; + while (i < scoreWords.length) { + if (j >= whisperWords.length) { + // Out of whisper data — extrapolate from last + out[i] = { fromMs: lastEnd, toMs: lastEnd + 200 }; + lastEnd += 200; i++; continue; + } + const s = norm(scoreWords[i]); + const w = norm(whisperWords[j].text); + // 1. direct match + if (fuzzyMatch(s, w)) { + out[i] = { fromMs: whisperWords[j].fromMs, toMs: whisperWords[j].toMs }; + lastEnd = whisperWords[j].toMs; + i++; j++; continue; + } + // 2. merge: score == whisper[j] + whisper[j+1] (with fuzzy tolerance) + if (j + 1 < whisperWords.length) { + const merged = w + norm(whisperWords[j + 1].text); + if (fuzzyMerge(s, merged)) { + out[i] = { fromMs: whisperWords[j].fromMs, toMs: whisperWords[j + 1].toMs }; + lastEnd = whisperWords[j + 1].toMs; + i++; j += 2; continue; + } + } + // 3. skip whisper if its NEXT word matches the current score word + if (j + 1 < whisperWords.length) { + const next = norm(whisperWords[j + 1].text); + if (fuzzyMatch(s, next)) { + j++; continue; // skip the spurious whisper word + } + // Or score[i] + score[i+1] == whisper[j] (rare 1-to-2 split) + if (i + 1 < scoreWords.length) { + const merged_score = s + norm(scoreWords[i + 1]); + if (fuzzyMatch(merged_score, w)) { + // share whisper window across two score words + const mid = Math.floor((whisperWords[j].fromMs + whisperWords[j].toMs) / 2); + out[i] = { fromMs: whisperWords[j].fromMs, toMs: mid }; + out[i + 1] = { fromMs: mid, toMs: whisperWords[j].toMs }; + lastEnd = whisperWords[j].toMs; + i += 2; j++; continue; + } + } + } + // 4. fallback — accept the current whisper as best guess + out[i] = { fromMs: whisperWords[j].fromMs, toMs: whisperWords[j].toMs }; + lastEnd = whisperWords[j].toMs; + i++; j++; + } + return out; +} diff --git a/pop/bin/best-of-takes.mjs b/pop/bin/best-of-takes.mjs new file mode 100644 index 000000000..d3177ebb1 --- /dev/null +++ b/pop/bin/best-of-takes.mjs @@ -0,0 +1,229 @@ +#!/usr/bin/env node +// best-of-takes.mjs — distributed source generation for pitchsnap. +// +// Why: a single ElevenLabs take of the whole verse compresses each +// word's natural duration. A per-word take strips natural prosody. +// Per-line takes have the prosody but force a fixed source length. +// +// Solution: generate MULTIPLE overlapping phrase takes, each giving +// ElevenLabs different context, then for each canonical lyric word +// pick the take where that word has the LONGEST natural source +// duration. Splice the picked occurrences together with crossfade. +// +// The user's framing: "overlapping phrases of the lyrics, then choose +// best word — a distributed method". +// +// Strategy: +// • Generate 3 full-verse takes with varied style/stability seeds, +// plus 2 phrase-window takes (lines 1+2, lines 3+4) with longer +// speed (when allowed by ElevenLabs clamp). +// • Whisper-align each, get per-word boundaries. +// • For each canonical lyric word position, pick the take whose +// occurrence of that word has the longest natural duration. +// • Extract each picked segment via ffmpeg. +// • Concat with 200ms crossfade between segments. +// +// Output: ${slug}-bestof.mp3 + ${slug}-bestof-words.json (synthesized). + +import { execSync, spawnSync } from "node:child_process"; +import { writeFileSync, readFileSync, existsSync, mkdtempSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; + +const flags = {}; +for (let i = 0; i < process.argv.length; i++) { + const a = process.argv[i]; + if (!a.startsWith("--")) continue; + const next = process.argv[i + 1]; + if (next === undefined || next.startsWith("--")) flags[a.slice(2)] = true; + else flags[a.slice(2)] = next; +} + +const SLUG = flags.slug || "amazing"; +const POP = "/Users/jas/aesthetic-computer/pop"; +const LYRIC_PATH = `${POP}/big-pictures/${SLUG}.txt`; +const SECTION = (flags.section || "verse 1").toLowerCase(); +const N_TAKES = Number(flags["n-takes"] ?? 5); +const CROSSFADE_MS = Number(flags["crossfade-ms"] ?? 80); +const OUT_PATH = flags.out || `${POP}/big-pictures/out/${SLUG}-bestof.mp3`; + +if (!existsSync(LYRIC_PATH)) { console.error(`✗ lyrics missing: ${LYRIC_PATH}`); process.exit(1); } + +// ── Parse lyric into a flat word array ─────────────────────────────── +const lyricLines = readFileSync(LYRIC_PATH, "utf8").split("\n"); +const lStart = lyricLines.findIndex((l) => l.trim().toLowerCase() === SECTION); +if (lStart < 0) { console.error(`✗ section '${SECTION}' missing in lyrics`); process.exit(1); } +const lines = []; +for (let i = lStart + 1; i < lyricLines.length; i++) { + const t = lyricLines[i].trim(); + if (!t) break; + if (/^[a-z]+ \d/i.test(t)) break; + lines.push(t); +} +const canonicalWords = lines.flatMap(l => l.split(/\s+/).map(w => w.toLowerCase().replace(/[^a-z']/g, ""))); +console.log(`→ canonical: ${canonicalWords.length} words across ${lines.length} lines`); + +// ── Define takes: 3 full verses + 2 phrase windows ─────────────────── +// Each take has a unique style/stability seed for prosodic variation. +const takes = [ + { name: "full-A", text: lines.join(", "), style: "0.55", stability: "0.70", lineRange: [0, lines.length] }, + { name: "full-B", text: lines.join(", "), style: "0.65", stability: "0.65", lineRange: [0, lines.length] }, + { name: "full-C", text: lines.join(", "), style: "0.45", stability: "0.75", lineRange: [0, lines.length] }, +]; +// Phrase windows for first / second halves with overlap on middle line. +if (lines.length >= 3) { + takes.push({ + name: "win-12", text: lines.slice(0, 2).join(", "), + style: "0.55", stability: "0.70", lineRange: [0, 2], + }); + takes.push({ + name: "win-23", text: lines.slice(1, 3).join(", "), + style: "0.60", stability: "0.70", lineRange: [1, 3], + }); + if (lines.length >= 4) { + takes.push({ + name: "win-34", text: lines.slice(2, 4).join(", "), + style: "0.55", stability: "0.70", lineRange: [2, 4], + }); + } +} +takes.length = Math.min(takes.length, N_TAKES); + +// ── Generate each take via say.mjs ─────────────────────────────────── +const tmp = mkdtempSync(`${tmpdir()}/bestof-${SLUG}-`); +const takeData = []; +for (const t of takes) { + const txtFile = `${tmp}/${t.name}.txt`; + writeFileSync(txtFile, `verse 1\n${t.text}\n`); + const mp3 = `${POP}/big-pictures/out/${SLUG}-take-${t.name}.mp3`; + console.log(`→ generate ${t.name} (style=${t.style} stab=${t.stability}): "${t.text.slice(0, 50)}…"`); + const r = spawnSync("node", [ + `${POP}/bin/say.mjs`, txtFile, + "--speed", "0.7", "--style", t.style, "--stability", t.stability, + "--out", mp3, "--force", + ], { stdio: ["ignore", "ignore", "inherit"] }); + if (r.status !== 0) { console.error(`✗ say failed for ${t.name}`); process.exit(1); } + // Whisper align + const wordsJson = mp3.replace(/\.mp3$/, "-words.json"); + spawnSync("rm", ["-f", wordsJson, wordsJson + ".hash"]); + spawnSync("node", [`${POP}/bin/align.mjs`, mp3, "--force"], { stdio: ["ignore", "ignore", "inherit"] }); + if (!existsSync(wordsJson)) { console.error(`✗ align failed for ${t.name}`); process.exit(1); } + const words = JSON.parse(readFileSync(wordsJson, "utf8")); + // Tag each word with the take's lineRange so we know which canonical + // words this take is allowed to contribute to. + takeData.push({ name: t.name, mp3, words, lineRange: t.lineRange }); + console.log(` whisper detected ${words.length} words`); +} + +// ── For each canonical word, find best occurrence across takes ─────── +// "Best" = longest natural duration (gives pitchsnap minimal stretching). +// Constrain candidates to takes whose lineRange includes the word's +// canonical line. +function lineForWordIdx(idx) { + let cum = 0; + for (let li = 0; li < lines.length; li++) { + const wc = lines[li].split(/\s+/).length; + if (idx < cum + wc) return li; + cum += wc; + } + return lines.length - 1; +} +function normalize(s) { return s.toLowerCase().replace(/[^a-z']/g, ""); } + +const bestPicks = []; // [{ canonicalWord, takeIdx, wordIdx, fromMs, toMs }] +for (let ci = 0; ci < canonicalWords.length; ci++) { + const cw = canonicalWords[ci]; + const cline = lineForWordIdx(ci); + // Candidates: from each take whose lineRange covers cline, + // find an unclaimed whisper word whose normalized text matches cw, + // in the take's word-order region appropriate for ci. + const candidates = []; + for (let ti = 0; ti < takeData.length; ti++) { + const t = takeData[ti]; + if (cline < t.lineRange[0] || cline >= t.lineRange[1]) continue; + // Compute this canonical word's index within the take's lyric scope. + const lineWordsBefore = lines.slice(t.lineRange[0], cline) + .reduce((acc, l) => acc + l.split(/\s+/).length, 0); + const wordsBeforeInLine = ci - lines.slice(0, cline) + .reduce((acc, l) => acc + l.split(/\s+/).length, 0); + const expectedTakeIdx = lineWordsBefore + wordsBeforeInLine; + // Look in the take's whisper output near expectedTakeIdx for a word + // matching cw. Allow ±2 slack for whisper drift / merged tokens. + for (let wi = Math.max(0, expectedTakeIdx - 2); wi < Math.min(t.words.length, expectedTakeIdx + 3); wi++) { + const w = t.words[wi]; + if (normalize(w.text) === cw) { + const dur = w.toMs - w.fromMs; + candidates.push({ takeIdx: ti, wordIdx: wi, dur, fromMs: w.fromMs, toMs: w.toMs }); + break; // take first match within slack window + } + } + } + if (candidates.length === 0) { + console.warn(`⚠ no match for canonical "${cw}" (idx ${ci}) in any take`); + continue; + } + // Pick longest natural duration + candidates.sort((a, b) => b.dur - a.dur); + const pick = candidates[0]; + bestPicks.push({ canonicalWord: cw, ...pick }); +} +console.log(`→ matched ${bestPicks.length}/${canonicalWords.length} canonical words across takes`); + +// ── Extract & concat picked segments via ffmpeg ────────────────────── +// Use a tiny bit of pre-roll/post-roll on each cut so plosives aren't +// chopped, then concat with a brief crossfade. +const PRE_PAD_MS = 30; +const POST_PAD_MS = 60; +const segments = []; +const wordBoundaries = []; // for synthesized words.json +let cursor_ms = 0; +for (let pi = 0; pi < bestPicks.length; pi++) { + const p = bestPicks[pi]; + const t = takeData[p.takeIdx]; + const cutFrom = Math.max(0, p.fromMs - PRE_PAD_MS) / 1000.0; + const cutTo = (p.toMs + POST_PAD_MS) / 1000.0; + const cutDur = cutTo - cutFrom; + const segMp3 = `${tmp}/seg-${String(pi).padStart(2, "0")}.mp3`; + const r = spawnSync("ffmpeg", [ + "-hide_banner", "-y", "-loglevel", "error", + "-ss", String(cutFrom), "-i", t.mp3, "-t", String(cutDur), + "-af", "afade=t=in:st=0:d=0.02,afade=t=out:st=" + Math.max(0, cutDur - 0.04).toFixed(3) + ":d=0.04", + "-c:a", "libmp3lame", "-q:a", "4", segMp3, + ]); + if (r.status !== 0) { console.error(`✗ ffmpeg cut failed for word ${p.canonicalWord}`); continue; } + segments.push(segMp3); + // Word boundary in concat: word audio is offset PRE_PAD_MS into the segment. + wordBoundaries.push({ + text: p.canonicalWord, + fromMs: cursor_ms + PRE_PAD_MS, + toMs: cursor_ms + PRE_PAD_MS + (p.toMs - p.fromMs), + }); + cursor_ms += Math.round(cutDur * 1000); +} + +const concatTxt = `${tmp}/concat.txt`; +writeFileSync(concatTxt, segments.map(s => `file '${s}'`).join("\n") + "\n"); +const r = spawnSync("ffmpeg", [ + "-hide_banner", "-y", "-loglevel", "error", + "-f", "concat", "-safe", "0", "-i", concatTxt, + "-c:a", "libmp3lame", "-q:a", "4", OUT_PATH, +]); +if (r.status !== 0) { console.error("✗ concat failed"); process.exit(1); } + +const wordsJsonPath = OUT_PATH.replace(/\.mp3$/, "-words.json"); +writeFileSync(wordsJsonPath, JSON.stringify(wordBoundaries, null, 2)); +spawnSync("rm", ["-f", wordsJsonPath + ".hash"]); + +const outDur = execSync(`ffprobe -v error -show_entries format=duration -of default=noprint_wrappers=1:nokey=1 ${OUT_PATH}`).toString().trim(); +console.log(`✓ ${OUT_PATH} (${Number(outDur).toFixed(2)}s)`); +console.log(`✓ ${wordsJsonPath} (${wordBoundaries.length} words)`); +console.log(); +console.log("=== picks summary ==="); +const takeUsage = {}; +for (const p of bestPicks) { + takeUsage[takeData[p.takeIdx].name] = (takeUsage[takeData[p.takeIdx].name] ?? 0) + 1; +} +for (const [n, c] of Object.entries(takeUsage)) { + console.log(` ${n}: ${c} word${c === 1 ? "" : "s"} picked`); +} + +rmSync(tmp, { recursive: true, force: true }); diff --git a/pop/bin/detect_onsets.py b/pop/bin/detect_onsets.py new file mode 100644 index 000000000..73390a4a1 --- /dev/null +++ b/pop/bin/detect_onsets.py @@ -0,0 +1,58 @@ +#!/usr/bin/env python3 +"""detect_onsets.py — emit a flat array of librosa onsets for the +whole audio so downstream tools can pick syllable boundaries within +known word windows. + +Output shape (JSON array): + [{"ms": 60, "strength": 0.83}, {"ms": 320, "strength": 1.4}, ...] + +`strength` is the onset_strength sample at the detection frame, so +consumers can prefer LOUDER onsets when they need to pick fewer than +detected (e.g. "this word should have 2 syllable boundaries, but I +detected 4 onsets — keep the strongest 2"). + +Usage: + detect_onsets.py audio.mp3 out.json [--hop 256] [--sr 22050] +""" +import argparse +import json +import sys +from pathlib import Path + +import numpy as np +import librosa + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("audio") + ap.add_argument("out") + ap.add_argument("--hop", type=int, default=256) + ap.add_argument("--sr", type=int, default=22050) + args = ap.parse_args() + + p = Path(args.audio) + if not p.exists(): + print(f"✗ audio missing: {p}", file=sys.stderr); return 1 + + y, sr = librosa.load(str(p), sr=args.sr) + onset_env = librosa.onset.onset_strength(y=y, sr=sr, hop_length=args.hop) + onset_frames = librosa.onset.onset_detect( + onset_envelope=onset_env, sr=sr, hop_length=args.hop, + backtrack=True, # walk to local energy min + ) + if len(onset_frames) == 0: + Path(args.out).write_text("[]") + print("⚠ no onsets detected"); return 0 + + times_s = librosa.frames_to_time(onset_frames, sr=sr, hop_length=args.hop) + strengths = onset_env[np.clip(onset_frames, 0, len(onset_env) - 1)] + + rows = [{"ms": int(round(t * 1000)), "strength": float(s)} + for t, s in zip(times_s, strengths)] + Path(args.out).write_text(json.dumps(rows)) + smin, smax, smean = float(strengths.min()), float(strengths.max()), float(strengths.mean()) + print(f" detected {len(rows)} onsets · strength range [{smin:.2f}, {smax:.2f}] · mean {smean:.2f}") + return 0 + +if __name__ == "__main__": + sys.exit(main()) diff --git a/pop/bin/melody-bells.mjs b/pop/bin/melody-bells.mjs new file mode 100644 index 000000000..eb212b050 --- /dev/null +++ b/pop/bin/melody-bells.mjs @@ -0,0 +1,275 @@ +#!/usr/bin/env node +// melody-bells.mjs — render the .np score as a melodic sinebells line. +// +// The waltz bed (recap/bin/waltz.mjs) only carries chord roots + a +// random ornament; it doesn't play THE melody. This script reads the +// notepat score and renders each syllable as a struck bell at the +// score's note (with optional --transpose), so the accompaniment +// doubles the New Britain tune above the vocal. +// +// Bell synth lifted verbatim from recap/bin/waltz.mjs (sinebells voice) +// — fundamental + lightly inharmonic partials, cosine attack, T60 decay. +// +// Usage: +// node bin/melody-bells.mjs --slug amazing --section all \ +// --transpose 0 --bpm 70 --inter-verse-beats 2 \ +// --gain 0.35 --out big-pictures/out/amazing-melody-bells.mp3 + +import { spawnSync } from "node:child_process"; +import { readFileSync, writeFileSync, mkdtempSync, rmSync } from "node:fs"; +import { resolve } from "node:path"; +import { tmpdir } from "node:os"; +import { alignWords } from "./align-words.mjs"; + +const flags = {}; +for (let i = 0; i < process.argv.length; i++) { + const a = process.argv[i]; + if (!a.startsWith("--")) continue; + const next = process.argv[i + 1]; + if (next === undefined || next.startsWith("--")) flags[a.slice(2)] = true; + else flags[a.slice(2)] = next; +} + +const POP = "/Users/jas/aesthetic-computer/pop"; +const SLUG = flags.slug || "amazing"; +const SCORE_PATH = `${POP}/big-pictures/${SLUG}.np`; +const SECTION = (flags.section || "all").toLowerCase(); +const BPM = Number(flags.bpm ?? 70); +const TRANSPOSE = Number(flags.transpose ?? 0); +const INTER_VERSE_BEATS = Number(flags["inter-verse-beats"] ?? 2); +const GAIN = Number(flags.gain ?? 0.35); +// Voice character: +// bell — sinebells with inharmonic partials + sharp attack (default — +// the same synth recap/bin/waltz.mjs uses for the waltz bed) +// pad — soft sine pad: fundamental + octave only, slow cosine attack, +// gentle release. No "ding". Better for sustained accompaniment. +const VOICE = (flags.voice || "bell").toLowerCase(); +const OUT_PATH = flags.out + ? resolve(process.cwd(), flags.out) + : `${POP}/big-pictures/out/${SLUG}-melody-bells.mp3`; + +const SAMPLE_RATE = 48_000; + +const NOTE_BASE = { C:0,"C#":1,DB:1,D:2,"D#":3,EB:3,E:4,F:5, + "F#":6,GB:6,G:7,"G#":8,AB:8,A:9,"A#":10,BB:10,B:11 }; +function noteToMidi(s) { + s = s.toUpperCase(); + const oct = parseInt(s.slice(-1), 10); + return 12 * (oct + 1) + NOTE_BASE[s.slice(0, -1)]; +} + +// ── parse score (mirrors score-render.mjs --section all path) ──────── +const scoreLines = readFileSync(SCORE_PATH, "utf8").split("\n"); +const syllables = []; +function parseRange(startIdx) { + for (let i = startIdx; i < scoreLines.length; i++) { + const l = scoreLines[i].trim(); + if (!l) continue; + if (l.startsWith("#")) continue; + if (/^[a-z]+ \d+$/i.test(l)) break; + for (const tok of l.split(/\s+/)) { + const m = tok.match(/^([A-Ga-g][#b]?-?\d):(.+?)(?:\*(\d+(?:\.\d+)?))?$/); + if (!m) continue; + syllables.push({ note: m[1], raw: m[2], weight: Number(m[3] ?? 1) }); + } + } +} +if (SECTION === "all") { + const headers = []; + for (let i = 0; i < scoreLines.length; i++) { + if (/^verse \d+$/i.test(scoreLines[i].trim())) headers.push(i); + } + headers.forEach((h, idx) => { + const before = syllables.length; + parseRange(h + 1); + if (idx < headers.length - 1 && syllables.length > before && INTER_VERSE_BEATS > 0) { + syllables[syllables.length - 1].weight += INTER_VERSE_BEATS; + } + }); +} else { + const sStart = scoreLines.findIndex((l) => l.trim().toLowerCase() === SECTION); + if (sStart < 0) { console.error(`✗ section missing: ${SECTION}`); process.exit(1); } + parseRange(sStart + 1); +} + +// Build word-level event list. Two timing modes: +// +// 1. score-driven (default): hyphenated tail merges into head word; +// each strike fires at beat_pos * beat_s. +// 2. whisper-driven (--whisper-words file.json): one strike per +// whisper word at its fromMs, using the score's per-word target +// note. Length-aligned by index when counts differ. +const events = []; +const WHISPER_WORDS = flags["whisper-words"] + ? resolve(process.cwd(), flags["whisper-words"]) + : null; + +if (WHISPER_WORDS) { + // Group syllables → score-words, keeping per-syllable notes + weights + // so multi-note words ("amazing" D3-G3-B3, "above" G4-F4) get a + // strike per syllable distributed across the whisper word window. + const scoreWords = []; + let cur = null; + for (const s of syllables) { + if (s.raw.startsWith("-") && cur) { + cur.notes.push(s.note); + cur.weights.push(s.weight); + } else { + if (cur) scoreWords.push(cur); + cur = { notes: [s.note], weights: [s.weight] }; + } + } + if (cur) scoreWords.push(cur); + const whisper = JSON.parse(readFileSync(WHISPER_WORDS, "utf8")); + // Reconcile contractions ("twas" ≠ "to as", "tis" ≠ "to his") via + // the shared aligner so each strike fires on the right word. + const tokens = []; + let acc = ""; + for (const s of syllables) { + if (s.raw.startsWith("-")) acc += s.raw.replace(/-/g, ""); + else { if (acc) tokens.push(acc); acc = s.raw.replace(/-/g, ""); } + } + if (acc) tokens.push(acc); + const aligned = alignWords(tokens, whisper); + for (let i = 0; i < scoreWords.length; i++) { + const w = scoreWords[i]; + const winMs = Math.max(40, aligned[i].toMs - aligned[i].fromMs); + const totalWeight = w.weights.reduce((x, y) => x + y, 0); + let cum = 0; + for (let s = 0; s < w.notes.length; s++) { + const startSec = (aligned[i].fromMs + (cum / totalWeight) * winMs) / 1000; + const durSec = (w.weights[s] / totalWeight) * winMs / 1000; + const midi = noteToMidi(w.notes[s]) + TRANSPOSE; + events.push({ startSec, midi, durSec, gain: GAIN }); + cum += w.weights[s]; + } + } + var totalSec = whisper[whisper.length - 1].toMs / 1000; + console.log(`→ ${events.length} bell strikes · ${totalSec.toFixed(1)}s · whisper-driven · transpose=${TRANSPOSE}st`); +} else { + // Score-driven: ONE strike per syllable so multi-note words ring out + // every note. (Was previously skipping continuation syllables — that + // collapsed melismas to single tones.) + const beat_s = 60.0 / BPM; + let beat_pos = 0; + for (const s of syllables) { + const startSec = beat_pos * beat_s; + const durSec = s.weight * beat_s; + const midi = noteToMidi(s.note) + TRANSPOSE; + events.push({ startSec, midi, durSec, gain: GAIN }); + beat_pos += s.weight; + } + var totalSec = beat_pos * beat_s; + console.log(`→ ${events.length} strikes · ${totalSec.toFixed(1)}s · score-driven · transpose=${TRANSPOSE}st`); +} + +// ── voice presets ──────────────────────────────────────────────────── +// bell — sinebells with inharmonic clang (recap/bin/waltz.mjs synth) +// pad — soft sine pad: fundamental + octave, slow attack, gentle +// plateau, slow release. No "ding". Sustained accompaniment. +const BELL_PARTIALS = [ + { ratio: 0.5, amp: 0.28, decayT60: 5.5 }, + { ratio: 1.0, amp: 1.00, decayT60: 4.5 }, + { ratio: 2.0, amp: 0.32, decayT60: 2.6 }, + { ratio: 2.4, amp: 0.10, decayT60: 1.2 }, + { ratio: 3.0, amp: 0.09, decayT60: 1.0 }, + { ratio: 4.5, amp: 0.04, decayT60: 0.6 }, + { ratio: 5.4, amp: 0.02, decayT60: 0.4 }, +]; +const PAD_PARTIALS = [ + { ratio: 1.0, amp: 1.00 }, // fundamental + { ratio: 2.0, amp: 0.18 }, // octave (gentle warmth) +]; + +const RING_TAIL_BELL = 6.0; +const RING_TAIL_PAD = 1.2; +const BELL_GAIN = 0.42; +const PAD_GAIN = 0.55; + +function midiToFreq(m) { return 440 * Math.pow(2, (m - 69) / 12); } + +const ringTail = VOICE === "pad" ? RING_TAIL_PAD : RING_TAIL_BELL; +const lenSec = totalSec + ringTail + 1.0; +const out = new Float32Array(Math.ceil(lenSec * SAMPLE_RATE)); + +for (const ev of events) { + const startIdx = Math.floor(ev.startSec * SAMPLE_RATE); + const fundFreq = midiToFreq(ev.midi); + const twoPiOverSr = (2 * Math.PI) / SAMPLE_RATE; + + if (VOICE === "pad") { + // ADSR-style pad: 60ms cosine attack → flat sustain (event dur) → + // 250ms cosine release. Pure sine + octave. No ding. + const ATTACK_S = 0.060; + const RELEASE_S = 0.250; + const attackSamp = Math.floor(ATTACK_S * SAMPLE_RATE); + const sustainSamp = Math.floor(ev.durSec * SAMPLE_RATE); + const releaseSamp = Math.floor(RELEASE_S * SAMPLE_RATE); + const totalSamp = attackSamp + sustainSamp + releaseSamp; + const partials = PAD_PARTIALS.map((p) => ({ + omega: twoPiOverSr * fundFreq * p.ratio, + amp: p.amp, + })); + for (let i = 0; i < totalSamp; i++) { + const dst = startIdx + i; + if (dst < 0 || dst >= out.length) continue; + let env; + if (i < attackSamp) { + env = 0.5 - 0.5 * Math.cos((Math.PI * i) / attackSamp); + } else if (i < attackSamp + sustainSamp) { + env = 1.0; + } else { + const r = i - (attackSamp + sustainSamp); + env = 0.5 + 0.5 * Math.cos((Math.PI * r) / releaseSamp); + } + let s = 0; + for (const p of partials) s += Math.sin(p.omega * i) * p.amp; + out[dst] += s * env * ev.gain * PAD_GAIN; + } + } else { + // bell (default) — original sinebells decay synthesis + const ringSamples = Math.floor((ev.durSec + ringTail) * SAMPLE_RATE); + const ATTACK_SEC = 0.012; + const attackS = ATTACK_SEC * SAMPLE_RATE; + const partials = BELL_PARTIALS.map((p) => ({ + omega: twoPiOverSr * fundFreq * p.ratio, + amp: p.amp, + decay: Math.exp(-Math.log(1000) / (p.decayT60 * SAMPLE_RATE)), + })); + for (let i = 0; i < ringSamples; i++) { + const dst = startIdx + i; + if (dst < 0 || dst >= out.length) continue; + let s = 0; + for (const p of partials) { + const env = p.amp * Math.pow(p.decay, i); + if (env < 1e-5) continue; + s += Math.sin(p.omega * i) * env; + } + let att = 1; + if (i < attackS) att = 0.5 - 0.5 * Math.cos((Math.PI * i) / attackS); + out[dst] += s * att * ev.gain * BELL_GAIN; + } + } +} + +// peak normalize to -1.5 dBFS +let peak = 0; +for (let i = 0; i < out.length; i++) if (Math.abs(out[i]) > peak) peak = Math.abs(out[i]); +const tgt = Math.pow(10, -1.5 / 20); +const norm = peak > 0 ? Math.min(1, tgt / peak) : 1; +if (norm < 1) for (let i = 0; i < out.length; i++) out[i] *= norm; + +// ── write out via ffmpeg (raw f32 → mp3) ───────────────────────────── +const tmp = mkdtempSync(`${tmpdir()}/melody-bells-`); +const rawPath = `${tmp}/bells.f32.raw`; +writeFileSync(rawPath, Buffer.from(out.buffer, out.byteOffset, out.byteLength)); +const r = spawnSync("ffmpeg", [ + "-hide_banner", "-y", "-loglevel", "error", + "-f", "f32le", "-ar", String(SAMPLE_RATE), "-ac", "1", + "-i", rawPath, + "-c:a", "libmp3lame", "-q:a", "2", + OUT_PATH, +]); +rmSync(tmp, { recursive: true, force: true }); +if (r.status !== 0) { console.error("✗ ffmpeg encode failed"); process.exit(1); } +console.log(`✓ ${OUT_PATH} (peak norm ${norm.toFixed(3)} · ${events.length} strikes)`); diff --git a/pop/bin/mfa-align.mjs b/pop/bin/mfa-align.mjs new file mode 100644 index 000000000..8854f131a --- /dev/null +++ b/pop/bin/mfa-align.mjs @@ -0,0 +1,230 @@ +#!/usr/bin/env node +// mfa-align.mjs — end-of-pipeline forced alignment. +// +// PURPOSE: produce per-word timings on the FINAL (post-pitchsnap) audio +// with the correct lyric text — replacing the lossy whisper-on-distorted- +// audio path that currently misrecognizes words ("saved a wretch" → +// "same direct") because pitchsnap distorts formants. +// +// IDEAL: Montreal Forced Aligner (Kaldi-based, ~30ms phoneme accuracy) +// `pip install montreal-forced-aligner` → installs a python wrapper +// but the underlying _kalpy native module is conda-only on Python 3.14. +// On this 8GB machine with no conda, MFA does not install. +// +// AENEAS FALLBACK: requires espeak + numpy build chain that fails on +// Python 3.14 (numpy ImportError mid-build). +// +// PRAGMATIC SOLUTION (used here): "alignment by substitution". +// Whisper's word *boundaries* on the final audio are accurate even +// when its *transcription* is wrong (it segments correctly, just +// mishears phonemes). We: +// 1. Load the whisper words.json (correct timings, wrong text). +// 2. Load the lyric .txt (correct text, no timings). +// 3. Run Needleman-Wunsch to align the two word sequences. +// 4. For each lyric-word, find the matched whisper-word and +// substitute the canonical lyric text into that timing slot. +// 5. For lyric-words with no whisper match (insertions), interpolate +// timing from neighbors. +// +// This is what MFA would do, minus the phoneme model — and for +// ElevenLabs-generated speech (clean signal, known lyrics) the +// accuracy is comparable. +// +// OUTPUT: ${slug}-mfa-words.json with shape `[{text, fromMs, toMs}]` +// matching whisper words.json so timeline.py can pick it up +// transparently. +// +// USAGE: +// node bin/mfa-align.mjs --audio big-pictures/out/amazing-final.mp3 \ +// --text big-pictures/amazing.txt \ +// [--whisper big-pictures/out/amazing-final-words.json] + +import { readFileSync, writeFileSync, existsSync } from "node:fs"; +import { resolve, dirname, basename } from "node:path"; + +const argv = process.argv.slice(2); +const flags = {}; +for (let i = 0; i < argv.length; i++) { + const a = argv[i]; + if (a.startsWith("--")) { + const key = a.slice(2); + const next = argv[i + 1]; + if (next !== undefined && !next.startsWith("--")) { flags[key] = next; i++; } + else flags[key] = true; + } +} + +const AUDIO = flags.audio ? resolve(process.cwd(), flags.audio) : null; +const TEXT = flags.text ? resolve(process.cwd(), flags.text) : null; +let WHISPER = flags.whisper ? resolve(process.cwd(), flags.whisper) : null; +let OUT = flags.out ? resolve(process.cwd(), flags.out) : null; + +if (!AUDIO || !TEXT) { + console.error("usage: node bin/mfa-align.mjs --audio --text [--whisper ] [--out ]"); + process.exit(1); +} +if (!existsSync(AUDIO)) { console.error(`✗ audio not found: ${AUDIO}`); process.exit(1); } +if (!existsSync(TEXT)) { console.error(`✗ text not found: ${TEXT}`); process.exit(1); } + +// Auto-derive whisper + out paths from audio basename. +const audioDir = dirname(AUDIO); +const audioStem = basename(AUDIO).replace(/\.[^.]+$/, ""); +if (!WHISPER) { + for (const cand of [ + `${audioDir}/${audioStem}-words.json`, + `${audioDir}/${audioStem}.words.json`, + ]) { + if (existsSync(cand)) { WHISPER = cand; break; } + } +} +if (!OUT) { + // Match the slug by stripping common suffixes (-final, -world, etc.) + const slug = audioStem.replace(/-(final|world|vocal|perline|mix|snapped).*$/, ""); + OUT = `${audioDir}/${slug}-mfa-words.json`; +} + +// ── Lyric text → ordered list of canonical words ──────────────────── +// Mirror say.mjs's lyric parser: drop section headers, blank lines, and +// hyphens; lowercase for matching; keep original casing/punctuation in +// a parallel array for the output. +const HEADER_RE = /^(hook|verse \d+|outro|bridge|chorus|intro)$/i; +const lyricLines = readFileSync(TEXT, "utf8").split("\n") + .map((l) => l.trim()) + .filter((l) => l && !HEADER_RE.test(l)); +const lyricWordsRaw = lyricLines.join(" ").split(/\s+/).filter(Boolean); +const norm = (s) => s.toLowerCase().replace(/[^a-z0-9']/g, ""); +const lyricKeys = lyricWordsRaw.map(norm).filter(Boolean); + +// Re-derive lyricWordsRaw to match the filter (in case some tokens +// were pure punctuation). +const lyricWords = lyricWordsRaw.filter((w) => norm(w)); + +console.log(` lyric: ${lyricWords.length} words from ${TEXT}`); + +// ── Whisper words (timings only — text will be substituted) ──────── +let whisperWords = []; +if (WHISPER && existsSync(WHISPER)) { + whisperWords = JSON.parse(readFileSync(WHISPER, "utf8")); + console.log(` whisper: ${whisperWords.length} words from ${WHISPER}`); +} else { + console.error(`✗ no whisper words.json found near ${AUDIO}; cannot do alignment-by-substitution.`); + console.error(` Run pitchwords/whisper on the final.mp3 first.`); + process.exit(1); +} + +// ── Needleman-Wunsch alignment ────────────────────────────────────── +// Standard DP with substitution = -1 (different) / +2 (same after +// normalization), gap = -1. This handles: +// - lyric word missing from whisper (whisper dropped it) → gap in B +// - whisper word with no lyric counterpart (hallucination) → gap in A +// - both present but text differs (mishearing) → match +const A = lyricKeys; +const B = whisperWords.map((w) => norm(w.text)); +const M = A.length, N = B.length; + +const GAP = -1; +const SUB_DIFF = -1; +const SUB_SAME = 2; +// Phoneme-leniency: if first letter matches we treat as partial credit. +// Helps "saved" pair with "same" instead of dropping. +const SUB_PARTIAL = 0; + +const score = Array.from({ length: M + 1 }, () => new Int32Array(N + 1)); +const trace = Array.from({ length: M + 1 }, () => new Uint8Array(N + 1)); // 0=match, 1=delA, 2=delB +for (let i = 0; i <= M; i++) { score[i][0] = i * GAP; trace[i][0] = 1; } +for (let j = 0; j <= N; j++) { score[0][j] = j * GAP; trace[0][j] = 2; } +trace[0][0] = 0; + +for (let i = 1; i <= M; i++) { + for (let j = 1; j <= N; j++) { + const a = A[i - 1], b = B[j - 1]; + let s; + if (a === b) s = SUB_SAME; + else if (a[0] && b[0] && a[0] === b[0]) s = SUB_PARTIAL; + else s = SUB_DIFF; + const diag = score[i - 1][j - 1] + s; + const up = score[i - 1][j] + GAP; + const left = score[i][j - 1] + GAP; + if (diag >= up && diag >= left) { score[i][j] = diag; trace[i][j] = 0; } + else if (up >= left) { score[i][j] = up; trace[i][j] = 1; } + else { score[i][j] = left; trace[i][j] = 2; } + } +} + +// Walk back to build the alignment. +const pairs = []; // {ai, bj} where -1 = gap +let i = M, j = N; +while (i > 0 || j > 0) { + const t = trace[i][j]; + if (i > 0 && j > 0 && t === 0) { pairs.push({ ai: i - 1, bj: j - 1 }); i--; j--; } + else if (i > 0 && t === 1) { pairs.push({ ai: i - 1, bj: -1 }); i--; } + else { pairs.push({ ai: -1, bj: j - 1 }); j--; } +} +pairs.reverse(); + +// Stats +let matches = 0, mismatches = 0, gaps = 0; +for (const p of pairs) { + if (p.ai < 0 || p.bj < 0) gaps++; + else if (A[p.ai] === B[p.bj]) matches++; + else mismatches++; +} +console.log(` align: ${matches} exact / ${mismatches} mishearings-corrected / ${gaps} gaps`); + +// ── Build output: one entry per LYRIC word, with timing borrowed ───── +// from its matched whisper word (or interpolated when there's no match). +const out = new Array(lyricWords.length).fill(null); +for (const p of pairs) { + if (p.ai < 0) continue; // whisper-only insertion: ignore + const lyricIdx = p.ai; + if (p.bj < 0) continue; // lyric word with no whisper match: interpolate later + const w = whisperWords[p.bj]; + out[lyricIdx] = { + text: lyricWords[lyricIdx], + fromMs: w.fromMs, + toMs: w.toMs, + }; +} + +// Interpolate gaps. For a run of unmatched lyric-words, split the +// neighbor's window proportionally. +for (let li = 0; li < out.length; li++) { + if (out[li] !== null) continue; + // Find the previous matched anchor and the next matched anchor. + let prevIdx = li - 1; + while (prevIdx >= 0 && out[prevIdx] === null) prevIdx--; + let nextIdx = li + 1; + while (nextIdx < out.length && out[nextIdx] === null) nextIdx++; + const prev = prevIdx >= 0 ? out[prevIdx] : null; + const next = nextIdx < out.length ? out[nextIdx] : null; + // Count consecutive nulls in this run. + let runStart = li; + while (runStart > 0 && out[runStart - 1] === null) runStart--; + let runEnd = li; + while (runEnd < out.length - 1 && out[runEnd + 1] === null) runEnd++; + const runLen = runEnd - runStart + 1; + // Window: from prev.toMs to next.fromMs, or extrapolate. + let winStart, winEnd; + if (prev && next) { winStart = prev.toMs; winEnd = next.fromMs; } + else if (prev) { winStart = prev.toMs; winEnd = prev.toMs + 800 * runLen; } + else if (next) { winEnd = next.fromMs; winStart = Math.max(0, next.fromMs - 800 * runLen); } + else { winStart = 0; winEnd = 800 * runLen; } + const slot = (winEnd - winStart) / runLen; + const localIdx = li - runStart; + out[li] = { + text: lyricWords[li], + fromMs: Math.round(winStart + slot * localIdx), + toMs: Math.round(winStart + slot * (localIdx + 1)), + }; +} + +// Sanity: ensure no nulls remain. +const final = out.filter((w) => w !== null); +if (final.length !== lyricWords.length) { + console.warn(` ⚠ ${lyricWords.length - final.length} lyric words could not be timed`); +} + +writeFileSync(OUT, JSON.stringify(final, null, 2)); +console.log(`✓ ${OUT} (${final.length} words)`); +console.log(` first 5: ${final.slice(0, 5).map((w) => `${w.text}@${w.fromMs}ms`).join(" ")}`); +console.log(` last 3: ${final.slice(-3).map((w) => `${w.text}@${w.fromMs}ms`).join(" ")}`); diff --git a/pop/bin/midi_to_np.py b/pop/bin/midi_to_np.py new file mode 100644 index 000000000..aef6ff23c --- /dev/null +++ b/pop/bin/midi_to_np.py @@ -0,0 +1,145 @@ +#!/usr/bin/env python3 +"""midi_to_np.py — extract a melody track from a MIDI file and write it +out as a `.np` notepat score that the big-pictures pipeline can sing. + +Usage: + midi_to_np.py input.mid output.np \ + [--track CLARINET] \ + [--track-index 14] \ + [--lyrics "its a world of laughter ..."] \ + [--start-beat 111] [--end-beat 130] \ + [--bpm 122] + +You can either pass `--track NAME` (matched case-insensitive substring) +or `--track-index N`. If neither is set, the script prints a summary +of every track + range so you can pick by ear/eye, then exits. + +The `--lyrics` flag is a flat list of words (whitespace separated). The +script aligns one syllable per MIDI note in the chosen window — so the +word/note counts have to line up. Use `--start-beat` / `--end-beat` to +trim the window down to just the chorus or verse you need. + +Multi-syllable words use leading hyphens on continuation tokens +(notepat convention): "amazing" → "a- -ma- -zing". + +If --lyrics isn't passed, the script writes "n01 n02 n03 …" placeholder +syllables so the score is still playable; you can edit by hand. +""" +import argparse +import sys +from pathlib import Path + +import mido + +NAMES = ["C","C#","D","D#","E","F","F#","G","G#","A","A#","B"] +def midi_name(n): return f"{NAMES[n%12]}{n//12-1}" + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("midi"); ap.add_argument("out") + ap.add_argument("--track", help="case-insensitive substring of track name") + ap.add_argument("--track-index", type=int, help="numeric track index") + ap.add_argument("--lyrics", help="flat whitespace-separated syllables; use - for continuation") + ap.add_argument("--start-beat", type=float, default=0) + ap.add_argument("--end-beat", type=float, default=1e9) + ap.add_argument("--min-dur-beats", type=float, default=0.4, + help="ignore notes shorter than this (skips ornaments)") + ap.add_argument("--bpm", type=int, default=None, + help="override tempo annotation in the .np header") + args = ap.parse_args() + + m = mido.MidiFile(args.midi) + tpb = m.ticks_per_beat + + # Detect tempo for the header. + bpm = args.bpm + if bpm is None: + for tr in m.tracks: + for msg in tr: + if msg.type == "set_tempo": + bpm = round(60_000_000 / msg.tempo); break + if bpm: break + if not bpm: bpm = 120 + + # Pick the track. + tracks = list(m.tracks) + chosen = None + if args.track_index is not None: + chosen = tracks[args.track_index] + elif args.track: + needle = args.track.lower() + for tr in tracks: + if needle in (tr.name or "").lower(): + chosen = tr; break + if chosen is None: + # No selection — print summary and exit. + print(f"\n{Path(args.midi).name} · {tpb} ticks/beat · detected {bpm} bpm · {len(tracks)} tracks\n") + for i, tr in enumerate(tracks): + notes = [msg.note for msg in tr if msg.type=='note_on' and msg.velocity>0] + if not notes: continue + lo = midi_name(min(notes)); hi = midi_name(max(notes)) + med = midi_name(sorted(notes)[len(notes)//2]) + print(f" [{i:2d}] {(tr.name or ''):16s} {len(notes):4d} notes range {lo}-{hi} median {med}") + print("\npass --track NAME or --track-index N to extract a melody") + return 0 + + # Extract melodic notes from the chosen track. + t = 0 + notes = [] + active = {} + for msg in chosen: + t += msg.time + if msg.type == "note_on" and msg.velocity > 0: + active[msg.note] = t + elif (msg.type == "note_off") or (msg.type == "note_on" and msg.velocity == 0): + if msg.note in active: + start = active.pop(msg.note) + notes.append((start, msg.note, t - start)) + notes.sort() + notes = [(s, n, d) for s, n, d in notes + if d / tpb >= args.min_dur_beats + and args.start_beat <= s/tpb < args.end_beat] + if not notes: + print(f"✗ no notes in window beat {args.start_beat}-{args.end_beat}", file=sys.stderr) + return 1 + + # Build syllables: lyric tokens (split on whitespace) get matched + # 1:1 to extracted notes. If counts mismatch we just truncate to + # min(len) and warn. + syls = [] + if args.lyrics: + toks = args.lyrics.split() + if len(toks) != len(notes): + print(f"⚠ {len(toks)} lyric tokens vs {len(notes)} notes — truncating to {min(len(toks), len(notes))}", file=sys.stderr) + n = min(len(toks), len(notes)) + for i in range(n): + s, midi, d = notes[i] + syls.append((midi, toks[i], max(1, round(d / tpb)))) + else: + for i, (s, midi, d) in enumerate(notes): + syls.append((midi, f"n{i:02d}", max(1, round(d / tpb)))) + + # Emit .np + lines = [] + lines.append(f"# Extracted by midi_to_np.py from {Path(args.midi).name}") + lines.append(f"# track: {chosen.name or ''} · {len(syls)} syllables · {bpm} bpm") + if args.start_beat > 0 or args.end_beat < 1e9: + lines.append(f"# window: beats {args.start_beat} to {args.end_beat}") + lines.append("") + lines.append("verse 1") + line = [] + cum_beats = 0 + for midi, syl, beats in syls: + line.append(f"{midi_name(midi)}:{syl}*{beats}") + cum_beats += beats + # Wrap at ~12 beats per .np line for readability + if cum_beats >= 12: + lines.append(" ".join(line)); line = []; cum_beats = 0 + if line: lines.append(" ".join(line)) + + Path(args.out).write_text("\n".join(lines) + "\n") + print(f"✓ {args.out} · {len(syls)} notes · {bpm} bpm") + return 0 + +if __name__ == "__main__": + sys.exit(main()) diff --git a/pop/bin/os_to_np.py b/pop/bin/os_to_np.py new file mode 100644 index 000000000..bcf35fe81 --- /dev/null +++ b/pop/bin/os_to_np.py @@ -0,0 +1,152 @@ +#!/usr/bin/env python3 +"""os_to_np.py — extract a melody from an Online Sequencer URL. + +Online Sequencer encodes the song as a base64-protobuf blob inside the +HTML page (`var data = '...'`). This script: + 1. Fetches the HTML. + 2. Pulls the data blob. + 3. Parses the protobuf into per-instrument note lists. + 4. With no --instrument, prints a summary of all instruments. + 5. With --instrument N, emits a .np matching that voice. + +Usage: + os_to_np.py https://onlinesequencer.net/509647 out.np + os_to_np.py https://onlinesequencer.net/509647 out.np --instrument 14 + os_to_np.py https://onlinesequencer.net/509647 out.np \ + --instrument 14 --start-time 8 --end-time 200 \ + --lyrics "its a small world af- -ter all ..." +""" +import argparse, base64, re, struct, sys, urllib.request +from collections import defaultdict +from pathlib import Path + +NOTE = ['C','C#','D','D#','E','F','F#','G','G#','A','A#','B'] +def note_name(p): return f"{NOTE[p%12]}{p//12-1}" + +def read_varint(b, i): + n = 0; shift = 0 + while True: + if i >= len(b): return None, i + v = b[i]; i += 1 + n |= (v & 0x7f) << shift + if v < 0x80: return n, i + shift += 7 + +def parse_msg(b, end): + out = {}; i = 0 + while i < end: + tag, i = read_varint(b, i) + if tag is None: break + wire = tag & 7; fid = tag >> 3 + if wire == 0: + v, i = read_varint(b, i); out.setdefault(fid, []).append(v) + elif wire == 5: + if i + 4 > end: break + v = struct.unpack(' end: break + out.setdefault(fid, []).append(b[i:i+length]); i += length + elif wire == 1: + if i + 8 > end: break + i += 8 + else: return out + return out + +def fetch_data(url): + req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0"}) + html = urllib.request.urlopen(req, timeout=30).read().decode("utf-8", errors="ignore") + m = re.search(r"var data = '([^']+)'", html) + if not m: raise RuntimeError("no `var data` blob found in page") + blob = m.group(1) + # Add padding for stray base64 if needed + pad = len(blob) % 4 + if pad: blob += "=" * (4 - pad) + return base64.b64decode(blob, validate=False) + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("url") + ap.add_argument("out") + ap.add_argument("--instrument", type=int, help="instrument id (run without to list)") + ap.add_argument("--start-time", type=float, default=0, + help="OS time units (16th-notes) — skip earlier notes") + ap.add_argument("--end-time", type=float, default=1e12) + ap.add_argument("--lyrics", help="whitespace-separated syllables") + ap.add_argument("--bpm", type=int, default=120) + ap.add_argument("--units-per-beat", type=int, default=4, + help="OS time unit → beat divisor (default 4 = sixteenth-notes)") + args = ap.parse_args() + + blob = fetch_data(args.url) + top = parse_msg(blob, len(blob)) + by_instr = defaultdict(list) + for n_raw in top.get(2, []): + m = parse_msg(n_raw, len(n_raw)) + p = m.get(1, [None])[0] + s = m.get(2, [0.0])[0] + l = m.get(3, [None])[0] + instr = m.get(4, [None])[0] + if p is not None: by_instr[instr].append((s, p, l)) + + if args.instrument is None: + print(f"\n{args.url}\n{len(by_instr)} instruments\n") + for instr in sorted(by_instr): + notes = sorted(by_instr[instr]) + ps = [n[1] for n in notes] + lo, hi = min(ps), max(ps) + med = sorted(ps)[len(ps)//2] + first = notes[0] + print(f" instr {instr:3d}: {len(notes):4d} notes range {note_name(lo)}-{note_name(hi)} median {note_name(med)} first {note_name(first[1])}@t={first[0]:.0f}") + print("\npass --instrument N to extract that voice") + return 0 + + notes = sorted(by_instr.get(args.instrument, [])) + notes = [n for n in notes if args.start_time <= n[0] < args.end_time] + if not notes: + print(f"✗ no notes for instrument {args.instrument}", file=sys.stderr); return 1 + + # OS uses sixteenth-notes (units_per_beat=4). Convert each note's + # length to .np beat weights by rounding to the nearest integer + # ≥ 1. Inflate weights to absorb rests so the next syllable lands + # at the right beat. + upb = args.units_per_beat + syls = [] + for i, (s, p, l) in enumerate(notes): + # Effective length = max(note length, gap until next note start) + if i + 1 < len(notes): + next_s = notes[i + 1][0] + eff = next_s - s + else: + eff = l or 0 + beats = max(1, round(eff / upb)) + syls.append((p, beats)) + + toks = (args.lyrics.split() if args.lyrics else + [f"n{i:02d}" for i in range(len(syls))]) + if args.lyrics and len(toks) != len(syls): + print(f"⚠ {len(toks)} lyric tokens vs {len(syls)} notes — using min({min(len(toks), len(syls))})", file=sys.stderr) + n = min(len(toks), len(syls)) + + lines = [] + lines.append(f"# Extracted by os_to_np.py from {args.url}") + lines.append(f"# instrument {args.instrument} · {n} syllables · {args.bpm} bpm") + if args.start_time or args.end_time < 1e12: + lines.append(f"# window: t={args.start_time}-{args.end_time} (16th-notes)") + lines.append("") + lines.append("verse 1") + line, cum = [], 0 + for i in range(n): + p, beats = syls[i] + line.append(f"{note_name(p)}:{toks[i]}*{beats}") + cum += beats + if cum >= 12: + lines.append(" ".join(line)); line, cum = [], 0 + if line: lines.append(" ".join(line)) + Path(args.out).write_text("\n".join(lines) + "\n") + print(f"✓ {args.out} · {n} syllables · bpm {args.bpm}") + return 0 + +if __name__ == "__main__": + sys.exit(main()) diff --git a/pop/bin/perline.mjs b/pop/bin/perline.mjs new file mode 100644 index 000000000..5e36e8ce8 --- /dev/null +++ b/pop/bin/perline.mjs @@ -0,0 +1,232 @@ +#!/usr/bin/env node +// perline.mjs — generate per-line ElevenLabs takes with score-aware +// final-syllable elongation. +// +// Why: phrase-end held syllables in the score (e.g. row.np `dream*4`, +// amazing.np `me*5` / `sound*5` / `found*5` / `see*5`) need to actually +// SUSTAIN in the source vocal so pitchsnap doesn't have to do a 6× +// rubberband stretch (which exceeds WORLD's clean range, and confuses +// whisper into merging the held vowel with the next line's onset — +// "me I" boundary collapsed to zero gap, etc.). +// +// What: read the .np score + .txt lyric file, identify each line's +// final word's score weight, repeat the final vowel cluster +// `(weight - 1)` times to coax ElevenLabs into a longer sustain, then +// generate one say.mjs call per line with a long inter-line silence. +// Concat into `${slug}-perline.mp3`. +// +// Usage: +// node bin/perline.mjs --slug amazing +// [--bpm 70] [--gap 1.5] inter-line silence in seconds +// [--style 0.6] [--stability 0.7] [--speed 0.7] + +import { execSync, spawnSync } from "node:child_process"; +import { writeFileSync, readFileSync, existsSync, mkdtempSync, rmSync } from "node:fs"; +import { resolve, dirname } from "node:path"; +import { tmpdir } from "node:os"; + +const flags = {}; +for (let i = 0; i < process.argv.length; i++) { + const a = process.argv[i]; + if (!a.startsWith("--")) continue; + const next = process.argv[i + 1]; + if (next === undefined || next.startsWith("--")) flags[a.slice(2)] = true; + else flags[a.slice(2)] = next; +} + +const SLUG = flags.slug || "amazing"; +const POP = "/Users/jas/aesthetic-computer/pop"; +const SCORE_PATH = `${POP}/big-pictures/${SLUG}.np`; +const LYRIC_PATH = `${POP}/big-pictures/${SLUG}.txt`; +const SECTION = (flags.section || "verse 1").toLowerCase(); +// Default to v1 settings: 0.3s gap, no elongation. Elongation looked +// promising on paper but ElevenLabs ignores doubled vowels, so it +// only inflated the source duration without sustaining held notes — +// regression in audible output. Opt in with --elongate to retry. +const GAP_S = Number(flags.gap ?? 0.3); +const STYLE = flags.style ?? "0.55"; +const STABILITY = flags.stability ?? "0.7"; +const SPEED = flags.speed ?? "0.7"; +const ELONGATE = flags.elongate === true || flags.elongate === "true"; +const OUT_PATH = flags.out + ? resolve(process.cwd(), flags.out) + : `${POP}/big-pictures/out/${SLUG}-perline.mp3`; + +if (!existsSync(SCORE_PATH)) { console.error(`✗ score missing: ${SCORE_PATH}`); process.exit(1); } +if (!existsSync(LYRIC_PATH)) { console.error(`✗ lyrics missing: ${LYRIC_PATH}`); process.exit(1); } + +// ── Parse the score: collect (line-index, word-index, weight) tuples ── +// Score lines map 1:1 to lyric lines under the same section header. +const scoreLines = readFileSync(SCORE_PATH, "utf8").split("\n"); +const scoreSecStart = scoreLines.findIndex( + (l) => l.trim().toLowerCase() === SECTION, +); +if (scoreSecStart < 0) { console.error(`✗ section '${SECTION}' missing in score`); process.exit(1); } + +// scoreWords[lineIdx] = [{ raw, weight }, ...] — one per token. +const scoreWords = []; +for (let i = scoreSecStart + 1; i < scoreLines.length; i++) { + const l = scoreLines[i].trim(); + if (!l) break; + if (l.startsWith("#")) continue; + if (/^[a-z]+ \d/i.test(l)) break; + const lineEntries = []; + for (const tok of l.split(/\s+/)) { + const m = tok.match(/^[A-Ga-g][#b]?-?\d:(.+?)(?:\*(\d+(?:\.\d+)?))?$/); + if (!m) continue; + lineEntries.push({ raw: m[1], weight: Number(m[2] ?? 1) }); + } + scoreWords.push(lineEntries); +} +console.log(`→ score has ${scoreWords.length} line(s) under '${SECTION}'`); + +// ── Parse lyric file: per-section line array ────────────────────────── +const lyricText = readFileSync(LYRIC_PATH, "utf8"); +const lyricLines = lyricText.split("\n"); +const lyricSecStart = lyricLines.findIndex( + (l) => l.trim().toLowerCase() === SECTION, +); +if (lyricSecStart < 0) { console.error(`✗ section '${SECTION}' missing in lyrics`); process.exit(1); } +const lines = []; +for (let i = lyricSecStart + 1; i < lyricLines.length; i++) { + const t = lyricLines[i].trim(); + if (!t) break; + if (/^[a-z]+ \d/i.test(t)) break; + lines.push(t); +} +console.log(`→ ${lines.length} lyric line(s)`); + +if (lines.length !== scoreWords.length) { + console.warn(`⚠ score lines (${scoreWords.length}) != lyric lines (${lines.length}) — final-word elongation may misalign`); +} + +// ── Per-word elongation by score weight ─────────────────────────────── +// Any score syllable with weight >= 3 gets its final vowel cluster +// repeated (weight - 2) times so ElevenLabs actually sustains it. +// Phrase-end held syllables AND mid-line held notes both qualify +// (e.g. amazing.np line 3 has D4:now*3 which is mid-line — without +// elongation whisper merges 'now' into adjacent words and pitchsnap +// has to do a 20× stretch). +// +// Also inserts a comma + space ", " between any two consecutive *N>=3 +// words on the same line so whisper doesn't merge them across vowel +// boundaries. +function elongateOneWord(word, weight) { + if (weight < 3) return word; + const extras = Math.max(1, Math.round(weight - 2)); + const trailMatch = word.match(/^([a-zA-Z']+)([.,!?;:]*)$/); + if (!trailMatch) return word; + const w = trailMatch[1], trail = trailMatch[2]; + const vowels = [...w.matchAll(/[aeiouy]+/gi)]; + if (vowels.length === 0) return word; + const v = vowels[vowels.length - 1]; + const vEnd = v.index + v[0].length; + return w.slice(0, vEnd) + v[0].slice(-1).repeat(extras) + w.slice(vEnd) + trail; +} + +function buildLine(lyricLine, scoreLine) { + // Each scoreLine entry maps 1:1 to a SYLLABLE, but lyricLine has + // WORDS. Group consecutive syllables that share a hyphen-marker + // chain into one word, then assign that word the MAX weight of its + // syllable group (so multi-syllable words like 'amazing'/'merrily' + // get their held syllable's weight). + const groups = []; + let cur = null; + for (const e of scoreLine) { + const r = e.raw; + const startsContinuation = r.startsWith("-"); + if (cur && startsContinuation) { + cur.weight = Math.max(cur.weight, e.weight); + cur.syls.push(r); + } else { + cur = { syls: [r], weight: e.weight }; + groups.push(cur); + } + } + // Group weight = MAX of its constituent syllable weights. + const groupWeights = groups.map(g => g.weight); + + const tokens = lyricLine.split(/\s+/); + if (tokens.length !== groupWeights.length) { + console.warn(` ⚠ word count ${tokens.length} != score-group count ${groupWeights.length} for "${lyricLine}"`); + } + const out = []; + for (let j = 0; j < tokens.length; j++) { + const w = groupWeights[j] ?? 1; + const elongated = elongateOneWord(tokens[j], w); + // If this word AND the next are both *3+ holds, force a comma so + // whisper doesn't merge them across the sustained-vowel boundary. + const nextW = groupWeights[j + 1] ?? 1; + const trail = (w >= 3 && nextW >= 3) ? "," : ""; + out.push(elongated + trail); + } + return out.join(" "); +} + +const elongatedLines = lines.map((line, i) => { + if (!ELONGATE) { + console.log(` L${i + 1} "${line}"`); + return line; + } + const sl = scoreWords[i] || []; + const out = buildLine(line, sl); + const changed = out !== line; + console.log(` L${i + 1} ${changed ? "→ " : " "} "${out}"`); + return out; +}); + +// ── Generate one ElevenLabs take per line via say.mjs ──────────────── +// Per-line style/stability variation gives different prosody per phrase +// (the v1 "much better" recipe). +const PER_LINE_STYLE = ["0.55", "0.45", "0.65", "0.50"]; +const PER_LINE_STABILITY = ["0.70", "0.75", "0.65", "0.70"]; + +const tmp = mkdtempSync(`${tmpdir()}/perline-${SLUG}-`); +const lineMp3s = []; +for (let i = 0; i < elongatedLines.length; i++) { + const idx = i + 1; + const lineFile = `${tmp}/${SLUG}-l${idx}.txt`; + writeFileSync(lineFile, `verse 1\n${elongatedLines[i]}\n`); + const lineOut = `${POP}/big-pictures/out/${SLUG}-l${idx}.mp3`; + const lstyle = flags.style != null ? STYLE : (PER_LINE_STYLE[i] ?? STYLE); + const lstab = flags.stability != null ? STABILITY : (PER_LINE_STABILITY[i] ?? STABILITY); + console.log(`→ say.mjs L${idx}: speed=${SPEED} style=${lstyle} stability=${lstab}`); + const r = spawnSync("node", [ + `${POP}/bin/say.mjs`, lineFile, + "--speed", String(SPEED), + "--style", String(lstyle), + "--stability", String(lstab), + "--out", lineOut, + "--force", + ], { stdio: "inherit" }); + if (r.status !== 0) { console.error(`✗ say.mjs failed on L${idx}`); process.exit(1); } + lineMp3s.push(lineOut); +} + +// ── Concat with ${GAP_S}s silence between lines ─────────────────────── +const silWav = `${tmp}/silence.mp3`; +spawnSync("ffmpeg", [ + "-hide_banner", "-y", "-loglevel", "error", + "-f", "lavfi", "-t", String(GAP_S), "-i", "anullsrc=r=44100:cl=mono", + silWav, +], { stdio: "inherit" }); + +const concatTxt = `${tmp}/concat.txt`; +const concatLines = []; +for (let i = 0; i < lineMp3s.length; i++) { + concatLines.push(`file '${lineMp3s[i]}'`); + if (i < lineMp3s.length - 1) concatLines.push(`file '${silWav}'`); +} +writeFileSync(concatTxt, concatLines.join("\n") + "\n"); + +const r = spawnSync("ffmpeg", [ + "-hide_banner", "-y", "-loglevel", "error", + "-f", "concat", "-safe", "0", "-i", concatTxt, + "-c:a", "libmp3lame", "-q:a", "4", + OUT_PATH, +], { stdio: "inherit" }); +if (r.status !== 0) { console.error("✗ concat failed"); process.exit(1); } + +rmSync(tmp, { recursive: true, force: true }); +const dur = execSync(`ffprobe -v error -show_entries format=duration -of default=noprint_wrappers=1:nokey=1 ${OUT_PATH}`).toString().trim(); +console.log(`✓ ${OUT_PATH} (${Number(dur).toFixed(2)}s, ${elongatedLines.length} lines, ${GAP_S}s gaps)`); diff --git a/pop/bin/perword.mjs b/pop/bin/perword.mjs new file mode 100644 index 000000000..1e00db0b7 --- /dev/null +++ b/pop/bin/perword.mjs @@ -0,0 +1,269 @@ +#!/usr/bin/env node +// perword.mjs — generate per-WORD ElevenLabs takes (one API call per +// lyric word) and concat them with tiny gaps between. +// +// Why: per-line takes ("that saved a wretch like me") give ElevenLabs +// the whole phrase to perform, which means held phrase-end syllables +// (`me*5` = 4.29s target) only get ~0.2s of natural source duration. +// Pitchsnap then has to do 18× stretch, which exceeds WORLD's clean +// range and produces garbled vowel sustain. The user can't hear "me" +// in the rendered audio. +// +// Per-word takes give each syllable ~0.5-1.5s of natural ElevenLabs +// prosody (a single word read aloud has natural release/decay), so +// pitchsnap needs only 3-5× stretch — clean WORLD range, audible word. +// +// Trade-off: more API calls (26 instead of 4 for amazing's verse 1). +// At jeffrey-pvc rates, ~$0.30 vs $0.05 per regen. +// +// Usage: +// node bin/perword.mjs --slug amazing +// [--gap-ms 80] [--style 0.55] [--stability 0.7] [--speed 0.7] + +import { execSync, spawnSync } from "node:child_process"; +import { writeFileSync, readFileSync, existsSync, mkdtempSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; + +const flags = {}; +for (let i = 0; i < process.argv.length; i++) { + const a = process.argv[i]; + if (!a.startsWith("--")) continue; + const next = process.argv[i + 1]; + if (next === undefined || next.startsWith("--")) flags[a.slice(2)] = true; + else flags[a.slice(2)] = next; +} + +const SLUG = flags.slug || "amazing"; +const POP = "/Users/jas/aesthetic-computer/pop"; +const SCORE_PATH = `${POP}/big-pictures/${SLUG}.np`; +const LYRIC_PATH = `${POP}/big-pictures/${SLUG}.txt`; +const SECTION = (flags.section || "verse 1").toLowerCase(); +const BPM = Number(flags.bpm ?? 70); +// 80ms was too tight — whisper merged single-char words ('i', 'a') +// into adjacent words during the source-side segmentation, causing +// pitchsnap to skip those score positions and shift every subsequent +// word by one slot. 250ms minimum keeps word boundaries distinct. +const GAP_MS = Number(flags["gap-ms"] ?? 250); +const HELD_GAP_MS = Number(flags["held-gap-ms"] ?? 400); +const STYLE = flags.style ?? "0.55"; +const STABILITY = flags.stability ?? "0.7"; +const SPEED = flags.speed ?? "0.7"; +// Validator + retry: each generated word must have natural duration +// >= max(MIN_NATURAL_S, target_dur / MAX_STRETCH). For a *5 hold at +// BPM=70, target=4.29s; with MAX_STRETCH=4 we need natural >= 1.07s. +// Short *1 words just need >= 0.4s. Retries cycle through prosody +// strategies until threshold hits or we run out. +const MIN_NATURAL_S = Number(flags["min-natural"] ?? 0.40); +const MAX_STRETCH = Number(flags["max-stretch"] ?? 4.0); +const MAX_RETRIES = Number(flags["max-retries"] ?? 5); +const OUT_PATH = flags.out + ? flags.out + : `${POP}/big-pictures/out/${SLUG}-perword.mp3`; + +if (!existsSync(SCORE_PATH)) { console.error(`✗ score missing: ${SCORE_PATH}`); process.exit(1); } +if (!existsSync(LYRIC_PATH)) { console.error(`✗ lyrics missing: ${LYRIC_PATH}`); process.exit(1); } + +// ── Parse score: per-line list of {raw, weight} ─────────────────────── +const scoreLines = readFileSync(SCORE_PATH, "utf8").split("\n"); +const sStart = scoreLines.findIndex((l) => l.trim().toLowerCase() === SECTION); +if (sStart < 0) { console.error(`✗ section '${SECTION}' missing in score`); process.exit(1); } +const scoreSyls = []; // flat array of {raw, weight} for the section +for (let i = sStart + 1; i < scoreLines.length; i++) { + const l = scoreLines[i].trim(); + if (!l) break; + if (l.startsWith("#")) continue; + if (/^[a-z]+ \d/i.test(l)) break; + for (const tok of l.split(/\s+/)) { + const m = tok.match(/^[A-Ga-g][#b]?-?\d:(.+?)(?:\*(\d+(?:\.\d+)?))?$/); + if (!m) continue; + scoreSyls.push({ raw: m[1], weight: Number(m[2] ?? 1) }); + } +} + +// Group syllables into words by hyphen markers — each lyric word maps +// to one or more score syllables. Take MAX weight as the word's hold +// (so multi-syllable words like 'a-ma-zing' inherit the held syllable's +// weight if any). +const wordWeights = []; +let cur = null; +for (const s of scoreSyls) { + if (s.raw.startsWith("-") && cur) { + cur.weight = Math.max(cur.weight, s.weight); + } else { + cur = { weight: s.weight }; + wordWeights.push(cur); + } +} +console.log(`→ score: ${scoreSyls.length} syllables grouped into ${wordWeights.length} words`); + +// ── Parse lyric file: flat array of words ───────────────────────────── +const lyricLines = readFileSync(LYRIC_PATH, "utf8").split("\n"); +const lStart = lyricLines.findIndex((l) => l.trim().toLowerCase() === SECTION); +if (lStart < 0) { console.error(`✗ section '${SECTION}' missing in lyrics`); process.exit(1); } +const lyricWords = []; +for (let i = lStart + 1; i < lyricLines.length; i++) { + const t = lyricLines[i].trim(); + if (!t) break; + if (/^[a-z]+ \d/i.test(t)) break; + for (const w of t.split(/\s+/)) lyricWords.push(w); +} +console.log(`→ lyrics: ${lyricWords.length} words`); + +if (lyricWords.length !== wordWeights.length) { + console.warn(`⚠ word counts mismatch (lyrics=${lyricWords.length}, score=${wordWeights.length})`); +} + +// ── Helpers ────────────────────────────────────────────────────────── +function probeDur(path) { + try { + const out = execSync( + `ffprobe -v error -show_entries format=duration -of default=noprint_wrappers=1:nokey=1 ${path}`, + { encoding: "utf8" } + ).trim(); + return Number(out); + } catch { return 0; } +} + +function elongateVowel(word, n) { + const vowels = [...word.matchAll(/[aeiouy]+/gi)]; + if (vowels.length === 0) return word + "e".repeat(n); + const v = vowels[vowels.length - 1]; + const vEnd = v.index + v[0].length; + return word.slice(0, vEnd) + v[0].slice(-1).repeat(n) + word.slice(vEnd); +} + +// Retry strategies — each tweaks one or more knobs of the request. +// Returns [{label, text, style, stability}, ...] +function retryStrategies(word) { + return [ + { label: "ellipsis", text: `${word}...`, style: STYLE, stability: STABILITY }, + { label: "double-vowel",text: elongateVowel(word, 2) + "...", style: STYLE, stability: STABILITY }, + { label: "soft-low-stab",text: `${word}, ${word}.`, style: "0.45", stability: "0.55" }, + { label: "many-dots", text: `${word}............`, style: "0.50", stability: STABILITY }, + { label: "echo+vowel", text: `${elongateVowel(word, 4)}.`, style: "0.60", stability: "0.55" }, + ]; +} + +// ── Generate one ElevenLabs take per word with validate+retry ──────── +const tmp = mkdtempSync(`${tmpdir()}/perword-${SLUG}-`); +const wordMp3s = []; +for (let i = 0; i < lyricWords.length; i++) { + const word = lyricWords[i].replace(/[.,!?;:]/g, ""); + const wt = (wordWeights[i] && wordWeights[i].weight) ?? 1; + const targetDur = wt * (60 / BPM); + // Required natural duration: at least MIN_NATURAL_S, but if that's + // not enough to keep stretch <= MAX_STRETCH, raise the bar. + const needed = Math.max(MIN_NATURAL_S, targetDur / MAX_STRETCH); + + // Try: original first, then ellipsis-augmented, then progressively + // more aggressive prosody hints. Keep the LONGEST that meets `needed`, + // or the longest of all attempts if none do. + const wordOut = `${POP}/big-pictures/out/${SLUG}-w${String(i).padStart(2, "0")}.mp3`; + const attempts = [ + { label: "plain", text: word, style: STYLE, stability: STABILITY }, + ...retryStrategies(word).slice(0, MAX_RETRIES), + ]; + let best = { dur: 0, attempt: -1, path: null, label: null }; + for (let a = 0; a < attempts.length; a++) { + const att = attempts[a]; + const candPath = `${tmp}/${SLUG}-w${String(i).padStart(2, "0")}-a${a}.mp3`; + const candFile = `${tmp}/${SLUG}-w${String(i).padStart(2, "0")}-a${a}.txt`; + writeFileSync(candFile, `verse 1\n${att.text}\n`); + const r = spawnSync("node", [ + `${POP}/bin/say.mjs`, candFile, + "--speed", String(SPEED), + "--style", String(att.style), + "--stability", String(att.stability), + "--out", candPath, + "--force", + ], { stdio: ["ignore", "ignore", "inherit"] }); + if (r.status !== 0) continue; + const d = probeDur(candPath); + if (d > best.dur) best = { dur: d, attempt: a, path: candPath, label: att.label }; + if (d >= needed) break; // good enough — stop trying + } + if (!best.path) { console.error(`✗ all retries failed on "${word}"`); process.exit(1); } + + // Copy winner to canonical location + spawnSync("cp", [best.path, wordOut]); + const ok = best.dur >= needed ? "✓" : "⚠"; + console.log( + `${ok} [${String(i + 1).padStart(2, " ")}/${lyricWords.length}] "${word.padEnd(8)}" hold=${wt}× ` + + `target=${targetDur.toFixed(2)}s need=${needed.toFixed(2)}s ` + + `got=${best.dur.toFixed(2)}s [a${best.attempt}:${best.label}]` + ); + wordMp3s.push({ path: wordOut, weight: wt }); +} + +// ── Concat with per-word silence gaps ──────────────────────────────── +// Held words get longer LEADING silence so whisper keeps them as +// separate utterances when aligning the concat. +function silMp3(durS, name) { + const path = `${tmp}/${name}.mp3`; + spawnSync("ffmpeg", [ + "-hide_banner", "-y", "-loglevel", "error", + "-f", "lavfi", "-t", String(durS), "-i", "anullsrc=r=44100:cl=mono", + path, + ]); + return path; +} +const shortSil = silMp3(GAP_MS / 1000, "sil-short"); +const heldSil = silMp3(HELD_GAP_MS / 1000, "sil-held"); + +const concatTxt = `${tmp}/concat.txt`; +const lines = []; +for (let i = 0; i < wordMp3s.length; i++) { + const w = wordMp3s[i]; + if (i > 0) { + // Use the LONGER gap before any word that's part of a held syllable + // (matches user pattern: held notes need clean entry boundary). + lines.push(`file '${w.weight >= 3 ? heldSil : shortSil}'`); + } + lines.push(`file '${w.path}'`); +} +writeFileSync(concatTxt, lines.join("\n") + "\n"); + +const r = spawnSync("ffmpeg", [ + "-hide_banner", "-y", "-loglevel", "error", + "-f", "concat", "-safe", "0", "-i", concatTxt, + "-c:a", "libmp3lame", "-q:a", "4", OUT_PATH, +], { stdio: "inherit" }); +if (r.status !== 0) { console.error("✗ concat failed"); process.exit(1); } + +rmSync(tmp, { recursive: true, force: true }); +const dur = execSync(`ffprobe -v error -show_entries format=duration -of default=noprint_wrappers=1:nokey=1 ${OUT_PATH}`).toString().trim(); +console.log(`✓ ${OUT_PATH} (${Number(dur).toFixed(2)}s, ${lyricWords.length} words, gap=${GAP_MS}ms / held-gap=${HELD_GAP_MS}ms)`); + +// ── Synthesize words.json from KNOWN word durations + gaps ─────────── +// Don't rely on whisper to segment — it routinely drops short words +// like 'i' and 'a' even with generous inter-word gaps, which then +// causes pitchsnap to skip score syllables and shift every subsequent +// word by one slot ("me/i confusion"). Since we generated each word +// as its own mp3 with known gaps, the boundaries are deterministic. +const wordsJsonPath = OUT_PATH.replace(/\.mp3$/, "-words.json"); +const wordsJsonHash = wordsJsonPath + ".hash"; +const synthWords = []; +let cursor_ms = 0; +for (let i = 0; i < wordMp3s.length; i++) { + const w = wordMp3s[i]; + const wd_s = probeDur(w.path); + const wd_ms = Math.round(wd_s * 1000); + // Apply LEADING gap (this word's entry pause) + if (i > 0) { + const isHeld = w.weight >= 3; + cursor_ms += isHeld ? HELD_GAP_MS : GAP_MS; + } + synthWords.push({ + text: lyricWords[i].replace(/[.,!?;:]/g, ""), + fromMs: cursor_ms, + toMs: cursor_ms + wd_ms, + }); + cursor_ms += wd_ms; +} +writeFileSync(wordsJsonPath, JSON.stringify(synthWords, null, 2)); +// Drop any stale whisper hash so downstream callers don't think this +// is a whisper-aligned file. +if (existsSync(wordsJsonHash)) { + try { execSync(`rm ${wordsJsonHash}`); } catch (_) {} +} +console.log(` ✓ synthesized ${synthWords.length} word boundaries → ${wordsJsonPath}`); diff --git a/pop/bin/realign-from-whisper.mjs b/pop/bin/realign-from-whisper.mjs new file mode 100644 index 000000000..9c6385ec2 --- /dev/null +++ b/pop/bin/realign-from-whisper.mjs @@ -0,0 +1,116 @@ +#!/usr/bin/env node +// realign-from-whisper.mjs — use whisper's RAW timing on the final mp3 +// to retime each storyboard slide.start to where audio actually fires. +// +// Why: pitchsnap reports snappedStart as the INTENDED placement, but +// WORLD synthesis introduces per-word timing offsets (±1.5s observed). +// Whisper detects actual audio onsets — even when its TEXT recognition +// is wrong (autotune smear: "amazing"→"Crazy", "wretch"→"turn"), +// the BOUNDARIES are at the right wall-clock positions. +// +// Approach: walk whisper words and canonical lyric words in parallel, +// skipping whisper non-word artifacts (♪ etc), and map each canonical +// word to its corresponding whisper-detected onset. +// +// Usage: node bin/realign-from-whisper.mjs --slug amazing + +import { readFileSync, writeFileSync, existsSync } from "node:fs"; + +const flags = {}; +for (let i = 0; i < process.argv.length; i++) { + const a = process.argv[i]; + if (!a.startsWith("--")) continue; + const next = process.argv[i + 1]; + if (next === undefined || next.startsWith("--")) flags[a.slice(2)] = true; + else flags[a.slice(2)] = next; +} + +const SLUG = flags.slug || "amazing"; +const POP = "/Users/jas/aesthetic-computer/pop"; +const SB = flags.storyboard || `${POP}/big-pictures/out/${SLUG}.storyboard.json`; +const WORDS = flags.words || `${POP}/big-pictures/out/${SLUG}-final-words.json`; + +if (!existsSync(SB)) { console.error(`✗ storyboard missing: ${SB}`); process.exit(1); } +if (!existsSync(WORDS)) { console.error(`✗ words missing: ${WORDS}`); process.exit(1); } + +const sb = JSON.parse(readFileSync(SB, "utf8")); +const whisper = JSON.parse(readFileSync(WORDS, "utf8")); + +// Filter out non-word whisper detections (musical notes, stray symbols). +const realWords = whisper.filter(w => /[a-zA-Z]/.test(w.text)); +console.log(`→ ${whisper.length} whisper entries → ${realWords.length} real words`); + +// Group score syllables into canonical words (a-/-ma-/-zing → 1 word). +function isWordStart(slide) { + const r = (slide.rawText ?? slide.text ?? "").trim(); + return !r.startsWith("-"); +} +const wordGroups = []; +for (let i = 0; i < sb.slides.length; i++) { + if (isWordStart(sb.slides[i]) || wordGroups.length === 0) { + wordGroups.push({ slideIdxs: [i] }); + } else { + wordGroups[wordGroups.length - 1].slideIdxs.push(i); + } +} +console.log(`→ ${sb.slides.length} slides grouped into ${wordGroups.length} canonical words`); + +// Map canonical word → whisper-detected onset by ORDER (positional). +// Whisper may have more or fewer entries (mishearings produce extra +// tokens), so we use min of the two and warn on mismatch. +const n = Math.min(wordGroups.length, realWords.length); +if (wordGroups.length !== realWords.length) { + console.warn(`⚠ count mismatch (canonical=${wordGroups.length} whisper=${realWords.length}); aligning ${n}`); +} + +const total = sb.audioDuration ?? sb.duration ?? 0; +let drift_sum = 0; +for (let wi = 0; wi < n; wi++) { + const grp = wordGroups[wi]; + const wd = realWords[wi]; + const wordStart = wd.fromMs / 1000; + const wordEnd = wd.toMs / 1000; + const sylN = grp.slideIdxs.length; + for (let si = 0; si < sylN; si++) { + const slide = sb.slides[grp.slideIdxs[si]]; + const newStart = wordStart + (wordEnd - wordStart) * (si / sylN); + drift_sum += Math.abs(newStart - slide.start) * 1000; + slide.start = Number(newStart.toFixed(3)); + } +} + +// Recompute slide.end as next-slide's start (contiguous). +for (let i = 0; i < sb.slides.length; i++) { + const s = sb.slides[i]; + const next = sb.slides[i + 1]; + s.end = Number((next ? next.start : total).toFixed(3)); + s.duration = Number((s.end - s.start).toFixed(3)); +} + +// Min-duration clamp for single-char and zero-duration slides. +const MIN_MULTI = 0.18, MIN_SINGLE = 0.55; +for (let i = 0; i < sb.slides.length; i++) { + const s = sb.slides[i]; + const txt = (s.text ?? "").replace(/[^a-zA-Z]/g, ""); + const minS = txt.length <= 1 ? MIN_SINGLE : MIN_MULTI; + if (s.duration >= minS) continue; + const prev = sb.slides[i - 1]; + if (!prev) continue; + const prevTxt = (prev.text ?? "").replace(/[^a-zA-Z]/g, ""); + const prevMin = prevTxt.length <= 1 ? MIN_SINGLE : MIN_MULTI; + const need = minS - s.duration; + const takeable = Math.min(need, Math.max(0, prev.duration - prevMin)); + if (takeable <= 0) continue; + prev.end = Number((prev.end - takeable).toFixed(3)); + prev.duration = Number((prev.end - prev.start).toFixed(3)); + s.start = prev.end; + s.duration = Number((s.end - s.start).toFixed(3)); +} + +console.log(`→ avg slide.start drift: ${(drift_sum / n).toFixed(0)}ms per word`); +writeFileSync(SB, JSON.stringify(sb, null, 2)); +console.log(`✓ ${SB}`); +console.log(` first 6:`); +for (const s of sb.slides.slice(0, 6)) { + console.log(` ${String(s.i).padStart(2)} '${s.text.padEnd(8)}' ${s.start.toFixed(2)}-${s.end.toFixed(2)}s (${s.duration.toFixed(2)}s)`); +} diff --git a/pop/bin/realign-storyboard.mjs b/pop/bin/realign-storyboard.mjs new file mode 100644 index 000000000..86f0e92ef --- /dev/null +++ b/pop/bin/realign-storyboard.mjs @@ -0,0 +1,140 @@ +#!/usr/bin/env node +// realign-storyboard.mjs — rewrite a storyboard.json so each slide.start +// matches the ACTUAL audio onset position from MFA-aligned word data, +// instead of the rigid beat × 60/BPM grid that pitchsnap targets. +// +// Why: pitchsnap places each word's WORLD-snapped position at the score's +// beat boundary, but the audible vowel attack lands ~200-1300ms later +// (consonant onset, cross-fade ramps, formant warm-up). Slides timed to +// the score grid therefore lead the audio noticeably — words show up on +// screen before they're heard. Variable per-word, so a global shift +// can't fix it; we have to use per-word measurements. +// +// Approach: group consecutive score syllables into "words" by hyphen +// markers (a-/-ma-/-zing → 'amazing'), then index-map them onto the MFA +// word array (canonical lyric order matches). Each word's first +// syllable inherits the MFA fromMs as its slide.start; remaining +// syllables of multi-syllable words subdivide the MFA window evenly. +// slide.end = next slide's slide.start (so contiguous, no gaps). +// +// Usage: +// node bin/realign-storyboard.mjs --slug amazing +// [--storyboard ] [--mfa ] [--out ] + +import { readFileSync, writeFileSync, existsSync } from "node:fs"; +import { resolve } from "node:path"; + +const flags = {}; +for (let i = 0; i < process.argv.length; i++) { + const a = process.argv[i]; + if (!a.startsWith("--")) continue; + const next = process.argv[i + 1]; + if (next === undefined || next.startsWith("--")) flags[a.slice(2)] = true; + else flags[a.slice(2)] = next; +} + +const SLUG = flags.slug || "amazing"; +const POP = "/Users/jas/aesthetic-computer/pop"; +const SB = flags.storyboard || `${POP}/big-pictures/out/${SLUG}.storyboard.json`; +const MFA = flags.mfa || `${POP}/big-pictures/out/${SLUG}-mfa-words.json`; +const OUT = flags.out || SB; + +if (!existsSync(SB)) { console.error(`✗ storyboard missing: ${SB}`); process.exit(1); } +if (!existsSync(MFA)) { console.error(`✗ mfa words missing: ${MFA}`); process.exit(1); } + +const sb = JSON.parse(readFileSync(SB, "utf8")); +const mfa = JSON.parse(readFileSync(MFA, "utf8")); + +// Group slides into "words" by score-syllable hyphenation markers. +// raw text examples: "a-", "-ma-", "-zing", "grace", "lit-", "-tle" +function isWordStart(slide) { + const r = (slide.rawText ?? slide.text ?? "").trim(); + return !r.startsWith("-"); +} +const wordGroups = []; // [{slideIdxs: [...]}] +for (let i = 0; i < sb.slides.length; i++) { + const s = sb.slides[i]; + if (isWordStart(s) || wordGroups.length === 0) { + wordGroups.push({ slideIdxs: [i] }); + } else { + wordGroups[wordGroups.length - 1].slideIdxs.push(i); + } +} +console.log(`→ ${sb.slides.length} slides grouped into ${wordGroups.length} words`); +console.log(`→ ${mfa.length} mfa words to map`); + +if (wordGroups.length !== mfa.length) { + console.warn(`⚠ word count mismatch (${wordGroups.length} score vs ${mfa.length} mfa) — falling back to length-min mapping`); +} + +const n = Math.min(wordGroups.length, mfa.length); +const totalDur = sb.audioDuration ?? sb.duration ?? 0; + +let drift_ms_sum = 0; +let drift_count = 0; +for (let wi = 0; wi < n; wi++) { + const grp = wordGroups[wi]; + const mw = mfa[wi]; + const wordStart = mw.fromMs / 1000; + const wordEnd = mw.toMs / 1000; + const sylN = grp.slideIdxs.length; + // Sub-divide the mfa word's window evenly across its syllables. + // This gives multi-syllable words like 'amazing' three slide.starts + // anchored to one mfa onset. + for (let si = 0; si < sylN; si++) { + const slide = sb.slides[grp.slideIdxs[si]]; + const newStart = wordStart + (wordEnd - wordStart) * (si / sylN); + const oldStart = slide.start; + drift_ms_sum += Math.abs(newStart - oldStart) * 1000; + drift_count++; + slide.start = Number(newStart.toFixed(3)); + } +} + +// Now rewrite slide.end to be the next slide's slide.start (contiguous). +// Last slide: end = audioDuration. +for (let i = 0; i < sb.slides.length; i++) { + const s = sb.slides[i]; + const next = sb.slides[i + 1]; + s.end = Number((next ? next.start : totalDur).toFixed(3)); + s.duration = Number((s.end - s.start).toFixed(3)); +} + +// Fix too-short slides. Happens when: +// - MFA can't separate adjacent words (e.g. 'saved a wretch' merged +// to 'same direct', so 'a' got 0s window) +// - Single-character words ('a', 'i') just have very short natural +// durations even when correctly detected (~0.2-0.3s in source) +// User feedback: single-char words like 'a' need to be HELD longer, +// like score-marked held notes — they're too short to read otherwise. +// Steal time from the previous slide to expand short ones. +const MIN_SLIDE_MULTI = 0.18; // multi-char words: brief but readable +const MIN_SLIDE_SINGLE = 0.55; // single-char words ('a', 'i'): hold like a beat +for (let i = 0; i < sb.slides.length; i++) { + const s = sb.slides[i]; + const txt = (s.text ?? "").replace(/[^a-zA-Z]/g, ""); + const minS = txt.length <= 1 ? MIN_SLIDE_SINGLE : MIN_SLIDE_MULTI; + if (s.duration >= minS) continue; + const prev = sb.slides[i - 1]; + if (!prev) continue; + const need = minS - s.duration; + // Don't steal more than what leaves prev with its own minimum. + const prevMin = (prev.text ?? "").replace(/[^a-zA-Z]/g, "").length <= 1 ? MIN_SLIDE_SINGLE : MIN_SLIDE_MULTI; + const takeable = Math.min(need, Math.max(0, prev.duration - prevMin)); + if (takeable <= 0) continue; + prev.end = Number((prev.end - takeable).toFixed(3)); + prev.duration = Number((prev.end - prev.start).toFixed(3)); + s.start = prev.end; + s.duration = Number((s.end - s.start).toFixed(3)); + console.log(` → expanded slide ${i} '${s.text}' to ${minS.toFixed(2)}s by stealing from prev`); +} + +const avgDrift = drift_ms_sum / Math.max(1, drift_count); +console.log(`→ avg drift correction: ${avgDrift.toFixed(0)}ms per slide`); + +writeFileSync(OUT, JSON.stringify(sb, null, 2)); +console.log(`✓ ${OUT}`); +console.log(` first 5 slides:`); +for (const s of sb.slides.slice(0, 5)) { + console.log(` ${String(s.i).padStart(2)} '${s.text.padEnd(8)}' ${s.start.toFixed(2)}-${s.end.toFixed(2)}s (${s.duration.toFixed(2)}s)`); +} diff --git a/pop/bin/refine_words.py b/pop/bin/refine_words.py new file mode 100644 index 000000000..786373ddb --- /dev/null +++ b/pop/bin/refine_words.py @@ -0,0 +1,102 @@ +#!/usr/bin/env python3 +"""refine_words.py — snap whisper word boundaries to librosa onsets. + +Whisper-cli on synthesized speech recognizes WORDS reliably but its +boundaries can drift 50-150 ms from the actual energy onset. librosa's +onset_strength + onset_detect operate directly on the audio and find +where each new sound actually fires (~30 ms precision). + +Strategy: + 1. Load whisper words.json: [{text, fromMs, toMs}, ...] + 2. Run librosa onset detection on the same audio. + 3. For each whisper word, find the closest detected onset within + ±SEARCH_WIN_MS of its fromMs. If found, replace fromMs with the + onset; bump toMs to keep dur >= MIN_DUR_MS. + 4. Write refined words.json with same schema. + +Usage: + refine_words.py [--win-ms 200] +""" +import argparse +import json +import sys +from pathlib import Path + +import numpy as np +import librosa + +DEFAULT_WIN_MS = 200 # search window around whisper.fromMs +MIN_DUR_MS = 50 # never let a word collapse to zero length +SR = 22050 # downsample target — onset detection is fine here + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("audio") + ap.add_argument("words") + ap.add_argument("out") + ap.add_argument("--win-ms", type=int, default=DEFAULT_WIN_MS) + args = ap.parse_args() + + audio = Path(args.audio) + words_path = Path(args.words) + out_path = Path(args.out) + if not audio.exists(): + print(f"✗ audio missing: {audio}", file=sys.stderr); return 1 + if not words_path.exists(): + print(f"✗ words missing: {words_path}", file=sys.stderr); return 1 + + words = json.loads(words_path.read_text()) + if not words: + out_path.write_text("[]") + print("⚠ empty words.json"); return 0 + + # Detect onsets. backtrack=True walks back to the nearest local energy + # minimum so we land on the consonant attack, not the vowel peak. + y, sr = librosa.load(str(audio), sr=SR) + onset_env = librosa.onset.onset_strength(y=y, sr=sr, hop_length=256) + onsets_s = librosa.onset.onset_detect( + onset_envelope=onset_env, sr=sr, hop_length=256, + backtrack=True, units="time", + ) + onsets_ms = (onsets_s * 1000).astype(int) + print(f" librosa: {len(onsets_ms)} onsets across {len(y)/sr:.1f}s") + + # Sort once so we can binary-search per whisper word. + onsets_ms.sort() + + refined = [] + n_snapped = 0 + n_drift_ms = [] + for w in words: + from_ms = int(w["fromMs"]) + to_ms = int(w["toMs"]) + # Find nearest onset within window + i = np.searchsorted(onsets_ms, from_ms) + candidates = [] + if i > 0: candidates.append(onsets_ms[i - 1]) + if i < len(onsets_ms): candidates.append(onsets_ms[i]) + best = None + best_diff = args.win_ms + 1 + for c in candidates: + d = abs(int(c) - from_ms) + if d < best_diff: best_diff = d; best = int(c) + if best is not None: + new_from = best + new_to = max(to_ms, new_from + MIN_DUR_MS) + n_drift_ms.append(new_from - from_ms) + n_snapped += 1 + refined.append({"text": w["text"], "fromMs": new_from, "toMs": new_to}) + else: + refined.append({"text": w["text"], "fromMs": from_ms, "toMs": to_ms}) + + out_path.write_text(json.dumps(refined)) + if n_snapped: + avg = sum(n_drift_ms) / len(n_drift_ms) + amax = max(abs(d) for d in n_drift_ms) + print(f" snapped {n_snapped}/{len(words)} (avg drift {avg:+.0f} ms · max |drift| {amax} ms)") + else: + print(f" no snaps — every word outside ±{args.win_ms}ms window") + return 0 + +if __name__ == "__main__": + sys.exit(main()) diff --git a/pop/bin/render_frames.py b/pop/bin/render_frames.py index afac90e87..bd564e873 100644 --- a/pop/bin/render_frames.py +++ b/pop/bin/render_frames.py @@ -271,38 +271,28 @@ def camera_x(t, slides, lane_starts, lane_widths, screen_w, peek_px, return 0.0 cs = lambda i: lane_starts[i] + lane_widths[i] / 2.0 - screen_w / 2.0 LANE_W = lane_widths[0] if lane_widths else screen_w - # Build keyframe list — 2 per slide (arrival + held-centered). - # in_end is at slide.start + td: the camera ARRIVES centered td - # AFTER the word's audio begins. So when a word starts being sung - # the slide slides in from the right. The camera then holds at - # cs(i) for the FULL duration of the slide (no peek-drift). The - # transition cs(i)→cs(i+1) happens during the NEXT slide's - # opening td seconds. + # Build keyframe list — slide-in happens DURING the first td of + # the new word's audio (not before — user feedback: "slides + # slide before they're spoken" was bad). Camera arrives centered + # at slide.start + td. Then dead-still hold (no drift) until + # slide.end. With LANE_W = 1.85*W, the wide lane spacing makes + # the slide scroll fully off-screen left before the next one + # enters from the right — no "doubling" two words on screen. kfs = [] for i, s in enumerate(slides): d = s["end"] - s["start"] td = transition_dur(d) in_end = s["start"] + td - # Held-centered keyframe at slide.end so the camera stays - # rooted on the slide's letters for the full sustain, even - # for *4/*5 long-held notes — no drifting away while the - # word is still being sung. - kfs.append((in_end, cs(i))) - kfs.append((s["end"], cs(i))) - # Phantom pre-roll keyframe at t=0: place the camera at cs(0) - - # LANE_W (≡ cs(n-1) mod n*LANE_W) so frame 0 shows slide_(n-1) - # centered and the transition to cs(0) plays out over the first - # slide's opening td. This makes the very first slide also slide - # in from the right exactly as the first word begins. - if n > 0 and slides[0]["start"] < 0.001: - kfs.insert(0, (0.0, cs(0) - LANE_W)) + kfs.append((in_end, cs(i))) # arrival, td after audio start + kfs.append((s["end"], cs(i))) # held dead-still until end if loop_end_t is not None and loop_end_t > kfs[-1][0]: - # During trailing silence (after the last slide ends but before - # loop_end_t), keep the camera CENTERED on the last slide. Frame - # N-1 lands at cs(n-1); frame 0 of the next iteration starts at - # cs(0)-LANE_W ≡ cs(n-1) mod — seamless wrap with no off-center - # drift while the final word's tail is still ringing. - kfs.append((loop_end_t, cs(n - 1))) + # Loop closure: from cs(n-1) at s_{n-1}.end, slide back to slide_0. + # cs(0)+n*LANE_W ≡ cs(0) mod n*LANE_W, so the camera lands on the + # FIRST slide centered at loop_end_t. Frame 0 of the next loop + # iteration is also at cs(0) — perfect seam, slide_0 is centered + # on both sides of the wrap. The slide-back motion plays out over + # the trailing silence, satisfying "loop back to the first frame". + kfs.append((loop_end_t, cs(0) + n * LANE_W)) first_t, first_pos = kfs[0] last_t, last_pos = kfs[-1] if t <= first_t: @@ -324,9 +314,12 @@ def camera_x(t, slides, lane_starts, lane_widths, screen_w, peek_px, # high and zooms up. Neighbors bounce/zoom less. Like macOS dock hover # but the cursor moves left-to-right through the word over the slide's # duration. -DOCK_WINDOW = 1.5 # how many "char slots" of influence each side -MAX_BOUNCE_PX = 64 # peak bounce — more energetic, more responsive -MAX_ZOOM = 1.00 +DOCK_WINDOW = 1.8 # influence each side — wider so neighbors join in +MAX_BOUNCE_PX = 140 # peak vertical bounce (was 64) — way more dramatic +MAX_ZOOM = 1.45 # peak scale-up at the cursor (was 1.0 = no zoom) +MAX_ROTATION_DEG = 14 # peak rotation wiggle, ± degrees +MAX_X_WIGGLE_PX = 22 # peak horizontal wiggle +WIGGLE_HZ = 7.5 # wiggle frequency for rotation+x — fast enough to feel jittery def char_emphasis(t, slide, char_idx, n_chars, amp_now): # Time window over which the singing-cursor traverses the word — @@ -341,9 +334,9 @@ def char_emphasis(t, slide, char_idx, n_chars, amp_now): if distance > DOCK_WINDOW: return 0.0 weight = 0.5 + 0.5 * np.cos(np.pi * distance / DOCK_WINDOW) - # Amp-driven response — minimal floor so quiet moments are quiet, - # peak moments really pop. - return weight * (0.10 + 0.90 * amp_now) + # Amp-driven response — minimal floor so quiet moments stay quiet, + # peak moments really pop. Bias toward amp so loud peaks dominate. + return weight * (0.15 + 0.85 * amp_now) def char_bounce_y(emphasis): return int(emphasis * MAX_BOUNCE_PX * -1) # negative = up @@ -351,6 +344,21 @@ def char_bounce_y(emphasis): def char_zoom(emphasis): return 1.0 + emphasis * (MAX_ZOOM - 1.0) +def char_rotation(emphasis, t, char_idx): + # Each char wiggles on its own phase (offset by index) so the word + # doesn't rock as a single block — letters look alive. + if emphasis <= 0: + return 0.0 + phase = 2 * np.pi * (WIGGLE_HZ * t + char_idx * 0.37) + return float(emphasis * MAX_ROTATION_DEG * np.sin(phase)) + +def char_x_wiggle(emphasis, t, char_idx): + if emphasis <= 0: + return 0 + # Cosine offset against rotation's sine — gives a slight orbital feel + phase = 2 * np.pi * (WIGGLE_HZ * t + char_idx * 0.37) + np.pi / 2 + return int(emphasis * MAX_X_WIGGLE_PX * np.cos(phase)) + # ── Live-waveform connector polyline ───────────────────────────────── # Renders a single thick scrolling waveform polyline between two words. @@ -483,7 +491,13 @@ def main(): KERN_SPACING = 4 # air between letters within a word BASE_SCALE = 0.85 # bigger now since one word per screen FOCUS_BOOST = 1.18 # spotlight only mildly above general - LANE_W = W # one word per screen-width + LANE_W = int(W * 1.85) # wider lane spacing so during slide-in + # transitions, slide_i scrolls fully off + # left BEFORE slide_(i+1) enters from + # right — eliminates the "two words at + # once" doubling. Was W (one word per + # screen width) which guaranteed overlap + # at the transition midpoint. MIN_PAD = 80 # generous bg-color padding around each word # MAX_WORD_W must account for FOCUS_BOOST so even 7-letter words # fit within the viewport at peak spotlight scale, with comfortable @@ -492,7 +506,13 @@ def main(): MAX_WORD_W = int((LANE_W - 2 * SIDE_AIR_PX) / FOCUS_BOOST) TARGET_WORD_H = int(H * 0.10) # ~192px tall — one-word focus PEEK_PX = int(W * 0.14) # ~150px of next word peeks in by hold-end - CONSUME_ZONE = 0.40 # whole left ~40% of screen disintegrates + CONSUME_ZONE = 0.13 # just the leftmost ~13% — only letters + # actually about to scroll off the + # screen disintegrate. Anything wider + # eats into the active word's first + # letter during slide-in transitions + # and makes the active syllable look + # faded/glitched while still being sung. LOOKAHEAD_ZONE = 0.82 # Per-word display scale — height-normalized, then capped to fit @@ -644,7 +664,8 @@ def main(): def paint_lane_glyphs(img, glyphs, lane_screen_x, lane_w, word_scale, focus_mult, baseline_src, per_char_zoom=None, per_char_bounce=None, - slide_idx=0, suppress_consume=False): + slide_idx=0, suppress_consume=False, + per_char_rotation=None, per_char_x_wiggle=None): if not glyphs: return n_chars = len(glyphs) @@ -684,11 +705,27 @@ def main(): else: glyph_img = g["img"] bnc = per_char_bounce[ci] if per_char_bounce else 0 + xwig = per_char_x_wiggle[ci] if per_char_x_wiggle else 0 + rot_deg = per_char_rotation[ci] if per_char_rotation else 0 + # Per-character rotation. PIL rotates around the image + # center, so the bbox grows; expand=True so corners aren't + # clipped, then we recenter on the original glyph anchor. + if abs(rot_deg) > 0.1: + pre_w, pre_h = glyph_img.size + glyph_img = glyph_img.rotate(rot_deg, resample=Image.BICUBIC, expand=True) + # Re-binarize alpha after bicubic to keep edges crisp. + arr_g = np.array(glyph_img) + arr_g[:, :, 3] = np.where(arr_g[:, :, 3] > 96, 255, 0).astype(np.uint8) + glyph_img = Image.fromarray(arr_g, "RGBA") + rot_dx = (glyph_img.size[0] - pre_w) // 2 + rot_dy = (glyph_img.size[1] - pre_h) // 2 + else: + rot_dx = rot_dy = 0 # Glyph bottom relative to baseline = below_src * s (descender) # so glyph's bottom in screen = baseline_y + below_src*s. # For most letters below_src = 0, so bottom = baseline_y. glyph_bottom = baseline_y - int((baseline_src - g["y1"]) * s) + bnc - gy = glyph_bottom - gh + gy = glyph_bottom - gh - rot_dy # Per-glyph consume effect: only fires once a slide is in # the past (its word being sung is over). Suppressed on the # current/active slide so words don't break up while sung. @@ -699,7 +736,7 @@ def main(): if glyph_img is None: x_cur += gw + KERN_SPACING continue - img.paste(glyph_img, (int(x_cur), gy), glyph_img) + img.paste(glyph_img, (int(x_cur + xwig - rot_dx), gy), glyph_img) x_cur += gw + KERN_SPACING for f in range(n_frames): @@ -764,16 +801,18 @@ def main(): suppress = not is_past if i == cur_idx and loop_off == 0 and glyphs: n_chars = len(glyphs) - zooms = [] - bounces = [] + zooms, bounces, rots, xwigs = [], [], [], [] for ci in range(n_chars): em = char_emphasis(t, slide_real, ci, n_chars, amp_now) zooms.append(char_zoom(em)) bounces.append(char_bounce_y(em)) + rots.append(char_rotation(em, t, ci)) + xwigs.append(char_x_wiggle(em, t, ci)) paint_lane_glyphs(img, glyphs, lane_screen_x, lane_widths[i], per_word_scales[i], scale_mult, baseline_src, zooms, bounces, slide_idx=i, - suppress_consume=suppress) + suppress_consume=suppress, + per_char_rotation=rots, per_char_x_wiggle=xwigs) else: paint_lane_glyphs(img, glyphs, lane_screen_x, lane_widths[i], per_word_scales[i], scale_mult, baseline_src, @@ -813,14 +852,20 @@ def main(): # 3. Chromatic aberration (R/B channels split) # 4. Cosine-eased alpha fade to bg gradient zone_w = int(W * CONSUME_ZONE) + # PIX_DIV-snap: round UP to a multiple of PIX_DIV so the + # downscale-then-upscale always lands at exactly zone_w (the + # pixelation step otherwise gives pw*PIX_DIV which can be + # smaller than zone_w, creating shape mismatches downstream). + PIX_DIV = 6 if zone_w > 0: + zone_w = ((zone_w + PIX_DIV - 1) // PIX_DIV) * PIX_DIV arr = np.array(img) arr_orig = arr[:, :zone_w, :].copy() # untouched left zone zone = arr_orig.copy() # 1. PIXELATION: downscale via BOX (anti-aliased average) - # then NEAREST upscale for chunky pixels - PIX_DIV = 6 + # then NEAREST upscale for chunky pixels. + # PIX_DIV is defined above (before zone_w snap). pw, ph = max(1, zone_w // PIX_DIV), max(1, H // PIX_DIV) zone_img = Image.fromarray(zone) small = zone_img.resize((pw, ph), Image.BOX) diff --git a/pop/bin/say.mjs b/pop/bin/say.mjs index 0393c2aec..634de93e6 100755 --- a/pop/bin/say.mjs +++ b/pop/bin/say.mjs @@ -69,6 +69,11 @@ const STYLE = flags.style !== undefined ? Number(flags.style) : null; // 0-1 sty const STABILITY = flags.stability !== undefined ? Number(flags.stability) : null; // 0-1 const SIMILARITY = flags.similarity !== undefined ? Number(flags.similarity) : null; // 0-1 const FORCE = flags.force === true; +// `--timestamps` opts into ElevenLabs `/with-timestamps` endpoint, which +// returns per-character alignment alongside the audio. Lossless, exact, +// free — the source-of-truth replacement for whisper STT word boundaries. +// When set, writes `${OUT_PATH}.alignment.json` next to the mp3. +const TIMESTAMPS = flags.timestamps === true; const OUT_PATH = expandHome(flags.out) || `${ROOT}/big-pictures/out/${slug}${SECTION ? `-${SECTION.replace(/\s+/g, "_")}` : ""}-vocal.mp3`; @@ -156,26 +161,36 @@ if (SPEED !== 1.0) body.speed = Math.max(0.7, Math.min(1.2, SPEED)); if (STYLE !== null && Number.isFinite(STYLE)) body.style = Math.max(0, Math.min(1, STYLE)); if (STABILITY !== null && Number.isFinite(STABILITY)) body.stability = Math.max(0, Math.min(1, STABILITY)); if (SIMILARITY !== null && Number.isFinite(SIMILARITY)) body.similarity = Math.max(0, Math.min(1, SIMILARITY)); +if (TIMESTAMPS) body.withTimestamps = true; const inputHash = createHash("sha256") .update(JSON.stringify(body)) .digest("hex").slice(0, 16); const hashFile = `${OUT_PATH}.hash`; +const ALIGNMENT_PATH = `${OUT_PATH}.alignment.json`; mkdirSync(dirname(OUT_PATH), { recursive: true }); if (!FORCE && existsSync(OUT_PATH) && existsSync(hashFile)) { const cached = readFileSync(hashFile, "utf8").trim(); - if (cached === inputHash) { + // When --timestamps was requested we also need the alignment sidecar. + // If the mp3 is cached but the alignment isn't, force a re-fetch. + const alignmentReady = !TIMESTAMPS || existsSync(ALIGNMENT_PATH); + if (cached === inputHash && alignmentReady) { const size = (readFileSync(OUT_PATH).length / 1024).toFixed(0); console.log(`✓ ${OUT_PATH} cached (${size} KB · hash ${inputHash}) — skipping /api/say`); + if (TIMESTAMPS) console.log(` alignment: ${ALIGNMENT_PATH}`); process.exit(0); } } -console.log(`→ POST /api/say · ${narration.length} chars · ${PROVIDER}/${VOICE_ID}` + (SPEED !== 1.0 ? ` · speed=${SPEED}` : "") + (STYLE !== null ? ` · style=${STYLE}` : "") + (SECTION ? ` · section=${SECTION}` : "")); +console.log(`→ POST /api/say · ${narration.length} chars · ${PROVIDER}/${VOICE_ID}` + (SPEED !== 1.0 ? ` · speed=${SPEED}` : "") + (STYLE !== null ? ` · style=${STYLE}` : "") + (SECTION ? ` · section=${SECTION}` : "") + (TIMESTAMPS ? " · with-timestamps" : "")); console.log(` preview: ${narration.split("\n")[0].slice(0, 80)}…`); -const res = await fetch("https://aesthetic.computer/api/say", { +// SAY_ENDPOINT lets the pipeline target a locally-run say endpoint +// (e.g. pop/chillwave/bin/say-local.mjs) when the production host is +// unreachable. Defaults to production. +const SAY_URL = process.env.SAY_ENDPOINT || "https://aesthetic.computer/api/say"; +const res = await fetch(SAY_URL, { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify(body), @@ -187,7 +202,61 @@ if (!res.ok) { process.exit(1); } -const buf = Buffer.from(await res.arrayBuffer()); -writeFileSync(OUT_PATH, buf); -writeFileSync(hashFile, inputHash + "\n"); -console.log(`✓ ${OUT_PATH} (${(buf.length / 1024).toFixed(0)} KB · hash ${inputHash})`); +if (TIMESTAMPS) { + // Server returned JSON `{audio, alignment, normalizedAlignment}`. + // Write mp3 to OUT_PATH (same as the non-timestamp path) and the + // alignment sidecar to OUT_PATH.alignment.json. + const ct = res.headers.get("content-type") || ""; + if (!ct.includes("application/json")) { + console.error(`✗ --timestamps was set but server returned ${ct}; was the server patched?`); + process.exit(1); + } + const json = await res.json(); + const buf = Buffer.from(json.audio, "base64"); + writeFileSync(OUT_PATH, buf); + writeFileSync(hashFile, inputHash + "\n"); + + // Build word-level boundaries from the per-character alignment. + // Words are runs of consecutive non-space characters; word time is + // [first-char start, last-char end]. + const a = json.alignment || {}; + const chars = a.characters || []; + const starts = a.character_start_times_seconds || []; + const ends = a.character_end_times_seconds || []; + const words = []; + let i = 0; + while (i < chars.length) { + if (/\s/.test(chars[i])) { i++; continue; } + const wStart = starts[i]; + let txt = ""; + let lastEnd = ends[i]; + while (i < chars.length && !/\s/.test(chars[i])) { + txt += chars[i]; + lastEnd = ends[i]; + i++; + } + words.push({ + text: txt, + fromMs: Math.round(wStart * 1000), + toMs: Math.round(lastEnd * 1000), + }); + } + + const alignmentDoc = { + source: "elevenlabs/with-timestamps", + text: narration, + voice: json.voice || PROVIDER, + characters: chars, + char_starts_s: starts, + char_ends_s: ends, + words, + }; + writeFileSync(ALIGNMENT_PATH, JSON.stringify(alignmentDoc, null, 2)); + console.log(`✓ ${OUT_PATH} (${(buf.length / 1024).toFixed(0)} KB · hash ${inputHash})`); + console.log(`✓ ${ALIGNMENT_PATH} (${chars.length} chars · ${words.length} words)`); +} else { + const buf = Buffer.from(await res.arrayBuffer()); + writeFileSync(OUT_PATH, buf); + writeFileSync(hashFile, inputHash + "\n"); + console.log(`✓ ${OUT_PATH} (${(buf.length / 1024).toFixed(0)} KB · hash ${inputHash})`); +} diff --git a/pop/bin/score-pitch.mjs b/pop/bin/score-pitch.mjs new file mode 100644 index 000000000..0eff607e0 --- /dev/null +++ b/pop/bin/score-pitch.mjs @@ -0,0 +1,207 @@ +#!/usr/bin/env node +// score-pitch.mjs — pitch-snap a natural-paced vocal to a .np score +// in ONE WORLD pass (no time-stretching). +// +// Why this exists: score-render.mjs slices the source per-word and +// rubberband-stretches each clip to the score's hymn-paced beats. +// When source words are 0.2-0.5s and target durs are 4-6s, stretches +// reach 10-25× and the time-stretch artifacts swamp the pitch +// replacement — listener hears smear, not melody. +// +// This script keeps the source timing intact: it asks WORLD to apply +// a target-f0 curve across the whole vocal, anchored to whisper word +// start times. The output is the same length as the input, but every +// voiced frame sings the score's pentatonic note. Jeffrey's prosody +// + ElevenLabs character preserved. +// +// Usage: +// node bin/score-pitch.mjs \ +// --slug amazing --section all \ +// --vocal big-pictures/out/amazing-7verse-vocal.mp3 \ +// --words big-pictures/out/amazing-7verse-vocal-words.json \ +// --transpose -5 \ +// --out big-pictures/out/amazing-7verse-pitched.mp3 + +import { spawnSync } from "node:child_process"; +import { readFileSync, writeFileSync, existsSync, mkdtempSync, rmSync } from "node:fs"; +import { resolve } from "node:path"; +import { tmpdir } from "node:os"; +import { alignWords } from "./align-words.mjs"; + +const flags = {}; +for (let i = 0; i < process.argv.length; i++) { + const a = process.argv[i]; + if (!a.startsWith("--")) continue; + const next = process.argv[i + 1]; + if (next === undefined || next.startsWith("--")) flags[a.slice(2)] = true; + else flags[a.slice(2)] = next; +} + +const POP = "/Users/jas/aesthetic-computer/pop"; +const SLUG = flags.slug || "amazing"; +const SECTION = (flags.section || "all").toLowerCase(); +const TRANSPOSE = Number(flags.transpose ?? -5); +const VOCAL_PATH = flags.vocal + ? resolve(process.cwd(), flags.vocal) + : `${POP}/big-pictures/out/${SLUG}-7verse-vocal.mp3`; +const WORDS_PATH = flags.words + ? resolve(process.cwd(), flags.words) + : `${POP}/big-pictures/out/${SLUG}-7verse-vocal-words.json`; +const SCORE_PATH = flags.score + ? resolve(process.cwd(), flags.score) + : `${POP}/big-pictures/${SLUG}.np`; +const OUT_PATH = flags.out + ? resolve(process.cwd(), flags.out) + : `${POP}/big-pictures/out/${SLUG}-7verse-pitched.mp3`; + +if (!existsSync(VOCAL_PATH)) { console.error(`✗ vocal missing: ${VOCAL_PATH}`); process.exit(1); } +if (!existsSync(WORDS_PATH)) { console.error(`✗ words missing: ${WORDS_PATH}`); process.exit(1); } +if (!existsSync(SCORE_PATH)) { console.error(`✗ score missing: ${SCORE_PATH}`); process.exit(1); } + +// ── parse score (mirrors score-render.mjs --section all path) ──────── +const NOTE_BASE = { C:0,"C#":1,DB:1,D:2,"D#":3,EB:3,E:4,F:5,"F#":6, + GB:6,G:7,"G#":8,AB:8,A:9,"A#":10,BB:10,B:11 }; +function noteToMidi(s) { + s = s.toUpperCase(); + return 12 * (parseInt(s.slice(-1), 10) + 1) + NOTE_BASE[s.slice(0, -1)]; +} +function midiToNote(m) { + const oct = Math.floor(m / 12) - 1; + const n = ["C","C#","D","D#","E","F","F#","G","G#","A","A#","B"][m % 12]; + return `${n}${oct}`; +} + +const scoreLines = readFileSync(SCORE_PATH, "utf8").split("\n"); +const syllables = []; +function parseRange(startIdx) { + for (let i = startIdx; i < scoreLines.length; i++) { + const l = scoreLines[i].trim(); + if (!l) continue; + if (l.startsWith("#")) continue; + if (/^[a-z]+ \d+$/i.test(l)) break; + for (const tok of l.split(/\s+/)) { + const m = tok.match(/^([A-Ga-g][#b]?-?\d):(.+?)(?:\*(\d+(?:\.\d+)?))?$/); + if (!m) continue; + syllables.push({ note: m[1], raw: m[2], weight: Number(m[3] ?? 1) }); + } + } +} +if (SECTION === "all") { + const headers = []; + for (let i = 0; i < scoreLines.length; i++) { + if (/^verse \d+$/i.test(scoreLines[i].trim())) headers.push(i); + } + headers.forEach((h) => parseRange(h + 1)); +} else { + const sStart = scoreLines.findIndex((l) => l.trim().toLowerCase() === SECTION); + if (sStart < 0) { console.error(`✗ section missing: ${SECTION}`); process.exit(1); } + parseRange(sStart + 1); +} + +// Group syllables → words (hyphenated chains merge). Each word keeps +// its full per-syllable note + weight list so we can drive a melisma: +// "amazing" (D3,G3,B3) hits all three notes within the word window, +// not just D3 the whole way. +const scoreWords = []; +let cur = null; +for (const s of syllables) { + if (s.raw.startsWith("-") && cur) { + cur.notes.push(s.note); + cur.weights.push(s.weight); + } else { + if (cur) scoreWords.push(cur); + cur = { + text: s.raw.replace(/^-|-$/g, "").toLowerCase(), + notes: [s.note], + weights: [s.weight], + }; + } +} +if (cur) scoreWords.push(cur); +const totalSyl = scoreWords.reduce((n, w) => n + w.notes.length, 0); +console.log(`→ score: ${scoreWords.length} words across ${totalSyl} syllables`); + +// ── load whisper words + run alignment ─────────────────────────────── +const whisper = JSON.parse(readFileSync(WORDS_PATH, "utf8")); +console.log(`→ whisper: ${whisper.length} words spanning ${(whisper[whisper.length-1].toMs/1000).toFixed(1)}s`); + +const aligned = alignWords(scoreWords.map((w) => w.text), whisper); +const matched = aligned.filter((a) => a).length; +console.log(`→ aligner: ${matched}/${scoreWords.length} score words placed in source time`); + +// Build per-SYLLABLE notes + starts. For each whisper-aligned word +// window [fromMs, toMs], distribute its syllables proportional to +// score weights so multi-note words sing the actual melisma. +const notes = []; +const starts = []; +for (let i = 0; i < scoreWords.length; i++) { + const w = scoreWords[i]; + const a = aligned[i]; + const window_ms = Math.max(40, a.toMs - a.fromMs); + const totalWeight = w.weights.reduce((x, y) => x + y, 0); + let cum = 0; + for (let s = 0; s < w.notes.length; s++) { + const tStart = a.fromMs + (cum / totalWeight) * window_ms; + notes.push(midiToNote(noteToMidi(w.notes[s]) + TRANSPOSE)); + starts.push((tStart / 1000).toFixed(3)); + cum += w.weights[s]; + } +} + +// ── decode mp3 → wav, run WORLD, encode wav → mp3 ──────────────────── +const tmp = mkdtempSync(`${tmpdir()}/score-pitch-${SLUG}-`); +const inWav = `${tmp}/in.wav`; +const outWav = `${tmp}/out.wav`; + +console.log(`→ decode mp3 → wav`); +let r = spawnSync("ffmpeg", [ + "-hide_banner", "-y", "-loglevel", "error", + "-i", VOCAL_PATH, "-ar", "44100", "-ac", "1", inWav, +]); +if (r.status !== 0) { console.error("✗ ffmpeg decode failed"); process.exit(1); } + +console.log(`→ WORLD f0 replacement · ${notes.length} notes (transpose ${TRANSPOSE >= 0 ? "+" : ""}${TRANSPOSE}st)`); +console.log(` first 6 notes: ${notes.slice(0, 6).join(", ")}`); +console.log(` first 6 starts: ${starts.slice(0, 6).join("s, ")}s`); +const XFADE_MS = String(flags["xfade-ms"] ?? 30); +const VOICING_RAMP_MS = String(flags["voicing-ramp-ms"] ?? 20); +const VIBRATO_HZ = String(flags["vibrato-hz"] ?? 0); +const VIBRATO_CENTS = String(flags["vibrato-cents"] ?? 0); +const RETAIN = String(flags["retain"] ?? 1.0); +r = spawnSync(`${POP}/.venv/bin/python`, [ + `${POP}/bin/pitchsnap_world.py`, inWav, outWav, + "--notes", notes.join(","), + "--note-starts", starts.join(","), + "--retain", RETAIN, + "--xfade-ms", XFADE_MS, + "--voicing-ramp-ms", VOICING_RAMP_MS, + "--vibrato-hz", VIBRATO_HZ, + "--vibrato-cents", VIBRATO_CENTS, +], { stdio: "inherit" }); +if (r.status !== 0) { console.error("✗ WORLD failed"); process.exit(1); } + +console.log(`→ encode wav → mp3`); +r = spawnSync("ffmpeg", [ + "-hide_banner", "-y", "-loglevel", "error", + "-i", outWav, "-c:a", "libmp3lame", "-q:a", "2", OUT_PATH, +]); +if (r.status !== 0) { console.error("✗ ffmpeg encode failed"); process.exit(1); } + +rmSync(tmp, { recursive: true, force: true }); + +// Emit alignment sidecar (used by track-poster.py + downstream tools). +// Includes per-syllable note + weight breakdown so consumers can render +// melismatic moments correctly. +const alignmentPath = OUT_PATH.replace(/\.mp3$/, "-alignment.json"); +writeFileSync(alignmentPath, JSON.stringify(scoreWords.map((w, i) => ({ + text: w.text, + note: w.notes[0], + notes: w.notes, + weights: w.weights, + midi: noteToMidi(w.notes[0]) + TRANSPOSE, + midis: w.notes.map((n) => noteToMidi(n) + TRANSPOSE), + fromMs: aligned[i].fromMs, + toMs: aligned[i].toMs, +})), null, 2)); +console.log(`✓ ${OUT_PATH}`); +console.log(`✓ ${alignmentPath} (${scoreWords.length} words · ${totalSyl} syllables)`); diff --git a/pop/bin/score-render.mjs b/pop/bin/score-render.mjs new file mode 100644 index 000000000..dd76200d5 --- /dev/null +++ b/pop/bin/score-render.mjs @@ -0,0 +1,299 @@ +#!/usr/bin/env node +// score-render.mjs — score-as-truth audio render. +// +// Pivot from the previous "stretch a whole-verse take to fit the score +// then realign whisper transcripts" pipeline. The score already knows +// every word's start time, target duration, and target pitch — render +// each word independently and place it at score time. No transcription +// step needed; the score IS the alignment. +// +// Pipeline: +// 1. Parse score (.np) → flat array of words with (start_s, dur_s, +// target_midi). Group multi-syllable score tokens into one word. +// 2. For each word, take its per-word clip from best-of-takes' picks +// (already cached as ${slug}-take-*.mp3 + per-word selections). +// Or just use ${slug}-perword.mp3 segments via the words.json +// that perword.mjs synthesized. +// 3. Rubberband each clip: +// --time = target_dur / source_dur (stretch to score duration) +// --pitch = target_midi - source_midi (semitone shift) +// 4. Place each stretched clip at score_start_s using ffmpeg adelay. +// 5. Mix all clips into one track (amix or just sequential delay+concat). +// +// Output: ${slug}-score-rendered.mp3 — clean, score-aligned, no whisper. +// +// Usage: +// node bin/score-render.mjs --slug amazing +// [--source perword|bestof] pick the per-word source set +// [--bpm 70] +// [--ref-note C3] source baritone reference +// [--out path.mp3] + +import { execSync, spawnSync } from "node:child_process"; +import { readFileSync, existsSync, mkdtempSync, rmSync } from "node:fs"; +import { resolve } from "node:path"; +import { tmpdir } from "node:os"; + +const flags = {}; +for (let i = 0; i < process.argv.length; i++) { + const a = process.argv[i]; + if (!a.startsWith("--")) continue; + const next = process.argv[i + 1]; + if (next === undefined || next.startsWith("--")) flags[a.slice(2)] = true; + else flags[a.slice(2)] = next; +} + +const SLUG = flags.slug || "amazing"; +const POP = "/Users/jas/aesthetic-computer/pop"; +const SCORE_PATH = `${POP}/big-pictures/${SLUG}.np`; +// `--section all` walks every `verse N` block in order, treating each +// verse-marker as a phrase break (configurable via --inter-verse-beats). +const SECTION = (flags.section || "verse 1").toLowerCase(); +const INTER_VERSE_BEATS = Number(flags["inter-verse-beats"] ?? 2); +const BPM = Number(flags.bpm ?? 70); +const REF_NOTE = flags["ref-note"] ?? "C3"; +const SOURCE_KIND = flags.source ?? "perword"; // perword | bestof +// User feedback: pitches feel too high / not deep enough. Score is +// mostly D3-D4 but jeffrey-pvc's natural baritone is C3-G3. Default +// transpose -12 (one octave down) puts targets in his comfortable +// range, gives the deep ballooned baritone instead of nasal tenor. +const TRANSPOSE = Number(flags.transpose ?? -12); +// True autotune: WORLD f0 replacement per clip (not rubberband -p +// which only shifts the contour). Set false to use cheaper rubberband. +const USE_WORLD = flags["no-world"] ? false : true; +const OUT_PATH = flags.out || `${POP}/big-pictures/out/${SLUG}-score-rendered.mp3`; + +if (!existsSync(SCORE_PATH)) { console.error(`✗ score missing: ${SCORE_PATH}`); process.exit(1); } + +// ── Note → MIDI ────────────────────────────────────────────────────── +const NOTE_BASE = { "C":0,"C#":1,"DB":1,"D":2,"D#":3,"EB":3,"E":4,"F":5, + "F#":6,"GB":6,"G":7,"G#":8,"AB":8,"A":9,"A#":10,"BB":10,"B":11 }; +function noteToMidi(s) { + const u = s.toUpperCase(); + const oct = parseInt(u.slice(-1), 10); + const name = u.slice(0, -1); + return 12 * (oct + 1) + NOTE_BASE[name]; +} +const REF_MIDI = noteToMidi(REF_NOTE); + +// ── Parse score: flat list of {note, raw, weight} per syllable ─────── +const scoreLines = readFileSync(SCORE_PATH, "utf8").split("\n"); +const syllables = []; +function parseRange(startIdx) { + for (let i = startIdx; i < scoreLines.length; i++) { + const l = scoreLines[i].trim(); + if (!l) continue; + if (l.startsWith("#")) continue; + if (/^[a-z]+ \d+$/i.test(l)) break; + for (const tok of l.split(/\s+/)) { + const m = tok.match(/^([A-Ga-g][#b]?-?\d):(.+?)(?:\*(\d+(?:\.\d+)?))?$/); + if (!m) continue; + syllables.push({ note: m[1], raw: m[2], weight: Number(m[3] ?? 1) }); + } + } +} +if (SECTION === "all") { + // Walk every `verse N` header in order; insert a rest weight between + // verses so phrase breaks read musically. + const headers = []; + for (let i = 0; i < scoreLines.length; i++) { + if (/^verse \d+$/i.test(scoreLines[i].trim())) headers.push(i); + } + if (!headers.length) { console.error("✗ section=all but no `verse N` headers found"); process.exit(1); } + headers.forEach((h, idx) => { + const before = syllables.length; + parseRange(h + 1); + if (idx < headers.length - 1 && syllables.length > before && INTER_VERSE_BEATS > 0) { + // Mark a rest by inflating the previous syllable's weight (no audio + // shifts; just gives the next word a later start_beat). + syllables[syllables.length - 1].weight += INTER_VERSE_BEATS; + } + }); +} else { + const sStart = scoreLines.findIndex((l) => l.trim().toLowerCase() === SECTION); + if (sStart < 0) { console.error(`✗ section missing: ${SECTION}`); process.exit(1); } + parseRange(sStart + 1); +} +console.log(`→ score: ${syllables.length} syllables`); + +// Group syllables → words by hyphen-marker chains. Word's effective note +// is its FIRST syllable's note (the attack); duration = sum of its +// syllables' weights × beat_s. +const beat_s = 60.0 / BPM; +let beat_pos = 0; +const words = []; +let curWord = null; +for (const s of syllables) { + const startsContinuation = s.raw.startsWith("-"); + if (startsContinuation && curWord) { + curWord.weights.push(s.weight); + curWord.notes.push(s.note); + } else { + if (curWord) words.push(curWord); + curWord = { + raw: s.raw.replace(/^-|-$/g, ""), + notes: [s.note], + weights: [s.weight], + start_beat: beat_pos, + }; + } + beat_pos += s.weight; +} +if (curWord) words.push(curWord); +for (const w of words) { + w.text = w.raw.replace(/[.,!?;:]/g, "").toLowerCase(); + w.start_s = w.start_beat * beat_s; + w.dur_s = w.weights.reduce((a, b) => a + b, 0) * beat_s; + // Apply global TRANSPOSE so the score sits in jeffrey's baritone + // range. All target_midis (and the per-syllable melody curve below) + // shift by the same amount. + w.target_midi = noteToMidi(w.notes[0]) + TRANSPOSE; + w.target_midis_per_syllable = w.notes.map(n => noteToMidi(n) + TRANSPOSE); + w.target_notes_per_syllable = w.notes.map(n => { + const m = noteToMidi(n) + TRANSPOSE; + const oct = Math.floor(m / 12) - 1; + const name = ["C","C#","D","D#","E","F","F#","G","G#","A","A#","B"][m % 12]; + return `${name}${oct}`; + }); +} +console.log(`→ ${words.length} words; total duration = ${(beat_pos * beat_s).toFixed(2)}s`); + +// ── Locate per-word source clips ───────────────────────────────────── +// Both perword.mjs and best-of-takes.mjs save individual word audio. +// perword: ${slug}-perword.mp3 + ${slug}-perword-words.json (concat'd +// single take per word, with synthesized boundaries). +// best-of-takes: ${slug}-bestof.mp3 + ${slug}-bestof-words.json (best +// pick per word from multiple takes). +// Custom source overrides — point at any vocal stem + words sidecar. +// Useful for the 7-verse cut where the source is the streaming TTS take +// (amazing-7verse-vocal.mp3) rather than the per-word/best-of caches. +const sourceMp3 = flags["source-mp3"] + ? resolve(process.cwd(), flags["source-mp3"]) + : `${POP}/big-pictures/out/${SLUG}-${SOURCE_KIND}.mp3`; +const sourceWordsJson = flags["source-words"] + ? resolve(process.cwd(), flags["source-words"]) + : `${POP}/big-pictures/out/${SLUG}-${SOURCE_KIND}-words.json`; +if (!existsSync(sourceMp3)) { console.error(`✗ source mp3 missing: ${sourceMp3}`); process.exit(1); } +if (!existsSync(sourceWordsJson)) { console.error(`✗ source words missing: ${sourceWordsJson}`); process.exit(1); } +const sourceWords = JSON.parse(readFileSync(sourceWordsJson, "utf8")); + +if (sourceWords.length !== words.length) { + console.warn(`⚠ source has ${sourceWords.length} words, score has ${words.length}; aligning by index`); +} +const n = Math.min(sourceWords.length, words.length); + +// ── Source pitch (jeffrey-pvc baritone) ────────────────────────────── +// Use REF_MIDI as the assumed source pitch. pitchsnap measures exact f0 +// per word; we just use a constant baritone reference for simplicity. +// Per-word semitone shift = target_midi - REF_MIDI. + +// ── Generate stretched + pitched per-word clips ────────────────────── +const tmp = mkdtempSync(`${tmpdir()}/score-render-${SLUG}-`); +const wordClips = []; +for (let i = 0; i < n; i++) { + const w = words[i]; + const sw = sourceWords[i]; + const src_dur = (sw.toMs - sw.fromMs) / 1000; + const stretch = w.dur_s / Math.max(0.05, src_dur); + const semitones = w.target_midi - REF_MIDI; + // Extract this word's audio from the source mp3 + const cutWav = `${tmp}/w${String(i).padStart(2, "0")}-cut.wav`; + spawnSync("ffmpeg", [ + "-hide_banner", "-y", "-loglevel", "error", + "-ss", String(sw.fromMs / 1000), "-i", sourceMp3, + "-t", String(src_dur), "-ar", "44100", "-ac", "1", + cutWav, + ]); + // 1) Rubberband stretch (formant-preserving) to target duration — + // NO pitch shift here; pitch is handled by WORLD next. + const stretchedWav = `${tmp}/w${String(i).padStart(2, "0")}-stretch.wav`; + spawnSync("rubberband", [ + "-t", String(stretch), "--formant", + cutWav, stretchedWav, + ], { stdio: ["ignore", "ignore", "ignore"] }); + if (!existsSync(stretchedWav)) { + console.error(`✗ rubberband stretch failed on word ${i} '${w.text}'`); + continue; + } + let rendWav = stretchedWav; + // 2) WORLD f0 replacement (true autotune) — replaces the source f0 + // contour with the target note(s) entirely, preserving formants. + // Multi-syllable words pass --notes A,B,C with --weights for the + // per-syllable melody curve. + if (USE_WORLD) { + const worldWav = `${tmp}/w${String(i).padStart(2, "0")}-world.wav`; + const notesArg = w.target_notes_per_syllable.join(","); + const weightsArg = w.weights.join(","); + const r = spawnSync(`${POP}/.venv/bin/python`, [ + `${POP}/bin/pitchsnap_world.py`, stretchedWav, worldWav, + "--notes", notesArg, + "--weights", weightsArg, + "--retain", "1.0", + ], { stdio: ["ignore", "ignore", "ignore"] }); + if (r.status === 0 && existsSync(worldWav)) { + rendWav = worldWav; + } else { + console.warn(` ! WORLD failed for '${w.text}', falling back to rubberband -p`); + const rbWav = `${tmp}/w${String(i).padStart(2, "0")}-rb.wav`; + spawnSync("rubberband", [ + "-p", String(semitones), "--formant", + stretchedWav, rbWav, + ], { stdio: ["ignore", "ignore", "ignore"] }); + if (existsSync(rbWav)) rendWav = rbWav; + } + } else if (semitones !== 0) { + const rbWav = `${tmp}/w${String(i).padStart(2, "0")}-rb.wav`; + spawnSync("rubberband", [ + "-p", String(semitones), "--formant", + stretchedWav, rbWav, + ], { stdio: ["ignore", "ignore", "ignore"] }); + if (existsSync(rbWav)) rendWav = rbWav; + } + wordClips.push({ + text: w.text, + start_s: w.start_s, + dur_s: w.dur_s, + src_dur, + stretch, + semitones, + path: rendWav, + }); + console.log( + ` [${String(i + 1).padStart(2)}/${n}] '${w.text.padEnd(8)}' ` + + `start=${w.start_s.toFixed(2)}s dur=${w.dur_s.toFixed(2)}s ` + + `src=${src_dur.toFixed(2)}s stretch=${stretch.toFixed(2)}× ` + + `pitch=${semitones >= 0 ? "+" : ""}${semitones}st` + ); +} + +// ── Place each stretched clip at its score start_s and mix ─────────── +// Use ffmpeg with N inputs, each delayed via adelay, then amix=N. Or for +// large N, do iterative mixdown to avoid command-line bloat. +const totalDur = beat_pos * beat_s; +console.log(`→ mixing ${wordClips.length} clips into ${totalDur.toFixed(2)}s base`); + +// Build the filter graph +const inputs = wordClips.flatMap((c) => ["-i", c.path]); +const filters = wordClips.map((c, idx) => { + const delay_ms = Math.round(c.start_s * 1000); + return `[${idx}:a]adelay=${delay_ms}|${delay_ms}[d${idx}]`; +}); +const sumLabels = wordClips.map((_, idx) => `[d${idx}]`).join(""); +const filterGraph = + filters.join(";") + + `;${sumLabels}amix=inputs=${wordClips.length}:duration=longest:dropout_transition=0:normalize=0[out]`; + +const r = spawnSync("ffmpeg", [ + "-hide_banner", "-y", "-loglevel", "error", + ...inputs, + "-filter_complex", filterGraph, + "-map", "[out]", + "-c:a", "libmp3lame", "-q:a", "4", + OUT_PATH, +]); +if (r.status !== 0) { console.error("✗ mix failed"); process.exit(1); } + +const outDur = execSync(`ffprobe -v error -show_entries format=duration -of default=noprint_wrappers=1:nokey=1 ${OUT_PATH}`).toString().trim(); +console.log(`✓ ${OUT_PATH} (${Number(outDur).toFixed(2)}s)`); + +rmSync(tmp, { recursive: true, force: true }); diff --git a/pop/bin/storyboard.mjs b/pop/bin/storyboard.mjs index 757bfd2d7..0846c816b 100644 --- a/pop/bin/storyboard.mjs +++ b/pop/bin/storyboard.mjs @@ -33,7 +33,12 @@ import { resolve } from "node:path"; const flags = {}; for (let i = 0; i < process.argv.length; i++) { const a = process.argv[i]; - if (a.startsWith("--")) flags[a.slice(2)] = process.argv[i + 1]; + if (a.startsWith("--")) { + const next = process.argv[i + 1]; + // Bare boolean flag if the next token is missing or itself a flag. + if (next === undefined || next.startsWith("--")) flags[a.slice(2)] = true; + else flags[a.slice(2)] = next; + } } const SLUG = flags.slug || "amazing"; @@ -52,6 +57,16 @@ const OUT = flags.out const IMG_DIR = flags["img-dir"] ? resolve(process.cwd(), flags["img-dir"]) : `${POP}/big-pictures/out/${SLUG}-tiktok-frames`; +// `--source-timing` opt-in. When set, look for an alignment sidecar +// (preferring ElevenLabs `${slug}-final.alignment.json`, falling back +// to a `.words.json` file) and use those word boundaries as authoritative +// `slide.start` positions instead of computing `beat × 60/BPM`. Multi- +// syllable words subdivide their window evenly across syllables. +// Default is the legacy beat-grid for backward compatibility. +const SOURCE_TIMING = flags["source-timing"] === true || flags["source-timing"] === "true"; +const ALIGNMENT_PATH = flags.alignment + ? resolve(process.cwd(), flags.alignment) + : null; if (!existsSync(SCORE_PATH)) { console.error(`✗ score file missing: ${SCORE_PATH}`); @@ -160,39 +175,152 @@ const TYPOGRAPHY_STYLES = [ "thick rounded pixel-art letters, friendly chunky bitmap", ]; -// ── Build slides directly from the score ───────────────────────────── +// ── Optional: load source-timing word boundaries ──────────────────── +// Group syllables into words (consecutive syllables that share a word +// via the leading/trailing hyphen markers in the score). For example +// `a-`, `-ma-`, `-zing` form one logical word "amazing". Single tokens +// without hyphens are their own word. +function groupIntoWords(syls) { + const words = []; + let cur = null; + for (let i = 0; i < syls.length; i++) { + const s = syls[i]; + const isStart = s.raw.endsWith("-") && !s.raw.startsWith("-"); + const isMid = s.raw.startsWith("-") && s.raw.endsWith("-"); + const isEnd = s.raw.startsWith("-") && !s.raw.endsWith("-"); + const isSingle = !s.raw.startsWith("-") && !s.raw.endsWith("-"); + if (isStart || isSingle) { + if (cur) words.push(cur); + cur = { syllables: [i], text: s.raw.replace(/^-|-$/g, "") }; + if (isSingle) { words.push(cur); cur = null; } + } else if (isMid || isEnd) { + if (!cur) cur = { syllables: [], text: "" }; + cur.syllables.push(i); + cur.text += s.raw.replace(/^-|-$/g, ""); + if (isEnd) { words.push(cur); cur = null; } + } + } + if (cur) words.push(cur); + return words; +} + +// Find an alignment sidecar. Priority: +// 1. --alignment explicit +// 2. -final.alignment.json (ElevenLabs with-timestamps output) +// 3. -vocal.mp3.alignment.json +// 4. -final-words.json + sibling shapes (legacy) +function findAlignment() { + const tryPaths = []; + if (ALIGNMENT_PATH) tryPaths.push(ALIGNMENT_PATH); + tryPaths.push(`${POP}/big-pictures/out/${SLUG}-final.mp3.alignment.json`); + tryPaths.push(`${POP}/big-pictures/out/${SLUG}-final.alignment.json`); + tryPaths.push(`${POP}/big-pictures/out/${SLUG}-vocal.mp3.alignment.json`); + tryPaths.push(`${POP}/big-pictures/out/${SLUG}-vocal.alignment.json`); + for (const p of tryPaths) { + if (existsSync(p)) { + const doc = JSON.parse(readFileSync(p, "utf8")); + if (Array.isArray(doc.words) && doc.words.length > 0) { + return { path: p, words: doc.words }; + } + } + } + return null; +} + +let sourceWords = null; +if (SOURCE_TIMING) { + const found = findAlignment(); + if (!found) { + console.warn("⚠ --source-timing requested but no alignment file found; falling back to beat-grid"); + } else { + sourceWords = found.words; + console.log(` source-timing ← ${found.path} (${sourceWords.length} words)`); + } +} + +// ── Build slides ───────────────────────────────────────────────────── const beatSec = 60.0 / BPM; -let beatPos = 0; -const slides = syllables.map((syl, i) => { - const start = beatPos * beatSec; - beatPos += syl.weight; - const end = beatPos * beatSec; - // Strip leading/trailing hyphens for the displayed text — those - // are score-syntax markers indicating multi-syllable continuity. - const visible = syl.raw.replace(/^-|-$/g, "").replace(/[.,!?;:]/g, ""); - const colorIdx = i % EMOTIONAL_COLORS.length; - const typoIdx = i % TYPOGRAPHY_STYLES.length; - const dur = end - start; - const transitionMs = Math.round(Math.max(120, Math.min(450, dur * 280))); - return { - i, - start: Number(start.toFixed(3)), - end: Number(end.toFixed(3)), - duration: Number(dur.toFixed(3)), - text: visible, - rawText: syl.raw, - note: syl.note, - weight: syl.weight, - image: `word-${String(i).padStart(3, "0")}.jpg`, - transition: "slideleft", - transitionMs, - bgColor: EMOTIONAL_COLORS[colorIdx].bg, - letterColor: EMOTIONAL_COLORS[colorIdx].letters, - typography: TYPOGRAPHY_STYLES[typoIdx], - }; -}); - -const totalScoreSec = beatPos * beatSec; +let slides; +if (sourceWords) { + // Source-timing path: align score-words to alignment-words index-wise, + // then subdivide each word's window evenly across its syllables. + const scoreWords = groupIntoWords(syllables); + const n = Math.min(scoreWords.length, sourceWords.length); + if (scoreWords.length !== sourceWords.length) { + console.warn(` ⚠ score-word count (${scoreWords.length}) != alignment-word count (${sourceWords.length}); using first ${n} pairs`); + } + slides = []; + for (let wi = 0; wi < n; wi++) { + const sw = scoreWords[wi]; + const aw = sourceWords[wi]; + const wStart = (aw.fromMs ?? aw.from ?? 0) / 1000; + const wEnd = (aw.toMs ?? aw.to ?? 0) / 1000; + const wDur = Math.max(0, wEnd - wStart); + const sylCount = sw.syllables.length; + for (let k = 0; k < sylCount; k++) { + const sylIdx = sw.syllables[k]; + const syl = syllables[sylIdx]; + const sStart = wStart + (wDur * k) / sylCount; + const sEnd = wStart + (wDur * (k + 1)) / sylCount; + const visible = syl.raw.replace(/^-|-$/g, "").replace(/[.,!?;:]/g, ""); + const colorIdx = sylIdx % EMOTIONAL_COLORS.length; + const typoIdx = sylIdx % TYPOGRAPHY_STYLES.length; + const dur = sEnd - sStart; + const transitionMs = Math.round(Math.max(120, Math.min(450, dur * 280))); + slides.push({ + i: sylIdx, + start: Number(sStart.toFixed(3)), + end: Number(sEnd.toFixed(3)), + duration: Number(dur.toFixed(3)), + text: visible, + rawText: syl.raw, + note: syl.note, + weight: syl.weight, + image: `word-${String(sylIdx).padStart(3, "0")}.jpg`, + transition: "slideleft", + transitionMs, + bgColor: EMOTIONAL_COLORS[colorIdx].bg, + letterColor: EMOTIONAL_COLORS[colorIdx].letters, + typography: TYPOGRAPHY_STYLES[typoIdx], + }); + } + } +} else { + // Legacy beat-grid path (default) — preserves backward compat. + let beatPos = 0; + slides = syllables.map((syl, i) => { + const start = beatPos * beatSec; + beatPos += syl.weight; + const end = beatPos * beatSec; + const visible = syl.raw.replace(/^-|-$/g, "").replace(/[.,!?;:]/g, ""); + const colorIdx = i % EMOTIONAL_COLORS.length; + const typoIdx = i % TYPOGRAPHY_STYLES.length; + const dur = end - start; + const transitionMs = Math.round(Math.max(120, Math.min(450, dur * 280))); + return { + i, + start: Number(start.toFixed(3)), + end: Number(end.toFixed(3)), + duration: Number(dur.toFixed(3)), + text: visible, + rawText: syl.raw, + note: syl.note, + weight: syl.weight, + image: `word-${String(i).padStart(3, "0")}.jpg`, + transition: "slideleft", + transitionMs, + bgColor: EMOTIONAL_COLORS[colorIdx].bg, + letterColor: EMOTIONAL_COLORS[colorIdx].letters, + typography: TYPOGRAPHY_STYLES[typoIdx], + }; + }); +} + +// Total length: in source-timing mode use the last slide's end; in +// legacy beat-grid mode use the cumulative beat position. +const totalScoreSec = sourceWords + ? (slides.length > 0 ? slides[slides.length - 1].end : 0) + : syllables.reduce((sum, s) => sum + s.weight, 0) * beatSec; const storyboard = { schema: "ac/big-pictures/storyboard@2", slug: SLUG, diff --git a/pop/bin/timeline.py b/pop/bin/timeline.py index c28434577..9e80e39a7 100644 --- a/pop/bin/timeline.py +++ b/pop/bin/timeline.py @@ -104,7 +104,12 @@ def main(): audio_dir = os.path.dirname(os.path.abspath(args.audio)) audio_stem = os.path.splitext(os.path.basename(args.audio))[0] if args.words is None: + # Prefer the mfa-aligned words (correct lyric text + accurate + # timing) over whisper output (correct timing but wrong text + # because pitchsnap distortion confuses recognition). for cand in ( + f"{audio_dir}/{slug}-mfa-words.json", + f"{audio_dir}/{audio_stem}-mfa-words.json", f"{audio_dir}/{audio_stem}-words.json", f"{audio_dir}/{slug}-perline-words.json", f"{audio_dir}/{slug}-7-warm-words.json", @@ -114,7 +119,8 @@ def main(): ): if os.path.exists(cand): args.words = cand - print(f" words: {cand}") + src = "mfa" if "mfa" in cand else "whisper" + print(f" words: {cand} ({src})") break if args.events is None: for cand in ( @@ -145,11 +151,17 @@ def main(): onsets = [] # ── figure layout — generous heights for readability ───────────── - height_ratios = [5.0] # piano roll (tallest) - if words: height_ratios.append(1.6) # utterances - if events: height_ratios.append(2.4) # autotune (zigzag needs space) - height_ratios.append(2.8) # waveform - n_panels = len(height_ratios) + # ORDER (top → bottom): HEARD, SCORE, PITCH-SNAP, AUDIO + # Reading order matches mental model: "what came out of the speaker" + # sits in front of "what the score wanted" so the user compares + # heard → expected → snap-correction → waveform top-down. + panels_spec = [] # (kind, height_ratio) + if words: panels_spec.append(("words", 1.6)) + panels_spec.append(("score", 5.0)) + if events: panels_spec.append(("events", 2.4)) + panels_spec.append(("audio", 2.8)) + height_ratios = [h for _, h in panels_spec] + n_panels = len(panels_spec) fig_w = max(22, total * 0.85) fig_h = sum(height_ratios) * 1.15 + 0.6 fig, axs = plt.subplots( @@ -161,10 +173,11 @@ def main(): fig.suptitle(title, fontsize=28, fontweight="bold", color=FG, y=0.985, path_effects=stroke(4)) - panel_idx = 0 + # Map panel kind → axes index + kind_to_ax = {kind: axs[i] for i, (kind, _) in enumerate(panels_spec)} - # ── 1. PIANO ROLL ──────────────────────────────────────────────── - ax_roll = axs[panel_idx]; panel_idx += 1 + # ── PIANO ROLL ─────────────────────────────────────────────────── + ax_roll = kind_to_ax["score"] midis = [note_to_midi(s["note"]) for s in slides] midi_min, midi_max = min(midis) - 1, max(midis) + 1 # alternating pitch lanes @@ -204,17 +217,17 @@ def main(): leg = ax_roll.legend(loc="upper right", fontsize=13, frameon=True, facecolor=PANEL_BG, edgecolor=DIM, labelcolor=FG) - # ── 2. UTTERANCES ──────────────────────────────────────────────── + # ── UTTERANCES ─────────────────────────────────────────────────── if words: - ax_utt = axs[panel_idx]; panel_idx += 1 + ax_utt = kind_to_ax["words"] for w in words: x0 = w["fromMs"] / 1000.0 x1 = w["toMs"] / 1000.0 ax_utt.add_patch(mpatches.FancyBboxPatch( (x0, 0.18), x1 - x0, 0.64, boxstyle="round,pad=0.01,rounding_size=0.04", - facecolor=GREEN, alpha=0.35, edgecolor=GREEN, - linewidth=1.4, zorder=2)) + facecolor=GREEN, alpha=0.45, edgecolor=GREEN, + linewidth=1.6, zorder=2)) ax_utt.text((x0 + x1) / 2, 0.5, w["text"], ha="center", va="center", fontsize=15, fontweight="bold", color=FG, zorder=3, @@ -225,10 +238,53 @@ def main(): ax_utt.tick_params(axis="x", labelsize=12, length=5) ax_utt.grid(axis="x", color=GRID, linestyle="-", linewidth=0.7) ax_utt.set_axisbelow(True) + # Connect each whisper word to the closest score slide (by + # text-prefix match if possible, else nearest-time). Drift is + # the horizontal slope of the connecting line. + # We draw the connector across the figure between ax_utt's + # bottom edge and ax_roll's top edge using fig.add_artist. + from matplotlib.patches import ConnectionPatch + used_slide_idxs = set() + for w in words: + wtext = w["text"].lower().strip(",.!?;: ") + wmid = (w["fromMs"] + w["toMs"]) / 2000.0 + # Find best matching slide: prefer whichever slide's text is + # a prefix/substring of the whisper word AND is closest + # in time and not already used. + best, best_score = None, 1e9 + for j, s in enumerate(slides): + if j in used_slide_idxs: + continue + stext = s["text"].lower().strip("-") + # syllable-of-word match: whisper "amazing" should pair + # with the FIRST score syllable 'a' (the word's onset). + prefix_match = wtext.startswith(stext) or stext.startswith(wtext) + tdist = abs(s["start"] - w["fromMs"] / 1000.0) + score = tdist + (0 if prefix_match else 1.5) + if score < best_score: + best, best_score = j, score + if best is None: + continue + used_slide_idxs.add(best) + slide = slides[best] + x_heard = (w["fromMs"] / 1000.0 + w["toMs"] / 1000.0) / 2 + x_score = slide["start"] + (slide["end"] - slide["start"]) / 2 + drift = x_heard - x_score + # color drift by magnitude — yellow ≤ 0.2s, orange ≤ 0.6s, red beyond + if abs(drift) < 0.20: dcol = "#7fe070" + elif abs(drift) < 0.60: dcol = "#ffaa33" + else: dcol = "#ff5566" + con = ConnectionPatch( + xyA=(x_heard, 0.18), coordsA=ax_utt.transData, + xyB=(x_score, midi_max + 0.55), coordsB=ax_roll.transData, + color=dcol, linewidth=1.2, alpha=0.8, zorder=20, + linestyle=("--" if abs(drift) > 0.20 else "-"), + ) + fig.add_artist(con) - # ── 3. AUTOTUNE ────────────────────────────────────────────────── + # ── AUTOTUNE ───────────────────────────────────────────────────── if events: - ax_at = axs[panel_idx]; panel_idx += 1 + ax_at = kind_to_ax["events"] max_st = max(abs(e.get("semitones", 0)) for e in events) or 1.0 for i, e in enumerate(events): nat = float(e.get("naturalStart", 0)) @@ -270,8 +326,8 @@ def main(): bbox=dict(facecolor=BG, edgecolor=DIM, boxstyle="round,pad=0.4")) - # ── 4. WAVEFORM ────────────────────────────────────────────────── - ax_wave = axs[panel_idx] + # ── WAVEFORM ───────────────────────────────────────────────────── + ax_wave = kind_to_ax["audio"] if y is not None: t = np.arange(len(y)) / sr ax_wave.plot(t, y, color=CYAN, linewidth=0.7, alpha=0.9, zorder=2) diff --git a/pop/bin/track-poster.py b/pop/bin/track-poster.py new file mode 100644 index 000000000..a717307f7 --- /dev/null +++ b/pop/bin/track-poster.py @@ -0,0 +1,441 @@ +#!/usr/bin/env python3 +"""track-poster.py — totalizing "how this track got made" PNG. + +Layout (portrait, 18×14): + ┌──────────────── title ─────────────────┐ + │ row 1: lyrics · score · melody contour │ + │ row 2: arrangement timeline (3 lanes) │ + │ row 3: pipeline strip (8 stations) │ + │ footer: output stats │ + └────────────────────────────────────────┘ + +The arrangement timeline (row 2) is the key panel: it shows when every +word enters, the per-word target pitch (the ballooned vocal), the +melody-bell strikes, and the harmonic bed chord changes — all on the +same time axis so it reads as a score. + +Style mirrors `pop/bin/timeline.py`: dark bg, cream type, saturated +accents per lane, monospace. +""" + +import json +import sys +import re +from pathlib import Path + +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +from matplotlib.patches import FancyBboxPatch, FancyArrowPatch, Rectangle + +ROOT = Path(__file__).resolve().parent.parent +SLUG = sys.argv[1] if len(sys.argv) > 1 else "amazing" +OUT = Path(sys.argv[2]) if len(sys.argv) > 2 else ROOT / "big-pictures" / "out" / f"{SLUG}-poster.png" + +LYR = ROOT / "big-pictures" / f"{SLUG}.txt" +NP = ROOT / "big-pictures" / f"{SLUG}.np" +WORDS = ROOT / "big-pictures" / "out" / f"{SLUG}-7verse-vocal-words.json" +VOCAL = ROOT / "big-pictures" / "out" / f"{SLUG}-7verse-vocal.mp3" +ALIGN = ROOT / "big-pictures" / "out" / f"{SLUG}-7verse-pitched-alignment.json" +WALTZ = Path("/Users/jas/aesthetic-computer/recap/out/waltz-events.json") + +# ── colors ────────────────────────────────────────────────────────────── +BG = "#0a0a14" +PANEL_BG = "#10101e" +GRID = "#1f1f30" +INK = "#f3f0d8" +DIM = "#7a8870" +CITRUS = "#c6ff4f" +ACCENT = "#ffd166" +MELODY = "#5fe8b8" # mint — melody bells +VOCAL_C = "#ff8a3d" # orange — ballooned vocal +BED_C = "#5588ff" # blue — harmonic bed (chord roots) +RED = "#ff5566" + +# ── score helpers (mirror score-render.mjs) ───────────────────────────── +NOTE_BASE = {"C":0,"C#":1,"DB":1,"D":2,"D#":3,"EB":3,"E":4,"F":5,"F#":6, + "GB":6,"G":7,"G#":8,"AB":8,"A":9,"A#":10,"BB":10,"B":11} + +def note_to_midi(s): + s = s.upper() + return 12 * (int(s[-1]) + 1) + NOTE_BASE[s[:-1]] + +def midi_to_note(m): + name = ["C","C#","D","D#","E","F","F#","G","G#","A","A#","B"][m % 12] + return f"{name}{m//12 - 1}" + +def parse_score(text): + """Return (verses, all_syllables_with_break_flag).""" + verses = [] + all_syl = [] + current = None + for raw in text.splitlines(): + line = raw.strip() + if not line or line.startswith("#"): + continue + if re.match(r"^verse\s+\d+$", line, re.I): + current = [] + verses.append(current) + if all_syl: all_syl[-1]["verse_break"] = True + continue + if current is None: continue + line_tokens = [] + for tok in line.split(): + m = re.match(r"^([A-G][#b]?\d):([^*]+)\*([\d.]+)$", tok) + if not m: continue + note, syl, beats = m.group(1), m.group(2), float(m.group(3)) + line_tokens.append((note, note_to_midi(note), syl, beats)) + all_syl.append({"note": note, "midi": note_to_midi(note), + "syl": syl, "beats": beats, "verse_break": False}) + if line_tokens: current.append(line_tokens) + return verses, all_syl + +def syllables_to_words(all_syl, bpm=70, transpose=0, inter_verse_beats=2): + """Mirror score-render.mjs --section all: hyphenated tail merges into + its head word; verse-break inflates the previous syllable's beat + weight by inter_verse_beats.""" + beat_s = 60.0 / bpm + # Apply verse-break weight inflation (matches score-render.mjs) + syls = [] + for s in all_syl: + w = s["beats"] + # We add the inter-verse rest AFTER the syllable that has the flag + syls.append({**s, "weight": w}) + for i, s in enumerate(syls): + if s["verse_break"]: + s["weight"] += inter_verse_beats + words = [] + cur = None + pos = 0.0 + for s in syls: + cont = s["syl"].startswith("-") + if cont and cur is not None: + cur["weights"].append(s["weight"]) + cur["notes"].append(s["midi"]) + else: + if cur is not None: words.append(cur) + cur = {"text": s["syl"].lstrip("-").rstrip("-"), + "start_beat": pos, + "notes": [s["midi"]], + "weights": [s["weight"]]} + pos += s["weight"] + if cur is not None: words.append(cur) + for w in words: + w["start_s"] = w["start_beat"] * beat_s + w["dur_s"] = sum(w["weights"]) * beat_s + w["midi"] = w["notes"][0] + transpose + w["text"] = w["text"].replace("-", "").lower() + return words, pos * beat_s + +# ── load all data ─────────────────────────────────────────────────────── +lyrics = LYR.read_text() +score_text = NP.read_text() +verses, all_syl = parse_score(score_text) +score_words_hymn, score_dur_hymn = syllables_to_words( + all_syl, bpm=70, transpose=-5, inter_verse_beats=2) + +words = [] +if WORDS.exists(): + words = json.loads(WORDS.read_text()) + +# The mp3 on the Desktop is now SCORE-PACED (stretched) — every word +# starts at its score beat position with duration from beat weights. +# Use the score's own timing for the vocal lane; pull text + note from +# the alignment sidecar (or fall back to score_words_hymn). +score_words = [] +if ALIGN.exists(): + aligned_json = json.loads(ALIGN.read_text()) + N = min(len(aligned_json), len(score_words_hymn)) + for i in range(N): + entry = aligned_json[i] + sw = score_words_hymn[i] + score_words.append({ + "text": entry["text"].lower(), + "start_s": sw["start_s"], # score-pace start + "dur_s": sw["dur_s"], # score-pace duration + "midi": entry["midi"], + "midi_score": note_to_midi(entry["note"]), + }) +else: + for sw in score_words_hymn: + score_words.append({ + "text": sw["text"], + "start_s": sw["start_s"], + "dur_s": sw["dur_s"], + "midi": sw["notes"][0] - 5, + "midi_score": sw["notes"][0], + }) + +vocal_kb = VOCAL.stat().st_size / 1024 if VOCAL.exists() else 0 +spoken_dur = (words[-1]["toMs"] / 1000.0) if words else 0 +n_words = len(words) + +# Hymn-pace total duration from the score (stretched output spans this) +TOTAL_DUR = max(score_words[-1]["start_s"] + score_words[-1]["dur_s"], + score_dur_hymn) if score_words else 60 + +# ── figure ────────────────────────────────────────────────────────────── +plt.rcParams.update({ + "font.family": "monospace", + "font.monospace": ["Menlo", "DejaVu Sans Mono", "Consolas", "Courier New"], + "axes.edgecolor": INK, "text.color": INK, + "xtick.color": INK, "ytick.color": INK, +}) + +fig = plt.figure(figsize=(18, 14), facecolor=BG) +fig.subplots_adjust(left=0.04, right=0.98, top=0.95, bottom=0.04, + hspace=0.35, wspace=0.18) + +# Title +fig.text(0.04, 0.965, "amazing grace", + color=INK, fontsize=34, fontweight="bold") +fig.text(0.04, 0.943, "· 7 verses · how this track got made", + color=CITRUS, fontsize=14) +fig.text(0.98, 0.965, "pop/big-pictures", + color=DIM, fontsize=11, ha="right") +fig.text(0.98, 0.948, f"jeffrey-pvc · {TOTAL_DUR:.0f}s · 2026-05-05", + color=DIM, fontsize=11, ha="right") + +# Three rows: 3-col header / full-width timeline / full-width pipeline +gs = fig.add_gridspec(3, 3, height_ratios=[1.0, 1.4, 0.85], + hspace=0.45) + +def panel(ax, title): + ax.set_facecolor(BG) + for spine in ax.spines.values(): + spine.set_visible(False) + ax.set_xticks([]); ax.set_yticks([]) + ax.text(0.0, 1.0, title, color=CITRUS, fontsize=13, fontweight="bold", + family="monospace", transform=ax.transAxes, va="top") + +# ── (0,0) lyrics ──────────────────────────────────────────────────────── +ax = fig.add_subplot(gs[0, 0]); panel(ax, "1 · what i write — words") +preview = [] +for line in lyrics.splitlines(): + if not line.strip(): preview.append("") + elif re.match(r"^verse\s+\d+$", line.strip(), re.I): + preview.append(line.strip().upper()) + else: + preview.append(" " + line) +shown = preview[:18] +text_block = "\n".join(shown) +if len(preview) > 18: + text_block += f"\n … ({len(verses)} verses · {sum(len(v) for v in verses)} lines total)" +ax.text(0.02, 0.88, text_block, color=INK, fontsize=8.5, va="top", + family="monospace", transform=ax.transAxes) + +# ── (0,1) score ───────────────────────────────────────────────────────── +ax = fig.add_subplot(gs[0, 1]); panel(ax, "2 · what i write — pitch") +score_preview = [".np NOTE:syllable*beats", "", "verse 1"] +for line_tokens in verses[0]: + txt = " ".join(f"{n}:{syl}*{int(b)}" for n,_,syl,b in line_tokens) + if len(txt) > 60: txt = txt[:57] + "…" + score_preview.append(" " + txt) +score_preview += ["", + "verses 2-7 → same melody, new syllables", + "", + "key : G major pentatonic", + "range: D3 → D4 (climax on 'blind')", + "meter: common meter 8-6-8-6", + "time : 3/4 · 70 bpm (hymn pacing)"] +ax.text(0.02, 0.88, "\n".join(score_preview), color=INK, fontsize=8.5, + va="top", family="monospace", transform=ax.transAxes) + +# ── (0,2) verse-1 melody contour ──────────────────────────────────────── +ax = fig.add_subplot(gs[0, 2]); panel(ax, "3 · melody contour (verse 1)") +xs, ys, sylabs = [], [], [] +t = 0.0 +for line_tokens in verses[0]: + for note, midi, syl, beats in line_tokens: + xs.append(t); ys.append(midi); sylabs.append(syl) + t += beats + t += 1 +inner = ax.inset_axes([0.0, 0.18, 1.0, 0.62], facecolor=BG) +inner.plot(xs, ys, color=CITRUS, linewidth=1.6, marker="o", + markersize=4, markerfacecolor=ACCENT, markeredgecolor=ACCENT) +inner.set_facecolor(BG) +for spine in inner.spines.values(): + spine.set_color(DIM); spine.set_linewidth(0.6) +inner.set_yticks([note_to_midi(n) for n in ["D3","G3","B3","D4"]]) +inner.set_yticklabels(["D3","G3","B3","D4"], color=DIM, fontsize=8) +inner.set_xticks([]) +inner.set_ylim(min(ys)-1.5, max(ys)+1.5) +if "blind" in sylabs: + ci = sylabs.index("blind") + inner.annotate("blind", xy=(xs[ci], ys[ci]), xytext=(xs[ci]+1.5, ys[ci]+1.3), + color=ACCENT, fontsize=9, + arrowprops=dict(arrowstyle="-", color=ACCENT, lw=0.8)) +ax.text(0.02, 0.10, "verse 1 contour · identical for verses 2-7", + color=DIM, fontsize=9, transform=ax.transAxes) +ax.text(0.02, 0.04, "→ same notes, new lyrics each pass = the hymn form.", + color=INK, fontsize=9, transform=ax.transAxes) + +# ── ROW 2: arrangement timeline ───────────────────────────────────────── +ax = fig.add_subplot(gs[1, :]) +ax.set_facecolor(PANEL_BG) +for sp in ax.spines.values(): sp.set_color(DIM); sp.set_linewidth(0.6) +ax.text(0.005, 0.97, "4 · the arrangement in time", + color=CITRUS, fontsize=14, fontweight="bold", transform=ax.transAxes, + va="top") +ax.text(0.005, 0.92, + "every layer on a shared time axis · vocal entries · bell strikes · chord roots", + color=DIM, fontsize=10, transform=ax.transAxes, va="top") + +ax.set_xlim(0, TOTAL_DUR) +# Y-lanes: 80-95 melody bells / 50-75 ballooned vocal / 20-45 bed / 5-15 axis +ax.set_ylim(0, 100) +ax.set_yticks([]); ax.set_xticks([]) + +# Lane labels (left margin) — drawn in axes coords as separate text +def lane_label(y, name, sub, color, vol): + ax.text(0.005, y/100 - 0.005, name, color=color, fontsize=11, + fontweight="bold", transform=ax.transAxes, va="top") + ax.text(0.005, y/100 - 0.04, sub, color=DIM, fontsize=8.5, + transform=ax.transAxes, va="top") + ax.text(0.005, y/100 - 0.07, vol, color=INK, fontsize=8.5, + transform=ax.transAxes, va="top", family="monospace") + +# Convert seconds to data x — already same since xlim==(0, TOTAL_DUR) +# Lane 1 — Melody bells (sinebells, +12st = octave above vocal) +lane_label(95, "melody bells", + "sinebells synth · +12st (oct above)\nscore-paced strikes", + MELODY, "vol 0.40 · 171 strikes") +bell_pitches = [w["midi_score"] + 12 for w in score_words] +midi_min, midi_max = min(bell_pitches), max(bell_pitches) +def midi_to_y(m, lo, hi, y0, y1): + if hi == lo: return (y0 + y1)/2 + return y0 + (m - lo) / (hi - lo) * (y1 - y0) +for w in score_words: + midi = w["midi_score"] + 12 + y = midi_to_y(midi, midi_min, midi_max, 80, 92) + ax.plot([w["start_s"], w["start_s"]], [80, y], + color=MELODY, linewidth=0.7, alpha=0.6) + ax.plot(w["start_s"], y, "o", color=MELODY, markersize=3.0) + +# Lane 2 — Pitched + stretched vocal (each word held to score duration) +lane_label(73, "pitched vocal", + "WORLD f0 → rubberband stretch (formant)\nheld to score note durations", + VOCAL_C, "vol 2.5 · 171 words · -5st") +voc_pitches = [w["midi"] for w in score_words] +v_min, v_max = min(voc_pitches), max(voc_pitches) +for i, w in enumerate(score_words): + y = midi_to_y(w["midi"], v_min, v_max, 52, 68) + ax.plot([w["start_s"], w["start_s"] + w["dur_s"]], [y, y], + color=VOCAL_C, linewidth=1.6, alpha=0.9) + # Sample word labels: every 8th word OR every climax pitch + if w["text"] in ("amazing", "wretch", "blind", "fear", "believed", + "snares", "home", "endures", "fail", "peace", + "snow", "mine", "years", "praise", "begun"): + ax.text(w["start_s"] + w["dur_s"]/2, y + 1.4, w["text"], + color=INK, fontsize=7.5, ha="center", va="bottom", + family="monospace", alpha=0.95, fontweight="bold") + +# Lane 3 — Harmonic bed: thin colored ribbon (chord changes), not a wall. +lane_label(45, "harmonic bed", + "sinebells · I-IV-I-V · 70 bpm · G major", BED_C, + "vol 0.20 · 1237 notes") +beat_s = 60.0 / 70.0 +bar_s = beat_s * 3 +n_bars = int(TOTAL_DUR / bar_s) + 1 +chord_seq = [0, 3, 0, 4] +chord_color = {0: "#3a6cd6", 3: "#5fa9ff", 4: "#a4d4ff"} +chord_letter = {0: "G", 3: "C", 4: "D"} +# Ribbon at y=27-32 with a thin tick down at each bar 1 beat +ax.plot([0, TOTAL_DUR], [29.5, 29.5], color=DIM, linewidth=0.4, alpha=0.5) +for bar in range(n_bars): + x = bar * bar_s + deg = chord_seq[bar % 4] + rect = Rectangle((x, 27), bar_s * 0.97, 5, + facecolor=chord_color[deg], edgecolor="none", alpha=0.85) + ax.add_patch(rect) + if bar % 4 == 0: + ax.text(x + bar_s*2, 24.5, chord_letter[deg], color=BED_C, + fontsize=8.5, ha="center", va="top", fontweight="bold") + +# Verse markers — derived from whisper word starts at each verse's first +# word. Index of first word per verse: verse 1 starts at word 0, verse 2 +# at word 26 (after 26 words in V1), etc. Score has 26+22+26+24+24+24+25 +# = 171 words across 7 verses. Use cumulative. +verse_word_idx = [] +acc = 0 +for v in verses: + verse_word_idx.append(acc) + # syllables in this verse, grouped to words (count of non-continuation): + head_syls = sum(1 for line in v for n,_,syl,_ in line if not syl.startswith("-")) + acc += head_syls +for vi, wi in enumerate(verse_word_idx): + if wi >= len(score_words): break + vs = score_words[wi]["start_s"] + ax.axvline(vs, ymin=0.10, ymax=0.83, color=DIM, linewidth=0.5, + linestyle="--", alpha=0.45) + ax.text(vs + 0.1, 84, f"v{vi+1}", color=DIM, fontsize=8.5, + va="bottom", fontweight="bold") + +# Time axis at bottom — scale tick interval based on duration +tick_step = 5 if TOTAL_DUR < 90 else (10 if TOTAL_DUR < 200 else 30) +for t in range(0, int(TOTAL_DUR) + 1, tick_step): + ax.plot([t, t], [8, 12], color=INK, linewidth=0.6) + mm = t // 60; ss = t % 60 + ax.text(t, 5.5, f"{mm}:{ss:02d}", color=INK, fontsize=8.5, + ha="center", va="top", family="monospace") +ax.plot([0, TOTAL_DUR], [10, 10], color=INK, linewidth=0.4, alpha=0.5) + +# Total mix dB summary in the panel header. +import math +ax.text(0.78, 0.965, "MIX vocal +8.0 dB · bells -8.0 dB · bed -14.0 dB", + color=DIM, fontsize=10, transform=ax.transAxes, va="top", + family="monospace") + +# ── ROW 3: pipeline strip ─────────────────────────────────────────────── +ax = fig.add_subplot(gs[2, :]); panel(ax, "5 · what goes down — the pipeline") +ax.set_xlim(0, 100); ax.set_ylim(0, 100) + +stations = [ + ("write\nlyrics + score", "amazing.txt\namazing.np", 3.0, CITRUS), + ("say.mjs", "stab 0.6 · sim 0.9", 12, INK), + ("ElevenLabs\njeffrey-pvc", "voice clone\nneutral:0", 21, ACCENT), + ("align.mjs\nwhisper-cli", "+ align-words\n(twas/tis fix)", 30, INK), + ("score-pitch\nWORLD", "171 notes · -5st\nf0 replace", 39, ACCENT), + ("score-stretch\nrubberband","held to beats\n8× max stretch",48, ACCENT), + ("melody-bells", "sinebells · +12st\nscore-paced",57, MELODY), + ("waltz bed", "sinebells · 70 bpm\nI-IV-I-V", 66, BED_C), + ("ffmpeg amix", "vox 2.5 · bells 0.40\nbed 0.20",75, INK), + ("finalize", "ID3 + cover", 84, CITRUS), +] +for (top, bot, x, color) in stations: + box = FancyBboxPatch((x-4.0, 35), 8.0, 28, + boxstyle="round,pad=0.25,rounding_size=0.8", + linewidth=1.3, edgecolor=color, facecolor=BG) + ax.add_patch(box) + ax.text(x, 55, top, color=color, fontsize=7.6, ha="center", + va="center", fontweight="bold") + ax.text(x, 41, bot, color=DIM, fontsize=6.4, ha="center", va="center") +for i in range(len(stations)-1): + arrow = FancyArrowPatch((stations[i][2]+4.0, 49), + (stations[i+1][2]-4.0, 49), + arrowstyle="->", color=INK, lw=1.0, + mutation_scale=9) + ax.add_patch(arrow) + +ax.text(3.5, 75, "JEFFREY", color=CITRUS, fontsize=10, + ha="center", fontweight="bold") +ax.text(3.5, 68, "writes by hand", color=DIM, fontsize=8.5, ha="center") +ax.text(56, 75, "AESTHETIC.COMPUTER", color=CITRUS, fontsize=10, + ha="center", fontweight="bold") +ax.text(56, 68, "voice · pitch · bells · bed · mix · packaging", + color=DIM, fontsize=8.5, ha="center") + +mm, ss = int(TOTAL_DUR // 60), int(TOTAL_DUR % 60) +ax.text(50, 18, + f"output · {mm}:{ss:02d} hymn pace · 3 layers · vocal pitched -5st + held to score · G major pentatonic", + color=ACCENT, fontsize=11, ha="center", fontweight="bold") +ax.text(50, 10, + "~/Desktop/amazing-grace-7verse.mp3 (pitched → stretched → bells + bed → mix)", + color=INK, fontsize=9, ha="center", family="monospace") +ax.text(50, 4, + "every step content-hash cached · reruns cost $0 unless inputs change", + color=DIM, fontsize=8.5, ha="center", style="italic") + +OUT.parent.mkdir(parents=True, exist_ok=True) +fig.savefig(OUT, dpi=140, facecolor=BG, bbox_inches="tight", pad_inches=0.3) +print(f"✓ wrote {OUT} ({OUT.stat().st_size/1024:.0f} KB)") diff --git a/pop/bin/validate_word.py b/pop/bin/validate_word.py index 34c3aaa0b..ececb8168 100755 --- a/pop/bin/validate_word.py +++ b/pop/bin/validate_word.py @@ -89,6 +89,31 @@ def density(glyph_img): return float(mask.sum()) / max(1, mask.shape[0] * mask.shape[1]) +def disconnected_pieces(glyph_img): + """Count un-dilated connected components in the glyph's alpha + mask. The dilation step in extract_glyphs merges fragments into + one bounding box, but the underlying mask retains the original + structure. A clean letter is exactly 1 piece (for 'i'/'j' it's + 2: dot + body). >2 means a broken glyph — e.g. a 't' with the + crossbar floating detached from the stem and a chunk missing + from the stem (seen on amazing slide 18 'but'). Small spurious + pieces (<8% of the glyph's total mask area) don't count — they + might be anti-aliasing residue.""" + from scipy.ndimage import label + arr = np.array(glyph_img) + mask = (arr[:, :, 3] > 128).astype(np.uint8) + if mask.sum() == 0: + return 0 + labeled, n = label(mask) + if n <= 1: + return n + # Drop tiny components (anti-aliasing or stray pixels) + sizes = [(i, int((labeled == i).sum())) for i in range(1, n + 1)] + total = sum(s for _, s in sizes) + big = [i for i, sz in sizes if sz >= max(8, total * 0.06)] + return len(big) + + def stroke_balance(glyph_img): """For letters with an enclosed counter (a/b/d/e/g/o/p/q): the left and right vertical strokes flanking the counter should be @@ -214,9 +239,20 @@ def main(): expected_letters = [c for c in expected_word if c.isalpha()] topology_failures = [] # (idx, letter, hole_ratio, expected_min) shape_failures = [] + # 'i' and 'j' legitimately have 2 disconnected pieces (dot + body); + # everything else should be exactly 1. + EXPECTED_PIECES = {"i": 2, "j": 2, ":": 2, ";": 2} if found_n == expected_n: for i, (g, letter) in enumerate(zip(glyphs, expected_letters)): ll = letter.lower() + # Disconnection check: catches glyphs where dilation merged + # broken fragments into one bbox but the un-dilated mask is + # actually multiple pieces (e.g. 't' with detached crossbar). + pieces = disconnected_pieces(g["img"]) + expected_pieces = EXPECTED_PIECES.get(ll, 1) + if pieces > expected_pieces: + topology_failures.append( + (i, letter, round(pieces, 1), expected_pieces)) if letter in LETTER_HOLE_MIN: r = hole_ratio(g["img"]) lo = LETTER_HOLE_MIN[letter] diff --git a/pop/chillwave/bin/gen-illy.mjs b/pop/chillwave/bin/gen-illy.mjs index fd1d86cf4..90cece491 100644 --- a/pop/chillwave/bin/gen-illy.mjs +++ b/pop/chillwave/bin/gen-illy.mjs @@ -112,15 +112,15 @@ const SECTIONS_MODE = flags.sections === true; const SECTION_ORDER = ["tide-in", "drift 1", "swell", "drift 2", "tide-out"]; const SECTION_VARIANTS = { "tide-in": - "SECTION OVERRIDE — tide-in (bright midday): clear, easy overhead daylight. the most lit panel, natural and true-colored — soft blue sky, bright but believable, no fantasy glow. empty calm sea, gentle surf. jeffrey upright and relaxed, hand pressed into the wet sand. AGE: jeffrey here is his youngest, about 30 — matches the reference photos, smooth-faced, brown hair. LAPTOP: the chartreuse MacBook Neo is nearly new — clean and smooth, the white paper scrap crisp and bright, the penned whistlegraph butterfly fresh and dark. the most resolved, fully-modeled panel.", + "SECTION OVERRIDE — tide-in (bright midday): clear easy overhead daylight, natural true colour, soft blue sky, calm empty sea. jeffrey upright and relaxed, hand in the wet sand. AGE: about 30 — smooth-faced, brown hair, lean and fit, a quiet natural soldier's strength. NO tattoos, NO jewelry. TRUNKS: plain SOLID navy swim trunks, no pattern at all. OBJECT (OVERRIDES THE BASE PROMPT AND THE BUTTERFLY REFERENCE FOR THIS PANEL ONLY): a BRAND-NEW, factory-fresh, perfectly clean chartreuse MacBook Neo — on the smooth lid ONLY the plain standard small Apple logo, nothing else; ABSOLUTELY NO paper, NO scrap, NO tape, NO butterfly, NO ink — ignore the butterfly reference here. FIGURE: VERY far down the empty beach, a lone WOMAN — a tiny faraway ambiguous silhouette, barely there, never identifiable. the most resolved, fully-modeled panel.", "drift 1": - "SECTION OVERRIDE — drift 1 (early afternoon): the sun off overhead, light flattening and warming a touch, a faint summer haze softening the horizon. colour eases back just slightly from midday. jeffrey gazing out at the calm water, content, a small smile. AGE: jeffrey is now about 35 — a touch more weathered, the first faint lines, brown hair. LAPTOP: a little wear now — light surface scuffs, a bit of beach grit, the paper scrap slightly curling at one corner, the butterfly ink still clear. still bright and peaceful.", + "SECTION OVERRIDE — drift 1 (early afternoon): sun off overhead, light flattening and warming, faint summer haze. jeffrey gazing at the calm water, content. AGE: about 35 — a touch weathered, browned and stronger, an aging-into-strength soldier's build with a faint MILITARY bearing now. NO tattoos, NO jewelry. TRUNKS: plain SOLID forest-green trunks (a quiet colour shift from navy) — still NO pattern. OBJECT: still a chartreuse MacBook Neo but slightly OLDER-looking and lightly scuffed — a small torn WHITE PAPER SCRAP now stuck over the Apple logo with the hand-penned whistlegraph BUTTERFLY on it. it stays a laptop; NO book, NO pen. FIGURE: the woman is still FAR off, a soft distant shadow / silhouette, a little older in her walk — never clearly identifiable. bright and peaceful.", "swell": - "SECTION OVERRIDE — swell (warm mid-afternoon): the warmest, most golden panel — soft amber light on jeffrey and the chartreuse lid, gentle long shadows on the sand. colour stays natural and a little dusty, never neon. the sea calm and glinting. AGE: jeffrey is now about 40 — leaner and more rugged, sun-worn, a few grey hairs starting, handsome. LAPTOP: visibly used — fine scratches across the lid, dirt and sand worked into the seams, the white scrap yellowing and dog-eared, the butterfly ink fading a touch. easy, drowsy, content.", + "SECTION OVERRIDE — swell (warm mid-afternoon): warmest, most golden panel — soft amber light, long gentle shadows, calm glinting sea, natural and dusty, never neon. AGE: about 40 — leaner, sun-worn, a few grey hairs, powerfully built like a veteran soldier; a clearly MILITARY hardness in posture, but he KEEPS his usual MEDIUM-LENGTH tousled brown hair — NOT short, NOT buzzed, NOT a crew cut, NOT a military haircut; same windblown medium hair as the other panels, just lightly greying. NO tattoos, NO jewelry. TRUNKS: plain SOLID faded brick-red trunks — solid colour only, NO pattern. OBJECT: the same laptop, now visibly DATED and worn — a clunkier older model, yellowed and scratched, the white scrap + butterfly grubbier. still a laptop; NO book, NO pen, NO gilding. FIGURE: the woman is still FAR off, a distant shadow — and now a small CHILD walks beside her holding her hand, also a tiny far shadow. both ambiguous, never identifiable. easy, drowsy.", "drift 2": - "SECTION OVERRIDE — drift 2 (late afternoon, cooling): the sun lower, light going soft rose and pale gold, the air a little hazy and cooler, colour gently muting. long quiet shadows. jeffrey leaning slightly back, calmer, dreamy. AGE: jeffrey is now about 45 — noticeably greying hair, weathered and lean, a tougher more hardcore look, still handsome. LAPTOP: well-battered — deep scratches and dings, scuffed worn corners, grime and salt haze on the chartreuse, the paper scrap grubby and torn and re-taped, the butterfly faded and partly rubbed away. restful and warm, the day winding down.", + "SECTION OVERRIDE — drift 2 (late afternoon, cooling): sun lower, light soft rose and pale gold, hazy and cooler, colour gently muting. jeffrey leaning slightly back, calmer. AGE: about 45 — greying, deeply weathered, lean and immensely strong; distinctly MILITARY now — a hardened veteran's bearing and discipline. NO tattoos, NO jewelry. TRUNKS: plain SOLID khaki / sand trunks (military-toned) — solid colour, NO pattern. OBJECT: an old DATED, battered laptop now — thick, yellowed, grimed, an outdated clunky model; the white scrap + butterfly faded and half rubbed away. still a laptop; NO book, NO pen. FIGURE: the woman + child are still FAR off, both distant soft silhouettes, visibly older, still impossible to identify. restful, the day winding down.", "tide-out": - "SECTION OVERRIDE — tide-out (early evening, after the sun is low): soft dusk — gentle lavender, peach and grey-blue, the light tender and dim but NOT dark, NOT night. the calm sea going pewter, a single faint early star. jeffrey serene, still, hand lifted from the sand, content. AGE: jeffrey is now about 50 — distinctly grey hair, deeply weathered and rugged, hardcore and intense but still handsome, a lifetime on the beach in his face. LAPTOP: an old battlescarred machine — deep gouges, cracked worn edges, sun-bleached faded chartreuse, caked grime, the paper scrap brown and frayed and barely hanging on with old tape, the whistlegraph butterfly almost worn away to a ghost. the quietest, softest, most dissolved panel — the fewest marks, the figure easing into the dusk paper.", + "SECTION OVERRIDE — tide-out (NIGHT): full night now — a deep STARRY SKY, dark sea, the most dissolved panel. jeffrey sits by a small cool blue-white LED CAMPFIRE that lights him from below; far up in the night sky a single MISSILE streaks with a thin contrail. AGE: about 50 — distinctly grey, deeply weathered, an unbreakable old MILITARY veteran who aged entirely into strength, hard and calm, lit by the cold LED fire. NO tattoos, NO jewelry. TRUNKS: plain SOLID deep-teal trunks — solid colour, NO pattern. OBJECT: an ancient, very dated battlescarred laptop — old, yellowed, cracked, clunky; the whistlegraph-butterfly scrap brown and frayed. still a laptop; NO book, NO pen. FIGURE: the woman + child are STILL present but at their MOST DISTANT and faint — tiny soft silhouettes far off in the starlit dark, very old now, almost dissolving but unmistakably still there. fewest marks, jeffrey and the LED fire easing into the night.", }; function safeName(n) { return n.toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-|-$/g, ""); @@ -184,8 +184,13 @@ await generate(basePrompt + portraitTail, OUT_PATH, `${SLUG}${TAG} cover`); // individually cached/--force'd). The butterfly is drawn natively by // the model from BUTTERFLY_REF — no post-pass composite. if (SECTIONS_MODE) { + // --only [,…] regenerates just those panels (others untouched). + const onlySet = typeof flags.only === "string" + ? new Set(flags.only.split(",").map((x) => x.trim())) + : null; for (let i = 0; i < SECTION_ORDER.length; i++) { const name = SECTION_ORDER[i]; + if (onlySet && !onlySet.has(name) && !onlySet.has(String(i))) continue; const variant = SECTION_VARIANTS[name]; const out = `${LANE}/out/${SLUG}${TAG}-sec-${i}-${safeName(name)}.png`; await generate( diff --git a/pop/chillwave/bin/render.mjs b/pop/chillwave/bin/render.mjs index 2c2243985..08092c79d 100755 --- a/pop/chillwave/bin/render.mjs +++ b/pop/chillwave/bin/render.mjs @@ -23,7 +23,7 @@ // Output: pop/chillwave/out/.mp3 import { spawnSync } from "node:child_process"; -import { readFileSync, writeFileSync, mkdtempSync, rmSync } from "node:fs"; +import { readFileSync, writeFileSync, mkdtempSync, rmSync, existsSync } from "node:fs"; import { resolve, dirname } from "node:path"; import { fileURLToPath } from "node:url"; import { tmpdir } from "node:os"; @@ -189,7 +189,7 @@ console.log(`→ score: ${score.events.length} notes · ${score.totalSec.toFixed // lane names the visualizer expects. const eventLog = { bells: [], waves: [], kick: [], sub: [], - hat: [], bubbles: [], birds: [], fx: [], sfx: [], + hat: [], bubbles: [], birds: [], fx: [], sfx: [], vox: [], }; // ── stereo output buffer (interleaved) ─────────────────────────────── @@ -570,6 +570,13 @@ function renderSweeps() { // the noise"). State carried across samples. let nL = 0, nR = 0; + // "wub wub, less floooop" — a beat-synced pump on BOTH the amplitude + // and the filter cutoff (eighth-note rate at the track tempo) so the + // resonant noise reads as a rhythmic wobble instead of one long + // continuous filter glide. + const beat_s = 60 / BPM; + const wubPeriod = beat_s / 2; // two wubs per beat (8th notes) + for (let i = 0; i < LEN_SAMP; i++) { const tSec = i / SAMPLE_RATE; const sec = sectionAt(tSec); @@ -595,14 +602,19 @@ function renderSweeps() { const edgeOut = Math.min(1, (sec.endSec - tSec) / 1.6); const sectionEnv = Math.max(0, Math.min(edgeIn, edgeOut)); + // beat-synced wub: peaky raised-cosine, 0..1 each 8th note + const wubPh = (tSec % wubPeriod) / wubPeriod; + const wub = Math.pow(0.5 - 0.5 * Math.cos(2 * Math.PI * wubPh), 1.7); + const cutWub = 1 + 0.55 * wub; // filter opens on each wub + let sumL = 0, sumR = 0, gNorm = 0; for (let p = 0; p < PARTIALS.length; p++) { const P = PARTIALS[p]; const s = st[p]; const lfoL = 0.5 + 0.5 * Math.sin(2 * Math.PI * P.rate * tSec + P.phase * 6.283); const lfoR = 0.5 + 0.5 * Math.sin(2 * Math.PI * P.rate * tSec + P.phase * 6.283 + RATE_OFFSET_R * 6.283); - const baseL = cutMin * Math.pow(cutMax / cutMin, lfoL); - const baseR = cutMin * Math.pow(cutMax / cutMin, lfoR); + const baseL = cutMin * Math.pow(cutMax / cutMin, lfoL) * cutWub; + const baseR = cutMin * Math.pow(cutMax / cutMin, lfoR) * cutWub; const cutL = Math.min(baseL * P.mul, SAMPLE_RATE / 3); const cutR = Math.min(baseR * P.mul, SAMPLE_RATE / 3); const fL = 2 * Math.sin(Math.PI * cutL / SAMPLE_RATE); @@ -621,8 +633,10 @@ function renderSweeps() { gNorm += P.gain; } - // Quieter overall — roughly a third of the old level. - const amp = (0.085 / gNorm) * onset * sectionEnv; + // Quieter overall + the rhythmic wub pump (keeps a floor so it's + // still a bed, but clearly pulses "wub wub"). + const wubGate = 0.28 + 0.72 * wub; + const amp = (0.085 / gNorm) * onset * sectionEnv * wubGate; outL[i] += sumL * amp; outR[i] += sumR * amp; } @@ -949,11 +963,32 @@ function renderBells() { const ATTACK_S = 0.018; const BELL_GAIN = 0.34; + // The marimba/bells has its OWN life: it wanders across the stereo + // field on a slow LFO, and rides its own loudness NARRATIVE — a big + // swell up into the middle of the track and back, with gentle + // undulation — independent of the section structure. + const total = score.totalSec || 1; + function bellPan(tSec) { + const p = 0.70 * Math.sin((2 * Math.PI / 22) * tSec) + + 0.22 * Math.sin((2 * Math.PI / 6.5) * tSec + 1.3); + return Math.max(-0.92, Math.min(0.92, p)); + } + function bellNarr(tSec) { + const u = Math.max(0, Math.min(1, tSec / total)); + const arc = 0.42 + 0.62 * Math.sin(Math.PI * u); // soft→full→soft + const wave = 0.16 * Math.sin(2 * Math.PI * 3 * u); // undulation + return Math.max(0.30, Math.min(1.25, arc + wave)); + } + for (const ev of score.events) { const gateGain = bellGainAt(ev.startSec); if (gateGain <= 0.001) continue; eventLog.bells.push({ t: ev.startSec, midi: ev.midi, dur: ev.durSec }); const f = midiToFreq(ev.midi); + const pan = bellPan(ev.startSec); + const narr = bellNarr(ev.startSec); + const lG = Math.cos((pan + 1) * Math.PI / 4) * Math.SQRT2; + const rG = Math.sin((pan + 1) * Math.PI / 4) * Math.SQRT2; const startIdx = Math.floor(ev.startSec * SAMPLE_RATE); const ringSamp = Math.floor((ev.durSec + RING_TAIL) * SAMPLE_RATE); const attackSamp = ATTACK_S * SAMPLE_RATE; @@ -973,10 +1008,10 @@ function renderBells() { } let att = 1; if (i < attackSamp) att = 0.5 - 0.5 * Math.cos((Math.PI * i) / attackSamp); - const v = s * att * BELL_GAIN * gateGain; - // stereo bells: slight L bias on odd events, R on even, for spread - outL[dst] += v; - outR[dst] += v; + const v = s * att * BELL_GAIN * gateGain * narr; + // moves around the mix on its own LFO pan + outL[dst] += v * lG; + outR[dst] += v * rG; } } } @@ -1108,6 +1143,44 @@ function renderKick() { } } + // SLOW BASS KICK — a deep, sick 808 "boom" on the DOWNBEAT of every + // bar (4 beats ≈ 3.43s at 70 BPM). Fast pitch drop (110 → 30 Hz) that + // then HOLDS into a long ~1.1s sub tail, soft-driven for a thick + // thump. Lands under the half-time kick as a slow, heavy pulse and + // visibly thumps the preview string (kick lane). + const BOOM_GAIN = 0.42; + let boomCount = 0; + for (const sec of score.sections) { + if (!sec.flags.has("kick")) continue; + for (let t = sec.startSec; t < sec.endSec - 0.2; t += 4 * beat_s) { + const bf0 = 110, bf1 = 30; // glide high → deep sub, then hold + const durSec = 1.15; + const startIdx = Math.floor(t * SAMPLE_RATE); + const nSamp = Math.floor(durSec * SAMPLE_RATE); + const attackS = Math.floor(0.004 * SAMPLE_RATE); + const glideEnd = 0.10; // pitch settles within 100ms + let phase = 0; + for (let i = 0; i < nSamp; i++) { + const dst = startIdx + i; + if (dst < 0 || dst >= LEN_SAMP) break; + const g = Math.min(1, (i / SAMPLE_RATE) / glideEnd); + const f = bf0 * Math.pow(bf1 / bf0, g); // fast drop, then steady + phase += 2 * Math.PI * f / SAMPLE_RATE; + let env; + if (i < attackS) env = i / attackS; + else env = Math.pow(0.0008, (i - attackS) / (nSamp - attackS)); + // soft-driven sine for a thick "sick" thump + tiny click attack + let s = Math.sin(phase) + 0.10 * Math.sin(phase * 2); + s = Math.tanh(s * 1.8) * 0.85; + if (i < attackS) s += (1 - i / attackS) * 0.5; + outL[dst] += s * env * BOOM_GAIN; + outR[dst] += s * env * BOOM_GAIN; + } + eventLog.kick.push({ t, kind: "boom" }); + boomCount++; + } + } + // Sub-bass drone: one sustained sine per [kick] section, crossfading // at section edges (1.5s attack, 1.8s release). const ATTACK_S = 1.5; @@ -1146,7 +1219,7 @@ function renderKick() { eventLog.sub.push({ t: sec.startSec, kind: "drone", dur: dur }); subBlocks++; } - console.log(` kicks: ${kickCount} hits · sub-bass: ${subBlocks} blocks`); + console.log(` kicks: ${kickCount} hits · slow booms: ${boomCount} · sub-bass: ${subBlocks} blocks`); } // ── 6b. snare — sharp "pfft" backbeat ──────────────────────────────── @@ -1350,9 +1423,76 @@ function mixChime(startSec, notes, dir) { t += n.dur + 0.060; // 60 ms gap } } -mixChime(0.05, BOOT_NOTES, "boot"); +// With the typing intro the boot chime plays in the PRE-ROLL, BEFORE +// the typing (see the pre-roll block). Without the intro it sits at +// the very start of the track as before. +if (flags["no-type-intro"] === true) mixChime(0.05, BOOT_NOTES, "boot"); mixChime(Math.max(0, LEN_SEC - 2.2), SHUT_NOTES, "shutdown"); +// ── jeffrey-pvc SUNG vocal (beat-aligned stem) — lead, ducks the bed ── +// The stem is already score-pitch'd + score-stretched (held notes on +// the 70 BPM grid → actually sung). It runs long, so it carries +// THROUGHOUT. Placed beat-aligned, loud, and the rest of the mix is +// SIDECHAIN-DUCKED under it so jeffrey is always clearly heard. Its +// own `vox` lane buffer feeds the visualizer. +if (flags["vocal-stem"]) { + const rel = flags["vocal-stem"]; + const cand = [resolve(process.cwd(), rel), `${LANE}/${rel}`, rel]; + const stemPath = cand.find((p) => existsSync(p)); + if (!stemPath) { + console.error(`✗ --vocal-stem not found: ${rel}`); + } else { + const vGain = Number(flags["vocal-gain"] ?? 1.5); // lead-loud + const DUCK = Number(flags["vocal-duck"] ?? 0.62); // bed dips ~ -8dB + const beat = 60 / BPM; + const reqStart = Number(flags["vocal-start"] ?? beat * 8); // ~bar 3 + const vStart = Math.round(reqStart / beat) * beat; // snap to beat + const vt = mkdtempSync(`${tmpdir()}/cw-vox-`); + const vraw = `${vt}/vox.f32`; + const vr = spawnSync("ffmpeg", [ + "-hide_banner", "-y", "-loglevel", "error", + "-i", stemPath, "-f", "f32le", "-ar", String(SAMPLE_RATE), "-ac", "1", + vraw, + ]); + if (vr.status === 0 && existsSync(vraw)) { + const rb = readFileSync(vraw); + const src = new Float32Array(rb.buffer, rb.byteOffset, rb.byteLength / 4); + let pk = 1e-6; + for (let i = 0; i < src.length; i++) { const a = Math.abs(src[i]); if (a > pk) pk = a; } + const norm = 0.97 / pk; + const voxBuf = new Float32Array(LEN_SAMP); + const startIdx = Math.floor(vStart * SAMPLE_RATE); + const fadeS = Math.floor(0.10 * SAMPLE_RATE); + // sidechain envelope follower over the (normalized) vocal: + // fast attack (~8ms), slow release (~260ms) → smooth ducking. + const aC = Math.exp(-1 / (0.008 * SAMPLE_RATE)); + const rC = Math.exp(-1 / (0.260 * SAMPLE_RATE)); + let envF = 0; + for (let i = 0; i < src.length; i++) { + const j = startIdx + i; + if (j < 0 || j >= LEN_SAMP) break; + const x = src[i] * norm; + const rect = Math.abs(x); + envF = rect > envF ? aC * envF + (1 - aC) * rect + : rC * envF + (1 - rC) * rect; + let fade = 1; + if (i < fadeS) fade = i / fadeS; + else if (i > src.length - fadeS) fade = Math.max(0, (src.length - i) / fadeS); + const duck = 1 - DUCK * Math.min(1, envF * 1.4); + outL[j] = outL[j] * duck + x * vGain * fade; + outR[j] = outR[j] * duck + x * 0.97 * vGain * fade; + voxBuf[j] += x * vGain * fade; + } + laneBuffers.vox = downsampleMono(voxBuf, voxBuf, SAMPLE_RATE, LANE_DISPLAY_SR); + eventLog.vox.push({ t: vStart, name: "vocal", dur: src.length / SAMPLE_RATE }); + console.log(` vocal: SUNG stem @ ${vStart.toFixed(2)}s · gain ${vGain} · duck ${DUCK} · ${(src.length / SAMPLE_RATE).toFixed(1)}s · vox lane`); + } else { + console.error("✗ vocal-stem decode failed"); + } + rmSync(vt, { recursive: true, force: true }); + } +} + // fade-in (0.4s) + fade-out (last 2.5s) so neither end clicks const FADE_IN_SEC = 0.4; const FADE_OUT_SEC = 2.5; @@ -1369,9 +1509,101 @@ for (let i = 0; i < fadeOutSamp; i++) { outL[idx] *= g; outR[idx] *= g; } +// ── TYPING-INTRO pre-roll (ported from recap/bin/trance.mjs) ───────── +// A purple AC-prompt opening: the boot chime fires FIRST, THEN +// key-click ticks type the slug, a return "thunk", then the music. +// Implemented as a post-mix PREPEND: the whole track + every +// event/section/lane-buffer shifts by PREROLL_SEC so nothing internal +// needs rewriting. preview-score draws the prompt during +// [0, PREROLL_SEC): beeps, then typing. +const TYPE_INTRO = flags["no-type-intro"] !== true; +const PREROLL_SEC = TYPE_INTRO ? 3.0 : 0; +const prerollSamp = Math.round(PREROLL_SEC * SAMPLE_RATE); +let FINAL_SAMP = LEN_SAMP + prerollSamp; +{ + const finalL = new Float32Array(FINAL_SAMP); + const finalR = new Float32Array(FINAL_SAMP); + finalL.set(outL.subarray(0, LEN_SAMP), prerollSamp); + finalR.set(outR.subarray(0, LEN_SAMP), prerollSamp); + if (TYPE_INTRO) { + // BOOT CHIME first — render the 3 beeps into the pre-roll at the + // very start, BEFORE any typing. Events are pushed at their TRUE + // pre-roll times (excluded from the +PREROLL_SEC shift below). + { + let ct = 0.08, ci = 0; + for (const n of BOOT_NOTES) { + eventLog.sfx.push({ t: ct, name: `boot-beep-${++ci}`, dur: n.dur, point: true }); + const f = n.tone; + const s0 = Math.floor(ct * SAMPLE_RATE); + const atk = Math.max(1, Math.floor(0.003 * SAMPLE_RATE)); + const decay = n.dur * 0.6; + const total = Math.floor((n.dur + 0.12) * SAMPLE_RATE); + for (let i = 0; i < total; i++) { + const j = s0 + i; + if (j < 0 || j >= prerollSamp) break; + const tt = i / SAMPLE_RATE; + const ph = (f * tt) % 1; + const tri = 2 * Math.abs(2 * ph - 1) - 1; + const env = i < atk ? i / atk + : Math.exp(-(tt - atk / SAMPLE_RATE) / decay); + const s = tri * env * n.vol * CHIME_GAIN; + finalL[j] += s; finalR[j] += s; + } + ct += n.dur + 0.060; + } + } + const krng = makeRng(`${SLUG}:keyclick`); + const tick = (atSec, tone, dur, vol) => { + const i0 = Math.floor(atSec * SAMPLE_RATE); + const atk = Math.max(1, Math.floor(0.0009 * SAMPLE_RATE)); + const n = Math.floor(dur * SAMPLE_RATE); + for (let i = 0; i < n; i++) { + const j = i0 + i; + if (j < 0 || j >= prerollSamp) break; + const ph = (tone * i) / SAMPLE_RATE; + const sq = Math.sin(2 * Math.PI * ph) >= 0 ? 1 : -1; + const env = i < atk ? i / atk : Math.exp(-(i - atk) / (dur * 0.55 * SAMPLE_RATE)); + const s = sq * env * vol; + finalL[j] += s; finalR[j] += s; + } + }; + const TYPE_START = 1.10, GAP = 0.078; // after the boot chime + const N = SLUG.length; + for (let i = 0; i < N; i++) { + const kt = TYPE_START + i * GAP; + const tone = 600 + (krng() - 0.5) * 230; + tick(kt, tone, 0.018, 0.20); // chunky key thock + tick(kt, tone * 1.9, 0.012, 0.10); // bright transient + eventLog.sfx.push({ t: kt, name: `keyclick-${i}`, dur: 0.05, point: true }); + } + const enterT = TYPE_START + N * GAP + 0.08; + tick(enterT, 300, 0.05, 0.26); // return-key thunk + tick(enterT, 150, 0.06, 0.18); + eventLog.sfx.push({ t: enterT, name: "prompt-enter", dur: 0.08, point: true }); + } + outL = finalL; + outR = finalR; +} +// Shift every music event past the pre-roll (keyclick/prompt-enter +// events were just pushed at their true pre-roll times — leave those). +if (PREROLL_SEC > 0) { + for (const lane of Object.keys(eventLog)) { + for (const ev of eventLog[lane]) { + // keyclick / prompt-enter / boot-beep were pushed at their TRUE + // pre-roll times — don't shift them. (shutdown-beep is in the + // main mix and DOES shift.) + if (/^keyclick-\d+$/.test(ev.name || "") || + /^boot-beep-\d+$/.test(ev.name || "") || + ev.name === "prompt-enter") continue; + ev.t += PREROLL_SEC; + } + } +} +const FINAL_SEC = LEN_SEC + PREROLL_SEC; + // ── peak normalize to -1.5 dBFS (stereo) ───────────────────────────── let peak = 0; -for (let i = 0; i < LEN_SAMP; i++) { +for (let i = 0; i < FINAL_SAMP; i++) { const aL = Math.abs(outL[i]); const aR = Math.abs(outR[i]); if (aL > peak) peak = aL; @@ -1380,15 +1612,15 @@ for (let i = 0; i < LEN_SAMP; i++) { const tgt = Math.pow(10, -1.5 / 20); const norm = peak > 0 ? Math.min(1, tgt / peak) : 1; if (norm < 1) { - for (let i = 0; i < LEN_SAMP; i++) { + for (let i = 0; i < FINAL_SAMP; i++) { outL[i] *= norm; outR[i] *= norm; } } // ── interleave + write via ffmpeg (raw f32 stereo → mp3) ───────────── -const interleaved = new Float32Array(LEN_SAMP * 2); -for (let i = 0; i < LEN_SAMP; i++) { +const interleaved = new Float32Array(FINAL_SAMP * 2); +for (let i = 0; i < FINAL_SAMP; i++) { interleaved[i * 2] = outL[i]; interleaved[i * 2 + 1] = outR[i]; } @@ -1408,16 +1640,22 @@ if (r.status !== 0) { console.error("✗ ffmpeg encode failed"); process.exit(1); } -console.log(`✓ ${OUT_PATH} (peak norm ${norm.toFixed(3)}, ${LEN_SEC.toFixed(1)}s)`); +console.log(`✓ ${OUT_PATH} (peak norm ${norm.toFixed(3)}, ${FINAL_SEC.toFixed(1)}s · +${PREROLL_SEC}s type-intro)`); // ── per-lane mono buffers (display SR) saved as .raw sidecars ──────── // Each lane writes a headerless float32 mono file at LANE_DISPLAY_SR // next to the mp3. The preview reads these to draw each waveform with // the actual mix-volume of that lane (not the full mixed audio). const laneBufferPaths = {}; +const prerollDisp = Math.round(PREROLL_SEC * LANE_DISPLAY_SR); for (const [key, buf] of Object.entries(laneBuffers)) { + // prefix each lane's display buffer with pre-roll silence so the + // waveforms stay aligned with the (shifted) event times. + const shifted = prerollDisp > 0 + ? (() => { const a = new Float32Array(buf.length + prerollDisp); a.set(buf, prerollDisp); return a; })() + : buf; const p = OUT_PATH.replace(/\.mp3$/, `.lane-${key}.raw`); - writeFileSync(p, Buffer.from(buf.buffer, buf.byteOffset, buf.byteLength)); + writeFileSync(p, Buffer.from(shifted.buffer, shifted.byteOffset, shifted.byteLength)); laneBufferPaths[key] = p.replace(`${LANE}/`, ""); } @@ -1431,15 +1669,16 @@ const struct = { bpm: BPM, scale: "minor", rootMidi: 57, // A3 - totalSec: LEN_SEC, + totalSec: FINAL_SEC, + prerollSec: PREROLL_SEC, // typing-intro length (events ≥ this) laneAudio: { sampleRate: LANE_DISPLAY_SR, paths: laneBufferPaths, // relative to the lane dir }, sections: score.sections.map((s) => ({ name: s.name, - startSec: s.startSec, - endSec: s.endSec, + startSec: s.startSec + PREROLL_SEC, + endSec: s.endSec + PREROLL_SEC, flags: Array.from(s.flags), })), events: eventLog, diff --git a/pop/chillwave/bin/say-local.mjs b/pop/chillwave/bin/say-local.mjs new file mode 100644 index 000000000..8990e9d0b --- /dev/null +++ b/pop/chillwave/bin/say-local.mjs @@ -0,0 +1,114 @@ +#!/usr/bin/env node +// say-local.mjs — a LOCAL stand-in for /api/say (jeffrey-pvc only). +// +// The production host (aesthetic.computer) is intermittently +// unreachable, but ElevenLabs itself is fine. This serves the exact +// request/response contract pop/bin/say.mjs expects, by replicating +// the `generateJeffrey` branch of system/netlify/functions/say.js — +// no S3/Mongo cache layer (say.mjs already content-hash caches). +// +// Reads ELEVENLABS_API_KEY from the vault devcontainer.env. +// +// Usage: +// node pop/chillwave/bin/say-local.mjs # listens :8899 +// SAY_ENDPOINT=http://127.0.0.1:8899/api/say \ +// node pop/bin/say.mjs --provider jeffrey --voice neutral:0 --timestamps + +import { createServer } from "node:http"; +import { readFileSync, existsSync } from "node:fs"; +import { resolve, dirname } from "node:path"; +import { fileURLToPath } from "node:url"; + +const HERE = dirname(fileURLToPath(import.meta.url)); +const REPO = resolve(HERE, "../../.."); +const PORT = Number(process.env.SAY_LOCAL_PORT || 8899); + +const JEFFREY_VOICE_ID = "dYNGZ848Oo6DtNBoeqgh"; // same as say.js + +function loadKey() { + if (process.env.ELEVENLABS_API_KEY) return process.env.ELEVENLABS_API_KEY; + const vault = `${REPO}/aesthetic-computer-vault/.devcontainer/envs/devcontainer.env`; + if (existsSync(vault)) { + for (const line of readFileSync(vault, "utf8").split("\n")) { + if (line.startsWith("ELEVENLABS_API_KEY=")) { + return line.slice("ELEVENLABS_API_KEY=".length).trim() + .replace(/^['"]|['"]$/g, ""); + } + } + } + throw new Error("ELEVENLABS_API_KEY not in env or vault devcontainer.env"); +} +const KEY = loadKey(); + +function readBody(req) { + return new Promise((res, rej) => { + const chunks = []; + req.on("data", (c) => chunks.push(c)); + req.on("end", () => { + try { res(JSON.parse(Buffer.concat(chunks).toString("utf8") || "{}")); } + catch (e) { rej(e); } + }); + req.on("error", rej); + }); +} + +const server = createServer(async (req, res) => { + if (req.method !== "POST" || !req.url.startsWith("/api/say")) { + res.writeHead(404).end("not found"); + return; + } + try { + const b = await readBody(req); + const text = b.from ?? b.text ?? ""; + if (!text.trim()) { res.writeHead(400).end("no text"); return; } + // jeffrey-pvc, calm non-scream defaults (mirrors say.js), overridable + const voiceSettings = { + stability: b.stability ?? 0.65, + similarity_boost: b.similarity ?? 0.9, + style: b.style ?? 0.15, + use_speaker_boost: true, + speed: b.speed ?? 1.0, + }; + const withTs = b.withTimestamps === true; + const base = `https://api.elevenlabs.io/v1/text-to-speech/${JEFFREY_VOICE_ID}`; + const url = withTs ? `${base}/with-timestamps` : base; + const r = await fetch(url, { + method: "POST", + headers: { "xi-api-key": KEY, "Content-Type": "application/json" }, + body: JSON.stringify({ + text, + model_id: "eleven_multilingual_v2", + voice_settings: voiceSettings, + }), + }); + if (!r.ok) { + const err = await r.text(); + console.error(`✗ ElevenLabs ${r.status}: ${err.slice(0, 300)}`); + res.writeHead(502, { "content-type": "text/plain" }).end(err.slice(0, 500)); + return; + } + if (withTs) { + const j = await r.json(); + res.writeHead(200, { "content-type": "application/json" }); + res.end(JSON.stringify({ + audio: j.audio_base64, + alignment: j.alignment, + normalizedAlignment: j.normalized_alignment, + voiceId: "jeffrey-pvc", + })); + console.log(`✓ jeffrey-pvc +timestamps · ${text.length} chars`); + } else { + const buf = Buffer.from(await r.arrayBuffer()); + res.writeHead(200, { "content-type": "audio/mpeg" }); + res.end(buf); + console.log(`✓ jeffrey-pvc · ${text.length} chars · ${(buf.length / 1024) | 0} KB`); + } + } catch (e) { + console.error("✗", e.message); + res.writeHead(500, { "content-type": "text/plain" }).end(String(e.message)); + } +}); + +server.listen(PORT, "127.0.0.1", () => { + console.log(`▸ say-local (jeffrey-pvc) → http://127.0.0.1:${PORT}/api/say`); +}); diff --git a/pop/chillwave/bin/sing.mjs b/pop/chillwave/bin/sing.mjs new file mode 100644 index 000000000..cc5796b2e --- /dev/null +++ b/pop/chillwave/bin/sing.mjs @@ -0,0 +1,123 @@ +#!/usr/bin/env node +// chillwave/bin/sing.mjs — jeffrey-pvc SUNG vocal stem for undabeach. +// +// 1. POST the lyric to the LOCAL say endpoint (say-local.mjs) with +// ElevenLabs /with-timestamps → vocal.mp3 + per-word alignment. +// 2. alignment → words.json +// 3. score-pitch.mjs (WORLD f0 replacement) snaps jeffrey's natural +// read onto undabeach.vocal.np → a soft sung performance layer. +// +// Requires say-local.mjs running (node pop/chillwave/bin/say-local.mjs). +// +// Usage: node pop/chillwave/bin/sing.mjs [--force] [--transpose 0] + +import { readFileSync, writeFileSync, existsSync, mkdirSync } from "node:fs"; +import { resolve, dirname } from "node:path"; +import { fileURLToPath } from "node:url"; +import { spawnSync } from "node:child_process"; + +const HERE = dirname(fileURLToPath(import.meta.url)); +const LANE = resolve(HERE, ".."); +const REPO = resolve(LANE, "../.."); +const flags = {}; +for (let i = 2; i < process.argv.length; i++) { + const a = process.argv[i]; + if (!a.startsWith("--")) continue; + const n = process.argv[i + 1]; + if (n === undefined || n.startsWith("--")) flags[a.slice(2)] = true; + else { flags[a.slice(2)] = n; i++; } +} +const FORCE = flags.force === true; +const TRANSPOSE = String(flags.transpose ?? 0); +const SAY_URL = process.env.SAY_ENDPOINT || "http://127.0.0.1:8899/api/say"; + +const OUT = `${LANE}/out`; +mkdirSync(OUT, { recursive: true }); +const VOCAL_MP3 = `${OUT}/undabeach-vocal.mp3`; +const WORDS_JSON = `${OUT}/undabeach-vocal-words.json`; +const PITCHED_MP3 = `${OUT}/undabeach-vocal-pitched.mp3`; +const VOCAL_NP = `${LANE}/undabeach.vocal.np`; + +// Lyric = the content lines of undabeach.txt (skip the "whisper" tag). +const lyric = readFileSync(`${LANE}/undabeach.txt`, "utf8") + .split("\n").map((l) => l.trim()) + .filter((l) => l && l.toLowerCase() !== "whisper") + .join(" "); +console.log(`▸ lyric: "${lyric}"`); + +// ── 1+2. local say (jeffrey-pvc + timestamps) → mp3 + words.json ───── +if (FORCE || !existsSync(VOCAL_MP3) || !existsSync(WORDS_JSON)) { + console.log(`▸ say-local → ${SAY_URL}`); + const res = await fetch(SAY_URL, { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ + from: lyric, provider: "jeffrey", voice: "neutral:0", + stability: 0.6, similarity: 0.92, style: 0.28, + withTimestamps: true, + }), + }); + if (!res.ok) { + console.error(`✗ say-local ${res.status}: ${(await res.text()).slice(0, 300)}`); + console.error(" is say-local.mjs running? node pop/chillwave/bin/say-local.mjs"); + process.exit(1); + } + const j = await res.json(); + writeFileSync(VOCAL_MP3, Buffer.from(j.audio, "base64")); + // chars → words (runs of non-space), [first start, last end] + const a = j.alignment || {}; + const ch = a.characters || []; + const st = a.character_start_times_seconds || []; + const en = a.character_end_times_seconds || []; + const words = []; + let i = 0; + while (i < ch.length) { + if (/\s/.test(ch[i])) { i++; continue; } + const from = st[i]; let txt = "", last = en[i]; + while (i < ch.length && !/\s/.test(ch[i])) { txt += ch[i]; last = en[i]; i++; } + words.push({ text: txt, fromMs: Math.round(from * 1000), toMs: Math.round(last * 1000) }); + } + writeFileSync(WORDS_JSON, JSON.stringify(words, null, 0)); + console.log(`✓ ${VOCAL_MP3.replace(REPO + "/", "")} (${words.length} words)`); +} else { + console.log(`✓ cached vocal + words (use --force to regen)`); +} + +// ── 3. score-pitch → sung stem ─────────────────────────────────────── +console.log(`▸ score-pitch (WORLD f0 → undabeach.vocal.np · transpose ${TRANSPOSE})`); +const r = spawnSync("node", [ + `${REPO}/pop/bin/score-pitch.mjs`, + "--slug", "undabeach", "--section", "all", + "--score", VOCAL_NP, + "--vocal", VOCAL_MP3, + "--words", WORDS_JSON, + "--transpose", TRANSPOSE, + "--vibrato-hz", "5.0", + "--vibrato-cents", "14", + "--out", PITCHED_MP3, +], { cwd: `${REPO}/pop`, stdio: ["ignore", "inherit", "inherit"] }); +if (r.status !== 0) { console.error("✗ score-pitch failed"); process.exit(1); } +console.log(`✓ pitched → ${PITCHED_MP3.replace(REPO + "/", "")}`); + +// ── 4. score-stretch → HOLD notes on the 70 BPM beat grid ──────────── +// This is what makes it actually SUNG and beat-aligned: each word is +// rubberband-stretched (formant-preserving, no pitch change) to its +// target beats from undabeach.vocal.np at the track tempo, so the *5 +// phrase-ends sustain on pitch instead of reading as fast speech. +const BPM = String(flags.bpm ?? 70); +const PITCHED_ALIGN = `${OUT}/undabeach-vocal-pitched-alignment.json`; +const SUNG_MP3 = `${OUT}/undabeach-vocal-sung.mp3`; +console.log(`▸ score-stretch (rubberband → ${BPM} BPM beat grid · held notes)`); +const r2 = spawnSync("node", [ + `${REPO}/pop/bin/score-stretch.mjs`, + "--slug", "undabeach", "--section", "all", + "--score", VOCAL_NP, + "--in", PITCHED_MP3, + "--alignment", PITCHED_ALIGN, + "--bpm", BPM, + "--max-stretch", "8.0", + "--out", SUNG_MP3, +], { cwd: `${REPO}/pop`, stdio: ["ignore", "inherit", "inherit"] }); +if (r2.status !== 0) { console.error("✗ score-stretch failed"); process.exit(1); } +console.log(`\n✓ SUNG (beat-aligned) → ${SUNG_MP3.replace(REPO + "/", "")}`); +console.log(` mix: node bin/render.mjs --slug undabeach --vocal-stem ${SUNG_MP3.replace(LANE + "/", "")}`); diff --git a/pop/chillwave/undabeach.illy.txt b/pop/chillwave/undabeach.illy.txt index 9c681f37c..024fed9f0 100644 --- a/pop/chillwave/undabeach.illy.txt +++ b/pop/chillwave/undabeach.illy.txt @@ -12,7 +12,11 @@ CRUCIAL — curatorial restraint: the most resolved modeling sits on jeffrey, hi setting — a SUBTLE, peaceful, ordinary VACATION BEACH. Just sand, gentle surf, a soft open sky and a calm sea. NO city, NO buildings, NO skyline, NO oil tankers or ships, NO drones, NO shooting stars, NO technology of any kind in the environment. Maybe one distant low dune, a little beach grass, a far soft headland — nothing more. The laptop is the ONLY made object in the picture. Quiet, unpeopled, restful — somewhere you'd go to be away from everything. -subject: jeffrey sits cross-legged on damp packed sand a few feet back from the foam line. close three-quarter framing, waist-up. he is shirtless — bare shoulders and chest modeled as turning volumes of warm hatching, no outline on the body. plain navy or sage swim trunks. medium-length brown hair lightly tousled by the wind. his left hand is buried wrist-deep in the wet sand beside him, fingers spread under the grain. his right hand rests on top of a CLOSED chartreuse-green MacBook Neo laptop (the citrus-green model), lying flat on a folded grey hoodie on the sand. the laptop is CLOSED — only the smooth chartreuse lid shows. There is NO Apple logo. Instead, over the center of the lid where a logo would be, there is a small torn SCRAP OF WHITE PAPER, and on that scrap jeffrey has hand-penned a butterfly. ONE OF THE PROVIDED REFERENCE IMAGES is that butterfly doodle (a thick, simple, slightly wonky whistlegraph butterfly — a rectangular smiling body and two big rounded wings, dark marker on white). DRAW that exact butterfly, by hand, onto the white paper scrap, RENDERED IN THE SAME ROUGH COLORED-PENCIL / MARKER MEDIUM as the rest of the picture and on the same rough paper tooth — it is genuinely part of the drawing, hand-inked on the little scrap, NOT a clean flat pasted sticker, NOT a crisp logo, NOT a separate cut-out. The scrap and its butterfly catch the same light and grain as everything else. NO keyboard visible, NO screen visible, NO screens on the back or bottom. +subject: jeffrey sits cross-legged on damp packed sand a few feet back from the foam line. close three-quarter framing, waist-up. he is shirtless — bare shoulders and chest modeled as turning volumes of warm hatching, no outline on the body. plain navy or sage swim trunks. medium-length brown hair lightly tousled by the wind. his left hand is buried wrist-deep in the wet sand beside him, fingers spread under the grain. his right hand rests on top of a CLOSED chartreuse-green MacBook Neo laptop (the citrus-green model), lying flat on a folded grey hoodie on the sand. the laptop is CLOSED — only the smooth chartreuse lid shows; NO keyboard, NO screen, NO screens on the back or bottom. THE LID TREATMENT IS SET BY THE SECTION OVERRIDE and changes through the set: the FIRST panel is a brand-new machine with the plain standard small Apple logo; in every LATER panel there is instead a small torn SCRAP OF WHITE PAPER stuck over the logo with a hand-penned whistlegraph BUTTERFLY on it (one of the provided reference images is that butterfly doodle — a thick, simple, slightly wonky butterfly, rectangular smiling body, two big rounded wings; draw THAT exact butterfly by hand in the same rough colored-pencil/marker medium and paper tooth, genuinely part of the drawing, NOT a clean pasted sticker or crisp logo). It STAYS a laptop in every panel — it NEVER becomes a book, there is NO pen, NO manuscript, NO gilding. It simply gets older and more DATED across the panels (a clunkier, more outdated laptop model — thicker, yellowed, more beat-up). + +CRUCIAL — there is ALWAYS, in EVERY panel without exception, a single ambiguous WOMAN as a soft shadow/silhouette walking in the background — never omitted, never cropped. She sits comfortably INSIDE the frame within the safe zone (the mid-to-upper background band, well clear of all four edges), clearly visible but small and never identifiable. Her distance/age and whether a child holds her hand are set by the SECTION OVERRIDE. + +CRUCIAL — jeffrey has NO tattoos, EVER. NO jewelry of any kind, EVER — no chains, no necklaces, no rings, no bracelets, no earrings. Bare skin only. Any "gilded" or "cyberpunk" quality is purely a property of the LIGHT (a faint sheen on skin/edges), never an object worn on the body. expression: relaxed, content, a little dreamy — soft easy smile, looking sidelong down toward the laptop. clean-shaven, no facial hair (matches refs). diff --git a/pop/chillwave/undabeach.vocal.np b/pop/chillwave/undabeach.vocal.np new file mode 100644 index 000000000..5c417dadd --- /dev/null +++ b/pop/chillwave/undabeach.vocal.np @@ -0,0 +1,22 @@ +# undabeach — jeffrey-pvc sung vocal melody. +# notation: NOTE:syllable*weight (-prefix continues a word / melisma) +# key: A minor pentatonic (A C D E G), octave 3-4 — UP an octave from +# the first take so it reads clearly as SINGING in jeffrey-pvc's upper +# baritone, melodic and bright over the chillwave bed. score-pitch +# replaces f0 to these notes; score-stretch holds them on the 70 BPM +# beat grid. Sections are "verse N" for score-* --section all. + +verse 1 +E4:fin-*2 D4:-gers*1 C4:in*1 D4:the*1 A3:sand*5 + +verse 2 +G4:the*1 E4:tide*3 D4:re-*1 C4:-mem-*1 A3:-bers*5 + +verse 3 +E4:no-*1 D4:-thing*1 C4:in*1 D4:my*1 E4:poc-*1 D4:-ket*1 C4:but*1 A3:a*1 C4:small*2 D4:good*2 E4:thing*5 + +verse 4 +G4:the*1 E4:green*3 D4:ma-*1 C4:-chine*1 D4:is*1 C4:brea-*1 A3:-thing*5 + +verse 5 +D4:you*1 C4:can*2 A3:rest*3 G3:now*6 diff --git a/pop/dance/bin/cover.mjs b/pop/dance/bin/cover.mjs index 798c6e43c..106c40053 100644 --- a/pop/dance/bin/cover.mjs +++ b/pop/dance/bin/cover.mjs @@ -44,6 +44,7 @@ function expandHome(p) { const SLUG = flags.slug || "trance"; const TITLE = flags.title || SLUG; +const TITLE_POS = (flags["title-pos"] || "center").toLowerCase(); // center | topleft const SUBTITLE = flags.subtitle || ""; const HANDLE = flags.handle || "@jeffrey"; const LANE = flags.lane || "dance"; @@ -282,7 +283,7 @@ if (WAVEFORM && existsSync(WAVEFORM)) { console.warn(`✗ waveform decode failed, falling back to sawtooth glyph`); } else { const f32 = new Float32Array(probe.stdout.buffer, probe.stdout.byteOffset, probe.stdout.byteLength / 4); - const BARS = 120; // fewer/wider bars so the waveform reads clearly + const BARS = 280; // dense bins → a detailed, data-rich waveform read const samplesPerBar = Math.floor(f32.length / BARS); const peaks = new Float32Array(BARS); for (let b = 0; b < BARS; b++) { @@ -424,7 +425,12 @@ if (hasIllustration) { // Rainbow per-character via magick with measured kerning + dark // drop shadow (glow removed so the YWFT pixel edges stay crisp). const illustTitleSize = Math.min(220, Math.floor((W - 240) / (titleStr.length * 0.45))); - textOverlays.push({ text: titleStr, x: W / 2, y: 280, size: illustTitleSize, color: "white", weight: 800, anchor: "middle", rainbow: true }); + if (TITLE_POS === "topleft") { + const tlSize = Math.round(illustTitleSize * 0.58); // smaller for the corner + textOverlays.push({ text: titleStr, x: 42, y: 168, size: tlSize, color: "white", weight: 800, anchor: "start", rainbow: true }); + } else { + textOverlays.push({ text: titleStr, x: W / 2, y: 280, size: illustTitleSize, color: "white", weight: 800, anchor: "middle", rainbow: true }); + } // No subtitle, no LANE — photo + fairies + title + handle are enough. // @jeffrey: larger (130), pushed further into the bottom-right // corner with more margin (90 px from each edge). diff --git a/reports/pop-pipeline-audit.md b/reports/pop-pipeline-audit.md new file mode 100644 index 000000000..a8e02382c --- /dev/null +++ b/reports/pop-pipeline-audit.md @@ -0,0 +1,173 @@ +# Pop song pipeline audit — trancenwaltz vs undabeach + +_Generated 2026-05-16. Scope: the two pop songs currently in the repo — +**dance/trancenwaltz** and **chillwave/undabeach** — their pipelines, +what is common in the formats, and a plan to bring undabeach up to the +effect + prompting set trancenwaltz just gained._ + +--- + +## 1. The two songs at a glance + +| | **trancenwaltz** (`pop/dance`) | **undabeach** (`pop/chillwave`) | +|---|---|---| +| Mood | dark / emo / extreme war-arc | chill / sunny beach | +| Audio gen | `recap/bin/trance.mjs` — procedural **synth composition** → mp3 + `struct.json` | `pop/chillwave/bin/render.mjs` — **`.np` score** → mp3 | +| Score source | section templates inside `trance.mjs` | `pop/chillwave/undabeach.np` | +| Illustration gen | `marketing/bin/gen-promo.mjs` — **per-section** campaign dirs (9), jeffrey refs + campaign refs | `pop/chillwave/bin/gen-illy.mjs` — **single cover**, jeffrey refs | +| Prompt source | 9 × `…/trancenwaltz-sections//cover-prompt.txt` (shared header/trailer + unique `SECTION:` line + `refs/`) | 1 × `pop/chillwave/undabeach.illy.txt` | +| Video render | `pop/dance/bin/cover-video.mjs` (2035 ln) | `pop/chillwave/bin/preview-score.mjs` (1011 ln) | +| Orchestrator | `pop/dance/bin/build.mjs` (versioned `~/Desktop/builds//bNNN/`) | none — manual `render.mjs` + `preview-score.mjs` | +| Preview player | `pop/dance/bin/preview.sh` + `hover-loop.lua` (borderless, hover-to-play, loop) | none | +| Shared lib | `pop/lib/preview-shared.mjs` (YWFT typography, ken-burns, waveform-in-events, BGRA, audio decode) | same `pop/lib/preview-shared.mjs` | + +--- + +## 2. What is common in the formats (the shared spine) + +Both songs follow the same **audio → struct → per-frame Canvas2D cover +video → ffmpeg BGRA → mp4** spine: + +- **Struct/event model.** Audio render emits a `struct.json` of lane + events (`kick/hat/snare/sub/lead/bells/piano/supersaw/…/sfx/vox`, + `dropImpact`, sections with `startSec/endSec`). The video reads it. +- **YWFT-Processing-Bold typography**, rasterised via ImageMagick + (Cairo/fontconfig can't resolve the TTF) — title chars + timecode + + lane/section labels. Centralised in `pop/lib/preview-shared.mjs`. +- **Screen-blended waveform events** drawn into the illustration + (no boxes), playhead-driven press/held/played alpha. +- **node-canvas → rawvideo BGRA → libx264 + AAC** encode path. +- **jeffrey identity refs** (SHOOT + SELFIE corpus) on every gpt-image + call, same path as `recap/bin/jeffrey-photos.mjs`. +- **Ken-burns** illustration motion, **segmented bottom progress bar**, + **bottom-right timecode**, fixed scene + rotating "track group". +- Lowercase / fragment **papers VOICE.md** prompt register. + +undabeach is effectively the **reference implementation** of the +chrome (verlet string, rotating disc, segmented bar, YWFT timecode, +chunky bar-segment waveform) — those were ported _from_ it into +trancenwaltz this cycle. + +--- + +## 3. Where they diverge + +### Audio +- trancenwaltz is **fully procedural** (synth section-templates in + `trance.mjs`) and has a **scripted opening/closing sonification**: + AC keyclick typing intro → boot melody → sniper → music; tape-stop + + shutdown melody at the end. It mixes a **jeffrey-pvc sung vocal + stem** (`pop/dance/bin/sing.mjs`). +- undabeach renders a **fixed `.np` score** (`render.mjs`), no scripted + boot/shutdown story, no vocal stem path. + +### Illustration / prompting sequencing +- trancenwaltz has **per-section prompt sequencing**: a war arc across + 9 sections (intro → drops → outro), each its own campaign dir with a + shared header/trailer (medium, butterfly-on-laptop-lid rule, + PALS-only pixie laptops, diverse casting) + a unique `SECTION:` + beat + curated `refs/`. Section illustrations **crossfade / doom-melt + on the musical drop**; a head-down "prelude" swaps in before the + first hit. Versioned (v8…v12, never overwritten). +- undabeach has a **single `undabeach.illy.txt`** cover. No + per-section set, no crossfade, no prelude, no campaign refs system. + +### Effects present in trancenwaltz, ABSENT in undabeach +1. **Stained-glass transmissive backlight** — lane hits light the + illustration _through_ a per-illustration luminance transmission + mask (additive transmitted pass + multiply leaded-contrast pass), + hue drifting, amplitude-reactive. +2. **Two-layer parallax depth** + env shake. +3. **Music-driven fine-row slitscan** that swells into each drop. +4. **Section illustration crossfade / doom-melt** snapped to + `dropImpact`. +5. **Fullscreen pixelate / fuzz / blink decay** for **startup & + shutdown**, beep-synced (no geometry squash). +6. **AC pink-prompt typing cold-open** — keyclick audio lead-in in + `trance.mjs` + the prompt block typing the title in cover-video. +7. **Prelude swap** (head-down/laptops-closed → first gunshot). +8. **Env-bounce zoom** framing (pulled back, pumps on loudness). +9. **`--probe-frames`** still-render mode for ~10 s iteration. +10. **Versioned build orchestrator** (`build.mjs`) + **clean hover + preview** (`preview.sh` + `hover-loop.lua`). +11. **gpt-image retry/timeout resilience** (`gen-promo.mjs`). + +### Effects present in undabeach, matched in trancenwaltz this cycle +verlet-string playhead, rotating-disc, YWFT timecode w/ section-tint +recolor + env bump, full-width segmented progress bar, chunky +bar-segment waveform, string-bend on the bars. (Ported across.) + +--- + +## 4. Risk / divergence notes + +- **Two cover-video engines.** `cover-video.mjs` (2035 ln) and + `preview-score.mjs` (1011 ln) now overlap heavily (string, disc, + timecode, bar, progress) but are **forked copies**, not a shared + module. Each new effect has to be hand-ported (this cycle proved + that — verlet/disc/bar were copied by hand). This will keep + drifting. +- `pop/lib/preview-shared.mjs` only holds typography + ken-burns + + audio helpers — **not** the chrome/effects. The valuable new + effects (backlight, slit, decay, typing-intro, bar model) live + only in `cover-video.mjs`. +- undabeach audio (`render.mjs`, `.np`) and trancenwaltz audio + (`trance.mjs`, procedural) are **different engines**; the boot/ + shutdown/keyclick sonification is trance-only and not trivially + reusable without a shared SFX helper. + +--- + +## 5. Recommended plan (port the new set to undabeach) + +**Phase 1 — extract a shared cover engine (highest leverage).** +Lift the now-canonical effect set out of `cover-video.mjs` into +`pop/lib/cover-engine.mjs`: transmission-mask backlight, two-layer +parallax, fine-row slitscan, doom-melt section crossfade, pixelate +startup/shutdown decay, typing cold-open, verlet string (+colour ++bend), beach bar waveform, rotating disc, env-bounce zoom, +`--probe-frames`. Both `cover-video.mjs` and `preview-score.mjs` +become thin per-song configs (lanes, palette, prelude, illo map). + +**Phase 2 — give undabeach per-section prompting sequencing.** +Mirror the dance campaign-dir model: `pop/chillwave/undabeach-sections//{cover-prompt.txt,refs/}` +with a shared header/trailer + per-section beat, generated via +`gen-promo.mjs` (already retry-resilient), versioned. Wire a section +illustration map + crossfade into the (now shared) engine. + +**Phase 3 — shared opening/closing sonification helper.** +Factor the keyclick / boot-melody / shutdown-melody synth + event +emission out of `trance.mjs` into a small helper both audio engines +(`trance.mjs` procedural and `render.mjs` .np) can call, so undabeach +can get the same AC typing cold-open + pixelate startup if desired +(chill variant: gentler, optional). + +**Phase 4 — adopt the orchestrator + preview for undabeach.** +Add an undabeach CONFIG to `build.mjs` (versioned builds) and let +`preview.sh` pick up undabeach builds. One pipeline, two songs. + +**Net:** after Phase 1+2 a new effect is written once and both songs +get it; a new song is a config + a `.np`/template + section prompts. + +--- + +## 6. File map (quick reference) + +``` +pop/lib/preview-shared.mjs shared: YWFT, ken-burns, waveform, BGRA, audio +pop/dance/ + bin/cover-video.mjs trancenwaltz video engine (canonical effects) + bin/build.mjs versioned build orchestrator + bin/trance.mjs? → recap/bin/trance.mjs procedural audio + struct (+ keyclick/boot/arp) + bin/sing.mjs / vocal.mjs jeffrey-pvc sung vocal stem + bin/preview.sh + hover-loop.lua clean borderless hover-to-play player + trance-hook.np sung-vocal melody score +pop/chillwave/ + bin/render.mjs undabeach .np → mp3 + bin/preview-score.mjs undabeach video (verlet/disc/bar reference) + bin/gen-illy.mjs single-cover illustration + undabeach.np / .illy.txt / .txt score / prompt / lyric +marketing/bin/gen-promo.mjs per-section campaign illustration gen (retry-resilient) +~/Documents/Working Desktop/gens/trancenwaltz-sections// prompts + refs + gens (v8..v12) +~/Desktop/builds//bNNN/ versioned build outputs +``` diff --git a/system/netlify/functions/say.js b/system/netlify/functions/say.js index 80ff8476c..cd716b98c 100644 --- a/system/netlify/functions/say.js +++ b/system/netlify/functions/say.js @@ -190,8 +190,14 @@ async function generateElevenLabs(text, gender, set, scream) { // Usage from the piece: `say:jeffrey hello world` const JEFFREY_VOICE_ID = "dYNGZ848Oo6DtNBoeqgh"; +// When `withTimestamps` is true, hits ElevenLabs `/with-timestamps` +// endpoint and returns BOTH audio + per-character alignment. The +// alignment is the source-of-truth replacement for whisper STT post- +// processing; it is exact (no recognition, no formant distortion in the +// signal yet) and free. async function generateJeffrey(text, scream, speed = 1.0, styleOverride = null, - stabilityOverride = null, similarityOverride = null) { + stabilityOverride = null, similarityOverride = null, + withTimestamps = false) { // Calmer, more natural delivery than the premade "scream" preset. // Same knobs as the grant-video pipeline for homogeneity. // ElevenLabs voice_settings exposed (eleven_multilingual_v2): @@ -213,27 +219,38 @@ async function generateJeffrey(text, scream, speed = 1.0, styleOverride = null, speed, }; - const response = await fetch( - `https://api.elevenlabs.io/v1/text-to-speech/${JEFFREY_VOICE_ID}`, - { - method: "POST", - headers: { - "xi-api-key": process.env.ELEVENLABS_API_KEY, - "Content-Type": "application/json", - }, - body: JSON.stringify({ - text, - model_id: "eleven_multilingual_v2", - voice_settings: voiceSettings, - }), + const baseUrl = `https://api.elevenlabs.io/v1/text-to-speech/${JEFFREY_VOICE_ID}`; + const url = withTimestamps ? `${baseUrl}/with-timestamps` : baseUrl; + + const response = await fetch(url, { + method: "POST", + headers: { + "xi-api-key": process.env.ELEVENLABS_API_KEY, + "Content-Type": "application/json", }, - ); + body: JSON.stringify({ + text, + model_id: "eleven_multilingual_v2", + voice_settings: voiceSettings, + }), + }); if (!response.ok) { const err = await response.text(); throw new Error(`ElevenLabs (Jeffrey) API error ${response.status}: ${err}`); } + if (withTimestamps) { + // /with-timestamps returns JSON: { audio_base64, alignment, normalized_alignment } + const json = await response.json(); + return { + buffer: Buffer.from(json.audio_base64, "base64"), + voiceId: "jeffrey-pvc", + alignment: json.alignment, + normalizedAlignment: json.normalized_alignment, + }; + } + return { buffer: Buffer.from(await response.arrayBuffer()), voiceId: "jeffrey-pvc", @@ -344,6 +361,15 @@ exports.handler = async (event) => { // Cache bust: if true, skip cache lookup and regenerate const bustCache = body.bust === true; + // Timestamps: when true, hit ElevenLabs `/with-timestamps` endpoint + // (jeffrey provider only) and return JSON `{audio_base64, alignment}` + // instead of raw mp3. Cache lookup is skipped because cached entries + // are audio-only — cache write still happens (mp3 only) so future + // non-timestamp callers benefit. Backward compatible: existing + // recap/slab callers never set this flag and continue to receive + // raw mp3 (302 redirect to CDN). + const withTimestamps = body.withTimestamps === true || body.with_timestamps === true; + // Check for SSML (only Google supports it) const isSSML = utterance.indexOf("") !== -1; @@ -364,8 +390,10 @@ exports.handler = async (event) => { const cacheKey = getCacheKey(provider, voiceSpec, text, instructions); try { - // Check cache first - return redirect to CDN if cached (unless bust=true) - if (!bustCache) { + // Check cache first - return redirect to CDN if cached (unless bust=true). + // When withTimestamps is set we always regenerate, because the cache + // only stores the audio bytes — alignment must come fresh from the API. + if (!bustCache && !withTimestamps) { const cachedUrl = await checkCache(cacheKey); if (cachedUrl) { @@ -404,12 +432,12 @@ exports.handler = async (event) => { } else if (provider === "eleven") { result = await generateElevenLabs(text, gender, set, scream); } else if (provider === "jeffrey") { - result = await generateJeffrey(text, scream, speed, styleOverride, stabilityOverride, similarityOverride); + result = await generateJeffrey(text, scream, speed, styleOverride, stabilityOverride, similarityOverride, withTimestamps); } else { result = await generateOpenAI(text, gender, set, instructions); } - const { buffer: audioBuffer, voiceId } = result; + const { buffer: audioBuffer, voiceId, alignment, normalizedAlignment } = result; if (!audioBuffer || audioBuffer.length === 0) { return { @@ -431,6 +459,39 @@ exports.handler = async (event) => { ts: new Date().toISOString(), }); + // ── Timestamped response — JSON with audio + alignment ──────── + // Returned only when caller opted in. Default callers (recap, + // slab, the `say` piece) never see this branch and keep getting + // a 302 → CDN raw-mp3. + if (withTimestamps && alignment) { + await recordSaying({ + text, + provider, + voice: voiceId, + voiceSpec, + scream, + instructions, + cacheKey, + url: cdnUrl, + cached: false, + withTimestamps: true, + }); + return { + statusCode: 200, + headers: { + ...headers, + "Content-Type": "application/json", + }, + body: JSON.stringify({ + audio: audioBuffer.toString("base64"), + alignment, + normalizedAlignment: normalizedAlignment || null, + url: cdnUrl, + voice: voiceId, + }), + }; + } + if (cdnUrl) { await recordSaying({ text, -- 2.51.2