diff --git a/pop/menuband/bin/chrome-reel.mjs b/pop/menuband/bin/chrome-reel.mjs index 3420f3bb59..44712246d9 100644 --- a/pop/menuband/bin/chrome-reel.mjs +++ b/pop/menuband/bin/chrome-reel.mjs @@ -10,10 +10,12 @@ // REEL-ONLY: no progress bar, no timecode (Reels bring their own // furniture; @jeffrey wants reels with nothing under them). Pals + title. // -// Post pass: sim.mjs writes out/base-menuband-reel.mp4 + meta-menuband-reel.json, -// this pipes every frame through node-canvas and re-encodes with the audio. +// Post pass: a sim writes out/base-.mp4 + meta-.json, this pipes +// every frame through node-canvas and re-encodes with the audio. // -// Usage: node pop/menuband/bin/chrome-reel.mjs +// Usage: node pop/menuband/bin/chrome-reel.mjs [slug] +// slug defaults to "menuband-reel" (the launch reel); the campaign sims +// use menuband-announce / menuband-features / menuband-chords. import { spawn, spawnSync } from "node:child_process"; import { readFileSync, existsSync, mkdirSync } from "node:fs"; @@ -30,9 +32,10 @@ const HERE = dirname(fileURLToPath(import.meta.url)); const LANE = resolve(HERE, ".."); const REPO = resolve(LANE, "..", ".."); const OUTDIR = `${LANE}/out`; -const BASE = `${OUTDIR}/base-menuband-reel.mp4`; -const META = JSON.parse(readFileSync(`${OUTDIR}/meta-menuband-reel.json`, "utf8")); -const OUT = `${OUTDIR}/menuband-reel.mp4`; +const SLUG = process.argv[2] || "menuband-reel"; +const BASE = `${OUTDIR}/base-${SLUG}.mp4`; +const META = JSON.parse(readFileSync(`${OUTDIR}/meta-${SLUG}.json`, "utf8")); +const OUT = `${OUTDIR}/${SLUG}.mp4`; const assetsDir = `${OUTDIR}/chrome-assets-reel`; mkdirSync(assetsDir, { recursive: true }); diff --git a/pop/menuband/bin/reel-lib.mjs b/pop/menuband/bin/reel-lib.mjs new file mode 100644 index 0000000000..1c77099a22 --- /dev/null +++ b/pop/menuband/bin/reel-lib.mjs @@ -0,0 +1,451 @@ +// menuband/bin/reel-lib.mjs — shared helpers for the Menu Band promo-reel +// sims (sim-announce / sim-features / sim-chords), factored out of sim.mjs so +// the launch reel's look carries across the whole campaign: the pale-lilac +// light-mode desktop, the note-particle system, the framed real-capture +// windows, and — new here — the STRIP RIG. +// +// The strip rig re-lights the REAL captured menu-bar piano WITHOUT running the +// MenuBand binary: sim.mjs already cached one capture per lit note +// (out/menubar-frames/mb-.png, notes 67..83) plus the idle strip. Each +// single-note capture differs from idle only in that key's pixels, so diffing +// finds every key's rectangle + its ROYGBIV color, and ANY chord is composited +// by blitting those real lit-key regions onto the real idle strip. Same +// pixels the app draws — no binary spawn, no guessed geometry. + +import { readdirSync, readFileSync, existsSync, mkdirSync, writeFileSync } from "node:fs"; +import { resolve, dirname } from "node:path"; +import { fileURLToPath } from "node:url"; +import { once } from "node:events"; +import { createCanvas, registerFont, loadImage } from "canvas"; +import { spawnFFmpegEncode } from "../../lib/preview-shared.mjs"; + +const HERE = dirname(fileURLToPath(import.meta.url)); +export const LANE = resolve(HERE, ".."); +export const OUT = `${LANE}/out`; +export const W = 1080, H = 1920, FPS = 30; + +// macOS system fonts — registered before any canvas is created. +try { registerFont("/System/Library/Fonts/SFNS.ttf", { family: "MBSans" }); } catch {} +try { registerFont("/System/Library/Fonts/SFNSRounded.ttf", { family: "MBSansRounded" }); } catch {} + +// The exact Menu Band icon keyboard palette (AboutWindow.swift). +export const KEY_COLORS = [ + [255, 77, 107], [255, 153, 46], [255, 214, 56], [51, 209, 179], [97, 158, 255], +]; +export const INK = "rgba(20,18,28,0.92)"; +export const INK_RGB = [20, 18, 28]; + +export const rgb = (a, al = 1) => `rgba(${a[0]},${a[1]},${a[2]},${al})`; +export const easeOut = (u) => 1 - Math.pow(1 - Math.max(0, Math.min(1, u)), 3); +export const clamp01 = (u) => Math.max(0, Math.min(1, u)); + +export function makeStage() { + const canvas = createCanvas(W, H); + return { canvas, ctx: canvas.getContext("2d") }; +} + +export function roundRect(ctx, x, y, w, h, r) { + const rr = Math.min(r, w / 2, h / 2); + ctx.beginPath(); + ctx.moveTo(x + rr, y); + ctx.arcTo(x + w, y, x + w, y + h, rr); + ctx.arcTo(x + w, y + h, x, y + h, rr); + ctx.arcTo(x, y + h, x, y, rr); + ctx.arcTo(x, y, x + w, y, rr); + ctx.closePath(); +} + +export function text(ctx, str, x, y, size, color, weight = 700, align = "center", rounded = true) { + ctx.fillStyle = color; ctx.textAlign = align; ctx.textBaseline = "middle"; + ctx.font = `${weight} ${size}px ${rounded ? "MBSansRounded" : "MBSans"}`; + ctx.fillText(str, x, y); ctx.textAlign = "left"; +} + +// The launch reel's pale-lilac light-mode desktop, verbatim. +export function drawDesktop(ctx) { + const g = ctx.createLinearGradient(0, 0, W, H); + g.addColorStop(0, "rgb(238,232,246)"); + g.addColorStop(0.5, "rgb(216,203,234)"); + g.addColorStop(1, "rgb(190,170,214)"); + ctx.fillStyle = g; ctx.fillRect(0, 0, W, H); + const s = ctx.createLinearGradient(0, H, W, 0); + s.addColorStop(0, "rgba(255,255,255,0)"); + s.addColorStop(0.6, "rgba(255,255,255,0.28)"); + s.addColorStop(1, "rgba(255,255,255,0)"); + ctx.fillStyle = s; ctx.fillRect(0, 0, W, H); +} + +export function vignette(ctx) { + const vg = ctx.createRadialGradient(W / 2, H / 2, H * 0.3, W / 2, H / 2, H * 0.75); + vg.addColorStop(0, "rgba(0,0,0,0)"); vg.addColorStop(1, "rgba(0,0,0,0.28)"); + ctx.fillStyle = vg; ctx.fillRect(0, 0, W, H); +} + +// ── the Menu Band app icon, drawn from scratch (AboutWindow.iconImage), +// lifted verbatim from sim.mjs — purple squircle + cream 5-key piano; +// litKeys (Set of 0..4) glow in the icon's own vivid palette. ────────────── +const SQ_TL = [211, 80, 196], SQ_BR = [67, 29, 113]; +const CREAM = [246, 242, 232], KEYBLACK = [27, 26, 25], EDGE = [40, 28, 22]; +export function drawIcon(ctx, x, y, px, litKeys) { + const R = (x0, y0, x1, y1) => [x + x0 * px, y + y0 * px, (x1 - x0) * px, (y1 - y0) * px]; + const margin = px * 0.098; + const sq = [x + margin, y + margin, px - 2 * margin, px - 2 * margin]; + const radius = sq[2] * 0.2235; + ctx.save(); + roundRect(ctx, sq[0], sq[1], sq[2], sq[3], radius); + ctx.clip(); + const g = ctx.createLinearGradient(sq[0], sq[1], sq[0] + sq[2], sq[1] + sq[3]); + g.addColorStop(0, rgb(SQ_TL)); g.addColorStop(1, rgb(SQ_BR)); + ctx.fillStyle = g; ctx.fillRect(sq[0], sq[1], sq[2], sq[3]); + ctx.restore(); + const plate = R(0.254, 0.428, 0.744, 0.570); + roundRect(ctx, plate[0], plate[1], plate[2], plate[3], px * 0.012); + ctx.fillStyle = rgb(CREAM); ctx.fill(); + const whites = [[0, 0.254, 0.417], [2, 0.417, 0.580], [4, 0.580, 0.744]]; + for (const [idx, x0, x1] of whites) { + if (!litKeys.has(idx)) continue; + ctx.fillStyle = rgb(KEY_COLORS[idx]); + roundRect(ctx, x + x0 * px + 2, y + 0.428 * px + 2, (x1 - x0) * px - 4, 0.142 * px - 4, px * 0.01); + ctx.fill(); + } + ctx.strokeStyle = rgb(EDGE); ctx.lineWidth = px * 0.006; + for (const xf of [0.417, 0.580]) { + ctx.beginPath(); + ctx.moveTo(x + xf * px, y + 0.432 * px); + ctx.lineTo(x + xf * px, y + 0.566 * px); + ctx.stroke(); + } + roundRect(ctx, plate[0], plate[1], plate[2], plate[3], px * 0.012); + ctx.lineWidth = px * 0.008; ctx.strokeStyle = rgb(EDGE); ctx.stroke(); + const blacks = [[1, 0.369, 0.463], [3, 0.535, 0.629]]; + for (const [idx, x0, x1] of blacks) { + ctx.fillStyle = litKeys.has(idx) ? rgb(KEY_COLORS[idx]) : rgb(KEYBLACK); + roundRect(ctx, x + x0 * px, y + 0.428 * px, (x1 - x0) * px, 0.098 * px, px * 0.01); + ctx.fill(); + } +} + +// ── note-particle system (vector eighth-notes), lifted from sim.mjs ──────── +export function makeParticles(ctx) { + const particles = []; + function drawEighthNote(ox, oy, fill) { + ctx.save(); ctx.translate(ox, oy); ctx.fillStyle = fill; + ctx.save(); ctx.translate(0, 14); ctx.rotate(-0.3); ctx.beginPath(); + ctx.ellipse(0, 0, 11, 8, 0, 0, Math.PI * 2); ctx.fill(); ctx.restore(); + ctx.fillRect(9, -22, 4.5, 36); + ctx.beginPath(); ctx.moveTo(13, -22); ctx.quadraticCurveTo(30, -14, 20, 2); + ctx.quadraticCurveTo(26, -10, 13, -10); ctx.closePath(); ctx.fill(); + ctx.restore(); + } + return { + spawnNote(cx, cy, color, down = false) { + particles.push({ + x: cx + (Math.sin(particles.length * 1.7) * 6), y: cy, + vx: (Math.sin(particles.length * 2.3)) * 60, + vy: down ? 40 + (particles.length % 5) * 12 : -260 - (particles.length % 5) * 18, + life: 0, ttl: 2.2, color, rot: (particles.length % 7 - 3) * 0.12, + }); + }, + stepAndDraw(dt) { + for (const p of particles) { + p.life += dt; + p.x += p.vx * dt; p.y += p.vy * dt; p.vy += 320 * dt; + } + for (let i = particles.length - 1; i >= 0; i--) if (particles[i].life > particles[i].ttl) particles.splice(i, 1); + for (const p of particles) { + const a = Math.max(0, 1 - p.life / p.ttl); + ctx.save(); + ctx.globalAlpha = a; ctx.translate(p.x, p.y); ctx.rotate(p.rot); + drawEighthNote(2.5, 3, "rgba(0,0,0,0.5)"); + drawEighthNote(0, 0, rgb(p.color)); + ctx.restore(); + } + }, + }; +} + +// ── the strip rig — real captured strip, re-lightable per note set ───────── +/// Midis the cache holds single-note captures for (the waltz's white keys, +/// G4..B5). Any melody is folded onto these before lighting. +export const STRIP_MIDIS = [67, 69, 71, 72, 74, 76, 77, 79, 81, 83]; +const PC_TO_STRIP = new Map([[0, 72], [2, 74], [4, 76], [5, 77], [7, 79], [9, 69], [11, 71]]); + +/// Fold any (white-key) midi onto a strip key: same pitch class, the cached +/// octave. G/A/B keep their low octave when they arrive below middle C's row. +export function foldToStrip(midi) { + if (STRIP_MIDIS.includes(midi)) return midi; + const pc = ((midi % 12) + 12) % 12; + const base = PC_TO_STRIP.get(pc); + if (base == null) return null; // black key — not cached + if (midi < base - 6 && STRIP_MIDIS.includes(base - 12)) return base - 12; + return base; +} + +export async function loadStripRig() { + const dir = `${OUT}/menubar-frames`; + const idlePath = `${dir}/mb-idle.png`; + if (!existsSync(idlePath)) throw new Error(`missing ${idlePath} — run sim.mjs once with the MenuBand binary available`); + const idle = await loadImage(idlePath); + const w = idle.width, h = idle.height; + + const scratch = createCanvas(w, h); + const sctx = scratch.getContext("2d"); + const pixels = (img) => { + sctx.clearRect(0, 0, w, h); + sctx.drawImage(img, 0, 0); + return sctx.getImageData(0, 0, w, h); + }; + const idlePx = pixels(idle); + + // Diff each single-note capture against idle → that key's rect + color. + const keys = new Map(); // midi -> { x0, x1, y0, y1, cx, color, img } + for (const midi of STRIP_MIDIS) { + const p = `${dir}/mb-${midi}.png`; + if (!existsSync(p)) continue; + const img = await loadImage(p); + const px = pixels(img); + const colDiff = new Int32Array(w); + let y0 = h, y1 = 0; + for (let y = 0; y < h; y++) { + for (let x = 0; x < w; x++) { + const i = (y * w + x) * 4; + const d = Math.abs(px.data[i] - idlePx.data[i]) + + Math.abs(px.data[i + 1] - idlePx.data[i + 1]) + + Math.abs(px.data[i + 2] - idlePx.data[i + 2]); + if (d > 36) { colDiff[x]++; if (y < y0) y0 = y; if (y > y1) y1 = y; } + } + } + // Take the widest contiguous run of differing columns — that's the key + // (guards against stray anti-aliasing noise elsewhere on the strip). + let best = null, runStart = -1; + for (let x = 0; x <= w; x++) { + const on = x < w && colDiff[x] > 2; + if (on && runStart < 0) runStart = x; + if (!on && runStart >= 0) { + if (!best || x - runStart > best.x1 - best.x0) best = { x0: runStart, x1: x }; + runStart = -1; + } + } + if (!best) continue; + // Average lit color inside the run (for particles). + let r = 0, g = 0, b = 0, n = 0; + for (let y = Math.max(0, y0); y <= Math.min(h - 1, y1); y++) { + for (let x = best.x0; x < best.x1; x++) { + if (colDiff[x] <= 2) continue; + const i = (y * w + x) * 4; + r += px.data[i]; g += px.data[i + 1]; b += px.data[i + 2]; n++; + } + } + keys.set(midi, { + x0: best.x0, x1: best.x1, y0, y1, + cx: (best.x0 + best.x1) / 2 / w, + color: n ? [Math.round(r / n), Math.round(g / n), Math.round(b / n)] : INK_RGB, + img, + }); + } + + // Composite cache: any subset of lit keys → one canvas. + const composites = new Map(); // "72,76,79" -> canvas + function imageFor(midis) { + const lit = [...new Set(midis.map(foldToStrip).filter((m) => m != null && keys.has(m)))].sort((a, b) => a - b); + if (lit.length === 0) return idle; + const key = lit.join(","); + if (composites.has(key)) return composites.get(key); + const c = createCanvas(w, h); + const cx = c.getContext("2d"); + cx.drawImage(idle, 0, 0); + for (const m of lit) { + const k = keys.get(m); + const pad = 2; // seam-safe blit + const bx = Math.max(0, k.x0 - pad), bw = Math.min(w, k.x1 + pad) - bx; + const by = Math.max(0, k.y0 - pad), bh = Math.min(h, k.y1 + pad) - by; + cx.drawImage(k.img, bx, by, bw, bh, bx, by, bw, bh); + } + composites.set(key, c); + return c; + } + + return { idle, keys, imageFor, aspect: w / h }; +} + +/// Draw the strip in a rect; returns nothing — callers own placement. +export function drawStrip(ctx, rig, midis, x, y, w, alpha = 1) { + const img = rig.imageFor(midis); + const h = w / rig.aspect; + ctx.save(); ctx.globalAlpha = alpha; + ctx.drawImage(img, x, y, w, h); + ctx.restore(); + return { x, y, w, h }; +} + +/// Where a folded note's key sits on a drawn strip rect (for particles). +export function stripKeyX(rig, midi, rect) { + const k = rig.keys.get(foldToStrip(midi)); + return rect.x + rect.w * (k ? k.cx : 0.5); +} +export function stripKeyColor(rig, midi) { + return rig.keys.get(foldToStrip(midi))?.color ?? INK_RGB; +} + +// ── framed macOS window around a real capture (from sim.mjs drawAbout) ───── +export const TITLEBAR_H = 72; +export function drawFramedWindow(ctx, img, { alpha = 1, yOff = 0, contentH = null, cx = W / 2, cyFrac = 0.5, borderless = false } = {}) { + if (!img) return null; + const tb = borderless ? 0 : TITLEBAR_H; + const ch = contentH ?? Math.min(H * 0.66, 1180); + const aw = ch * img.width / img.height; + const ah = ch + tb; + const ax = cx - aw / 2, ay = H * cyFrac - ah / 2 + yOff; + ctx.save(); ctx.globalAlpha = alpha; + ctx.shadowColor = "rgba(0,0,0,0.45)"; ctx.shadowBlur = 60; ctx.shadowOffsetY = 24; + roundRect(ctx, ax, ay, aw, ah, 26); ctx.fillStyle = "white"; ctx.fill(); + ctx.shadowColor = "transparent"; + ctx.save(); roundRect(ctx, ax, ay, aw, ah, 26); ctx.clip(); + if (!borderless) { + ctx.fillStyle = "rgb(238,235,245)"; ctx.fillRect(ax, ay, aw, tb); + const lights = ["rgb(255,95,86)", "rgb(255,189,46)", "rgb(39,201,63)"]; + for (let i = 0; i < 3; i++) { ctx.beginPath(); ctx.fillStyle = lights[i]; ctx.arc(ax + 44 + i * 46, ay + tb / 2, 15, 0, 7); ctx.fill(); } + } + ctx.drawImage(img, ax, ay + tb, aw, ch); + ctx.restore(); + ctx.restore(); + return { ax, ay, aw, ah, tb }; +} + +// ── karaoke captions (the sung reels) ────────────────────────────────────── +// sing-jingle.mjs writes out/.words.sung.json — absolute word timings +// on the reel clock, grouped by lyric line. The active line sits centered +// near the bottom; sung words are ink, the active word rides an ink pill in +// white (reel.mjs --song's karaoke look, drawn with node-canvas — no libass +// in this ffmpeg build). +export function loadSungWords(slug) { + const p = `${OUT}/${slug}.words.sung.json`; + if (!existsSync(p)) { console.error(`✗ missing ${p} — run sing-jingle.mjs first`); process.exit(1); } + return JSON.parse(readFileSync(p, "utf8")); +} + +export function makeKaraoke(words, { y = H * 0.925, size = 54 } = {}) { + const lines = []; + for (const w of words) { + (lines[w.line] ??= []).push(w); + } + const packed = lines.filter(Boolean).map((ws) => ({ + words: ws, + from: ws[0].fromMs / 1000, + to: ws[ws.length - 1].toMs / 1000, + })); + return { + draw(ctx, t) { + const li = packed.findIndex((L, i) => { + const nextFrom = i + 1 < packed.length ? packed[i + 1].from : Infinity; + return t >= L.from - 0.4 && t < Math.min(L.to + 0.6, nextFrom - 0.05); + }); + if (li < 0) return; + const L = packed[li]; + const a = easeOut(clamp01((t - (L.from - 0.4)) / 0.3)) * + (1 - easeOut(clamp01((t - (L.to + 0.25)) / 0.35))); + if (a <= 0) return; + + let sz = size; + ctx.font = `700 ${sz}px MBSansRounded`; + const gap = sz * 0.38; + const widths = L.words.map((w) => ctx.measureText(w.text).width); + let total = widths.reduce((s, w) => s + w, 0) + gap * (L.words.length - 1); + if (total > W * 0.92) { + sz = sz * (W * 0.92) / total; + ctx.font = `700 ${sz}px MBSansRounded`; + for (let i = 0; i < L.words.length; i++) widths[i] = ctx.measureText(L.words[i].text).width; + total = widths.reduce((s, w) => s + w, 0) + (sz * 0.38) * (L.words.length - 1); + } + const g = sz * 0.38; + let x = W / 2 - total / 2; + ctx.save(); + ctx.globalAlpha = a; + ctx.textAlign = "left"; + ctx.textBaseline = "middle"; + for (let i = 0; i < L.words.length; i++) { + const w = L.words[i]; + const from = w.fromMs / 1000, to = w.toMs / 1000; + const active = t >= from && t < to; + const sung = t >= from; + if (active) { + const pop = 1 + 0.10 * (1 - easeOut(clamp01((t - from) / 0.18))); + const pw = widths[i] * pop, ph = sz * 1.42; + const cx = x + widths[i] / 2, cy = y; + roundRect(ctx, cx - pw / 2 - sz * 0.22, cy - ph / 2, pw + sz * 0.44, ph, sz * 0.32); + ctx.fillStyle = INK; ctx.fill(); + ctx.save(); + ctx.translate(cx, cy); ctx.scale(pop, pop); ctx.translate(-cx, -cy); + ctx.fillStyle = "rgba(255,255,255,0.98)"; + ctx.fillText(w.text, x, y + 1); + ctx.restore(); + } else { + ctx.fillStyle = sung ? INK : "rgba(20,18,28,0.35)"; + ctx.fillText(w.text, x, y + 1); + } + x += widths[i] + g; + } + ctx.restore(); + }, + }; +} + +/// Shared --sung plumbing: audio/base/meta paths swing to the -sung variants. +export function sungMode() { + const sung = process.argv.includes("--sung"); + return { sung, suffix: sung ? "-sung" : "" }; +} + +// ── notes.json helpers ───────────────────────────────────────────────────── +export function loadScore(name) { + const p = `${OUT}/${name}.notes.json`; + if (!existsSync(p)) { console.error(`✗ missing ${p} — run render-jingles.mjs first`); process.exit(1); } + return JSON.parse(readFileSync(p, "utf8")); +} +export function leadOf(score) { return (score.notes || []).filter((n) => n.lane === "lead"); } +export function litAt(lead, t, minHold = 0.12) { + const out = []; + for (const n of lead) if (t >= n.t && t < n.t + Math.max(minHold, n.dur)) out.push(n.midi); + return out; +} +export function makeOnsets(lead) { + const sorted = [...lead].sort((a, b) => a.t - b.t); + let cursor = 0; + return (t0, t1) => { + const hits = []; + while (cursor < sorted.length && sorted[cursor].t <= t1) { + if (sorted[cursor].t >= t0) hits.push(sorted[cursor]); + cursor++; + } + return hits; + }; +} + +// ── the frame pump: drawFrame(t) → ffmpeg, muxed with the jingle ─────────── +export async function renderVideo({ canvas, audioPath, outPath, total, drawFrame, label = "sim" }) { + const FRAMES = Math.round(total * FPS); + console.log(`▸ ${label} · ${FRAMES} frames · ${total.toFixed(1)}s · ${W}x${H}@${FPS}`); + const enc = spawnFFmpegEncode({ audioPath, w: W, h: H, fps: FPS, outPath, crf: 18 }); + const t0 = Date.now(); + for (let fi = 0; fi < FRAMES; fi++) { + drawFrame(fi / FPS); + if (!enc.stdin.write(canvas.toBuffer("raw"))) await once(enc.stdin, "drain"); + if (fi % 150 === 0) console.log(` ${fi}/${FRAMES} · ${((Date.now() - t0) / 1000).toFixed(0)}s`); + } + enc.stdin.end(); + await new Promise((res, rej) => { enc.on("close", (c) => (c === 0 ? res() : rej(new Error(`ffmpeg exit ${c}`)))); }); + console.log(`✓ base ${outPath}`); +} + +export function writeMeta(slug, total, scenes) { + writeFileSync(`${OUT}/meta-${slug}.json`, JSON.stringify({ + total, + slides: scenes.map((s) => ({ name: s.name, from: s.from, to: s.to, tint: s.tint })), + }, null, 2)); +} + +/// Scale scene fractions to seconds, sim.mjs-style. +export function makeScenes(defs, total) { + const scenes = defs.map((s) => ({ ...s, from: s.from * total, to: s.to * total })); + return { scenes, sceneAt: (t) => scenes.find((s) => t >= s.from && t < s.to) ?? scenes.at(-1) }; +} diff --git a/pop/menuband/bin/render-jingles.mjs b/pop/menuband/bin/render-jingles.mjs new file mode 100644 index 0000000000..88b131d1fc --- /dev/null +++ b/pop/menuband/bin/render-jingles.mjs @@ -0,0 +1,203 @@ +#!/usr/bin/env node +// render-jingles.mjs — the three campaign jingles for the Menu Band promo +// reels (announce / features / chords), built on the same /pop lullaby engine +// as the launch waltz (render-waltz.mjs). Each jingle writes an mp3 + a +// notes.json its sim choreographs to; the chords jingle also writes a +// segment score (menuband-chords.score.json) so the audio and the on-screen +// modifier keycaps agree to the frame. +// +// EVERY lead note is a WHITE KEY: the strip rig re-lights the real captured +// menu-bar piano from the single-note captures sim.mjs cached (G4..B5, +// C-major scale), so staying diatonic keeps every lit key pixel-real. +// +// Run: node pop/menuband/bin/render-jingles.mjs (from repo root) + +import { dirname, resolve } from "node:path"; +import { fileURLToPath } from "node:url"; +import { writeFileSync } from "node:fs"; +import { renderLullaby, m } from "../../marimba/lullabies/lib/core.mjs"; + +const HERE = dirname(fileURLToPath(import.meta.url)); +const OUT_DIR = resolve(HERE, "..", "out"); + +function makeTrack() { + const events = []; + const push = (lane, preset, startSec, note, durSec, gain, pan = 0, decayMul = 1.2) => { + events.push({ lane, preset, startSec, midi: typeof note === "number" ? note : m(note), durSec, gain, pan, decayMul }); + }; + return { events, push }; +} + +function writeScore(name, events, extra, meta) { + const notes = events + .map((e) => ({ t: +e.startSec.toFixed(4), dur: +e.durSec.toFixed(4), midi: e.midi, vel: +e.gain.toFixed(3), lane: e.lane })) + .sort((a, b) => a.t - b.t || a.midi - b.midi); + writeFileSync(resolve(OUT_DIR, `${name}.notes.json`), JSON.stringify({ ...meta, notes, ...extra }, null, 2)); +} + +function master(name, events, opts) { + const { mp3, durationSec } = renderLullaby(events, { + name, here: HERE, out: resolve(OUT_DIR, `${name}.mp3`), + healing: false, reverb: { wet: 0.26, decay: 0.82, damp: 0.38 }, + fadeIn: 0.3, fadeOut: 1.8, tailSec: 2.2, ...opts, + }); + console.log(`✓ ${mp3} · ${durationSec.toFixed(1)}s`); + return durationSec; +} + +// ════════════════════════════════════════════════════════════════════════ +// 1 · ANNOUNCE — a 12-bar fanfare waltz in C (138 BPM): the launch waltz's +// voice (glockenspiel over oom-pah-pah) but shorter and more emphatic, +// opening with a rising call and cadencing clean on tonic for the +// "on the Mac App Store now" card. +// ════════════════════════════════════════════════════════════════════════ +{ + const BPM = 138, BEAT = 60 / BPM, BAR = 3 * BEAT; + const { events, push } = makeTrack(); + const CH = { + C: ["C2", ["E4", "G4"]], G: ["G2", ["D4", "G4"]], G7: ["G2", ["F4", "B3"]], + F: ["F2", ["A3", "C4"]], Am: ["A2", ["C4", "E4"]], Dm: ["D2", ["F4", "A3"]], + }; + const accompany = (bar, chord, g = 1) => { + const [bn, [a, b]] = CH[chord]; + const t = bar * BAR; + push("bass", "bass", t, bn, 1.05 * BEAT, 0.4 * g, 0); + push("harmony", "staccato", t + BEAT, a, 0.42 * BEAT, 0.15 * g, -0.16, 0.9); + push("harmony", "staccato", t + BEAT, b, 0.42 * BEAT, 0.15 * g, -0.16, 0.9); + push("harmony", "staccato", t + 2 * BEAT, a, 0.42 * BEAT, 0.14 * g, 0.16, 0.9); + push("harmony", "staccato", t + 2 * BEAT, b, 0.42 * BEAT, 0.14 * g, 0.16, 0.9); + }; + const melody = (startBar, cell, { gain = 0.5, pan = 0.04, ring = 1.25 } = {}) => { + let t = startBar * BAR; + for (const [note, beats] of cell) { + if (note != null) push("lead", "glockenspiel", t, note, beats * BEAT * ring, gain, pan, 1.3); + t += beats * BEAT; + } + }; + ["C", "C", "F", "C", "Am", "F", "G7", "G7", "C", "F", "G7", "C"].forEach((c, i) => accompany(i, c)); + melody(0, [ + // bars 0-1: the rising fanfare call — climb the tonic triad, ring high + ["C5", 1], ["E5", 1], ["G5", 1], + ["C6", 2], ["G5", 1], + // bars 2-3: answer over F — a bright turn and settle + ["A5", 1], ["G5", 0.5], ["F5", 0.5], ["E5", 1], + ["F5", 1], ["E5", 1], ["C5", 1], + // bars 4-5: lean minor, reach back up + ["E5", 1], ["A5", 1], ["C6", 1], + ["A5", 1], ["F5", 1], ["A5", 1], + // bars 6-7: dominant gather — poised, expectant + ["G5", 1], ["F5", 1], ["D5", 1], + ["B4", 1], ["D5", 1], ["G5", 1], + // bars 8-11: the arrival — tune rings home, authentic cadence on C + ["E5", 1], ["G5", 1], ["C6", 1], + ["A5", 1], ["C6", 1], ["F5", 1], + ["F5", 1], ["E5", 1], ["D5", 1], + ["C5", 1], ["E5", 0.5], ["G5", 0.5], ["C6", 2], + ]); + // a far sparkle over the final bar + push("harmony", "glockenspiel", 11 * BAR + BEAT, m("G6"), 2.2 * BEAT, 0.1, 0.32, 1.5); + const durationSec = master("menuband-announce", events, { title: "Menu Band Announce Fanfare" }); + writeScore("menuband-announce", events, {}, { + bpm: BPM, beatSec: +BEAT.toFixed(6), barSec: +BAR.toFixed(6), durationSec: +durationSec.toFixed(4), + }); +} + +// ════════════════════════════════════════════════════════════════════════ +// 2 · FEATURES — a 10-bar 4/4 tour groove (116 BPM): kalimba arpeggio bed, +// woodblock ticks, and a singable vibraphone hook on top (the hook is +// the `lead` lane, so it's what the strip lights). +// ════════════════════════════════════════════════════════════════════════ +{ + const BPM = 116, BEAT = 60 / BPM, BAR = 4 * BEAT; + const { events, push } = makeTrack(); + const ARPS = { // white-key arp tones per chord + C: ["C4", "E4", "G4", "E4"], Am: ["A3", "C4", "E4", "C4"], + F: ["F3", "A3", "C4", "A3"], G: ["G3", "B3", "D4", "B3"], + }; + const BASSN = { C: "C2", Am: "A2", F: "F2", G: "G2" }; + const PROG = ["C", "Am", "F", "G", "C", "Am", "F", "G", "C", "C"]; + PROG.forEach((c, i) => { + const t = i * BAR; + push("bass", "bass", t, BASSN[c], 1.4 * BEAT, 0.36, 0); + push("bass", "bass", t + 2 * BEAT, BASSN[c], 1.1 * BEAT, 0.24, 0); + for (let s = 0; s < 8; s++) { // 8th-note kalimba arps + push("harmony", "kalimba", t + s * BEAT * 0.5, ARPS[c][s % 4], 0.5 * BEAT, 0.16, s % 2 ? 0.22 : -0.22, 1.0); + } + for (const b of [1, 3]) push("perc", "woodblock", t + b * BEAT, "E5", 0.15, 0.11, 0.1); + }); + const hook = (bar, cell, gain = 0.42) => { + let t = bar * BAR; + for (const [note, beats] of cell) { + if (note != null) push("lead", "vibraphone", t, note, beats * BEAT * 1.15, gain, 0.05, 1.25); + t += beats * BEAT; + } + }; + hook(0, [[null, 1], ["E5", 1], ["G5", 1], ["C6", 1]]); + hook(1, [["A5", 1.5], ["E5", 1.5], ["C5", 1]]); + hook(2, [["F5", 1], ["A5", 1], ["C6", 1.5], [null, 0.5]]); + hook(3, [["B5", 1], ["G5", 1], ["D5", 2]]); + hook(4, [["E5", 1], ["G5", 1], ["C6", 1], ["E6", 1]]); + hook(5, [["C6", 1.5], ["A5", 1.5], ["E5", 1]]); + hook(6, [["A5", 1], ["C6", 1], ["F5", 2]]); + hook(7, [["D5", 1], ["G5", 1], ["B5", 1], ["D6", 1]]); + hook(8, [["C6", 2], ["G5", 1], ["E5", 1]]); + hook(9, [["C5", 1], ["E5", 0.5], ["G5", 0.5], ["C6", 2]]); + const durationSec = master("menuband-features", events, { title: "Menu Band Feature Tour" }); + writeScore("menuband-features", events, {}, { + bpm: BPM, beatSec: +BEAT.toFixed(6), barSec: +BAR.toFixed(6), durationSec: +durationSec.toFixed(4), + }); +} + +// ════════════════════════════════════════════════════════════════════════ +// 3 · CHORDS — the modifier-grammar demo. One segment per quality, exactly +// the app's own mapping (MenuBandController.chordQuality/chordIntervals): +// ⌘ major · ⌥ minor · ⌘⌥ sus2 · ⌥⌃ diminished · ⌘⌥⌃ sus4 — voiced on +// white keys so every chord's strip lighting is pixel-real. Ends on a +// quick I–vi–IV–V–I progression with ⌘/⌥ flipping live. +// ════════════════════════════════════════════════════════════════════════ +{ + const { events, push } = makeTrack(); + const SEG = 2.5; + const segs = []; + const chordHit = (t, notes, { gain = 0.3, ring = 1.9, restrike = true } = {}) => { + push("bass", "bass", t, notes[0] - 24, 1.3, 0.34, 0); + notes.forEach((n, i) => push("lead", "vibraphone", t + i * 0.06, n, ring, gain, (i - 1) * 0.14, 1.3)); + if (restrike) notes.forEach((n, i) => push("lead", "vibraphone", t + 1.3 + i * 0.05, n, 1.1, gain * 0.6, (i - 1) * 0.14, 1.2)); + }; + const seg = (i, label, mods, chordName, notes) => { + const t0 = 0.4 + i * SEG, t1 = t0 + SEG; + segs.push({ t0: +t0.toFixed(3), t1: +t1.toFixed(3), label, mods, chordName, notes }); + chordHit(t0 + 0.15, notes); + return t0; + }; + seg(0, "one key plays one note", [], "C", [m("C5")]); + seg(1, "hold ⌘ — major", ["cmd"], "C major", [m("C5"), m("E5"), m("G5")]); + seg(2, "hold ⌥ — minor", ["opt"], "A minor", [m("A4"), m("C5"), m("E5")]); + seg(3, "⌘ ⌥ — sus2", ["cmd", "opt"], "C sus2", [m("C5"), m("D5"), m("G5")]); + seg(4, "⌥ ⌃ — diminished", ["opt", "ctl"], "B dim", [m("B4"), m("D5"), m("F5")]); + seg(5, "⌘ ⌥ ⌃ — sus4", ["cmd", "opt", "ctl"], "C sus4", [m("C5"), m("F5"), m("G5")]); + + // finale — the grammar in motion: C · Am · F · G · C with mods flipping + const FIN0 = 0.4 + 6 * SEG; + const FINALE = [ + ["C major", ["cmd"], [m("C5"), m("E5"), m("G5")]], + ["A minor", ["opt"], [m("A4"), m("C5"), m("E5")]], + ["F major", ["cmd"], [m("F4"), m("A4"), m("C5")]], + ["G major", ["cmd"], [m("G4"), m("B4"), m("D5")]], + ["C major", ["cmd"], [m("C5"), m("E5"), m("G5"), m("C6")]], + ]; + FINALE.forEach(([name, mods, notes], k) => { + const t0 = FIN0 + k * 0.85; + const last = k === FINALE.length - 1; + const t1 = last ? t0 + 3.0 : t0 + 0.85; + segs.push({ t0: +t0.toFixed(3), t1: +t1.toFixed(3), label: "every chord under three keys", mods, chordName: name, notes }); + chordHit(t0, notes, { gain: last ? 0.34 : 0.28, ring: last ? 2.6 : 1.0, restrike: false }); + }); + push("harmony", "glockenspiel", FIN0 + 4 * 0.85 + 0.4, m("G6"), 2.0, 0.1, 0.3, 1.5); + + const durationSec = master("menuband-chords", events, { title: "Menu Band Chord Grammar" }); + writeScore("menuband-chords", events, {}, { durationSec: +durationSec.toFixed(4) }); + writeFileSync(resolve(OUT_DIR, "menuband-chords.score.json"), + JSON.stringify({ durationSec: +durationSec.toFixed(4), segs }, null, 2)); + console.log(` ${segs.length} chord segments`); +} diff --git a/pop/menuband/bin/sim-announce.mjs b/pop/menuband/bin/sim-announce.mjs new file mode 100644 index 0000000000..a167bfeae2 --- /dev/null +++ b/pop/menuband/bin/sim-announce.mjs @@ -0,0 +1,143 @@ +#!/usr/bin/env node +// sim-announce.mjs — the "Menu Band is on the Mac App Store NOW" reel base. +// +// Three acts over the announce fanfare (render-jingles.mjs): +// 1 · the REAL captured menu-bar piano falls in front-and-center and plays +// the fanfare (strip rig — real lit-key pixels), headline under it; +// 2 · the REAL About window floats up while the band climbs to the top and +// rains notes over it; +// 3 · the end card — drawn app icon (AboutWindow geometry, keys pulsing +// with the music), "now on the Mac App Store", the real scannable QR, +// menuband.app. +// +// Output: out/base-menuband-announce.mp4 + meta-menuband-announce.json — +// then `node pop/menuband/bin/chrome-reel.mjs menuband-announce` stamps the +// pals chrome. +// +// Usage: node pop/menuband/bin/sim-announce.mjs (render-jingles.mjs first) + +import { loadImage } from "canvas"; +import { + W, H, FPS, OUT, INK, easeOut, clamp01, rgb, KEY_COLORS, + makeStage, roundRect, text, drawDesktop, vignette, drawIcon, + makeParticles, loadStripRig, drawStrip, stripKeyX, stripKeyColor, foldToStrip, + drawFramedWindow, loadScore, leadOf, litAt, makeOnsets, + renderVideo, writeMeta, makeScenes, sungMode, loadSungWords, makeKaraoke, +} from "./reel-lib.mjs"; + +const SLUG = "menuband-announce"; +// --sung: jeffrey's sung vocal mix + karaoke captions (sing-jingle.mjs), +// writing the -sung base/meta so the originals stay untouched. +const { sung: SUNG, suffix: VAR } = sungMode(); +const karaoke = SUNG ? makeKaraoke(loadSungWords(SLUG)) : null; +const score = loadScore(SLUG); +const TOTAL = score.durationSec; +const BAR_SEC = score.barSec || 1.3; +const lead = leadOf(score); +const onsetsBetween = makeOnsets(lead); + +const { canvas, ctx } = makeStage(); +const rig = await loadStripRig(); +const aboutImg = await loadImage(`${OUT}/about-frames/about-en.png`); +const qrImg = await loadImage(`${OUT}/qr-menuband.png`); +const particles = makeParticles(ctx); + +const { scenes: SCENES, sceneAt } = makeScenes([ + { name: "menu", from: 0.00, to: 0.40, tint: [97, 158, 255] }, + { name: "about", from: 0.40, to: 0.66, tint: [167, 139, 250] }, + { name: "end", from: 0.66, to: 1.00, tint: [255, 77, 107] }, +], TOTAL); + +// ── hero strip choreography (sim.mjs heroRect, retuned) ──────────────────── +const HERO_W = W * 0.96, HERO_X = (W - HERO_W) / 2; +const HERO_BOB = 22, HERO_DROP = 1.2, HERO_TOP = 54, HERO_RISE = 1.1; +function heroRect(t) { + const h = HERO_W / rig.aspect; + const middle = H * 0.40 - h / 2; + const riseEnd = SCENES[0].to; + const rise = easeOut((t - (riseEnd - HERO_RISE)) / HERO_RISE); + const rest = middle + (HERO_TOP - middle) * rise; + const enter = easeOut(clamp01(t / HERO_DROP)); + const bob = Math.sin((t / BAR_SEC) * Math.PI * 2) * HERO_BOB * (1 - rise * 0.6); + const y = (-h - 40) + (rest - (-h - 40)) * enter + bob * enter; + return { x: HERO_X, y, w: HERO_W, h, risen: rise > 0.5 }; +} + +// ── the end card ─────────────────────────────────────────────────────────── +function drawEndCard(t, local, e) { + const cw = W * 0.82, chh = H * 0.62; + const cx = (W - cw) / 2, cy = H * 0.52 - chh / 2 + (1 - e) * H * 0.5; + ctx.save(); ctx.globalAlpha = e; + ctx.shadowColor = "rgba(0,0,0,0.45)"; ctx.shadowBlur = 60; ctx.shadowOffsetY = 24; + roundRect(ctx, cx, cy, cw, chh, 40); ctx.fillStyle = "rgba(250,249,253,0.99)"; ctx.fill(); + ctx.shadowColor = "transparent"; + + // drawn app icon, its 5 keys pulsing with the lit melody (pc % 5 fold — + // the same fold sim.mjs used on this icon) + const litIcon = new Set(litAt(lead, t).map((mm) => ((mm % 12) + 12) % 12 % 5)); + const ipx = 340; + drawIcon(ctx, W / 2 - ipx / 2, cy + 44, ipx, litIcon); + + text(ctx, "Menu Band", W / 2, cy + 460, 96, INK, 800); + text(ctx, "now on the Mac App Store", W / 2, cy + 552, 46, "rgba(60,50,80,0.9)", 700); + + // the real scannable QR on a white tile + const q = 300, qx = W / 2 - q / 2, qy = cy + 610; + roundRect(ctx, qx - 18, qy - 18, q + 36, q + 36, 22); + ctx.fillStyle = "white"; ctx.fill(); + ctx.lineWidth = 2; ctx.strokeStyle = "rgba(20,18,28,0.14)"; ctx.stroke(); + ctx.imageSmoothingEnabled = false; + ctx.drawImage(qrImg, qx, qy, q, q); + ctx.imageSmoothingEnabled = true; + + text(ctx, "menuband.app · free", W / 2, qy + q + 74, 44, "rgba(60,50,80,0.85)", 700); + ctx.restore(); +} + +function drawFrame(t) { + drawDesktop(ctx); + const sc = sceneAt(t); + const local = (sc.to - sc.from) > 0 ? (t - sc.from) / (sc.to - sc.from) : 1; + const dt = 1 / FPS; + + const hero = heroRect(t); + const hRect = drawStrip(ctx, rig, litAt(lead, t), hero.x, hero.y, hero.w); + const edge = hero.risen ? hRect.y + hRect.h + 6 : hRect.y - 6; + for (const n of onsetsBetween(t - dt, t)) { + particles.spawnNote(stripKeyX(rig, n.midi, hRect), edge, stripKeyColor(rig, n.midi), hero.risen); + } + + if (sc.name === "menu") { + // headline under the band, fading in once it has landed + const a = easeOut(clamp01((t - 1.3) / 0.7)); + if (a > 0) { + ctx.save(); ctx.globalAlpha = a; + text(ctx, "Menu Band", W / 2, H * 0.60, 128, INK, 800); + text(ctx, "your menu bar is a synthesizer", W / 2, H * 0.60 + 116, 46, "rgba(60,50,80,0.88)", 600); + text(ctx, "out NOW on the Mac App Store", W / 2, H * 0.60 + 196, 52, INK, 800); + ctx.restore(); + } + } else if (sc.name === "about") { + const e = easeOut(Math.min(1, local * 5)); + drawFramedWindow(ctx, aboutImg, { alpha: e, yOff: (H * 0.66) * (1 - e), cyFrac: 0.54, contentH: Math.min(H * 0.60, 1140) }); + const a = easeOut(clamp01((local - 0.25) / 0.3)); + if (a > 0 && !SUNG) { // sung mode: the karaoke line owns this band + ctx.save(); ctx.globalAlpha = a; + text(ctx, "for macOS · free", W / 2, H * 0.925, 46, INK, 700); + ctx.restore(); + } + } else { + const e = easeOut(Math.min(1, local * 5)); + drawEndCard(t, local, e); + } + + particles.stepAndDraw(dt); + vignette(ctx); + karaoke?.draw(ctx, t); +} + +await renderVideo({ + canvas, audioPath: `${OUT}/${SLUG}${VAR}.mp3`, outPath: `${OUT}/base-${SLUG}${VAR}.mp4`, + total: TOTAL, drawFrame, label: `menuband announce sim${VAR}`, +}); +writeMeta(`${SLUG}${VAR}`, TOTAL, SCENES); diff --git a/pop/menuband/bin/sim-chords.mjs b/pop/menuband/bin/sim-chords.mjs new file mode 100644 index 0000000000..a254f36c9c --- /dev/null +++ b/pop/menuband/bin/sim-chords.mjs @@ -0,0 +1,155 @@ +#!/usr/bin/env node +// sim-chords.mjs — the Menu Band CHORD GRAMMAR quick-hits reel base. +// +// The app's actual modifier grammar (MenuBandController.chordQuality / +// chordIntervals), one segment per quality, audio and visuals driven by the +// same score (out/menuband-chords.score.json from render-jingles.mjs): +// +// ⌘ major · ⌥ minor · ⌘⌥ sus2 · ⌥⌃ diminished · ⌘⌥⌃ sus4 +// +// The REAL captured strip lights every chord (strip-rig composites of the +// app's own lit-key captures), three drawn macOS keycaps press per segment, +// and a finale walks I–vi–IV–V–I with the modifiers flipping live. +// +// Output: out/base-menuband-chords.mp4 + meta-menuband-chords.json — +// then `node pop/menuband/bin/chrome-reel.mjs menuband-chords`. +// +// Usage: node pop/menuband/bin/sim-chords.mjs (render-jingles.mjs first) + +import { readFileSync } from "node:fs"; +import { + W, H, FPS, OUT, INK, easeOut, clamp01, rgb, + makeStage, roundRect, text, drawDesktop, vignette, + makeParticles, loadStripRig, drawStrip, stripKeyX, stripKeyColor, + loadScore, leadOf, litAt, makeOnsets, + renderVideo, writeMeta, makeScenes, sungMode, loadSungWords, makeKaraoke, +} from "./reel-lib.mjs"; + +const SLUG = "menuband-chords"; +// --sung: jeffrey sings the chord grammar (sing-jingle.mjs) + karaoke. +const { sung: SUNG, suffix: VAR } = sungMode(); +const karaoke = SUNG ? makeKaraoke(loadSungWords(SLUG), { y: 1920 * 0.945 }) : null; +const score = loadScore(SLUG); +const { segs } = JSON.parse(readFileSync(`${OUT}/${SLUG}.score.json`, "utf8")); +const TOTAL = score.durationSec; +const lead = leadOf(score); +const onsetsBetween = makeOnsets(lead); + +const { canvas, ctx } = makeStage(); +const rig = await loadStripRig(); +const particles = makeParticles(ctx); + +const { scenes: SCENES } = makeScenes([ + { name: "intro", from: 0.00, to: 0.12, tint: [97, 158, 255] }, + { name: "grammar", from: 0.12, to: 0.63, tint: [255, 214, 56] }, + { name: "finale", from: 0.63, to: 0.90, tint: [163, 230, 53] }, + { name: "end", from: 0.90, to: 1.00, tint: [255, 77, 107] }, +], TOTAL); + +function segAt(t) { + const s = segs.find((x) => t >= x.t0 && t < x.t1); + if (s) return s; + return t >= segs.at(-1).t0 ? segs.at(-1) : null; +} + +// ── the strip: parked upper-third, always playing the chord ──────────────── +const HERO_W = W * 0.96, HERO_X = (W - HERO_W) / 2; +function heroRect(t) { + const h = HERO_W / rig.aspect; + const enter = easeOut(clamp01(t / 1.0)); + const bob = Math.sin(t * 1.6) * 8 * enter; + const rest = H * 0.26 - h / 2; + return { x: HERO_X, y: (-h - 40) + (rest - (-h - 40)) * enter + bob, w: HERO_W, h }; +} + +// ── drawn macOS keycaps: ⌘ ⌥ ⌃, pressing per segment ────────────────────── +// The lit-key olive from the real strip captures, so pressed caps match the +// keys they light. +const CAP_LIT = [176, 190, 60]; +const CAPS = [ + { id: "cmd", sym: "⌘", name: "command" }, + { id: "opt", sym: "⌥", name: "option" }, + { id: "ctl", sym: "⌃", name: "control" }, +]; +/// Per-cap press amount eased across segment changes. +const pressState = { cmd: 0, opt: 0, ctl: 0 }; +function drawKeycaps(t, dt) { + const s = segAt(t); + const active = new Set(s?.mods ?? []); + const size = 236, gap = 46; + const rowW = CAPS.length * size + (CAPS.length - 1) * gap; + const x0 = (W - rowW) / 2, y0 = H * 0.40; + for (const cap of CAPS) { + const target = active.has(cap.id) ? 1 : 0; + pressState[cap.id] += (target - pressState[cap.id]) * Math.min(1, dt * 16); + } + CAPS.forEach((cap, i) => { + const p = pressState[cap.id]; + const x = x0 + i * (size + gap), y = y0 + p * 10; + ctx.save(); + ctx.shadowColor = "rgba(0,0,0,0.30)"; + ctx.shadowBlur = 26 * (1 - p * 0.7); ctx.shadowOffsetY = 12 * (1 - p * 0.7); + roundRect(ctx, x, y, size, size, 42); + // white cap → lit olive as it presses (the strip's own lit-key color) + const mix = (a, b) => Math.round(a + (b - a) * p); + ctx.fillStyle = `rgb(${mix(250, CAP_LIT[0])},${mix(249, CAP_LIT[1])},${mix(253, CAP_LIT[2])})`; + ctx.fill(); + ctx.shadowColor = "transparent"; + ctx.lineWidth = 3; ctx.strokeStyle = "rgba(20,18,28,0.22)"; ctx.stroke(); + const ink = p > 0.5 ? "rgba(20,18,28,0.95)" : "rgba(20,18,28,0.85)"; + text(ctx, cap.sym, x + size / 2, y + size * 0.44, 110, ink, 500, "center", false); + text(ctx, cap.name, x + size / 2, y + size * 0.80, 34, ink, 600, "center", false); + ctx.restore(); + }); +} + +function drawFrame(t) { + drawDesktop(ctx); + const dt = 1 / FPS; + const s = segAt(t); + + const hero = heroRect(t); + const hRect = drawStrip(ctx, rig, litAt(lead, t, 0.3), hero.x, hero.y, hero.w); + for (const n of onsetsBetween(t - dt, t)) { + particles.spawnNote(stripKeyX(rig, n.midi, hRect), hRect.y - 6, stripKeyColor(rig, n.midi), false); + } + + // kicker up top, all reel long + const ka = easeOut(clamp01((t - 0.5) / 0.6)); + if (ka > 0) { + ctx.save(); ctx.globalAlpha = ka; + text(ctx, "chords live in your modifier keys", W / 2, H * 0.115, 54, INK, 800); + ctx.restore(); + } + + drawKeycaps(t, dt); + + // segment readouts: the chord name huge, the grammar line under it + if (s) { + const segLocal = clamp01((t - s.t0) / Math.max(0.001, s.t1 - s.t0)); + const a = easeOut(clamp01(segLocal * 6)); + ctx.save(); ctx.globalAlpha = a; + text(ctx, s.chordName, W / 2, H * 0.645, 132, INK, 800); + text(ctx, s.label, W / 2, H * 0.645 + 118, 54, "rgba(60,50,80,0.9)", 700); + ctx.restore(); + } + + // end tag over the ringing final chord + const endA = easeOut(clamp01((t - (segs.at(-1).t0 + 1.2)) / 0.8)); + if (endA > 0) { + ctx.save(); ctx.globalAlpha = endA; + text(ctx, "menuband.app", W / 2, H * 0.815, 88, INK, 800); + text(ctx, "free on the Mac App Store", W / 2, H * 0.815 + 88, 46, "rgba(60,50,80,0.9)", 700); + ctx.restore(); + } + + particles.stepAndDraw(dt); + vignette(ctx); + karaoke?.draw(ctx, t); +} + +await renderVideo({ + canvas, audioPath: `${OUT}/${SLUG}${VAR}.mp3`, outPath: `${OUT}/base-${SLUG}${VAR}.mp4`, + total: TOTAL, drawFrame, label: `menuband chords sim${VAR}`, +}); +writeMeta(`${SLUG}${VAR}`, TOTAL, SCENES); diff --git a/pop/menuband/bin/sim-features.mjs b/pop/menuband/bin/sim-features.mjs new file mode 100644 index 0000000000..169f9f3bbb --- /dev/null +++ b/pop/menuband/bin/sim-features.mjs @@ -0,0 +1,171 @@ +#!/usr/bin/env node +// sim-features.mjs — the Menu Band FEATURE TOUR reel base. +// +// Three acts over the tour groove (render-jingles.mjs): +// 1 · the REAL popover drops from the band's ♪ status glyph — the popover +// piano, instrument readout and full GM grid, straight out of the app +// (PopoverCapture), captions cycling the feature beats; +// 2 · the REAL fullscreen keymap slides up and cycles all five instrument +// families (the cached --render-keymap captures: piano → guitar → +// violin → trumpet → flute), LED scope + big piano + QWERTY map; +// 3 · end tag — the strip rains notes over "menuband.app · free on the +// Mac App Store". +// +// Output: out/base-menuband-features.mp4 + meta-menuband-features.json — +// then `node pop/menuband/bin/chrome-reel.mjs menuband-features`. +// +// Usage: node pop/menuband/bin/sim-features.mjs (render-jingles.mjs first) + +import { loadImage } from "canvas"; +import { + W, H, FPS, OUT, INK, easeOut, clamp01, + makeStage, roundRect, text, drawDesktop, vignette, drawIcon, + makeParticles, loadStripRig, drawStrip, stripKeyX, stripKeyColor, + loadScore, leadOf, litAt, makeOnsets, + renderVideo, writeMeta, makeScenes, sungMode, loadSungWords, makeKaraoke, +} from "./reel-lib.mjs"; + +const SLUG = "menuband-features"; +// --sung: jeffrey's sung vocal mix + karaoke captions (sing-jingle.mjs). +const { sung: SUNG, suffix: VAR } = sungMode(); +const karaoke = SUNG ? makeKaraoke(loadSungWords(SLUG)) : null; +const score = loadScore(SLUG); +const TOTAL = score.durationSec; +const BAR_SEC = score.barSec || 2.07; +const lead = leadOf(score); +const onsetsBetween = makeOnsets(lead); + +const { canvas, ctx } = makeStage(); +const rig = await loadStripRig(); +const popoverImg = await loadImage(`${OUT}/popover-frames/popover-0.png`); +const KEYMAP_PROGRAMS = [0, 24, 40, 56, 73]; // piano guitar violin trumpet flute +const keymapImgs = []; +for (const p of KEYMAP_PROGRAMS) keymapImgs.push(await loadImage(`${OUT}/about-frames/keymap-${p}.png`)); +const particles = makeParticles(ctx); + +const { scenes: SCENES, sceneAt } = makeScenes([ + { name: "popover", from: 0.00, to: 0.42, tint: [97, 158, 255] }, + { name: "keymap", from: 0.42, to: 0.85, tint: [163, 230, 53] }, + { name: "end", from: 0.85, to: 1.00, tint: [255, 77, 107] }, +], TOTAL); + +// The band parks at the very top for the whole reel (it's the anchor the +// popover hangs from) — it just falls in during the first bar. +const HERO_W = W * 0.96, HERO_X = (W - HERO_W) / 2, HERO_TOP = 54; +function heroRect(t) { + const h = HERO_W / rig.aspect; + const enter = easeOut(clamp01(t / 1.0)); + const bob = Math.sin((t / BAR_SEC) * Math.PI * 2) * 7 * enter; + return { x: HERO_X, y: (-h - 40) + (HERO_TOP - (-h - 40)) * enter + bob, w: HERO_W, h }; +} + +// ── the popover, hanging off the ♪ glyph like the real NSPopover ─────────── +// (arrow tip on the status item, panel below — app-store-real.mjs geometry). +const POP_CAPTIONS = [ + ["the popover piano", "click the ♪ in your menu bar"], + ["128 instruments", "the full General MIDI palette"], + ["type to play", "QWERTY is the keyboard · it sends real MIDI"], +]; +function drawPopover(t, local, hero) { + const e = easeOut(clamp01(local * 4.2)); + const ph = H * 0.545; + const pw = ph * popoverImg.width / popoverImg.height; + const anchorX = hero.x + hero.w * 0.965; // the ♪ glyph + const px = Math.min(W - pw - 24, Math.max(24, anchorX - pw / 2)); + const py = hero.y + hero.h + 16 + (1 - e) * (-40); + ctx.save(); ctx.globalAlpha = e; + // callout arrow up at the glyph + ctx.beginPath(); + ctx.moveTo(anchorX, py - 18); + ctx.lineTo(anchorX - 22, py + 2); + ctx.lineTo(anchorX + 22, py + 2); + ctx.closePath(); + ctx.fillStyle = "rgba(250,249,253,0.99)"; ctx.fill(); + ctx.shadowColor = "rgba(0,0,0,0.42)"; ctx.shadowBlur = 54; ctx.shadowOffsetY = 20; + roundRect(ctx, px, py, pw, ph, 30); ctx.fillStyle = "rgba(250,249,253,0.99)"; ctx.fill(); + ctx.shadowColor = "transparent"; + ctx.save(); roundRect(ctx, px, py, pw, ph, 30); ctx.clip(); + ctx.drawImage(popoverImg, px, py, pw, ph); + ctx.restore(); + ctx.restore(); + + // caption beats under the popover + const idx = Math.min(POP_CAPTIONS.length - 1, Math.floor(local * POP_CAPTIONS.length)); + const capLocal = local * POP_CAPTIONS.length - idx; + const a = easeOut(clamp01(capLocal * 5)) * (1 - easeOut(clamp01((capLocal - 0.82) / 0.18))); + if (a > 0 && e > 0.8) { + ctx.save(); ctx.globalAlpha = a; + text(ctx, POP_CAPTIONS[idx][0], W / 2, H * 0.80, 84, INK, 800); + text(ctx, POP_CAPTIONS[idx][1], W / 2, H * 0.80 + 84, 42, "rgba(60,50,80,0.88)", 600); + ctx.restore(); + } +} + +// ── the keymap act — the five family captures, crossfading ───────────────── +function drawKeymapAct(t, local) { + const e = easeOut(Math.min(1, local * 5)); + const per = 1 / KEYMAP_PROGRAMS.length; + const idx = Math.min(KEYMAP_PROGRAMS.length - 1, Math.floor(local / per)); + const segLocal = (local - idx * per) / per; + const fade = easeOut(clamp01(segLocal * 4)); // crossfade into idx + + const aw = W * 0.94; + const ah = aw * keymapImgs[0].height / keymapImgs[0].width; + const drift = 1 + 0.015 * Math.sin(local * Math.PI); // subtle breath + const dw = aw * drift, dh = ah * drift; + const ax = (W - dw) / 2, ay = H * 0.46 - dh / 2 + (H * 0.5) * (1 - e); + ctx.save(); ctx.globalAlpha = e; + ctx.shadowColor = "rgba(0,0,0,0.45)"; ctx.shadowBlur = 60; ctx.shadowOffsetY = 24; + roundRect(ctx, ax, ay, dw, dh, 26); ctx.fillStyle = "white"; ctx.fill(); + ctx.shadowColor = "transparent"; + ctx.save(); roundRect(ctx, ax, ay, dw, dh, 26); ctx.clip(); + if (fade < 1 && idx > 0) ctx.drawImage(keymapImgs[idx - 1], ax, ay, dw, dh); + ctx.globalAlpha = e * fade; + ctx.drawImage(keymapImgs[idx], ax, ay, dw, dh); + ctx.restore(); + ctx.restore(); + + if (e > 0.8) { + const NAMES = ["Piano", "Guitar", "Violin", "Trumpet", "Flute"]; + text(ctx, "it goes fullscreen", W / 2, H * 0.80, 84, INK, 800); + text(ctx, `LED waveform · big piano · QWERTY map · ${NAMES[idx]}`, W / 2, H * 0.80 + 84, 40, "rgba(60,50,80,0.88)", 600); + } +} + +function drawFrame(t) { + drawDesktop(ctx); + const sc = sceneAt(t); + const local = (sc.to - sc.from) > 0 ? (t - sc.from) / (sc.to - sc.from) : 1; + const dt = 1 / FPS; + + const hero = heroRect(t); + const hRect = drawStrip(ctx, rig, litAt(lead, t), hero.x, hero.y, hero.w); + for (const n of onsetsBetween(t - dt, t)) { + particles.spawnNote(stripKeyX(rig, n.midi, hRect), hRect.y + hRect.h + 6, stripKeyColor(rig, n.midi), true); + } + + if (sc.name === "popover") { + drawPopover(t, local, hero); + } else if (sc.name === "keymap") { + drawKeymapAct(t, local); + } else { + const e = easeOut(Math.min(1, local * 5)); + ctx.save(); ctx.globalAlpha = e; + const ipx = 300; + drawIcon(ctx, W / 2 - ipx / 2, H * 0.30, ipx, new Set(litAt(lead, t).map((mm) => ((mm % 12) + 12) % 12 % 5))); + text(ctx, "Menu Band", W / 2, H * 0.52, 110, INK, 800); + text(ctx, "free on the Mac App Store", W / 2, H * 0.52 + 104, 50, "rgba(60,50,80,0.9)", 700); + text(ctx, "menuband.app", W / 2, H * 0.52 + 184, 58, INK, 800); + ctx.restore(); + } + + particles.stepAndDraw(dt); + vignette(ctx); + karaoke?.draw(ctx, t); +} + +await renderVideo({ + canvas, audioPath: `${OUT}/${SLUG}${VAR}.mp3`, outPath: `${OUT}/base-${SLUG}${VAR}.mp4`, + total: TOTAL, drawFrame, label: `menuband features sim${VAR}`, +}); +writeMeta(`${SLUG}${VAR}`, TOTAL, SCENES); diff --git a/pop/menuband/bin/sing-jingle.mjs b/pop/menuband/bin/sing-jingle.mjs new file mode 100644 index 0000000000..4ef5cafdbc --- /dev/null +++ b/pop/menuband/bin/sing-jingle.mjs @@ -0,0 +1,729 @@ +#!/usr/bin/env node +// sing-jingle.mjs — jeffrey SINGS the Menu Band campaign jingles. +// +// v3 — the spinging engine round. The line-continuous WORLD chain of v2 now +// lives in spinging/lib (this script is its first caller), grounded in TEXT +// rather than spectra alone. Per line: +// +// 1 · TTS through /api/say (provider "jeffrey", stability 0.55 — identity; +// lines are cached, never re-spent) +// 2 · whisper-cli -ml 1 word boundaries, reconciled by +// spinging/lib/align-words.mjs (+ the presplit/rescale/repair guards) +// 3 · CHORAL NOTATION (spinging/lib/notation.mjs): every note gets its +// syllable underlay + {onset, nucleus, coda} phonemes from curated IPA +// (Wiktionary → espeak fallback, cached in spinging/cache). +// 4 · spinging/lib/sing_line_world.py — guided alignment (expected +// consonants anchor burst detection), per-line octave transposition +// minimizing |f0 shift| from the spoken take, harvest clamped to the +// real 60–300 Hz range (de-kermit), unstretched vowel onsets with +// natural-pitch glides, note-edge contour tapers (no pre-transition +// dip), monotone source maps (no stutter), goalpost-conformed drift/ +// vibrato, airy sustains + a quiet self-choir (angelic), and a +// percentile-conformance + click-scan QA gate — the line re-renders +// with adjusted tweaks until it sits inside the reference bands. +// 5 · lines placed at absolute time → out/-vocal.wav, then a soft +// reverb halo (spinging/lib/vocal_bus.py) on the vocal bus +// 6 · MASTERED mix: vocal bus (highpass → 3:1 compression → de-ess) ducks +// the jingle bed, then two-pass ffmpeg loudnorm to -14 LUFS / -1 dBTP +// → out/-sung.mp3 +// +// Also writes out/.words.sung.json (never touching any existing +// words.json) — the karaoke caption timings the --sung sims burn in. +// +// Run: node pop/menuband/bin/sing-jingle.mjs menuband-announce +// (or menuband-features / menuband-chords / all) +// --harmony 0.875 0 = fully spoken contour, 1 = perfect pitch lock + +import { spawnSync } from "node:child_process"; +import { readFileSync, writeFileSync, existsSync, mkdirSync } from "node:fs"; +import { resolve, dirname } from "node:path"; +import { fileURLToPath } from "node:url"; +import { createHash } from "node:crypto"; +import { alignWords } from "../../../spinging/lib/align-words.mjs"; +import { buildLineScore, writeLineScore } from "../../../spinging/lib/notation.mjs"; +import { sourceCounts } from "../../../spinging/lib/pronounce.mjs"; +import { decodeAudioMono } from "../../lib/preview-shared.mjs"; + +const HERE = dirname(fileURLToPath(import.meta.url)); +const LANE = resolve(HERE, ".."); +const OUT = `${LANE}/out`; +const POP = resolve(LANE, ".."); +const REPO = resolve(POP, ".."); +const SPINGING = `${REPO}/spinging`; +const VENV_PY = `${POP}/.venv/bin/python`; +const WORLD_HELPER = `${SPINGING}/lib/sing_line_world.py`; +const VOCAL_BUS = `${SPINGING}/lib/vocal_bus.py`; +const GOALPOSTS = `${SPINGING}/cache/goalposts.json`; +const WHISPER_MODEL = `${process.env.HOME}/.whisper-models/ggml-base.en.bin`; +const SAY_URL = "https://aesthetic.computer/api/say"; +const STABILITY = 0.55; // >= 0.5 keeps jeffrey's identity +const SR = 48_000; +// --harmony: contour-retention as a first-class knob (0 = spoken, 1 = lock). +// Round 3 defaults to a ~0.875 lock — "drift more into the perfect harmony". +const hIdx = process.argv.indexOf("--harmony"); +const HARMONY = hIdx > 0 ? parseFloat(process.argv[hIdx + 1]) : 0.875; +const QA_PASSES = 3; // re-render budget per line (percentile gate) + +// ── the lyrics ───────────────────────────────────────────────────────────── +// "lead" mode: words consume the jingle's lead notes in order — [word, nSlots] +// where nSlots is the word's syllable count. Totals must equal the lead lane. +// "explicit" mode (chords): every syllable is hand-placed on the triad — +// [word, [[t, dur, midi], …]] with midis already in the baritone register. +const LYRICS = { + "menuband-announce": { + mode: "lead", transpose: -24, + lines: [ + { tts: "Menu Band sings out!", + words: [["menu", 2], ["band", 1], ["sings", 1], ["out", 1]] }, + { tts: "Right up in your menu bar.", + words: [["right", 1], ["up", 1], ["in", 1], ["your", 1], ["menu", 2], ["bar", 1]] }, + { tts: "A synthesizer now.", + words: [["a", 1], ["synthesizer", 4], ["now", 1]] }, + { tts: "Every key plays a note.", + words: [["every", 2], ["key", 1], ["plays", 1], ["a", 1], ["note", 1]] }, + { tts: "It's out now on the Mac App Store.", + words: [["it's", 1], ["out", 1], ["now", 1], ["on", 1], ["the", 1], ["mac", 1], ["app", 1], ["store", 1]] }, + { tts: "Menu band dot app!", + words: [["menu", 2], ["band", 1], ["dot", 1], ["app", 1]] }, + ], + }, + "menuband-features": { + mode: "lead", transpose: -24, + lines: [ + { tts: "Click the note. Piano!", + words: [["click", 1], ["the", 1], ["note", 1], ["piano", 3]] }, + { tts: "A hundred twenty eight instruments, yeah!", + words: [["a", 1], ["hundred", 2], ["twenty", 2], ["eight", 1], ["instruments", 3], ["yeah", 1]] }, + { tts: "Type to play.", + words: [["type", 1], ["to", 1], ["play", 1]] }, + { tts: "Goes fullscreen.", + words: [["goes", 1], ["fullscreen", 2]] }, + { tts: "Your keys make music for free.", + words: [["your", 1], ["keys", 1], ["make", 1], ["music", 2], ["for", 1], ["free", 1]] }, + { tts: "The Mac App Store.", + words: [["the", 1], ["mac", 1], ["app", 1], ["store", 1]] }, + ], + }, + // Chord hits land at seg.t0 + 0.15, restrikes at seg.t0 + 1.45 + // (render-jingles.mjs chordHit) — phrase fronts sit on the hit, the + // quality word answers on the restrike, sung on the actual triad tones. + "menuband-chords": { + mode: "explicit", + lines: [ + { tts: "One key. One note.", + words: [["one", [[0.55, 0.5, 48]]], ["key", [[1.15, 0.5, 48]]], + ["one", [[1.85, 0.5, 48]]], ["note", [[2.35, 0.5, 48]]]] }, + { tts: "Command. Major.", + words: [["command", [[3.05, 0.4, 48], [3.45, 0.55, 52]]], + ["major", [[4.35, 0.45, 55], [4.8, 0.55, 52]]]] }, + { tts: "Option. Minor.", + words: [["option", [[5.55, 0.4, 45], [5.95, 0.55, 48]]], + ["minor", [[6.85, 0.45, 52], [7.3, 0.55, 48]]]] }, + { tts: "Command option. Sus two.", + words: [["command", [[8.05, 0.32, 48], [8.37, 0.33, 50]]], + ["option", [[8.7, 0.32, 55], [9.02, 0.33, 50]]], + ["sus", [[9.35, 0.45, 48]]], ["two", [[9.8, 0.55, 50]]]] }, + { tts: "Option control. Diminished.", + words: [["option", [[10.55, 0.32, 47], [10.87, 0.33, 50]]], + ["control", [[11.2, 0.32, 53], [11.52, 0.33, 50]]], + ["diminished", [[11.85, 0.3, 47], [12.15, 0.3, 50], [12.45, 0.45, 53]]]] }, + { tts: "All three keys. Sus four.", + words: [["all", [[13.05, 0.38, 48]]], ["three", [[13.43, 0.4, 53]]], + ["keys", [[13.83, 0.45, 55]]], + ["sus", [[14.35, 0.45, 53]]], ["four", [[14.8, 0.55, 48]]]] }, + { tts: "Every chord under three keys.", + words: [["every", [[15.4, 0.4, 48], [15.8, 0.4, 52]]], + ["chord", [[16.25, 0.7, 45]]], + ["under", [[17.1, 0.35, 53], [17.45, 0.35, 57]]], + ["three", [[17.95, 0.6, 55]]], + ["keys", [[18.8, 1.6, 48]]]] }, + ], + }, +}; + +// ── helpers ──────────────────────────────────────────────────────────────── +// Percentile-gate feedback: nudge the engine's tweak knobs toward the +// reference bands (see spinging/lib/vocal_shapes.py conformance()). +function adjustTweaks(tweaks, conf) { + if (!conf) return; + const miss = (k) => (conf[k] && conf[k].pass === false ? conf[k] : null); + let m; + if ((m = miss("plateau_drift_cents"))) { + if (m.value > m.hi) { tweaks.drift_scale *= 0.55; tweaks.beta_scale *= 0.75; } + else tweaks.drift_scale *= 1.6; + } + if ((m = miss("onset_glide_ms")) || (m = miss("onset_glide_cents"))) { + tweaks.glide_scale *= m.value > m.hi ? 0.7 : 1.35; + } + if ((m = miss("release_cents"))) { + if (m.value > m.hi) tweaks.beta_scale *= 0.8; + } + if ((m = miss("vib_depth_cents"))) { + tweaks.vib_depth_scale *= m.value > m.hi ? 0.6 : 1.5; + } + if ((m = miss("hf_ratio"))) { + tweaks.air_scale *= m.value > m.hi ? 0.6 : 1.5; + } +} + +function measureLoudnorm(file) { + const r = sh("ffmpeg", ["-y", "-i", file, "-af", + "loudnorm=I=-14:TP=-1.5:LRA=11:print_format=json", "-f", "null", "-"]); + const m = r.stderr.toString().match(/\{[^{}]*"input_i"[\s\S]*?\}/); + if (!m) throw new Error(`loudnorm measure pass printed no JSON for ${file}`); + return JSON.parse(m[0]); +} + +const sh = (cmd, args, opts = {}) => { + const r = spawnSync(cmd, args, { encoding: "utf8", maxBuffer: 256 * 1024 * 1024, ...opts }); + if (r.status !== 0) throw new Error(`${cmd} ${args[0]}: ${r.stderr?.toString().slice(0, 400)}`); + return r; +}; + +function writeWavF32(path, samples, sr = SR) { + const n = samples.length; + const buf = Buffer.alloc(44 + n * 4); + buf.write("RIFF", 0); buf.writeUInt32LE(36 + n * 4, 4); buf.write("WAVE", 8); + buf.write("fmt ", 12); buf.writeUInt32LE(16, 16); buf.writeUInt16LE(3, 20); // float + buf.writeUInt16LE(1, 22); buf.writeUInt32LE(sr, 24); buf.writeUInt32LE(sr * 4, 28); + buf.writeUInt16LE(4, 32); buf.writeUInt16LE(32, 34); + buf.write("data", 36); buf.writeUInt32LE(n * 4, 40); + for (let i = 0; i < n; i++) buf.writeFloatLE(samples[i], 44 + i * 4); + writeFileSync(path, buf); +} + +async function ttsLine(text, outFile, attempt = 0) { + if (existsSync(outFile)) return; + try { + await ttsLineOnce(text, outFile); + } catch (e) { + if (attempt >= 2) throw e; + console.log(` … /api/say hiccup (${String(e.message).slice(0, 60)}), retrying`); + await new Promise((r) => setTimeout(r, 2500 * (attempt + 1))); + await ttsLine(text, outFile, attempt + 1); + } +} + +async function ttsLineOnce(text, outFile) { + const res = await fetch(SAY_URL, { + method: "POST", + headers: { "Content-Type": "application/json", Origin: "https://aesthetic.computer" }, + body: JSON.stringify({ from: text, provider: "jeffrey", stability: STABILITY }), + redirect: "follow", + }); + if (!res.ok) throw new Error(`/api/say ${res.status}: ${(await res.text()).slice(0, 300)}`); + const ctype = res.headers.get("content-type") || ""; + let buf; + if (ctype.includes("application/json")) { + const json = await res.json(); + if (json.audio) buf = Buffer.from(json.audio, "base64"); + else if (json.url) buf = Buffer.from(await (await fetch(json.url)).arrayBuffer()); + else throw new Error(`/api/say JSON without audio`); + } else buf = Buffer.from(await res.arrayBuffer()); + if (!buf || buf.length < 256) throw new Error(`/api/say tiny body (${buf?.length})`); + writeFileSync(outFile, buf); +} + +// whisper.cpp -ml 1 tokens → words (the leading-space merge — yc-ref/lib/words.mjs) +function wordsFromWhisper(jsonPath) { + const raw = JSON.parse(readFileSync(jsonPath, "utf8")); + const words = []; + for (const seg of raw.transcription) { + const piece = seg.text; + if (!piece || !piece.trim()) continue; + const text = piece.trim(); + if (piece.startsWith(" ") || words.length === 0) { + words.push({ text, fromMs: seg.offsets.from, toMs: seg.offsets.to }); + } else { + const prev = words[words.length - 1]; + prev.text += text; + prev.toMs = seg.offsets.to; + } + } + const merged = []; + for (const w of words) { + if (/^[^\w'"“”‘’(]+$/.test(w.text) && merged.length > 0) { + const prev = merged[merged.length - 1]; + prev.text += w.text; + prev.toMs = Math.max(prev.toMs, w.toMs); + } else merged.push(w); + } + return merged; +} + +const norm = (s) => (s || "").toLowerCase().replace(/[^a-z']/g, ""); + +// Whisper writes spoken numbers as digits ("a hundred twenty eight" → "128"), +// which norms to nothing and derails alignment. Expand a digit word back into +// its spoken words, splitting the window char-proportionally. +function numberToWords(n) { + const ones = ["zero", "one", "two", "three", "four", "five", "six", "seven", "eight", "nine", + "ten", "eleven", "twelve", "thirteen", "fourteen", "fifteen", "sixteen", "seventeen", + "eighteen", "nineteen"]; + const tens = ["", "", "twenty", "thirty", "forty", "fifty", "sixty", "seventy", "eighty", "ninety"]; + if (n < 20) return [ones[n]]; + if (n < 100) return n % 10 ? [tens[Math.floor(n / 10)], ones[n % 10]] : [tens[Math.floor(n / 10)]]; + if (n < 1000) { + const rest = n % 100 ? numberToWords(n % 100) : []; + return [ones[Math.floor(n / 100)], "hundred", ...rest]; + } + return [String(n)]; +} +function expandDigitWords(words) { + const out = []; + for (const w of words) { + const digits = w.text.replace(/[^\d]/g, ""); + if (!digits || !/^\W*\d[\d,]*\W*$/.test(w.text)) { out.push(w); continue; } + const parts = numberToWords(parseInt(digits, 10)); + const chars = parts.reduce((s, p) => s + p.length, 0); + let t0 = w.fromMs; + const span = w.toMs - w.fromMs; + for (const p of parts) { + const dw = (span * p.length) / chars; + out.push({ text: p, fromMs: Math.round(t0), toMs: Math.round(t0 + dw) }); + t0 += dw; + } + } + return out; +} +function editDist(a, b) { + const m = a.length, n = b.length; + let prev = Array.from({ length: n + 1 }, (_, j) => j); + for (let i = 1; i <= m; i++) { + const curr = [i]; + for (let j = 1; j <= n; j++) { + curr[j] = a[i - 1] === b[j - 1] ? prev[j - 1] : 1 + Math.min(prev[j - 1], prev[j], curr[j - 1]); + } + prev = curr; + } + return prev[n]; +} + +// Whisper loves to weld words together ("menu band dot app" → "menuband.app.", +// "a synthesizer" → "synthesizer") — align-words' midpoint split then hands a +// tiny function word HALF the audio. Pre-split any heard word whose norm is +// (fuzzily) the concatenation of the next 2-4 score words, char-proportionally, +// so every score word owns roughly its own audio before alignment runs. +function presplitHeard(scoreWords, heard) { + const out = []; + let i = 0; + for (const h of heard) { + const hn = norm(h.text); + // best k = the join of score words that lands CLOSEST to this heard word — + // and a merge only wins if it beats the single next score word outright + // (otherwise "synthesizer" would swallow the "now" after it). + const singleDist = i < scoreWords.length ? editDist(norm(scoreWords[i]), hn) : Infinity; + let best = 0, bestDist = singleDist; + for (let k = 2; k <= 4 && i + k <= scoreWords.length; k++) { + const cat = scoreWords.slice(i, i + k).map(norm).join(""); + const d = editDist(cat, hn); + // ties prefer the LONGER join — "menubanddot" and "menubanddotapp" are + // equidistant from a heard "menubandapp.", and only the full join + // leaves no score word stranded past the end of the audio. + if (d <= Math.max(1, Math.ceil(cat.length / 4)) && d <= bestDist && d < singleDist) { + best = k; bestDist = d; + } + } + if (best >= 2) { + const parts = scoreWords.slice(i, i + best).map(norm); + const chars = parts.reduce((s, p) => s + Math.max(1, p.length), 0); + let t0 = h.fromMs; + const span = h.toMs - h.fromMs; + for (const p of parts) { + const w = (span * Math.max(1, p.length)) / chars; + out.push({ text: p, fromMs: Math.round(t0), toMs: Math.round(t0 + w) }); + t0 += w; + } + i += best; + } else { + out.push(h); + if (i < scoreWords.length) { + const sn = norm(scoreWords[i]); + if (sn === hn || editDist(sn, hn) <= 2 || (sn.length >= 3 && hn.startsWith(sn))) i++; + } + } + } + return out; +} + +// Whisper stamps the words at a clip's hard end zero-width and buries their +// audio inside the PREVIOUS window ("Type" 110-1020 actually holds "type to +// play"). Clamp every window into the audio, then repair each RUN of +// degenerate windows by re-splitting the host span (the last good window +// through the run's end) at its quietest energy dips — the gaps between the +// words whisper welded together. +function energySplit(audio, aMs, bMs, n) { + const hop = Math.floor(SR / 100); // 10ms + const a = Math.max(0, Math.floor((aMs / 1000) * SR)); + const b = Math.min(audio.length, Math.floor((bMs / 1000) * SR)); + let env = []; + for (let s = a; s + hop <= b; s += hop) { + let e = 0; + for (let k = s; k < s + hop; k++) e += audio[k] * audio[k]; + env.push(Math.sqrt(e / hop)); + } + // Trim the span to actual SPEECH: a whisper window welded to the file end + // drags a silent tail in, and dips picked there would hand words silence. + const thr = Math.max(0.008, 0.15 * Math.max(...env, 0)); + let f = env.findIndex((e) => e >= thr); + let l = env.length - 1; + while (l > 0 && env[l] < thr) l--; + if (f > 0 || l < env.length - 1) { + f = Math.max(0, f - 2); l = Math.min(env.length - 1, l + 4); + aMs = aMs + f * 10; bMs = aMs + (l - f + 1) * 10; + env = env.slice(f, l + 1); + } + const bounds = [Math.round(aMs)]; + if (n > 1) { + const lo = Math.floor(env.length * 0.12), hi = Math.ceil(env.length * 0.94); + const minSep = Math.max(3, Math.floor(env.length / (n * 2))); + const cand = []; + for (let k = lo; k < hi; k++) cand.push([env[k], k]); + cand.sort((x, y) => x[0] - y[0]); + const picks = []; + for (const [, k] of cand) { + if (picks.length >= n - 1) break; + if (picks.every((p) => Math.abs(p - k) >= minSep)) picks.push(k); + } + while (picks.length < n - 1) picks.push(Math.floor(((picks.length + 1) / n) * env.length)); + picks.sort((x, y) => x - y); + for (const p of picks) bounds.push(Math.round(aMs + p * 10)); + } + bounds.push(Math.round(bMs)); + return bounds; +} + +// On short exclamation lines whisper sometimes smears the whole timestamp +// axis past the real speech ("sings out" placed in post-speech silence). If +// even the PENULTIMATE word ends after the audible speech does, the axis is +// stretched — linearly rescale every window onto the real speech span. +function rescaleHeard(heard, audio, lineLenMs) { + if (heard.length < 2) return heard; + const hop = Math.floor(SR / 100); + const env = []; + for (let s = 0; s + hop <= audio.length; s += hop) { + let e = 0; + for (let k = s; k < s + hop; k++) e += audio[k] * audio[k]; + env.push(Math.sqrt(e / hop)); + } + const thr = Math.max(0.008, 0.15 * Math.max(...env, 0)); + let f = env.findIndex((e) => e >= thr); + let l = env.length - 1; + while (l > 0 && env[l] < thr) l--; + const speechStart = Math.max(0, f * 10 - 20); + const speechEnd = Math.min(lineLenMs, (l + 1) * 10 + 40); + if (heard[heard.length - 2].toMs <= speechEnd + 120) return heard; + const wStart = heard[0].fromMs; + const wEnd = Math.max(...heard.map((h) => h.toMs)); + const scale = (speechEnd - speechStart) / Math.max(1, wEnd - wStart); + for (const h of heard) { + h.fromMs = Math.round(speechStart + (h.fromMs - wStart) * scale); + h.toMs = Math.round(speechStart + (h.toMs - wStart) * scale); + } + return heard; +} + +function repairWindows(windows, lineLenMs, audio) { + for (const w of windows) { + w.toMs = Math.min(w.toMs, lineLenMs); + w.fromMs = Math.min(w.fromMs, Math.max(0, w.toMs - 40)); + } + let i = 0; + while (i < windows.length) { + if (windows[i].toMs - windows[i].fromMs >= 60) { i++; continue; } + let j = i; + while (j < windows.length && windows[j].toMs - windows[j].fromMs < 60) j++; + const host = i > 0 ? windows[i - 1] : null; + const hostStart = host ? host.fromMs : 0; + const hostEnd = Math.min(lineLenMs, Math.max(windows[j - 1].toMs, host ? host.toMs : 0)); + const parts = (host ? 1 : 0) + (j - i); + const bounds = energySplit(audio, hostStart, hostEnd, parts); + let k = 0; + if (host) { host.fromMs = bounds[0]; host.toMs = bounds[1]; k = 1; } + for (let w = i; w < j; w++, k++) { windows[w].fromMs = bounds[k]; windows[w].toMs = bounds[k + 1]; } + i = j; + } + return windows; +} + +// ── build absolute slots per word from the spec ──────────────────────────── +function buildLines(slug) { + const spec = LYRICS[slug]; + if (!spec) throw new Error(`no lyrics for ${slug}`); + const score = JSON.parse(readFileSync(`${OUT}/${slug}.notes.json`, "utf8")); + const lines = []; + if (spec.mode === "lead") { + const lead = (score.notes || []).filter((n) => n.lane === "lead").sort((a, b) => a.t - b.t); + const need = spec.lines.reduce((s, l) => s + l.words.reduce((a, [, n]) => a + n, 0), 0); + if (need !== lead.length) throw new Error(`${slug}: lyric syllables ${need} != lead notes ${lead.length}`); + let k = 0; + for (const line of spec.lines) { + const words = line.words.map(([w, n]) => { + const slots = lead.slice(k, k + n).map((s) => ({ t: s.t, dur: s.dur, midi: s.midi + spec.transpose })); + k += n; + return { w, slots }; + }); + lines.push({ tts: line.tts, words }); + } + } else { + for (const line of spec.lines) { + lines.push({ + tts: line.tts, + words: line.words.map(([w, slots]) => ({ w, slots: slots.map(([t, dur, midi]) => ({ t, dur, midi })) })), + }); + } + } + return { lines, durationSec: score.durationSec }; +} + +// ── main ─────────────────────────────────────────────────────────────────── +async function singOne(slug) { + console.log(`\n▸ ${slug} — jeffrey sings (line-continuous WORLD)`); + const { lines, durationSec } = buildLines(slug); + const dir = `${OUT}/sung/${slug}`; + mkdirSync(`${dir}/words`, { recursive: true }); + if (!existsSync(WHISPER_MODEL)) throw new Error(`whisper model missing: ${WHISPER_MODEL}`); + if (!existsSync(VENV_PY)) throw new Error(`pop venv missing: ${VENV_PY}`); + + const master = new Float32Array(Math.ceil(durationSec * SR)); + const sungWords = []; + const report = []; + const qaLines = []; + if (!existsSync(GOALPOSTS)) { + throw new Error(`goalposts missing: ${GOALPOSTS} — build with spinging goalposts`); + } + + // flatten words across lines for next-word lookahead + const flat = []; + lines.forEach((line, li) => line.words.forEach((w) => flat.push({ ...w, li }))); + + for (let li = 0; li < lines.length; li++) { + const line = lines[li]; + const hash = createHash("sha1").update(line.tts).digest("hex").slice(0, 8); + const mp3 = `${dir}/line-${li}-${hash}.mp3`; + await ttsLine(line.tts, mp3); + + // 16k mono for whisper (unpadded — trailing silence makes whisper smear + // word timestamps into it; repairWindows handles the zero-width final + // word it stamps at a hard file end), 48k mono for the WORLD engine + const w16 = mp3.replace(/\.mp3$/, "-16k.wav"); + if (!existsSync(w16)) { + sh("ffmpeg", ["-y", "-v", "error", "-i", mp3, "-ac", "1", "-ar", "16000", w16]); + } + const w48 = mp3.replace(/\.mp3$/, "-48k.wav"); + if (!existsSync(w48)) { + sh("ffmpeg", ["-y", "-v", "error", "-i", mp3, "-ac", "1", "-ar", String(SR), w48]); + } + const wj = mp3.replace(/\.mp3$/, "-words"); + if (!existsSync(`${wj}.json`)) { + sh("whisper-cli", ["-m", WHISPER_MODEL, "-f", w16, "-ml", "1", "-oj", "-ojf", "-of", wj], + { stdio: ["ignore", "ignore", "pipe"] }); + } + const { audio: lineAudio } = decodeAudioMono(mp3, SR); + const lineLen = lineAudio.length / SR; + const mapWords = line.words.map((w) => w.w); + const heard = presplitHeard(mapWords, + rescaleHeard(expandDigitWords(wordsFromWhisper(`${wj}.json`)), lineAudio, lineLen * 1000)); + const windows = repairWindows(alignWords(mapWords, heard), lineLen * 1000, lineAudio); + + console.log(` line ${li}: "${line.tts}" · whisper heard "${heard.map((h) => h.text).join(" ")}"`); + + // ── plan the line for sing_line_world.py ─────────────────────────────── + const planWords = []; + for (let wi = 0; wi < line.words.length; wi++) { + const word = line.words[wi]; + const win = windows[wi]; + const slots = word.slots; + const tStart = slots[0].t; + const last = slots[slots.length - 1]; + const globalIdx = lines.slice(0, li).reduce((s, l) => s + l.words.length, 0) + wi; + const next = flat[globalIdx + 1]; + let tEnd = last.t + Math.min(last.dur, 1.8); + // stop before the next word — a bigger reserve across lines, where the + // next line's onset cluster (which python can't see) needs room + if (next) tEnd = Math.min(tEnd, next.slots[0].t - (next.li === li ? 0.01 : 0.12)); + if (tEnd <= tStart + 0.1) tEnd = tStart + 0.1; + + // padded source window — consonants live just outside whisper's window, + // clamped to the neighbouring words' midpoints + const prevWin = wi > 0 ? windows[wi - 1] : null; + const nextWin = wi + 1 < windows.length ? windows[wi + 1] : null; + let s0 = win.fromMs - 60; + let s1 = win.toMs + 100; + if (prevWin) s0 = Math.max(s0, (prevWin.toMs + win.fromMs) / 2); + if (nextWin) s1 = Math.min(s1, (win.toMs + nextWin.fromMs) / 2 + 20); + s0 = Math.max(0, s0); s1 = Math.min(lineLen * 1000, s1); + + // a phrase starts at the line top or after an audible source gap + const phraseStart = wi === 0 || (prevWin && win.fromMs - prevWin.toMs > 150); + + planWords.push({ + w: word.w, wordIndex: wi, srcFromMs: Math.round(s0), srcToMs: Math.round(s1), + slots, hardEnd: +tEnd.toFixed(4), phraseStart: !!phraseStart, + }); + sungWords.push({ + text: word.w, fromMs: Math.round(tStart * 1000), toMs: Math.round(tEnd * 1000), line: li, + }); + } + + // ── choral notation sidecar (part A2) — the aligner + synthesizer both + // consume this; phonemes come from curated IPA, not spectra. + const score = await buildLineScore({ + text: line.tts, + words: planWords.map((p) => ({ w: p.w, slots: p.slots, phraseStart: p.phraseStart })), + }); + const scorePath = `${dir}/words/line-${li}-score.json`; + writeLineScore(scorePath, score); + + const lineT0 = Math.max(0, planWords[0].slots[0].t - 0.35); + const lineT1 = Math.min(durationSec, planWords[planWords.length - 1].hardEnd + 0.4); + const outWav = `${dir}/words/line-${li}-sung.wav`; + const planPath = `${dir}/words/line-${li}-plan.json`; + + // ── render + percentile-gate: iterate until the line sits inside the + // reference bands (or the pass budget runs out) ──────────────────────── + const tweaks = { drift_scale: 1, glide_scale: 1, vib_depth_scale: 1, beta_scale: 1, air_scale: 1 }; + let stats = {}; + let passes = 0; + for (let pass = 1; pass <= QA_PASSES; pass++) { + passes = pass; + const plan = { + line_wav: w48, + out_wav: outWav, + lead_wav: `${dir}/words/line-${li}-lead.wav`, + phoneme_sidecar: mp3.replace(/\.mp3$/, ".phonemes.json"), + score: scorePath, goalposts: GOALPOSTS, + line_t0: +lineT0.toFixed(4), line_t1: +lineT1.toFixed(4), + harmony: HARMONY, seed: 7 + li, + f0_floor: 60, f0_ceil: 300, // jeffrey's real range — de-kermit + octave_opt: true, choir: true, + tweaks, + words: planWords, + }; + writeFileSync(planPath, JSON.stringify(plan, null, 1)); + const wr = sh(VENV_PY, [WORLD_HELPER, planPath]); + try { stats = JSON.parse(wr.stdout.trim().split("\n").pop()); } catch {} + if (stats.error) break; + const clean = stats.clicks && stats.clicks.clicks === 0 && stats.clicks.flux_spikes === 0; + if ((!stats.conformance || stats.conformance._pass) && clean) break; + if (pass === QA_PASSES) break; + adjustTweaks(tweaks, stats.conformance); + console.log(` ↻ pass ${pass}: out of band — retweak ` + + Object.entries(tweaks).map(([k, v]) => `${k}=${v.toFixed(2)}`).join(" ")); + } + if (stats.error) { + report.push({ slug, line: li, word: "(line)", note: stats.error }); + continue; + } + for (const w of stats.words || []) { + report.push({ + slug, line: li, word: w.word, target: w.targets.join(","), + detected: w.detected_midi ?? "—", shift: w.shift_st ?? "—", + onsetMs: w.onset_ms, codaMs: w.coda_ms, + note: w.sung ? undefined : "no voiced nuclei — sung unpitched", + }); + } + const confBits = stats.conformance + ? Object.entries(stats.conformance) + .filter(([k, v]) => !k.startsWith("_") && v && v.pass === false) + .map(([k, v]) => `${k}=${v.value}∉[${v.lo},${v.hi}]`) + : []; + report.push({ + slug, line: li, word: "(qa)", info: + `f0 jumps max ${stats.f0_jump_max_cents}¢ p95 ${stats.f0_jump_p95_cents}¢ · ` + + `oct ${stats.line_transpose >= 0 ? "+" : ""}${stats.line_transpose} · β ${stats.beta} · ` + + `passes ${passes} · conf ${stats.conformance ? (stats.conformance._pass ? "PASS" : "miss: " + confBits.join(" ")) : "n/a"} · ` + + `clicks ${stats.clicks?.clicks ?? "?"}/${stats.clicks?.flux_spikes ?? "?"}`, + }); + qaLines.push({ + line: li, text: line.tts, passes, tweaks, + lineTranspose: stats.line_transpose, beta: stats.beta, harmony: HARMONY, + f0JumpMaxCents: stats.f0_jump_max_cents, f0JumpP95Cents: stats.f0_jump_p95_cents, + conformance: stats.conformance, clicks: stats.clicks, + }); + + // place the whole line at its absolute time — one continuous render, + // no per-word fades to smear + const { audio: sung } = decodeAudioMono(outWav, SR); + const at = Math.floor(lineT0 * SR); + for (let i = 0; i < sung.length && at + i < master.length; i++) master[at + i] += sung[i]; + } + + // normalize the vocal, write it, then MASTER the mix + let peak = 0; + for (let i = 0; i < master.length; i++) peak = Math.max(peak, Math.abs(master[i])); + if (peak > 0) for (let i = 0; i < master.length; i++) master[i] *= 0.85 / peak; + const vocalWav = `${OUT}/${slug}-vocal.wav`; + writeWavF32(vocalWav, master); + + // soft reverb halo on the vocal bus (angelic) — quiet wet, short decay + const vocalWet = `${OUT}/${slug}-vocal-wet.wav`; + sh(VENV_PY, [VOCAL_BUS, "reverb", vocalWav, vocalWet, "-16", "1.1"]); + + const bed = `${OUT}/${slug}.mp3`; + const mix = `${OUT}/${slug}-sung.mp3`; + const bedDur = parseFloat(sh("ffprobe", ["-v", "error", "-show_entries", "format=duration", + "-of", "csv=p=0", bed]).stdout.trim()); + // Vocal bus: highpass → gentle 3:1 compression (slow-ish release) → + // de-harsh, then the bed ducks under it. apad+atrim pin the mix to the + // bed's exact length — the sims mux with -shortest, and a mix even a + // frame short would truncate the video. + const premaster = `${OUT}/${slug}-sung-premaster.wav`; + sh("ffmpeg", ["-y", "-v", "error", "-i", bed, "-i", vocalWet, "-filter_complex", + "[1:a]aformat=sample_rates=48000:channel_layouts=stereo,highpass=f=70," + + "acompressor=threshold=0.125:ratio=3:attack=12:release=250:makeup=2," + + "deesser=i=0.4,asplit=2[sc][v];" + + "[0:a]aformat=sample_rates=48000:channel_layouts=stereo[b];" + + "[b][sc]sidechaincompress=threshold=0.05:ratio=5:attack=12:release=220[duck];" + + "[duck][v]amix=inputs=2:duration=first:normalize=0:weights=1 1.25[m];" + + "[m]highpass=f=30,alimiter=limit=0.89:level=false,apad=pad_dur=2," + + `atrim=0:${bedDur.toFixed(4)},asetpts=PTS-STARTPTS[out]`, + "-map", "[out]", premaster]); + + // hard glitch gate on the premaster (waveform + spectral-flux click scan) + const mixScan = JSON.parse( + sh(VENV_PY, [VOCAL_BUS, "scan", premaster]).stdout.trim().split("\n").pop()); + if (mixScan.clicks > 0) { + console.log(` ⚠ click scan flagged ${mixScan.clicks} clicks at ${mixScan.positions_s}`); + } + + // MASTER: two-pass loudnorm to -14 LUFS integrated / -1 dBTP + const mjson = measureLoudnorm(premaster); + sh("ffmpeg", ["-y", "-v", "error", "-i", premaster, "-af", + `loudnorm=I=-14:TP=-1.5:LRA=11:measured_I=${mjson.input_i}:measured_TP=${mjson.input_tp}` + + `:measured_LRA=${mjson.input_lra}:measured_thresh=${mjson.input_thresh}` + + `:offset=${mjson.target_offset}:linear=true`, + "-ar", "48000", "-c:a", "libmp3lame", "-q:a", "2", mix]); + const verify = measureLoudnorm(mix); // what actually landed on the mp3 + console.log(` mastered: ${mjson.input_i} LUFS / ${mjson.input_tp} dBTP → ` + + `${verify.input_i} LUFS / ${verify.input_tp} dBTP on disk`); + + writeFileSync(`${OUT}/${slug}.words.sung.json`, JSON.stringify(sungWords, null, 2)); + writeFileSync(`${OUT}/${slug}-sung-qa.json`, JSON.stringify({ + slug, harmony: HARMONY, engine: "spinging/lib/sing_line_world.py (round 3)", + goalposts: GOALPOSTS, + pronunciationSources: { ...sourceCounts }, + lines: qaLines, + mixClickScan: mixScan, + mastered: { lufs: parseFloat(verify.input_i), truePeakDb: parseFloat(verify.input_tp) }, + }, null, 1)); + console.log(`✓ ${mix}`); + console.log(`✓ ${OUT}/${slug}.words.sung.json · ${sungWords.length} words`); + console.log(`✓ ${OUT}/${slug}-sung-qa.json`); + return report; +} + +const positional = process.argv.slice(2).filter((a, i, all) => + !a.startsWith("--") && all[i - 1] !== "--harmony"); +const arg = positional[0] || "all"; +const slugs = arg === "all" ? Object.keys(LYRICS) : [arg]; +console.log(`harmony lock ${HARMONY} (β = ${(1 - HARMONY).toFixed(3)})`); +const allReports = []; +for (const slug of slugs) allReports.push(...(await singOne(slug))); +console.log(`\npronunciations: ${JSON.stringify(sourceCounts)}`); +console.log("\nword map (target ← detected, shift in st, onset/coda pickups in ms):"); +for (const r of allReports) { + if (r.info) { console.log(` [${r.slug} L${r.line}] ${r.info}`); continue; } + if (r.note) console.log(` ⚠ [${r.slug} L${r.line}] "${r.word}" — ${r.note}`); + else console.log(` [${r.slug} L${r.line}] "${r.word}" → ${r.target} (det ${r.detected}, shift ${r.shift}, on ${r.onsetMs}ms, coda ${r.codaMs}ms)`); +} diff --git a/pop/menuband/bin/sing_line_world.py b/pop/menuband/bin/sing_line_world.py new file mode 100644 index 0000000000..37f4d5d88b --- /dev/null +++ b/pop/menuband/bin/sing_line_world.py @@ -0,0 +1,531 @@ +#!/usr/bin/env python3 +""" +sing_line_world.py — phoneme-aware, LINE-CONTINUOUS singing for the Menu Band +jingles. Successor to sing_word_world.py (which pitched one word at a time and +left the joins to audio-domain concatenation — the choppiness jeffrey heard). + +jeffrey's spec, verbatim intent: "map the plosives / consonants and vowels — +align and lengthen — master." So per lyric line: + + 1 · ONE WORLD analysis of the whole TTS take (harvest → stonemask → + cheaptrick → d4c, 5ms frames). + 2 · Phoneme segmentation (heuristic stack — no aligner in pop/.venv): + voiced mask + energy envelope + spectral-flux burst detection for + plosives, low/high-band balance for fricatives vs sonorants, and a + vowelness score (voiced · energy · low-band dominance) whose peaks are + the vowel NUCLEI. Cached as a *.phonemes.json sidecar per line. + 3 · The note lives on the vowel. Consonant clusters keep their NATURAL + duration: onsets are pickups ending where the vowel starts ON the note + time; codas attach at the vowel's end, stealing from the note tail. + Only the vowel nucleus is lengthened — in the WORLD frame domain + (fractional sp/ap frame interpolation, transitions kept at natural + rate, slight smooth jitter so held frames don't buzz). + 4 · ONE continuous synthesis per line: a single target-f0 curve — plateaus + on vowels, ~40-80ms log-space portamento through voiced joins, f0=0 on + unvoiced frames, onset scoop at phrase starts, ±12-cent slow drift, + vibrato only on holds ≥ 0.7s — and jeffrey's own micro-contour kept at + --beta. Unvoiced consonant runs are composited back from the ORIGINAL + take (they were never warped, so the samples line up 1:1). + +Driven by sing-jingle.mjs with a JSON plan: + + sing_line_world.py plan.json + +plan.json = { + "line_wav": …48k mono wav of the TTS take…, + "out_wav": …, "phoneme_sidecar": …, + "line_t0": …, "line_t1": …, # absolute reel-clock span to render + "beta": 0.25, "seed": 7, + "words": [ { "w": "menu", "srcFromMs": …, "srcToMs": …, + "slots": [{"t":…,"dur":…,"midi":…},…], # absolute times + "hardEnd": …, "phraseStart": true }, … ] +} + +Prints one JSON line of stats (per-word detected midi/shift, f0-continuity +metrics) for the caller's report. +""" +import json +import sys + +import numpy as np +import soundfile as sf +import pyworld as pw + +FRAME_S = 0.005 # WORLD frame period (5ms) +F0_FLOOR = 65.0 +F0_CEIL = 500.0 +MAX_ONSET_S = 0.22 # consonant pickup cap +MAX_CODA_S = 0.26 +MAX_BREATH_S = 0.45 +GLIDE_SIGMA_S = 0.022 # gaussian smoothing of the target curve → ~40-80ms glides +SCOOP_S = 0.06 # phrase-onset scoop length +SCOOP_ST = 1.0 # …starting a semitone below +DRIFT_CENTS = 12.0 # slow detune drift depth +VIB_HOLD_S = 0.7 +VIB_HZ = 5.0 +VIB_CENTS = 30.0 +VIB_ONSET_S = 0.30 + + +def midi_to_hz(m): + return 440.0 * (2.0 ** ((np.asarray(m, dtype=np.float64) - 69.0) / 12.0)) + + +def hz_to_midi(hz): + return 69.0 + 12.0 * np.log2(hz / 440.0) + + +# ── analysis ─────────────────────────────────────────────────────────────── + +def analyze(x, fs): + f0_raw, t = pw.harvest(x, fs, f0_floor=F0_FLOOR, f0_ceil=F0_CEIL, + frame_period=FRAME_S * 1000.0) + f0 = pw.stonemask(x, f0_raw, t, fs) + fft_size = pw.get_cheaptrick_fft_size(fs, f0_floor=F0_FLOOR) + sp = pw.cheaptrick(x, f0, t, fs, fft_size=fft_size, f0_floor=F0_FLOOR) + ap = pw.d4c(x, f0, t, fs, fft_size=fft_size) + return f0, t, sp, ap, fft_size + + +def frame_features(x, fs, f0, sp): + """Per-frame: rms energy, spectral flux, low-band dominance, hf ratio.""" + n = len(f0) + hop = int(round(fs * FRAME_S)) + win = hop * 2 + rms = np.zeros(n) + for i in range(n): + a = max(0, i * hop - win // 2) + b = min(len(x), i * hop + win // 2) + if b > a: + rms[i] = np.sqrt(np.mean(x[a:b] ** 2)) + # band edges on the sp bins + nbins = sp.shape[1] + nyq = fs / 2.0 + k1 = int(nbins * 1000.0 / nyq) # < 1 kHz — vowel/sonorant power + k4 = int(nbins * 4000.0 / nyq) # > 4 kHz — sibilance / bursts + tot = sp.sum(axis=1) + 1e-12 + low_dom = sp[:, :k1].sum(axis=1) / tot + hf_ratio = sp[:, k4:].sum(axis=1) / tot + # spectral flux on log-sp (positive changes only) + lsp = np.log(sp + 1e-12) + d = np.diff(lsp, axis=0) + flux = np.zeros(n) + flux[1:] = np.maximum(d, 0).mean(axis=1) + return rms, flux, low_dom, hf_ratio + + +def classify_frames(f0, rms, flux, low_dom, hf_ratio): + """Coarse per-frame phone class: sil / plosive / fric / vcons / vowel.""" + n = len(f0) + voiced = f0 > 0 + floor = max(1e-4, 0.05 * np.percentile(rms, 95)) + cls = np.array(["sil"] * n, dtype=object) + speech = rms > floor + # plosive bursts: strong flux spike with a quiet 30ms run just before + flux_thr = np.percentile(flux[speech], 80) if speech.any() else np.inf + for i in range(2, n): + if flux[i] > flux_thr and rms[max(0, i - 7):max(1, i - 1)].mean() < 2.5 * floor: + cls[i] = "plosive" + for i in range(n): + if cls[i] == "plosive": + continue + if not speech[i]: + continue + if not voiced[i]: + cls[i] = "fric" if hf_ratio[i] > 0.10 else "plosive" + else: + cls[i] = "vowel" if low_dom[i] > 0.35 else "vcons" + return cls, voiced, floor + + +def find_nuclei(w0, w1, n_slots, voiced, rms, low_dom): + """Vowel nuclei inside word frames [w0,w1) — one per note slot.""" + seg = slice(w0, w1) + v = np.where(voiced[seg], rms[seg] * (0.25 + low_dom[seg]), 0.0) + if v.max() <= 0: + # completely unvoiced take — spread slots evenly, caller flags it + step = max(1, (w1 - w0) // max(1, n_slots)) + return [(w0 + k * step, min(w1, w0 + (k + 1) * step)) for k in range(n_slots)], False + thr = 0.30 * v.max() + regions = [] + i = 0 + while i < len(v): + if v[i] > thr: + j = i + while j < len(v) and v[j] > thr: + j += 1 + if (j - i) * FRAME_S >= 0.02: + regions.append([w0 + i, w0 + j]) + i = j + else: + i += 1 + if not regions: + regions = [[w0, w1]] + # merge regions separated by < 30ms (same vowel split by a flicker) + merged = [regions[0]] + for a, b in regions[1:]: + if a - merged[-1][1] < 6: + merged[-1][1] = b + else: + merged.append([a, b]) + regions = merged + # match count to slots: too many → keep strongest n in time order; + # too few → split the longest until we have enough + while len(regions) > n_slots: + scores = [v[a - w0:b - w0].sum() for a, b in regions] + regions.pop(int(np.argmin(scores))) + while len(regions) < n_slots: + lens = [b - a for a, b in regions] + k = int(np.argmax(lens)) + a, b = regions[k] + mid = (a + b) // 2 + regions[k:k + 1] = [[a, mid], [mid, b]] + return [(int(a), int(b)) for a, b in regions], True + + +def trim_silence(a, b, rms, floor, from_left=True, from_right=True): + while from_left and a < b - 1 and rms[a] < floor: + a += 1 + while from_right and b > a + 1 and rms[b - 1] < floor: + b -= 1 + return a, b + + +# ── the line builder ─────────────────────────────────────────────────────── + +def smooth_runs(curve, mask, sigma_frames): + """Gaussian-smooth `curve` independently inside each contiguous True run.""" + out = curve.copy() + r = int(max(1, round(3 * sigma_frames))) + kernel = np.exp(-0.5 * (np.arange(-r, r + 1) / sigma_frames) ** 2) + kernel /= kernel.sum() + i = 0 + n = len(curve) + while i < n: + if not mask[i]: + i += 1 + continue + j = i + while j < n and mask[j]: + j += 1 + seg = curve[i:j] + if len(seg) > 2: + pad = np.pad(seg, (r, r), mode="edge") + out[i:j] = np.convolve(pad, kernel, mode="valid") + i = j + return out + + +def main(): + plan = json.loads(open(sys.argv[1]).read()) + x, fs = sf.read(plan["line_wav"], dtype="float64") + if x.ndim > 1: + x = x.mean(axis=1) + + f0, t, sp, ap, fft_size = analyze(x, fs) + n_src = len(f0) + rms, flux, low_dom, hf_ratio = frame_features(x, fs, f0, sp) + cls, voiced, floor = classify_frames(f0, rms, flux, low_dom, hf_ratio) + + # interpolated log-f0 through unvoiced gaps (for the residual contour) + vi = np.where(voiced)[0] + if len(vi) < 5: + sf.write(plan["out_wav"], np.zeros(16, dtype=np.float32), fs) + print(json.dumps({"error": "line has almost no voiced frames"})) + return 1 + log_src = np.interp(np.arange(n_src), vi, np.log(f0[vi])) + + words = plan["words"] + beta = float(plan.get("beta", 0.25)) + rng = np.random.default_rng(int(plan.get("seed", 7))) + + # ── segment every word: onset cluster / nuclei / medials / coda ──────── + segs = [] # per word dict + sidecar_words = [] + for w in words: + w0 = max(0, int(round(w["srcFromMs"] / 1000.0 / FRAME_S))) + w1 = min(n_src, int(round(w["srcToMs"] / 1000.0 / FRAME_S))) + if w1 <= w0 + 2: + w1 = min(n_src, w0 + 8) + w0t, w1t = trim_silence(w0, w1, rms, floor) + n_slots = len(w["slots"]) + nuclei, sung = find_nuclei(w0t, w1t, n_slots, voiced, rms, low_dom) + if not sung and segs: + # whisper dropped a word and alignment handed this one a window + # of pure consonants — the vowel usually sits just OUTSIDE, in + # the previous word's overwide window or the following silence. + # Re-search from past the previous word's last nucleus. + pa = min(segs[-1]["nuclei"][-1][1] + 2, w0t) + pb = min(n_src, w1t + int(0.10 / FRAME_S)) + if pb > pa + 4 and voiced[pa:pb].any(): + pa2, pb2 = trim_silence(pa, pb, rms, floor) + nuclei, sung = find_nuclei(pa2, pb2, n_slots, voiced, rms, low_dom) + w0t, w1t = pa2, pb2 + # the previous word must not keep sounding into the stolen + # region — pull its coda back before our material starts + ps = segs[-1] + ps["coda"] = (ps["coda"][0], min(ps["coda"][1], nuclei[0][0])) + ps["w1"] = min(ps["w1"], nuclei[0][0]) + onset = (w0t, nuclei[0][0]) + if (onset[1] - onset[0]) * FRAME_S > MAX_ONSET_S: + onset = (onset[1] - int(MAX_ONSET_S / FRAME_S), onset[1]) + medials = [(nuclei[k][1], nuclei[k + 1][0]) for k in range(n_slots - 1)] + coda = (nuclei[-1][1], w1t) + if (coda[1] - coda[0]) * FRAME_S > MAX_CODA_S: + coda = (coda[0], coda[0] + int(MAX_CODA_S / FRAME_S)) + segs.append({"w": w, "onset": onset, "nuclei": nuclei, + "medials": medials, "coda": coda, "sung": sung, + "w0": w0t, "w1": w1t}) + sidecar_words.append({ + "word": w["w"], + "onsetMs": [round(onset[0] * FRAME_S * 1000), round(onset[1] * FRAME_S * 1000)], + "nucleiMs": [[round(a * FRAME_S * 1000), round(b * FRAME_S * 1000)] for a, b in nuclei], + "codaMs": [round(coda[0] * FRAME_S * 1000), round(coda[1] * FRAME_S * 1000)], + "classes": "".join({"sil": ".", "plosive": "P", "fric": "F", + "vcons": "C", "vowel": "V"}[c] for c in cls[w0t:w1t]), + }) + + with open(plan["phoneme_sidecar"], "w") as fh: + json.dump({"framePeriodMs": FRAME_S * 1000, "words": sidecar_words}, fh, indent=1) + + # ── output timeline ──────────────────────────────────────────────────── + line_t0 = float(plan["line_t0"]) + line_t1 = float(plan["line_t1"]) + out_n = int(round((line_t1 - line_t0) / FRAME_S)) + src_pos = np.full(out_n, -1.0) # fractional source frame per out frame + natural = np.zeros(out_n, dtype=bool) # 1:1-mapped (consonants, breaths) + target_midi = np.full(out_n, np.nan) # note plateaus (pre-glide) + in_word = np.zeros(out_n, dtype=bool) + vib_gain = np.zeros(out_n) + scoop = np.zeros(out_n) # semitone offsets (phrase scoops) + word_med = np.zeros(out_n) # per-frame word median log-f0 (residual ref) + + def of(t_abs): # absolute seconds → out frame index + return int(round((t_abs - line_t0) / FRAME_S)) + + def place(o_start, s_a, s_b, midi=None, med=None): + """1:1 placement of source frames [s_a,s_b) at out frame o_start.""" + L = s_b - s_a + a = max(0, o_start) + b = min(out_n, o_start + L) + if b <= a: + return + idx = np.arange(a, b) + src_pos[idx] = s_a + (idx - o_start) + natural[idx] = True + in_word[idx] = True + if midi is not None: + target_midi[idx] = midi + if med is not None: + word_med[idx] = med + + # pre-compute each word's onset length (for the previous word's hard end) + onset_len = [(s["onset"][1] - s["onset"][0]) * FRAME_S for s in segs] + + stats_words = [] + for wi, s in enumerate(segs): + w = s["w"] + slots = w["slots"] + n_slots = len(slots) + nuc_v = [] + for a, b in s["nuclei"]: + nuc_v.extend(f0[a:b][f0[a:b] > 0].tolist()) + med_hz = float(np.median(nuc_v)) if nuc_v else float(np.exp(log_src[s["w0"]])) + med_log = np.log(med_hz) + detected = float(hz_to_midi(med_hz)) if nuc_v else None + + # where must this word stop? (next word's pickup steals the tail) + hard_end = float(w["hardEnd"]) + if wi + 1 < len(segs): + hard_end = min(hard_end, segs[wi + 1]["w"]["slots"][0]["t"] - onset_len[wi + 1] - 0.01) + hard_end = max(hard_end, slots[-1]["t"] + 0.10) + + # onset cluster — pickup ending ON the first note time + oa, ob = s["onset"] + place(of(slots[0]["t"]) - (ob - oa), oa, ob, midi=slots[0]["midi"], med=med_log) + + coda_len = (s["coda"][1] - s["coda"][0]) * FRAME_S + for k in range(n_slots): + na, nb = s["nuclei"][k] + v_start = slots[k]["t"] + if k + 1 < n_slots: + med_len = (s["medials"][k][1] - s["medials"][k][0]) * FRAME_S + med_len = min(med_len, max(0.0, slots[k + 1]["t"] - v_start - 0.03)) + v_end = slots[k + 1]["t"] - med_len + else: + v_end = hard_end - coda_len + v_end = max(v_end, v_start + 0.03) + o_a, o_b = of(v_start), of(v_end) + o_a = max(0, o_a) + o_b = min(out_n, max(o_b, o_a + 2)) + n_out = o_b - o_a + n_nuc = nb - na + # vowel mapping: transitions natural, middle stretched w/ jitter + head = min(6, n_nuc // 3) + tail = min(6, n_nuc // 3) + if n_out <= n_nuc: + pos = na + np.linspace(0, n_nuc - 1, n_out) + else: + pos = np.empty(n_out) + pos[:head] = na + np.arange(head) + pos[n_out - tail:] = nb - tail + np.arange(tail) + mid_out = n_out - head - tail + mid_a, mid_b = na + head, nb - tail - 1 + base = np.linspace(mid_a, max(mid_a, mid_b), mid_out) + jit = rng.standard_normal(max(2, mid_out // 4)) + jit = np.interp(np.linspace(0, 1, mid_out), + np.linspace(0, 1, len(jit)), jit) * 1.2 + pos[head:head + mid_out] = np.clip(base + jit, mid_a, max(mid_a, mid_b)) + idx = np.arange(o_a, o_b) + src_pos[idx] = pos[:len(idx)] + in_word[idx] = True + target_midi[idx] = slots[k]["midi"] + word_med[idx] = med_log + # vibrato only on true holds + hold = v_end - v_start + if hold >= VIB_HOLD_S: + tt = (idx - o_a) * FRAME_S + vib_gain[idx] = np.clip((tt - 0.15) / VIB_ONSET_S, 0, 1) + # phrase-onset scoop on the first vowel + if k == 0 and w.get("phraseStart"): + tt = (idx - o_a) * FRAME_S + scoop[idx] = -SCOOP_ST * np.clip(1.0 - tt / SCOOP_S, 0, 1) + # medial consonant cluster — pickup into the NEXT note + if k + 1 < n_slots: + ma, mb = s["medials"][k] + mlen = of(slots[k + 1]["t"]) - o_b + if mb - ma > 0 and mlen > 0: + place(o_b, mb - min(mb - ma, mlen), mb, + midi=slots[k + 1]["midi"], med=med_log) + + # coda at the vowel's end + ca, cb = s["coda"] + if cb > ca: + place(of(hard_end) - (cb - ca), ca, cb, midi=slots[-1]["midi"], med=med_log) + + # breath in the source gap before the NEXT phrase start → natural + if wi + 1 < len(segs) and segs[wi + 1]["w"].get("phraseStart"): + ga, gb = s["w1"], segs[wi + 1]["w0"] + ga, gb = trim_silence(ga, gb, rms, floor * 0.6) + if gb > ga + 4 and rms[ga:gb].mean() > floor * 0.6: + blen = min(gb - ga, int(MAX_BREATH_S / FRAME_S)) + nxt_on = of(segs[wi + 1]["w"]["slots"][0]["t"]) - \ + int(onset_len[wi + 1] / FRAME_S) + ba = nxt_on - blen - 4 + if ba > of(hard_end) + 2: + place(ba, gb - blen, gb) # unpitched — no midi + stats_words.append({ + "word": w["w"], "targets": [sl["midi"] for sl in slots], + "detected_midi": None if detected is None else round(detected, 2), + "shift_st": None if detected is None else + round(float(np.mean([sl["midi"] for sl in slots])) - detected, 2), + "onset_ms": round(onset_len[wi] * 1000), + "coda_ms": round(coda_len * 1000), + "nuclei": len(s["nuclei"]), "sung": s["sung"], + }) + + # ── continuous target-f0 curve ───────────────────────────────────────── + mapped = src_pos >= 0 + src_i = np.clip(np.round(src_pos).astype(int), 0, n_src - 1) + voiced_out = np.zeros(out_n, dtype=bool) + voiced_out[mapped] = voiced[src_i[mapped]] + + # fill target through voiced consonants that carry no note of their own + tm = target_midi.copy() + have = ~np.isnan(tm) + if have.any(): + ii = np.where(have)[0] + tm = np.interp(np.arange(out_n), ii, tm[ii]) + tm = tm + scoop + target_log = np.log(midi_to_hz(tm)) + # portamento: smooth the target INSIDE contiguous in-word runs only + target_log = smooth_runs(target_log, in_word, GLIDE_SIGMA_S / FRAME_S) + + # slow detune drift so plateaus never sound quantized + tt = np.arange(out_n) * FRAME_S + ph = rng.uniform(0, 2 * np.pi, 2) + drift = (DRIFT_CENTS / 1200.0) * np.log(2) * \ + (0.6 * np.sin(2 * np.pi * 0.23 * tt + ph[0]) + + 0.4 * np.sin(2 * np.pi * 0.61 * tt + ph[1])) + vib = (VIB_CENTS / 1200.0) * np.log(2) * vib_gain * np.sin(2 * np.pi * VIB_HZ * tt) + + resid = np.zeros(out_n) + fm = mapped & (word_med != 0) + frac = src_pos - np.floor(src_pos) + i0 = np.clip(np.floor(src_pos).astype(int), 0, n_src - 1) + i1 = np.clip(i0 + 1, 0, n_src - 1) + resid[fm] = ((1 - frac[fm]) * log_src[i0[fm]] + frac[fm] * log_src[i1[fm]]) - word_med[fm] + # octave-fold: vocal fry and harvest octave errors put ~1200-cent steps in + # the residual which beta would scale into audible lurches — fold them out, + # keep only the sub-octave micro-contour, then smooth it + resid = resid - np.log(2) * np.round(resid / np.log(2)) + resid = smooth_runs(resid, fm, 2.0) + + f0_out = np.exp(target_log + beta * resid + drift + vib) + f0_out[~voiced_out] = 0.0 + + # ── assemble sp/ap streams and synthesize ONCE ───────────────────────── + sp_out = np.full((out_n, sp.shape[1]), 1e-12) + ap_out = np.ones((out_n, ap.shape[1])) + w0f = (1 - frac[mapped])[:, None] + w1f = frac[mapped][:, None] + sp_out[mapped] = w0f * sp[i0[mapped]] + w1f * sp[i1[mapped]] + ap_out[mapped] = np.clip(w0f * ap[i0[mapped]] + w1f * ap[i1[mapped]], 0.0, 1.0) + + y = pw.synthesize(np.ascontiguousarray(f0_out), + np.ascontiguousarray(sp_out), + np.ascontiguousarray(ap_out), fs, + frame_period=FRAME_S * 1000.0) + + # ── unvoiced natural runs: composite the ORIGINAL samples back in ────── + hop = int(round(fs * FRAME_S)) + comp = natural & mapped & ~voiced_out + ramp = int(0.005 * fs) + i = 0 + while i < out_n: + if not comp[i]: + i += 1 + continue + j = i + while j < out_n and comp[j] and (j == i or abs(src_pos[j] - src_pos[j - 1] - 1) < 0.5): + j += 1 + if (j - i) >= 2: # ≥10ms + oa, ob = i * hop, min(len(y), j * hop) + sa = int(round(src_pos[i])) * hop + sb = sa + (ob - oa) + if sb <= len(x) and ob <= len(y): + seg = x[sa:sb].copy() + L = ob - oa + r = min(ramp, L // 2) + if r > 1: + fade = 0.5 - 0.5 * np.cos(np.pi * np.arange(r) / r) + mixin = np.ones(L) + mixin[:r] = fade + mixin[L - r:] = fade[::-1] + y[oa:ob] = mixin * seg + (1 - mixin) * y[oa:ob] + else: + y[oa:ob] = seg + i = j + # line-edge fades + ef = int(0.010 * fs) + if len(y) > 2 * ef: + y[:ef] *= np.linspace(0, 1, ef) + y[-ef:] *= np.linspace(1, 0, ef) + + sf.write(plan["out_wav"], y.astype(np.float32), fs) + + # f0-continuity metric: max & p95 cents jump between consecutive voiced + # frames inside words (the smoothness evidence) + j = np.abs(np.diff(np.log(np.where(f0_out > 0, f0_out, np.nan)))) * 1200 / np.log(2) + j = j[~np.isnan(j)] + print(json.dumps({ + "words": stats_words, + "f0_jump_max_cents": round(float(j.max()), 1) if len(j) else 0, + "f0_jump_p95_cents": round(float(np.percentile(j, 95)), 1) if len(j) else 0, + "out_frames": out_n, + })) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/pop/menuband/bin/sing_word_world.py b/pop/menuband/bin/sing_word_world.py new file mode 100644 index 0000000000..f830e466a1 --- /dev/null +++ b/pop/menuband/bin/sing_word_world.py @@ -0,0 +1,163 @@ +#!/usr/bin/env python3 +""" +sing_word_world.py — melodyne-style RELATIVE pitch mapping for one sung word. + +The companion to pop/bin/pitchsnap_world.py, but built around jeffrey's rule +for sung TTS: never absolute-snap his voice into a scale — DETECT the word's +actual fundamental in the take and shift it BY THE INTERVAL to the target +note(s). The word's own micro-contour (onset glides, declination, consonant +transitions) survives at a reduced depth (--beta), so it sings without losing +the baritone character. + + f0_new(t) = target_hz(t) * (f0(t) / median_f0) ** beta + +WORLD chain (harvest → stonemask → cheaptrick → d4c → synthesize) with the +same unvoiced-composite trick as pitchsnap_world.py: consonants and sibilants +pass through from the original take, only voiced frames are resynthesized. + +Usage: + sing_word_world.py in.wav out.wav --midis "48,52" --starts "0,0.4" + [--beta 0.25] [--f0-floor 70] [--xfade-ms 60] + [--vibrato-hz 5] [--vibrato-cents 0] [--vibrato-onset-ms 250] + +Prints one JSON line: {"detected_midi": …, "shift_st": …, "voiced_pct": …} +so the caller (sing-jingle.mjs) can log the per-word interval and flag +words whose pitch detection resisted. +""" +import argparse +import json +import sys +import numpy as np +import soundfile as sf +import pyworld as pw + + +def midi_to_hz(m): + return 440.0 * (2.0 ** ((m - 69.0) / 12.0)) + + +def hz_to_midi(hz): + return 69.0 + 12.0 * np.log2(hz / 440.0) + + +def main(): + p = argparse.ArgumentParser() + p.add_argument("in_wav") + p.add_argument("out_wav") + p.add_argument("--midis", required=True, help="comma-separated target MIDI notes") + p.add_argument("--starts", required=True, + help="comma-separated start times (sec, first 0) per note within the word") + p.add_argument("--beta", type=float, default=0.25, + help="how much of the word's own pitch contour survives (0=flat, 1=all)") + p.add_argument("--f0-floor", type=float, default=70.0) + p.add_argument("--f0-ceil", type=float, default=500.0) + p.add_argument("--xfade-ms", type=float, default=60.0) + p.add_argument("--vibrato-hz", type=float, default=0.0) + p.add_argument("--vibrato-cents", type=float, default=0.0) + p.add_argument("--vibrato-onset-ms", type=float, default=250.0) + args = p.parse_args() + + midis = np.array([float(m) for m in args.midis.split(",")], dtype=np.float64) + starts = np.array([float(s) for s in args.starts.split(",")], dtype=np.float64) + if len(midis) != len(starts): + print(json.dumps({"error": "midis/starts length mismatch"})) + return 1 + target_hzs = midi_to_hz(midis) + + x, fs = sf.read(args.in_wav, dtype="float64") + if x.ndim > 1: + x = x.mean(axis=1) + + f0_raw, t = pw.harvest(x, fs, f0_floor=args.f0_floor, f0_ceil=args.f0_ceil, + frame_period=5.0) + f0 = pw.stonemask(x, f0_raw, t, fs) + voiced = f0 > 0 + n_frames = len(t) + frame_period_s = (t[1] - t[0]) if n_frames > 1 else 0.005 + + # Not enough voiced material to pitch (a pure consonant burst) — pass the + # original through untouched and let the caller report it. + if voiced.sum() < 5: + sf.write(args.out_wav, x.astype(np.float32), fs) + print(json.dumps({"detected_midi": None, "shift_st": 0.0, + "voiced_pct": float(100.0 * voiced.mean())})) + return 0 + + med_hz = float(np.median(f0[voiced])) + detected = float(hz_to_midi(np.array(med_hz))) + target_mean = float(midis.mean()) + + # Per-frame target: hold each note from its start, crossfade in log space. + seg_starts = np.clip(np.round(starts / frame_period_s).astype(np.int64), 0, n_frames) + xfade_frames = max(1, int(args.xfade_ms / (frame_period_s * 1000.0))) + target_log = np.zeros(n_frames) + for i in range(n_frames): + seg = int(np.searchsorted(seg_starts[1:], i, side="right")) + seg = min(seg, len(midis) - 1) + center = np.log(target_hzs[seg]) + if seg + 1 < len(midis): + dist = seg_starts[seg + 1] - i + if dist < xfade_frames: + u = 1.0 - (dist / xfade_frames) + center = (1 - u) * center + u * np.log(target_hzs[seg + 1]) + target_log[i] = center + target_curve = np.exp(target_log) + + if args.vibrato_hz > 0 and args.vibrato_cents > 0: + time_sec = np.arange(n_frames) * frame_period_s + onset = args.vibrato_onset_ms / 1000.0 + fade = np.clip((time_sec - onset) / max(0.05, onset), 0.0, 1.0) + depth = (args.vibrato_cents / 100.0) / 12.0 + target_curve = target_curve * (2.0 ** (np.sin(2 * np.pi * args.vibrato_hz * time_sec) * depth * fade)) + + # Relative map: move the word's median onto the target, keep beta of the + # residual contour AROUND ITS OWN RAW MEDIAN. Using the raw median makes + # the output land on target even when harvest tracked a harmonic — frames + # and median double together, so the residual stays clean either way. + # Interpolate f0 through unvoiced gaps first so WORLD sees a continuous + # curve (no 0→target phase pops). + log_src = np.log(np.maximum(f0, 1e-6)) + vi = np.where(voiced)[0] + log_src_i = np.interp(np.arange(n_frames), vi, log_src[vi]) + resid = log_src_i - np.log(med_hz) # his contour around center + f0_synth = np.exp(np.log(target_curve) + args.beta * resid) + + fft_size = pw.get_cheaptrick_fft_size(fs, f0_floor=args.f0_floor) + sp = pw.cheaptrick(x, f0, t, fs, fft_size=fft_size, f0_floor=args.f0_floor) + ap = pw.d4c(x, f0, t, fs, fft_size=fft_size) + y = pw.synthesize(f0_synth, sp, ap, fs, frame_period=5.0) + + # Re-impose voiced/unvoiced: WORLD audio on voiced frames, the ORIGINAL + # take on unvoiced (consonants stay crisp), 5ms cosine ramps at edges. + spf = int(round(fs * 0.005)) + mask = np.repeat(voiced.astype(np.float64), spf) + if len(mask) < len(y): + mask = np.pad(mask, (0, len(y) - len(mask)), mode="edge") + mask = mask[:len(y)] + ramp = int(0.005 * fs) + if ramp > 1: + edges = np.diff(mask.astype(np.int8)) + for idx in np.where(edges == 1)[0]: + for k in range(ramp): + pos = idx + 1 + k + if pos < len(mask): + mask[pos] *= 0.5 - 0.5 * np.cos(np.pi * (k + 1) / ramp) + for idx in np.where(edges == -1)[0]: + for k in range(ramp): + pos = idx - k + if pos >= 0: + mask[pos] *= 0.5 - 0.5 * np.cos(np.pi * (k + 1) / ramp) + n = min(len(y), len(x), len(mask)) + out = mask[:n] * y[:n] + (1.0 - mask[:n]) * x[:n] + sf.write(args.out_wav, out.astype(np.float32), fs) + + print(json.dumps({ + "detected_midi": round(detected, 2), + "shift_st": round(target_mean - detected, 2), + "voiced_pct": round(float(100.0 * voiced.mean()), 1), + })) + return 0 + + +if __name__ == "__main__": + sys.exit(main())