From 778903e85e4090e1f8e4f2345ed3b380a6c582c5 Mon Sep 17 00:00:00 2001 From: "prompt.ac/@jeffrey" Date: Tue, 9 Jun 2026 18:37:27 -0700 Subject: [PATCH] notepat native: GM instrument HUD label + live digit entry, chord-tone highlight, shift-latch sustain - persistent on-screen instrument readout: 'NNN Name' (program 1-128) / 'MIDI PASSTHRU' / legacy 'wave '; shows the digit buffer live with a blinking caret as the program number is typed - chord modifiers now light up ALL chord-tone pads on the grid HUD (root full-bright, extensions dimmer), via chordToneMidi scan in drawGrid - shift-linger is now a true latch: lifting the key keeps the note(s) ringing (joins the heldKeys set, same as Enter-latch); same-key re-press toggles off; stopAll/exit clears. Chords latch as a unit. Plus the GM synthesis research dossier (docs/gm-synthesis/): per-instrument synthesis method + cited papers + performant C algorithm for all 128 GM programs (4 family files) and 00-stochasticism.md (bounded organic per-note variation calibrated to the percussion system). Implemented by subagents. --- .../docs/gm-synthesis/00-stochasticism.md | 385 ++++++++++ .../01-piano-mallet-organ-guitar.md | 662 +++++++++++++++++ .../02-bass-strings-ensemble-brass.md | 701 ++++++++++++++++++ .../03-reed-pipe-synthlead-synthpad.md | 477 ++++++++++++ .../04-synthfx-ethnic-percussive-soundfx.md | 688 +++++++++++++++++ fedac/native/pieces/notepat.mjs | 134 +++- 6 files changed, 3033 insertions(+), 14 deletions(-) create mode 100644 fedac/native/docs/gm-synthesis/00-stochasticism.md create mode 100644 fedac/native/docs/gm-synthesis/01-piano-mallet-organ-guitar.md create mode 100644 fedac/native/docs/gm-synthesis/02-bass-strings-ensemble-brass.md create mode 100644 fedac/native/docs/gm-synthesis/03-reed-pipe-synthlead-synthpad.md create mode 100644 fedac/native/docs/gm-synthesis/04-synthfx-ethnic-percussive-soundfx.md diff --git a/fedac/native/docs/gm-synthesis/00-stochasticism.md b/fedac/native/docs/gm-synthesis/00-stochasticism.md new file mode 100644 index 0000000000..5065dbe54a --- /dev/null +++ b/fedac/native/docs/gm-synthesis/00-stochasticism.md @@ -0,0 +1,385 @@ +# GM Synthesis Dossier 00 — Bounded Per-Note Stochasticism + +A cross-cutting design principle for the Aesthetic Computer native GM synthesis +library (`fedac/native/src/audio.c`). It governs *every* instrument in the four +family dossiers (01–04), the same way a single bow stroke is never bit-identical +to the last on a real instrument. + +> **The requirement (jeffrey).** Every sound in the GM system gets a *bit* of +> stochasticism so it reads as organic — but **not** so much that it changes the +> overall timbre or pitch. Just enough that the same voicing playing the same +> note doesn't produce the exact same wave profile every trigger, while staying +> **relatively consistent** — the way the AC percussion hi-hats and snares +> already vary. + +This document is the **calibration anchor** for that "bit." The numbers here are +not invented; they are measured from the existing AC percussion system +(`system/public/aesthetic.computer/lib/percussion.mjs`, bundled to the device as +`/lib/percussion.mjs`) and then mapped onto the GM synthesis levers. + +--- + +## 1. The calibration reference: how AC percussion already varies + +`percussion.mjs` fires each drum as a stack of `sound.synth()` voices. Two inline +helpers introduce all the per-hit variation (percussion.mjs:115–116): + +```js +// rj — jitter a CENTER value by a ± fraction (multiplicative). +const rj = (center, frac) => center * (1 + (Math.random() - 0.5) * 2 * frac); +// rn — uniform value in a range (used for pan offsets). +const rn = (min, max) => min + Math.random() * (max - min); +``` + +Measuring every call site across the 12 drums, the variation lands in tight bands: + +| Lever | What varies | Range used in `percussion.mjs` | +|---|---|---| +| **Per-voice amplitude** | `volume:` on each layer | `rj(v, 0.10)`–`rj(v, 0.18)`, a few tails at `0.20`–`0.25` → **±10–18% typical, ±25% max** | +| **Tail decay / duration** | `duration:` on the ring-out layers | `rj(d, 0.18)`–`rj(d, 0.25)` → **±18–25%** (only on *tails*, never the transient) | +| **Stereo pan** | base `downPan` + per-tail spread | base `rn(-0.02, 0.02)`…`rn(-0.06, 0.06)`; noise tails add `rn(-0.04, 0.04)` → **≈ ±0.02–0.06, up to ±0.10 on diffuse tails** | +| **Tonal frequency** | the drum's pitch (`tone:`) | **NOT JITTERED — zero.** | + +The last row is the whole point, and it is stated explicitly in `audio.c` at the +top of the native percussion recipes (audio.c:1827–1834): + +> *"Key principle: the iconic 808 frequencies (238 Hz snare, 540/800 Hz cowbell, +> 6-square hat cluster) are NOT jittered — they're what makes each drum sound +> like itself. Per-hit variation comes from TIMBRE jitter (volume balance, +> decay, attack, pan) not tonal jitter."* + +**The calibrated feel, distilled:** the percussion system varies *energy +distribution* (which partial is loudest this time, how long the tail rings, where +it sits in the stereo field) by **±10–25%**, and varies *pitch by exactly +nothing*. Variation lives in amplitude/decay/pan; identity lives in frequency. + +For pitched GM instruments we cannot hold pitch at *literally* zero variance — +phase-locked unison stacks need a hair of detune to avoid robotic phase-perfect +combing — but we keep the pitch lever an order of magnitude smaller than the +amplitude lever, well under audible mistuning. That is the single most important +calibration decision in this document, and it comes straight from the drums. + +### 1.1 How `notepat` invokes it (no jitter params passed) + +The native `notepat` calls the shared kernel through `triggerPercussionDown` → +`playPercussion` → `sharedPlayPercussion(sound, letter, { volume, pan, +pitchFactor, phase, holdVoices, onVoice })` (notepat.mjs:1763, 1847, 1857). It +passes **no** random or jitter parameters — the stochasticism is entirely +*internal* to the synth kernel. That is the model GM synthesis must follow: the +caller asks for "note N at velocity V," and the *engine* decides the micro-offsets +at note-on. Pieces never thread RNG state through the API. + +--- + +## 2. The engine already has the machinery + +Two facts from `audio.c` / `audio.h` make this cheap to implement: + +1. **A per-voice PRNG already exists.** `xorshift32(uint32_t *state)` + (audio.c:100–107) is the engine's standard PRNG, and every `ACVoice` carries + a `uint32_t noise_seed` (audio.h:87). It is already seeded per-trigger at + voice allocation (audio.c:3361): + + ```c + v->noise_seed = (uint32_t)(audio->next_id * 2654435761u); + ``` + + `audio->next_id` increments on every `audio_synth()` (audio.c:3356), so each + trigger gets a **distinct, deterministic** seed. This is exactly the property + we want: *deterministic within a trigger, varied across triggers.* Today only + `WAVE_NOISE/WHISTLE/GUN/HARP/PIANO` get seeded — the GM types must extend that + list (see §6). + +2. **The phase-increment rule is non-negotiable.** Sustained tones advance a + phase register and read a wavetable; never `sin(TAU*f*t)` per sample (audio.c + convention + `MEMORY.md`: *"long sine phase-increment not sin(TAU*f*t)"*). All + stochastic detune is therefore applied **once, at note-on**, baked into each + partial's `f_inc` (or KS delay length, FM ratio, filter cutoff). We do **not** + modulate pitch per-sample. This keeps the variation coherent within a note + (no white-noise warble) and costs nothing in the inner loop. + +--- + +## 3. The mechanism: a note-on PRNG draw + +**Principle: draw once, at note-on; bake into voice parameters; never re-roll +per sample.** Within a single note the micro-offsets are constant, so the note is +*coherent* (a fixed, slightly-unique voicing) rather than chaotic. Across notes +they differ because `noise_seed` differs. This is structurally identical to how +`percussion.mjs` calls `rj()`/`rn()` once per voice at fire time. + +Two distinct uses of the per-voice PRNG, which must not be confused: + +- **Structural noise** (KS excitation seed, breath/bow/reed turbulence, hammer + thump) — consumed *per-sample* in the inner loop. This already works via + `xorshift32(&v->noise_seed)` and is the bulk of "organic" for noise-driven + voices. A fresh `noise_seed` per trigger means a different excitation burst + every pluck — exactly the snare-tail behaviour, applied to the string. +- **Parametric jitter** (detune, per-partial amp/phase, decay, cutoff, pan) — + drawn *once at note-on* from the same PRNG, **before** the structural noise + loop starts consuming it. Draw all parametric jitter first so the structural + stream downstream is still effectively white. + +A global **`organic`** amount (0..1, default **0.6** — see §5) scales every +parametric lever. `organic = 0` reproduces the old bit-identical behaviour +(precise contexts: tuners, test tones, click tracks); `organic = 1` is the +maximum still-tasteful spread. The default 0.6 places the amplitude lever at +≈ ±9–15%, squarely inside the percussion band. + +--- + +## 4. What to vary, and by how much (bounded ranges) + +All ranges below are the spread at **`organic = 1.0`**; the realized spread is +`range × organic`. Bounds are chosen so that at the default `organic = 0.6` the +amplitude/decay levers sit inside the percussion ±10–25% band, and the pitch +lever stays an order of magnitude below audible mistuning. + +| # | Lever | Spread @ `organic=1` | @ default `0.6` | Rationale vs. perc reference | +|---|---|---|---|---| +| **a** | **Pitch micro-detune** (per voice, and per-partial for unison/phantom strings) | **±6 cents** | **±3.6 cents** | The drums jitter pitch by **0** for identity. We cannot use 0 (phase-locked partials comb), so we pick the smallest value that breaks phase-lock yet stays *well* under the ~±5–6 cent JND for melodic mistuning. A real piano's own 3-string detune is ±0.3–1.5 cents (dossier 01); ±3.6 cents is "alive," not "out of tune." **Hard ceiling: ±6 cents, never exceeded regardless of `organic`.** | +| **b** | **Per-partial / per-mode amplitude** (modal, additive, drawbar) | **±15%** (±1.4 dB) | **±9%** | Directly mirrors percussion's `rj(v, 0.10–0.18)`. This is *the* primary organic lever — it changes *which partial leads* this note without moving the spectral centroid enough to alter perceived timbre. Apply independently per partial; the sum stays level-normalized. | +| **c** | **Per-partial start phase** (additive/modal/drawbar) | **random 0..1** (full) | full | Free and identity-safe: start phase is inaudible in isolation but decorrelates stacked voices so two simultaneous same-notes don't phase-cancel. Percussion gets this implicitly (noise has no defined phase); pitched additive voices must do it explicitly. **Exception:** struck/plucked attacks where partials must align for a crisp transient — keep those phase-coherent (see §7 mallets). | +| **d** | **Excitation seed** (KS pluck, bow/breath/reed noise) | fresh `noise_seed` per trigger (already happens) | — | The structural analogue of a snare's noise burst differing every hit. No knob needed — it is on whenever the voice consumes `noise_seed`. `organic` does not gate this; turning it off would make plucks robotic. | +| **e** | **KS pluck-position** `β` (comb tap) | **±8% of β** | **±4.8%** | Models the player not hitting the identical spot twice. Shifts the comb notches slightly → timbral shimmer, no pitch change. Stays well inside the audible-but-subtle zone. | +| **f** | **FM index / ratio micro-jitter** | index **±6%**; ratio **±2 cents** | ±3.6% / ±1.2 cents | Index jitter ≈ the amplitude lever (it *is* a brightness/energy lever). Ratio jitter is held to the pitch ceiling (b) because an FM ratio error *is* a detune of the sidebands. | +| **g** | **Attack-time micro-jitter** | **±12%** | **±7.2%** | Percussion staggers multi-burst attacks deliberately (clap 0.005/0.015/0.025); generic attack jitter of ±7% reproduces the "no two strokes land identically" feel without smearing transients. Applied to `v->attack`. | +| **h** | **Velocity / amplitude micro-jitter** (whole-voice gain) | **±8%** (±0.7 dB) | **±4.8%** | The conservative end of percussion's volume band, applied to the *whole* voice (vs. per-partial in **b**). Keeps a repeated note from being a perfect amplitude copy. | +| **i** | **Decay-time micro-jitter** (T60 / per-mode tau) | **±15%** | **±9%** | Straight from percussion's tail jitter `rj(d, 0.18–0.25)`. Apply to per-partial/per-mode decay multipliers and to overall release. The ring-out length is *audibly* alive and very identity-safe. | +| **j** | **Filter cutoff micro-jitter** (subtractive) | **±5%** (≈ ±0.8 semitone of cutoff) | **±3%** | Analog filters drift; ±3% cutoff is "warm analog," not a different patch. Below the threshold where it reads as a timbre change. | +| **k** | **Stereo pan micro-jitter** (optional) | **±0.05** | **±0.03** | Exactly the percussion base pan spread `rn(-0.02..0.06)`. Adds spatial life. Optional — skip for centered mono leads. | + +### 4.1 The two invariants + +Whatever an instrument does with the levers above, two perceptual guarantees hold +at **all** values of `organic ∈ [0,1]`: + +1. **Pitch stays within ±5 cents** of nominal. The pitch lever (a) is capped at + ±6 cents *spread* (so ±3 cents from center at the default, ±6 cents absolute + worst case at `organic=1`) and never scales past that hard ceiling. Phantom/ + unison detune layers count against the same budget. +2. **Spectral centroid stays stable.** Per-partial amplitude jitter is + **zero-mean and independent** across partials, so it perturbs *which* partial + leads without shifting the centroid in expectation. Never apply a *correlated* + tilt (that *would* be a timbre change — it's a different instrument, not a + different stroke). + +--- + +## 5. The single global knob: `organic` + +```c +// In ACAudio (engine-global). Default tuned to the percussion feel. +double organic_amount; // 0.0 = bit-identical, 1.0 = max tasteful spread. + // DEFAULT 0.6 → amp lever ≈ ±9%, inside the + // percussion ±10–25% band; pitch ≈ ±3.6c. +``` + +- **Why 0.6 is the default.** At 0.6 the dominant lever (per-partial amplitude, + ±15% × 0.6 ≈ ±9%) lands at the *low* end of percussion's measured ±10–18% + band — deliberately a touch gentler than drums, because pitched sustained tones + expose variation more than transient drum hits do. Decay jitter at ±9% matches + drum tails. Pitch at ±3.6 cents is sub-JND. The result *feels* like the hats + and snares without ever sounding detuned. +- **One knob, global, exposed.** A single `audio_set_organic(double)` (or a JS + binding) scales every lever. Turn it down for precise contexts (tuner, test + tone, metronome); leave at 0.6 for music. Per-family multipliers (§7) ride on + top of this one global so the *relative* character is preserved as you scale. +- **It is bounded, seeded, and coherent — therefore "relatively consistent."** + Bounded: every lever has a hard ceiling (§4). Seeded: `noise_seed` is a pure + function of `next_id`, so a trigger's sound is reproducible given its seed. + Coherent: parametric jitter is drawn once at note-on and frozen for the note's + life — no drift, no warble, no runaway. This is precisely why percussion + sounds varied-but-consistent, and the same property carries over. + +--- + +## 6. Per-family application + +Each GM family leans on different levers (cross-referenced against dossiers 01, +02, 03, 04). `mul` is a per-family multiplier on the global `organic`, capturing +"how much variation suits this instrument." Pitch is *always* the tightest lever. + +| GM range | Family | Primary engine (per dossier) | Dominant organic levers | `mul` | Notes | +|---|---|---|---|---|---| +| 1–4 | Acoustic piano | modal additive + inharmonicity (01) | **b** per-partial amp, **d/hammer** thump seed, **a** 3-string phantom detune, **i** per-partial decay | 0.8 | Phantom-string detune (±0.3–1.5c, dossier 01) *is* lever (a) under the ±6c cap; fresh hammer-noise seed per strike. | +| 5–6 | Electric piano (FM) | 2–4 op FM tine/reed (01) | **f** index/ratio jitter, **g** attack (bark), **h** velocity | 0.7 | Index jitter = attack "bark" variation; ratio jitter stays at pitch ceiling. | +| 7–8 | Harpsichord / Clavi | extended KS (01) | **d** excitation seed, **e** pluck-position β, **b** | 0.9 | Plucked → excitation+β dominate, like a snare's noise burst per hit. | +| 9–15 | Chromatic perc (mallets/bells) | modal resonator bank (01) | **b** per-mode amp (strike-position), **i** per-mode decay, **d** mallet-click seed | 1.0 | Highest `mul`: these *are* percussion. Strike-position → per-mode amplitude jitter is the exact analogue of which drum partial leads. **Keep attack phase-coherent** for a crisp strike (§4 lever c exception). | +| 16 | Dulcimer | KS unison strings (01) | **a** unison detune, **d** excitation, **e** β | 0.9 | Detune across the 2–3 courses uses lever (a) under cap. | +| 17–19 | Hammond organ | additive drawbars + Leslie (01) | **least variation:** **c** per-drawbar phase, tiny **a** tonewheel detune, **g** key-click seed | 0.3 | Organs are the *most consistent* — an electromechanical machine. Mostly key-click variation + a hair of Leslie/tonewheel phase. Low `mul` by design. | +| 20–24 | Pipe / reed / accordion | additive ranks / free-reed subtractive (01) | **a** rank/reed detune (musette), **j** cutoff, breath/chiff seed | 0.5 | Musette detune is intrinsic to the patch; the *organic* part is the small extra wobble + chiff-noise seed. | +| 25–32 | Guitar (all) | extended KS + waveshaper (01) | **d** excitation, **e** β, **b**, **j** on driven variants | 0.9 | Pick-vs-finger excitation seed differs every pluck. Drive variants jitter the pre-gain a touch via **h**. | +| 33–40 | Bass | KS (dark) / subtractive synth (02) | KS: **d/e**; subtractive: **j** cutoff, **a** osc detune | 0.7 | Synth basses lean on cutoff drift (j); acoustic basses on excitation. | +| 41–47 | Strings (solo/section) | bowed waveguide / KS pizz / modal (02) | bow-noise **d**, **a** vibrato-rate + bow detune, **g** | 1.0 | Bowed strings are the headline organic case: fresh bow-noise seed + small vibrato-rate jitter per note = no two strokes alike. Pizz uses excitation+β. | +| 48 | Timpani | modal + pitch drop (02) | **b** per-mode amp, **i** decay, **d** strike seed | 1.0 | Membrane modes; same as mallets. | +| 49–55 | Ensemble (supersaw/strings) | supersaw + chorus (02) | **a** per-saw detune, **c** phase, **j** cutoff | 0.6 | Supersaw is *built* from detune; the organic add is per-note phase + small detune wobble on top of the fixed spread. | +| 52–54 | Choir / voice | formant synthesis (02) | breath/aspiration seed **d**, **a** per-voice detune, **g** onset | 0.9 | Per-singer detune + aspiration noise = a believable section. | +| 56 | Orchestra hit | cluster + transient (02) | **b**, **i**, **d** transient seed | 0.8 | Mostly transient/decay jitter. | +| 57–63 | Brass | brass waveguide / subtractive (02) | breath/lip-noise **d**, **g** attack (blat), **a** small detune, **j** | 0.9 | Attack-timing + breath-noise jitter give the human "blat"; section stacks add detune. | +| 64 | Reeds-as-brass / synth brass | subtractive (02) | **j** cutoff, **a** detune, **g** | 0.7 | Filter drift dominates. | +| 65–72 | Reed (sax/oboe/clarinet…) | reed waveguide nonlinearity (03) | reed-noise **d**, **g** attack chiff, **a** vibrato-rate, breath **h** | 1.0 | Reed turbulence seed per note is the core; vibrato-rate jitter prevents mechanical LFO lock. | +| 73–80 | Pipe (flute/whistle/recorder) | Cook flute waveguide (03) | breath-noise **d**, **a** vibrato-rate, **g** chiff, **h** breath | 1.0 | Already partly organic via `generate_whistle_sample` breath noise (audio.c:195); add vibrato-rate jitter + chiff variation. The flute's 5 Hz vibrato (audio.c:188) should get ±10% rate jitter. | +| 81–88 | Synth lead | subtractive named-waveforms (03) | **j** cutoff drift, **a** detune, **c** phase | 0.6 | Analog-style cutoff/detune drift; keep it tasteful so leads stay tuned. | +| 89–96 | Synth pad | detuned multi-osc + slow sweep (03) | **a** per-osc detune, **c** phase, **j** slow cutoff | 0.5 | Pads are wide already; small per-osc detune + phase decorrelation. | +| 97–104 | Synth FX | PhISEM / pads / FM bells (04) | **d** PhISEM particle seed, **b**, **i** | 0.9 | PhISEM is *inherently* stochastic (random droplet timing) — the seed is the whole point. | +| 105–112 | Ethnic | KS + sympathetic / modal / reed / bowed (04) | per-instrument: **d/e** (sitar/koto/banjo), **a** (sympathetic detune), reed **d** (bagpipe) | 0.9 | Sympathetic strings → small detune across them (a). Sitar sawari buzz benefits from excitation seed variation. | +| 113–120 | Percussive (bells/taiko/steelpan…) | modal banks / membrane modes (04) | **b** per-mode amp, **i** decay, **d** strike seed | 1.0 | Pure percussion — top `mul`, same calibration as the drum reference itself. | +| 121–128 | Sound FX (breath/seashore/bird/heli/applause/gunshot) | filtered noise / PhISEM / `WAVE_GUN` (04) | **d** seed is everything; **i**, pan **k** | 1.0 | These are *defined* by noise — a fresh seed per trigger is the entire organic story. Gunshot reuses `WAVE_GUN`, already seeded. | + +**Reading the table:** organs (0.3) and pads/leads (0.5–0.6) are deliberately the +*most consistent* — machine-like, electromechanical. Mallets, percussion, bowed +strings, reeds/pipes, and noise-FX sit at `mul ≈ 1.0` — these are where the human +hand and turbulent air live, and they carry the most variation, exactly as the +hi-hats and snares do in the reference kit. + +--- + +## 7. Engine-level guarantees (why it stays consistent, not drifting) + +1. **Bounded.** Every lever in §4 has a hard numeric ceiling. The pitch ceiling + (±6 cents) is enforced *after* `mul` and `organic` multiply, so no family or + knob setting can detune a note audibly. +2. **Seeded & reproducible.** `noise_seed = next_id * 2654435761u` is a pure + function of the trigger index — a given trigger is fully reproducible. Useful + for testing (assert two triggers differ, but a *replayed* trigger with the + same id is identical). +3. **Coherent within a note.** Parametric jitter is drawn **once** at note-on and + frozen. No per-sample re-rolling of pitch/amp/cutoff → no warble, no drift, no + divergence over a long held note. The only per-sample randomness is structural + excitation noise, which is *supposed* to be broadband. +4. **Zero-mean & independent.** Per-partial jitter averages to no spectral tilt, + so timbre identity is statistically preserved across many triggers even as any + single trigger is unique. +5. **One global escape hatch.** `organic = 0` restores exact, deterministic, + bit-identical synthesis for any context that needs it. + +--- + +## 8. Reusable C sketch (matches `audio.c` style) + +Add near the top of `audio.c`, alongside `xorshift32` (audio.c:100) and `clampd` +(audio.c:109). These are the helpers every per-instrument implementation calls at +note-on. They draw from the voice's own `noise_seed` so the draws are coherent +within the trigger and varied across triggers. + +```c +// ============================================================ +// Bounded per-note stochasticism (see docs/gm-synthesis/00-stochasticism.md) +// All draws come from the voice's per-trigger noise_seed (seeded in +// audio_synth from next_id). Call these ONCE at note-on and bake the +// result into voice params — never per-sample (phase-increment rule). +// ============================================================ + +// Global "organic" amount: 0 = bit-identical, 1 = max tasteful spread. +// Default 0.6 places the amplitude lever inside the percussion ±10-25% band. +// Lives on ACAudio; exposed via audio_set_organic(). A file-static mirror +// keeps the note-on helpers dependency-free in the inner code. +static double g_organic_amount = 0.6; + +void audio_set_organic(double amt) { + g_organic_amount = clampd(amt, 0.0, 1.0); +} + +// Uniform [0,1) from the voice PRNG. +static inline double voice_rand_unit(ACVoice *v) { + return (double)xorshift32(&v->noise_seed) / (double)UINT32_MAX; +} + +// Bipolar [-1,1] from the voice PRNG. The workhorse for every jitter lever. +static inline double voice_rand_bipolar(ACVoice *v) { + return voice_rand_unit(v) * 2.0 - 1.0; +} + +// Cents → frequency ratio. 1200 cents = 1 octave. cents_to_ratio(0)==1.0. +static inline double cents_to_ratio(double cents) { + return pow(2.0, cents / 1200.0); +} + +// ---- Lever helpers. `mul` is the per-family multiplier (§6, default 1.0). ---- + +// Bounded multiplicative jitter around `center` by ±`frac` (the percussion +// `rj` idiom): center * (1 ± frac*organic*mul*u). Use for amp, decay, attack, +// cutoff, FM index. Caller picks `frac` from the §4 table (amp 0.15, decay +// 0.15, attack 0.12, cutoff 0.05, fm index 0.06, ...). +static inline double voice_jitter(ACVoice *v, double center, + double frac, double mul) { + double u = voice_rand_bipolar(v); + return center * (1.0 + frac * g_organic_amount * mul * u); +} + +// Bounded pitch detune in cents → ratio, HARD-CAPPED at ±6 cents regardless +// of organic/mul so perceived pitch never moves audibly (invariant §4.1). +// `spread_cents` is the per-lever spread (default 6.0); pass the partial/voice +// frequency, get the detuned frequency back. +#define ORGANIC_MAX_CENTS 6.0 +static inline double voice_detune(ACVoice *v, double freq, + double spread_cents, double mul) { + double cents = spread_cents * g_organic_amount * mul * voice_rand_bipolar(v); + if (cents > ORGANIC_MAX_CENTS) cents = ORGANIC_MAX_CENTS; + if (cents < -ORGANIC_MAX_CENTS) cents = -ORGANIC_MAX_CENTS; + return freq * cents_to_ratio(cents); +} + +// Random start phase [0,1) for an additive/modal partial. Decorrelates +// stacked same-notes so they don't phase-cancel. Free + identity-safe. +static inline double voice_rand_phase(ACVoice *v) { + return voice_rand_unit(v); +} + +// Small bounded pan offset (±0.05 spread), added to the voice's base pan. +// Mirrors percussion's rn(-0.02..0.06) spread. Result clamped to [-1,1]. +static inline double voice_pan_jitter(ACVoice *v, double base_pan, double mul) { + double off = 0.05 * g_organic_amount * mul * voice_rand_bipolar(v); + return clampd(base_pan + off, -1.0, 1.0); +} +``` + +### 8.1 Usage at note-on (example: a modal mallet voice) + +```c +// In audio_synth(), after the base init, for WAVE_MALLET (family mul = 1.0): +// IMPORTANT: draw all PARAMETRIC jitter first, before any per-sample +// structural noise consumes the seed, so the downstream stream stays white. +const double mul = 1.0; // mallets/percussion: full organic (§6) +for (int m = 0; m < v->nmodes; m++) { + double fm = f0 * mode_ratio[m]; + fm = voice_detune(v, fm, 6.0, mul); // §4a ±≤6 cents + v->m_inc[m] = fm / sr; + v->m_amp[m] = voice_jitter(v, base_amp[m], 0.15, mul); // §4b ±15% + // Struck attack → keep partials phase-COHERENT for a crisp transient + // (§4 lever c exception). Use 0.0, NOT voice_rand_phase, for mallets. + v->m_phase[m] = 0.0; + double tau = voice_jitter(v, base_tau[m], 0.15, mul); // §4i ±15% decay + v->m_dec_mult[m] = exp(-1.0 / (tau * sr)); +} +v->pan = voice_pan_jitter(v, v->pan, mul); // §4k ±0.05 +// (Per-sample mallet-click noise then consumes noise_seed in the inner loop.) +``` + +For a sustained additive voice (organ, choir, supersaw) the only change is +`v->p_phase[k] = voice_rand_phase(v);` instead of `0.0`, because there is no +transient to keep coherent and phase decorrelation is desirable. + +--- + +## 9. Implementation checklist + +- [ ] Add `audio_set_organic()` + `g_organic_amount` (default 0.6) and a JS + binding so pieces / the global config can tune it. +- [ ] Extend the note-on seeding (audio.c:3359) so the new GM wave types + (`WAVE_EPIANO/MALLET/ORGAN/PLUCK/FM` etc.) also get `noise_seed` set. +- [ ] Add the §8 helpers next to `xorshift32`. +- [ ] In each new per-instrument note-on path, draw parametric jitter **first**, + apply the §6 family `mul`, and respect the §4 ranges. +- [ ] Keep all pitch jitter routed through `voice_detune` (hard ±6c cap). +- [ ] Verify the two invariants (§4.1): a tuner reads within ±5 cents; the + long-term spectral centroid is stable across many triggers. + +--- + +*This dossier (00) is the calibration spine for dossiers 01–04. Numbers anchored +to `percussion.mjs` (amp ±10–25%, decay ±18–25%, pan ±0.02–0.06, pitch 0) and the +engine's existing `xorshift32` + per-voice `noise_seed`. Cross-referenced against +all four family dossiers (01 piano/mallet/organ/guitar, 02 bass/strings/ensemble/ +brass, 03 reed/pipe/synthlead/synthpad, 04 synthfx/ethnic/percussive/soundfx).* diff --git a/fedac/native/docs/gm-synthesis/01-piano-mallet-organ-guitar.md b/fedac/native/docs/gm-synthesis/01-piano-mallet-organ-guitar.md new file mode 100644 index 0000000000..c1bbcf9e68 --- /dev/null +++ b/fedac/native/docs/gm-synthesis/01-piano-mallet-organ-guitar.md @@ -0,0 +1,662 @@ +# GM Synthesis Dossier 01 — Piano, Chromatic Percussion, Organ, Guitar + +Real-time **algorithmic** synthesis recipes for General MIDI programs 1–32, written +as an implementation spec for the Aesthetic Computer native C audio engine +(`fedac/native/src/audio.c`). No sample playback — every voice is generated from +oscillators, delay lines, and filters. Target: 44.1 / 48 / 192 kHz, 32-voice +polyphony, per-voice / per-sample render in `generate_sample()`. + +> **Scope note.** GM program numbers below are 1-based (Acoustic Grand = 1). The +> four families documented here are 1–32. Three later dossiers will cover the +> remaining 96 programs. + +--- + +## 0. How these fit the existing engine + +Before the per-instrument sections, here is the mapping to what already exists in +`audio.c`, so the implementer reuses primitives instead of inventing parallel +machinery. + +### 0.1 Existing voice infrastructure (read this first) + +The `ACVoice` struct (audio.h:68–228) already carries everything most of these +algorithms need: + +- **Phase accumulators** — `double phase`, advanced by `frequency / sample_rate` + in `generate_sample()`. The repo convention (and a hard-won lesson in + `MEMORY.md`: *"long sine phase-increment not sin(TAU*f*t)"*) is to advance a + phase register and read a **wavetable**, never call `sin()` per sample for + sustained tones. Additive partials must each carry their own phase register. +- **Delay-line buffers** — `float whistle_bore_buf[2048]` + `int whistle_bore_w` + (write cursor) and `float whistle_jet_buf[512]`. These are **reused per wave + type** (a voice is only ever one type at a time). The harp's Karplus-Strong + string lives in `whistle_bore_buf`; every plucked/struck-string instrument + here reuses the same field. There is also `whistle_frac_read()` for + fractional (interpolated) delay reads — essential for in-tune KS. +- **Filter state** — biquad fields (`noise_b0..a2`, `noise_x1..y2`) and one-pole + state (`harp_lp1`, `whistle_lp1`, `gun_bore_lp`). One-pole and biquad helpers + already exist (`setup_noise_filter`). The gun model shows the pattern for + **3 parallel biquad body modes** (`gun_body_a1[3]`, `gun_body_y1[3]`) — this + is *exactly* the modal-resonator pattern the mallet/bell family needs, just + more modes. +- **Envelopes** — `compute_envelope(v)` (attack / decay / duration / fade) and + per-sample exponential decay multipliers (the gun model's + `gun_*_decay_mult = exp(-1/(tau*sr))` idiom). Mallets and pianos want + per-partial exponential decay multipliers exactly like this. +- **Frequency smoothing** — `target_frequency` → `frequency` one-pole glide + already in the render loop. + +### 0.2 Reuse map (primitive → instrument family) + +| Existing primitive | Reused / extended by | +|---|---| +| `generate_harp_sample` (canonical KS + Jaffe-Smith stretch) | All guitars, harpsichord, clavi, dulcimer, music box damper | +| `whistle_bore_buf` fractional delay + `whistle_frac_read` | Every string model (KS / waveguide) + comb pluck-position filter | +| `gun_body_*[3]` parallel biquad bank | Mallets (vibraphone/marimba/glock/xylophone/tubular bells) modal banks, extended to 3–8 modes | +| Wavetable phase-increment oscillator | Organ drawbars (additive sines), FM operators (EPs, bells, clav) | +| `compute_envelope` + per-sample `exp` decay mult | Per-partial / per-mode decay on all struck voices | +| `setup_noise_filter` biquad | Hammer-noise thump, breath/wind for accordion/harmonica, pick-scrape | +| `drive_mix` / tanh soft-clip master FX | Overdrive/distortion guitar waveshaper (but apply **per-voice** for these) | + +### 0.3 Proposed new `WaveType` enum entries + +The cleanest fit is a small set of new families rather than 32 enum members. +Suggested additions (parameterised by a `gm_program` field on the voice that +selects the sub-variant — mirroring how `gun_preset` selects a `GunPresetParams` +row): + +```c +WAVE_EPIANO, // FM tine/reed (Rhodes, Wurli, DX, FM EP, clav-FM) +WAVE_MALLET, // modal bank (celesta..tubular bells, dulcimer) +WAVE_ORGAN, // additive drawbars + optional Leslie +WAVE_PLUCK, // extended-KS string (guitars, harpsichord, clavi) +WAVE_FM, // generic 2–4 op FM (bells/celesta/music-box voices) +``` + +Each carries a `const *Params` table row keyed by GM program, same pattern as +`gun_presets[GUN_PRESET_COUNT]`. Variant timbre lives in data, not code. + +--- + +# Family 1 — Piano (GM 1–8) + +**Family verdict:** The struck steel string is *the* hard case. Two viable +algorithmic routes: (a) **modal/additive with inharmonic stretched partials** +(cheap, the engine already had this — see audio.c:515 "Old modal-additive synth +removed"), or (b) **digital-waveguide / commuted synthesis** (Smith & Van Duyne +1995). For real-time 32-voice on the AC target, **modal additive with +inharmonicity** is the right default for acoustic pianos; the electric pianos +(5,6) want **FM tine models** (Chowning), and harpsichord/clavi (7,8) want +**extended Karplus-Strong** (Jaffe-Smith), because they are plucked/struck- +damped, not free-ringing struck strings. + +### Core acoustic-piano algorithm (shared by GM 1–4) + +**Inharmonicity.** A real piano string is stiff, so partials are stretched: + +``` +f_n = n · f0 · sqrt(1 + B·n²) (Fletcher & Rossing 1998, eq. 12.12) +``` + +`B` is the inharmonicity coefficient: ~0.0001 in the bass register growing to +~0.004 in the treble. This stretch is *the* perceptual signature of a piano vs. +a pure harmonic tone, and it's why detuned-octave "stretch tuning" sounds right. + +**Per-sample model** (10 partials, the count the engine previously used): + +```c +// state per voice: phase[10], amp[10], dec_mult[10], freq[10] +double s = 0.0; +for (int k = 0; k < NPART; k++) { + s += v->p_amp[k] * wt_sin(v->p_phase[k]); // wavetable sine, NOT sinf() + v->p_phase[k] += v->p_finc[k]; // precomputed inc = f_n/sr + if (v->p_phase[k] >= 1.0) v->p_phase[k] -= 1.0; + v->p_amp[k] *= v->p_dec_mult[k]; // exp decay, high partials die first +} +// + hammer thump: short LPF noise burst, ~5ms exp decay (reuse setup_noise_filter) +// + 3 mistuned "phantom" fundamentals (±0.3–1.5 cents) → inter-string beating +return (s + hammer) * compute_envelope(v); +``` + +Key parameters set at note-on (velocity → brightness): higher velocity adds +upper partials and increases hammer-noise gain. Decay multipliers: +`dec_mult[k] = exp(-1.0 / (tau_k * sr))` with `tau_k` shrinking for higher `k` +(treble partials T60 ≈ 0.3–1 s; fundamental T60 ≈ 5–20 s in bass). + +**References (already cited in audio.h:193–215):** +- Fletcher, H. & Rossing, T.D. (1998). *The Physics of Musical Instruments*, 2nd + ed., Springer, Ch. 12 (inharmonicity). +- Bank, B. (2000). *Physically-Based Sound Modeling of the Piano*, M.Sc. thesis, + BME Budapest. — modal/additive grand with stretched partials. +- Bank, B. & Välimäki, V. (2003). "Robust Loss Filter Design for Digital + Waveguide Synthesis of String Tones," *IEEE SP Letters* 10(1), 18–20. +- Smith, J.O. & Van Duyne, S.A. (1995). "Commuted Piano Synthesis," *Proc. ICMC*, + Banff. — hammer + soundboard impulse collapsed into the excitation. +- Smith, J.O. *Physical Audio Signal Processing*, CCRMA — https://ccrma.stanford.edu/~jos/pasp/ + +--- + +## GM 1 — Acoustic Grand Piano + +- **Method:** Modal additive, 10–12 inharmonic partials + hammer thump + 3-string + phantom detune. (Default `B` schedule above.) +- **Timbre params:** Bright attack, long sustain; `B` small in bass, full + velocity→partial map. Wide stereo from string position. +- **C sketch:** Shared core above; this is the reference voice. + +## GM 2 — Bright Acoustic Piano + +- **Method:** Same modal core. +- **Distinguishing params:** Boost upper-partial amplitudes (multiply `amp[k]` + for `k≥4` by ~1.3–1.6), slower treble decay, slightly **harder hammer** + (more noise-burst HF, shorter noise tau). Optional one-pole high-shelf on + output. + +## GM 3 — Electric Grand Piano + +- **Method:** Modal core but **fewer, cleaner partials** (6–8) with reduced + inharmonicity (`B` ≈ half of acoustic) — models a sampled/amplified CP-70-style + grand. Add a gentle tine-like 2nd-partial emphasis. +- **Distinguishing params:** Less hammer noise, tighter (more harmonic) spectrum, + a touch of the FM tine shimmer from GM 5 mixed at low level. Light tanh drive + (`drive_mix` per-voice ~0.1) for the "electrified" edge. + +## GM 4 — Honky-tonk Piano + +- **Method:** Modal core, **dual detuned voices**. The defining trait is two + strings per note deliberately mistuned. +- **Distinguishing params:** Render the partial set **twice** with the second + copy's `f_inc` scaled by ~±10–18 cents (`1.006`–`1.010`), summed. This + produces the characteristic slow chorus beating of a worn upright. Slightly + shorter sustain, more hammer clack. + +--- + +## GM 5 — Electric Piano 1 (Rhodes / tine) + +- **Method:** **FM tine model** (Chowning FM) — the authentic and cheap route. + The Rhodes tine is a struck asymmetric tuning-fork sensed by an electromagnetic + pickup whose **asymmetric nonlinearity** generates the bark. Two routes: + 1. *FM (recommended for CPU):* 2-operator FM. Carrier at `f0`, modulator at + `c:m ≈ 1:1` for body plus a **high-ratio modulator** (`m ≈ 14:1`) at low + index to inject the metallic "tine ping" on attack. Modulation index + **decays exponentially** so the bark is loud at attack then mellows to a + near-sine bell — exactly Chowning's bell envelope idea applied to a tine. + 2. *Physical (reference quality):* mass-spring tine + asymmetric pickup + waveshaper (Pfeifle & Bader, DAFx 2017). +- **FM recipe (per-sample):** + +```c +// carrier c, modulator m; both wavetable-sine, phase-increment +double mod = wt_sin(v->fm_mphase) * v->fm_index; // fm_index decays per-sample +double car = wt_sin(v->fm_cphase + mod); +v->fm_cphase += v->fm_cinc; v->fm_mphase += v->fm_minc; +v->fm_index *= v->fm_index_dec; // exp decay → mellowing +// + small attack "tine" operator at ~14*f0, its own fast-decaying index +return car * compute_envelope(v) * vel_gain; +``` + +- **Distinguishing params:** velocity → attack index (harder = more bark); + long bell-like decay; pickup asymmetry via mild `tanh(a + g·x)` (DC-offset bias + then DC-block) to add even harmonics — the Rhodes "growl." +- **Refs:** Chowning, J. (1973). "The Synthesis of Complex Audio Spectra by Means + of Frequency Modulation," *JAES* 21(7), 526–534. Pfeifle, F. & Bader, R. (2017). + "Real-Time Physical Model of a Wurlitzer and Rhodes Electric Piano," *Proc. + DAFx-17* — http://www.dafx17.eca.ed.ac.uk/papers/DAFx17_paper_79.pdf + +## GM 6 — Electric Piano 2 (Wurlitzer / DX-EP) + +- **Method:** Same FM-tine engine, **reed** flavour (Wurlitzer) or glassy DX-EP. +- **Distinguishing params vs GM 5:** Wurlitzer reed → **stronger asymmetric pickup + nonlinearity** (more even harmonics, hollow/reedy, faster decay), modulator + ratio nearer `c:m = 1:2`, more aggressive `tanh` bias. The "DX7 E.PIANO 1" + variant: 4-op stack, `c:m` whole-number ratios, crystalline attack, longer + release. Add a touch of the bell index for the FM-piano gloss. + +## GM 7 — Harpsichord + +- **Method:** **Extended Karplus-Strong** (plucked, not struck). Reuse + `generate_harp_sample` engine. The plectrum pluck is bright and the string is + hard-damped → short, even decay with characteristic **pluck-position comb**. +- **EKS additions over the bare harp:** + - *Pick-position comb:* `y = x - x[n - βN]`, `β ≈ 0.13` (pluck near the + bridge) → bright, nasal notches (Jaffe-Smith `H_β(z)=1 - z^{-⌊βN+0.5⌋}`). + - *Damping filter:* one-pole loop LPF with low cutoff so high partials persist + less than a guitar (brighter, drier than nylon). + - *Tuning allpass:* `H_η(z) = -(η - z^{-1})/(1 - η z^{-1})` for in-tune + fractional delay (already implicit in `whistle_frac_read`, but the allpass + is phase-flatter for sustained tones). +- **Distinguishing params:** very bright seed spectrum (don't pre-smooth the + noise as much as nylon), short-ish decay, the comb is the signature. Two + unison strings (8'+4') optional: render at `f0` and `2·f0`. +- **Refs:** Karplus & Strong 1983; Jaffe & Smith 1983 "Extensions of the + Karplus-Strong Plucked-String Algorithm," *CMJ* 7(2), 56–69; Smith *PASP* EKS + page — https://ccrma.stanford.edu/~jos/pasp/Extended_Karplus_Strong_Algorithm.html + +## GM 8 — Clavi (Clavinet) + +- **Method:** **Extended KS** with a struck-string excitation + **strong pickup + coloration**. A Clavinet string is struck by a rubber tangent against an anvil + → percussive, funky, twangy, with magnetic pickup like an electric guitar. +- **Distinguishing params:** pluck position near the very end (`β ≈ 0.05`) for a + thin bright tone; **fast decay + hard mute on note-off** (the Clavi's defining + staccato); pass the KS output through a `tanh` waveshaper for the wah-friendly + electric bite; optional FM-tine attack click. Velocity → brightness via the + dynamics LPF `H_L(z)=(1-R_L)/(1-R_L z^{-1})`, `R_L = e^{-πLT}`. + +--- + +# Family 2 — Chromatic Percussion (GM 9–16) + +**Family verdict:** Every one of these is a struck metal/wood bar, tube, or +string with a small number of strong, **inharmonic** modes and a percussive +exponential decay. The unanimous best method is **modal synthesis**: a bank of +parallel resonators (decaying sinusoids), one per measured mode, excited by a +short impulse/noise burst. This is the *same architecture as the gun body-mode +bank* (`gun_body_*[3]`) extended to N modes. It is extremely cheap (N phase- +increment sines × per-mode exp decay) and the **modal ratios are the entire +identity** of each instrument — so the data tables below are the spec. + +### Core mallet/modal algorithm (shared GM 9–16) + +```c +// Per voice: M modes. ratio[m] (×f0), amp[m], dec_mult[m], phase[m], inc[m]. +// Excitation: at note-on seed each mode's amplitude (= strike energy × mode +// participation); optionally add a 1–3 ms noise/mallet-click transient. +double s = 0.0; +for (int m = 0; m < v->nmodes; m++) { + s += v->m_amp[m] * wt_sin(v->m_phase[m]); + v->m_phase[m] += v->m_inc[m]; // inc = f0*ratio[m]/sr + if (v->m_phase[m] >= 1.0) v->m_phase[m] -= 1.0; + v->m_amp[m] *= v->m_dec_mult[m]; // exp decay; high modes faster +} +return s * compute_envelope(v); +``` + +> Alternative for very many modes / true impulse response: a parallel **biquad +> resonator bank** excited by a delta (the `gun_body` 2-pole form), each tuned to +> `f0·ratio[m]` with `Q` set from desired T60. Equivalent output; biquads are +> better when you want to re-excite (rolls). For pure struck one-shots the +> decaying-sinusoid additive form above is cheaper and click-free. + +**References (modal ratios):** +- Fletcher & Rossing (1998), *The Physics of Musical Instruments*, Ch. 18–21 + (bars, plates, bells). +- CCRMA Music 150/152 percussion notes — https://ccrma.stanford.edu/CCRMA/Courses/150/percussion.html +- Cook, P.R. (2002). *Real Sound Synthesis for Interactive Applications*, AK + Peters — modal synthesis chapters + STK `ModalBar`. +- Bilbao, S. (2009). *Numerical Sound Synthesis*, Wiley — bar/plate FD models + (reference, heavier than needed here). + +--- + +## GM 9 — Celesta + +- **Method:** Modal — struck steel bar over a resonator box. Near-harmonic, + bell-like but gentle. Equivalent: **2-op FM** (DX "CELESTA"). +- **Modal ratios:** approx `1.0, 4.0, ~10.8` (steel bar, like a soft + glockenspiel) with the upper modes weak. Decay moderate (T60 ~1–2 s). +- **FM alt:** `c:m = 1:4`, low index, fast index decay → clean bell ping. +- **Params:** soft mallet → small HF content; gentle attack click. + +## GM 10 — Glockenspiel + +- **Method:** Modal — short steel bar, brilliant. +- **Modal ratios (transverse bar, free-free):** **`1.0 : 2.76 : 5.40 : 8.90`** + (the classic free-free bar eigenvalue ratios; CCRMA). Only the fundamental + rings long; upper modes die fast. +- **Params:** very bright strike, high-pitched (G5–C8), short-to-medium decay, + hard mallet → strong attack transient. T60 of mode 0 ~1–3 s, others < 0.3 s. + +## GM 11 — Music Box + +- **Method:** Modal — plucked steel comb tooth. Like a tiny glockenspiel with a + **plucked** (not struck) excitation → softer attack, pure-ish tone. +- **Modal ratios:** dominant fundamental + weak `~6.3, ~17` overtones (cantilever + beam, clamped-free: `1.0 : 6.27 : 17.55 : ...`). Cantilever ratios, not + free-free — this is what makes a music box sound thinner/purer than a glock. +- **Params:** soft pluck transient, medium decay, slight per-note level jitter + for the mechanical feel; add faint comb/mechanism noise. + +## GM 12 — Vibraphone + +- **Method:** Modal — deeply undercut **aluminium** bar (long decay) + tremolo. +- **Modal ratios:** **`1.0 : 4.0 : ~9.6`** (first overtone tuned to *two octaves*, + the deliberate vibe tuning; CCRMA). Aluminium → very long decay. +- **Params:** long T60 (3–8 s), **amplitude tremolo** from the motor-driven + resonator discs — implement as a slow LFO on output gain (~4–7 Hz, depth + ~0.3), the vibraphone's signature. Soft yarn mallet → minimal HF. + +## GM 13 — Marimba + +- **Method:** Modal — undercut **rosewood/synthetic** bar + tube resonator. +- **Modal ratios:** **`1.0 : 4.0 : 9.2`** (second partial at two octaves, third + ~3 octaves + minor third; CCRMA). Wood → faster decay than vibe. +- **Params:** warm, dark (low HF), medium-short decay (T60 ~0.5–1.5 s); tube + resonator emphasises the fundamental → strong mode-0 amplitude, weak uppers. + +## GM 14 — Xylophone + +- **Method:** Modal — less-undercut wooden bar, bright and dry. +- **Modal ratios:** **`1.0 : 3.0 : ~6.0`** (first overtone at the *twelfth*, + ratio 3.0 — the xylophone tuning, vs marimba's 4.0; CCRMA). +- **Params vs marimba:** higher first-overtone (3.0 not 4.0) → harder, woodier; + shorter decay; brighter strike transient. This single ratio difference (3 vs 4) + is the marimba/xylophone distinction. + +## GM 15 — Tubular Bells (Chimes) + +- **Method:** Modal — long brass tube. **Strongly inharmonic**, with a *phantom* + strike pitch. +- **Modal ratios:** the audible strike tone arises from modes **4:5:6 ≈ 2:3:4**, + so the ear infers a fundamental an octave **below** mode 4 (missing-fundamental + effect). Practical mode set (relative to perceived pitch): approx + **`2.0 : 3.0 : 4.16 : 5.43 : 6.79 : 8.21`** (brass tube; the cluster is what + makes chimes shimmer). Very long decay. +- **FM alt (classic):** Chowning bell — inharmonic `c:m` ratio with + exponentially-decaying index; or Hind's 3-pair tubular-bell FM + (`c:m` pairs `2.0:5.0`, `0.6:4.8`, `0.22:0.83`). 2-op `m ≈ 3.5·c` is the + quick recipe (iastate / SOS). +- **Params:** long T60 (5–15 s), metallic clang transient, slow beating between + near-degenerate modes. + +## GM 16 — Dulcimer (Hammered Dulcimer) + +- **Method:** **Karplus-Strong** (struck string, multiple unison courses) — not + modal. Reuse `generate_harp_sample`. A hammered dulcimer is a struck *string*, + so it rings harmonically with a bright metallic attack. +- **Params:** 2–3 detuned unison strings per note (render KS 2–3× with ±2–6 cents + → shimmer/beating), hard-mallet bright seed, medium decay, no damping + (strings ring freely → use the long-sustain KS stretch `S ≈ 0.999`). Slight + comb from strike position. + +--- + +# Family 3 — Organ (GM 17–24) + +**Family verdict:** Two sub-mechanisms. The **electric/Hammond organs (17–19)** +are pure **additive synthesis of sine drawbars** (the original instrument *is* +additive synthesis) — cheap and exact with phase-increment wavetable sines, plus +key-click and optional Leslie. The **pipe/free-reed organs (20–24)** are +better as **subtractive** (bandlimited sawtooth/pulse through formant filters) +for church/reed pipes, and **free-reed additive+detune** for accordion/harmonica +(beating reed banks). All are **sustained** voices (no decay) — the cleanest fit +to the engine's existing sustained-oscillator path. + +### Core drawbar (Hammond) algorithm — shared GM 17–19 + +Nine drawbars, fixed harmonic ratios (Electric Druid): + +| Drawbar | 16′ | 5⅓′ | 8′ | 4′ | 2⅔′ | 2′ | 1⅗′ | 1⅓′ | 1′ | +|---|---|---|---|---|---|---|---|---|---| +| Ratio ×f0 | **0.5** | **1.5** | **1.0** | **2.0** | **3.0** | **4.0** | **5.0** | **6.0** | **8.0** | + +```c +// 9 sine partials at the fixed ratios above, each with a drawbar amplitude +// (registration). All phase-increment wavetable sines — sustained, no decay. +double s = 0.0; +for (int d = 0; d < 9; d++) { + s += v->drawbar_amp[d] * wt_sin(v->db_phase[d]); + v->db_phase[d] += v->db_inc[d]; // inc = f0*ratio[d]/sr + if (v->db_phase[d] >= 1.0) v->db_phase[d] -= 1.0; +} +// key-click: 2–5 ms noise/HF burst at note-on (contact bounce); percussion: see GM18 +return s * compute_envelope(v); // attack/release short; full sustain +``` + +**Refs:** Hammond drawbar/tonewheel: Electric Druid "Technical aspects of the +Hammond Organ" — https://electricdruid.net/technical-aspects-of-the-hammond-organ/ ; +Hammond organ (Wikipedia) — pseudo-harmonic tonewheel series. Leslie: Doppler ++ tremolo + sideband modulation (rotating horn fast ~6.9 Hz / slow ~0.8 Hz, +bass drum ~5.7 / 0.7 Hz). + +## GM 17 — Drawbar Organ + +- **Method:** Additive 9 drawbars. Classic full registration (e.g. `88 8000 000` + or `888 000 000`). Slight tonewheel leakage (faint inharmonic hum) + key-click. +- **Params:** registration table; small per-wheel detune for the tonewheel + "swirl"; optional slow Leslie. + +## GM 18 — Percussive Organ + +- **Method:** Drawbar additive + **Hammond percussion**: a single extra harmonic + (2nd `4′` or 3rd `2⅔′`) added at note-on with a **fast exponential decay** + (the "ping" of B3 percussion, ~0.2–0.6 s), non-retriggering on legato. +- **Params:** percussion harmonic select (2nd/3rd), fast/slow decay, soft/normal + volume — the four front-panel B3 percussion switches. + +## GM 19 — Rock Organ + +- **Method:** Drawbar additive driven into **overdrive** + fast Leslie. The rock + sound is a Hammond through a cranked Leslie/amp. +- **Params:** full drawbars + `tanh` waveshaping (per-voice `drive_mix` ~0.4–0.7) + for grit, fast Leslie (horn ~6.9 Hz) with chorus/vibrato. Add upper-drawbar + emphasis for bite. + +### Leslie rotary (shared, optional, for 17–19) + +Apply as a **per-voice or bus** effect: amplitude tremolo (LFO on gain) + +Doppler pitch wobble (tiny modulated delay line — the `whistle_jet_buf` or +`wobble_buf` ring is ideal) + a slight comb. Two rotors at different rates (horn +faster than bass drum), with fast/slow (chorale/tremolo) speed switch and +spin-up/spin-down inertia. + +## GM 20 — Church Organ (Pipe) + +- **Method:** **Additive** of (near-)harmonic pipe ranks, OR subtractive: sum of + several octave-spaced ranks (8′ + 4′ + 2′ + mixtures) each a soft sine/triangle, + with a breathy attack chiff. Flue-pipe tone is close to a few harmonics. +- **Params:** many ranks → big, slow attack (chiff = short noise burst), very + slight detune between ranks for the cathedral shimmer, long reverb-friendly + sustain, no tremolo. Add 2⅔′/1⅗′ mutation ranks for the bright plenum. + +## GM 21 — Reed Organ + +- **Method:** **Free-reed additive/subtractive** — buzzy sustained tone. Sum of + harmonics with a sawtooth-ish spectrum through a gentle lowpass + a hint of + reed beating (two slightly detuned copies). +- **Params:** static spectrum, soft attack, mild breath noise, no tremolo. Less + bright than accordion. + +## GM 22 — Accordion + +- **Method:** **Free-reed**: multiple reed banks slightly detuned (the "musette" + wet tuning). Subtractive sawtooth/pulse + detuned unison stack. +- **Params:** 2–3 detuned oscillators per note (±10–25 cents — the musette + shimmer is the identity), bellows-driven gentle amplitude swell, breath noise, + bright buzzy spectrum (saw → lowpass). + +## GM 23 — Harmonica + +- **Method:** **Free-reed**, single/dual reed, **breath-driven**. Subtractive + saw/pulse + strong breath-noise component + amplitude tremolo from breath. +- **Params:** prominent airy breath noise (filtered noise mixed in), expressive + attack/vibrato, slight pitch bend on attack ("draw" bend), narrow bright + spectrum, hand-wah lowpass optional. + +## GM 24 — Tango Accordion (Bandoneon) + +- **Method:** Same free-reed engine as GM 22, **drier/sharper** tuning. +- **Params vs accordion:** less wet detune (tighter musette or dry), more + reed-buzz HF, sharper bellows attack, the characteristic bandoneon "bite." + Often dual-reed at the octave. + +--- + +# Family 4 — Guitar (GM 25–32) + +**Family verdict:** All eight are **plucked steel/nylon strings** → the +unanimous best method is **Extended Karplus-Strong / digital waveguide** +(Jaffe-Smith), which the engine already implements as `WAVE_HARP`. The whole +family is one shared EKS string model plus per-variant **excitation +(pick/finger), pluck position, damping, and an output nonlinearity** (clean → +muted → overdrive → distortion is a *waveshaping continuum* on the same string). +This is the most reuse-dense family in the dossier: one engine, eight data rows. + +### Core guitar string (shared GM 25–32) — extends `generate_harp_sample` + +```c +// String = KS delay line in whistle_bore_buf, length = sr/f0 (fractional). +// Per sample (extends the harp loop): +double x = whistle_frac_read(buf, N, w, string_delay); // delayed sample +// -- pick-position comb (Jaffe-Smith H_β): subtract a tap at βN -- +double picked = x - v->pick_amt * read_tap(buf, w, beta*string_delay); +// -- loop damping LPF (one-pole; cutoff = brightness; lower => darker) -- +double damp = (1.0 - v->loop_b) * picked + v->loop_b * v->harp_lp1; +v->harp_lp1 = damp; +double y = v->stretch * damp; // S<1 decay (EKS) +buf[w] = (float)y; w = (w+1)%N; +// -- output nonlinearity (clean..distortion continuum) -- +double out = (v->drive > 0.0) ? tanh_wt(v->pre*y)*v->post : y; +return out * compute_envelope(v); +``` + +**Excitation at note-on:** seed the delay line with one wavelength of shaped +noise/impulse. *Pick* (steel/electric) = brighter, sharper seed (less pre- +smoothing). *Finger/nylon* = pre-smoothed, rounder seed (more lowpass passes on +the noise). Pick-direction lowpass `H_p(z)=(1-p)/(1-p z^{-1})`. + +**Refs:** Karplus & Strong 1983; Jaffe & Smith 1983 (EKS — pick position, +dynamics, stretch); Smith *PASP* — Karplus-Strong & EKS pages +(https://ccrma.stanford.edu/~jos/pasp/). Distortion/feedback via nonlinear +shaping of KS output: Karplus-Strong (Wikipedia, "electric guitar / feedback" +section) — output through tanh waveshaper modelling an overdriven amp, with a +feedback delay back into the string for sustained-feedback notes; Sullivan, C. +(1990). "Extending the Karplus-Strong Algorithm to Synthesize Electric Guitar +Timbres with Distortion and Feedback," *CMJ* 14(3), 26–37 (the canonical paper +for distortion/feedback guitar — *this is the key citation for GM 28–32*). + +## GM 25 — Acoustic Guitar (Nylon) + +- **Method:** EKS, **finger** excitation, body resonance. +- **Params:** pre-smoothed soft seed (mellow attack), moderate pluck position + (`β ≈ 0.13`), darker loop LPF, medium-long decay, **body resonance** (1–3 + biquad modes ~100/200 Hz — reuse `gun_body`) for the wooden box. No drive. + +## GM 26 — Acoustic Guitar (Steel) + +- **Method:** EKS, **pick** excitation, bright body. +- **Params vs nylon:** brighter seed (sharper pick), brighter loop LPF (more HF + sustain), pluck nearer bridge (`β ≈ 0.1`), metallic ring, stronger body + resonance / sympathetic shimmer. Optional pick-scrape noise transient. + +## GM 27 — Electric Guitar (Jazz) + +- **Method:** EKS, pick, **dark** magnetic-pickup tone, no drive. +- **Params:** warm/dark loop LPF (rolled-off highs — neck humbucker tone), + medium decay, light compression feel (softer dynamics curve), no body box + (solid/hollow electric → minimal acoustic resonance). Mellow, round. + +## GM 28 — Electric Guitar (Clean) + +- **Method:** EKS, pick, brighter than jazz, no drive. +- **Params:** brighter loop LPF than jazz (bridge-pickup sparkle), slightly + shorter decay, a hint of chorus optional. Glassy Strat-clean. + +## GM 29 — Electric Guitar (Muted) + +- **Method:** EKS, pick, **palm-mute** = very short decay + lowpass. +- **Params:** the muted variant = **strong loop damping + low stretch + `S`** (fast decay, T60 ~0.1–0.25 s) + aggressive lowpass → the percussive + "chunk." Pluck near bridge, short noisy attack. Often plays staccato. + +## GM 30 — Overdrive Guitar + +- **Method:** EKS string → **soft `tanh` waveshaper** (mild drive) → optional + feedback tap. +- **Params:** moderate `drive` (pre-gain ~3–5× into `tanh`, post-attenuate), + brighter sustain, slight feedback delay (a fraction of output back into the + string buffer → blooming sustain). Compression of dynamics. Per Sullivan 1990. + +## GM 31 — Distortion Guitar + +- **Method:** EKS string → **hard waveshaper** (heavy clipping) + feedback. +- **Params vs overdrive:** much higher pre-gain into `tanh`/hard-clip (saturated, + square-ish), pre-emphasis EQ before the shaper (boost mids), strong feedback + for self-sustaining notes, longer sustain. Power-chord friendly. Sullivan 1990 + is the reference (distortion **and** feedback). + +## GM 32 — Guitar Harmonics + +- **Method:** EKS string with the excitation **forced to a node** — a pluck- + position comb at a harmonic node kills the fundamental, leaving a pure high + partial (natural harmonic). Equivalently, seed the delay line at half/third + length, or set `β` to a node and boost the surviving partial. +- **Params:** very pure, bell-like high tone (touch a node at 12th/7th/5th fret → + 2nd/3rd/4th harmonic). Implement by driving the string at the harmonic + frequency with strong loop sustain and a node comb that suppresses the + fundamental; long ringing, glassy, slight chime. Often with light drive. + +--- + +## Appendix A — Consolidated modal-ratio tables (the data that *is* the timbre) + +| GM | Instrument | Mode ratios (×f0) | Notes | +|---|---|---|---| +| 10 | Glockenspiel | 1.0, 2.76, 5.40, 8.90 | free-free steel bar | +| 11 | Music Box | 1.0, 6.27, 17.55 | clamped-free (cantilever) tooth | +| 12 | Vibraphone | 1.0, 4.0, 9.6 | undercut Al, octave-octave tuning | +| 13 | Marimba | 1.0, 4.0, 9.2 | undercut wood, tube resonator | +| 14 | Xylophone | 1.0, 3.0, 6.0 | first overtone = twelfth (3.0) | +| 15 | Tubular Bells | 2.0, 3.0, 4.16, 5.43, 6.79, 8.21 | strike modes 4:5:6≈2:3:4, phantom f0 | +| 9 | Celesta | 1.0, 4.0, 10.8 | soft steel bar (weak uppers) | + +## Appendix B — Hammond drawbar ratios (GM 17–19) + +`16′=0.5, 5⅓′=1.5, 8′=1.0, 4′=2.0, 2⅔′=3.0, 2′=4.0, 1⅗′=5.0, 1⅓′=6.0, 1′=8.0` + +## Appendix C — Method selection summary + +| Family | Programs | Primary method | Engine basis | +|---|---|---|---| +| Acoustic piano | 1–4 | Modal additive + inharmonicity (`f_n=n·f0·√(1+Bn²)`) | wavetable partials + per-partial exp decay | +| Electric piano | 5–6 | 2–4 op FM tine/reed + asym. pickup `tanh` | FM operators, `drive_mix` | +| Harpsichord/Clavi | 7–8 | Extended Karplus-Strong (pluck-position comb) | `generate_harp_sample` | +| Chromatic perc. | 9–15 | Modal resonator bank (measured ratios) | `gun_body` biquads → N modes / decaying sines | +| Dulcimer | 16 | Karplus-Strong (struck unison strings) | `generate_harp_sample` | +| Hammond organ | 17–19 | Additive 9-drawbar sines + Leslie + click | wavetable sines + `wobble_buf` Leslie | +| Pipe/reed organ | 20–24 | Additive ranks / subtractive free-reed + detune | sines/saw + biquad formants | +| Guitar (all) | 25–32 | Extended KS string + per-voice waveshaper | `generate_harp_sample` + per-voice `tanh` + feedback | + +## Appendix D — Full reference list + +- Karplus, K. & Strong, A. (1983). "Digital Synthesis of Plucked-String and Drum + Timbres," *Computer Music Journal* 7(2), 43–55. +- Jaffe, D.A. & Smith, J.O. (1983). "Extensions of the Karplus-Strong Plucked- + String Algorithm," *CMJ* 7(2), 56–69. +- Smith, J.O. *Physical Audio Signal Processing* (online, CCRMA). KS: + https://ccrma.stanford.edu/~jos/pasp/Karplus_Strong_Algorithm.html ; EKS: + https://ccrma.stanford.edu/~jos/pasp/Extended_Karplus_Strong_Algorithm.html +- Sullivan, C.R. (1990). "Extending the Karplus-Strong Algorithm to Synthesize + Electric Guitar Timbres with Distortion and Feedback," *CMJ* 14(3), 26–37. +- Chowning, J.M. (1973). "The Synthesis of Complex Audio Spectra by Means of + Frequency Modulation," *JAES* 21(7), 526–534. +- Pfeifle, F. & Bader, R. (2017). "Real-Time Physical Model of a Wurlitzer and + Rhodes Electric Piano," *Proc. DAFx-17*, Edinburgh — + http://www.dafx17.eca.ed.ac.uk/papers/DAFx17_paper_79.pdf +- Fletcher, H. & Rossing, T.D. (1998). *The Physics of Musical Instruments*, 2nd + ed., Springer (inharmonicity Ch.12; bars/plates/bells Ch.18–21). +- Bank, B. (2000). *Physically-Based Sound Modeling of the Piano*, M.Sc. thesis, + BME Budapest. +- Bank, B. & Välimäki, V. (2003). "Robust Loss Filter Design for Digital + Waveguide Synthesis of String Tones," *IEEE Signal Processing Letters* 10(1), + 18–20. +- Smith, J.O. & Van Duyne, S.A. (1995). "Commuted Piano Synthesis," *Proc. ICMC*, + Banff. +- Cook, P.R. (2002). *Real Sound Synthesis for Interactive Applications*, AK + Peters (modal synthesis; STK ModalBar / Mandolin / Plucked). +- Välimäki, V., Pakarinen, J., Erkut, C. & Karjalainen, M. (2006). "Discrete-time + modelling of musical instruments," *Reports on Progress in Physics* 69, 1–78 + (survey covering KS, waveguides, modal, FM). +- Bilbao, S. (2009). *Numerical Sound Synthesis*, Wiley (FD bar/plate models). +- Hammond/Leslie: Electric Druid, "Technical aspects of the Hammond Organ" — + https://electricdruid.net/technical-aspects-of-the-hammond-organ/ ; Hammond + organ — Wikipedia (tonewheel pseudo-harmonic series). +- CCRMA Music 150/152 percussion modal notes — + https://ccrma.stanford.edu/CCRMA/Courses/150/percussion.html + +--- + +*32 / 32 GM programs documented (programs 1–32). Next dossier: 33–64 (Bass, +Strings, Ensemble, Brass).* diff --git a/fedac/native/docs/gm-synthesis/02-bass-strings-ensemble-brass.md b/fedac/native/docs/gm-synthesis/02-bass-strings-ensemble-brass.md new file mode 100644 index 0000000000..a22cd29805 --- /dev/null +++ b/fedac/native/docs/gm-synthesis/02-bass-strings-ensemble-brass.md @@ -0,0 +1,701 @@ +# GM Synthesis Dossier 02 — Bass, Strings, Ensemble, Brass (programs 33–64) + +Real-time algorithmic synthesis recipes for 32 General MIDI instruments, targeted +at the Aesthetic Computer native audio engine (`fedac/native/src/audio.c`). No +samples — everything here is a per-voice, per-sample state machine that runs +inside `generate_sample()` at up to 192 kHz across 32 voices. + +## How this maps onto the existing AC engine + +The engine already ships three physical-model generators that are the direct +templates for everything below. Read them first; the new instruments are +variations on these, not new infrastructure: + +| Existing generator | Technique | Reuse target | +| --- | --- | --- | +| `generate_whistle_sample` (audio.c:177) | Cook/STK waveguide flute: bore delay + jet delay + cubic limit-cycle NL + 1-pole loss LPF + DC blocker | **Bowed strings** (swap jet+cubic for a bow-table friction junction) and **brass** (swap cubic for a lip-filter quadratic NL) | +| `generate_harp_sample` (audio.c:300) | Karplus-Strong: noise-seeded delay line + 2-point averaging loss filter + Jaffe-Smith stretch | **All plucked/struck strings**: acoustic/electric/fretless/slap bass, pizzicato strings, harp | +| `generate_gun_sample` / classic layers (audio.c:1216) | Layered transient + parallel body biquads + Friedlander excitation | **Timpani** (modal biquads + noise transient) and **orchestra hit** (cluster + transient) | + +Key shared primitives already present and to be reused verbatim: + +- `whistle_frac_read(buf, N, w, delay)` — fractional-delay ring read (audio.c:140). The fundamental waveguide/KS read. +- `xorshift32(&v->noise_seed)` — cheap white noise. +- `compute_envelope(v)` — attack / sustain / decay / kill-fade envelope (audio.c:~110). +- The biquad pattern in `WAVE_NOISE` (audio.c:1485) and `setup_noise_filter` — a ready transposed-DF2 biquad with state `noise_x1/x2/noise_y1/y2`. +- Phase-increment oscillators (`v->phase += f/sr`) — the repo standard; **never** `sin(TAU*f*t)`. +- The shared `whistle_bore_buf[2048]` / `whistle_jet_buf[512]` ring buffers — a voice is exactly one wave type at a time, so these are the scratch delay lines for any new waveguide/KS model. + +Because the `ACVoice` struct is a tagged union-by-convention (each model squats on +the same buffers), new models cost almost no extra per-voice memory: a handful of +doubles plus a couple of small biquad state arrays. A practical plan is to add a +small number of new `WaveType` enum entries — e.g. `WAVE_BOWED`, `WAVE_BRASS`, +`WAVE_VOICE` (formant), `WAVE_SUPERSAW`, `WAVE_MODAL` — and select the GM-specific +*variant* via a per-voice `int gm_program` (or a `model_variant` byte) read inside +the generator. Variant differences (finger vs pick bass, muted vs open trumpet, +violin vs cello) are almost always just parameter values, not new code paths. + +### Performance budget + +At 192 kHz × 32 voices the per-sample cost ceiling is ~6.1 M voice-ticks/s. The +existing whistle/harp models already hit this with one fractional read + a 1-pole +filter + a cubic, so the headroom rule is: **one delay-line read, ≤2 biquads, ≤1 +transcendental (or a table lookup) per voice per sample.** Formant voices (the +most expensive here) use 3–4 parallel biquads — still fine for the ~8–16 voices a +choir patch realistically needs. Supersaw uses 7 phase-increment saws (trivial). +Where a `sin`/`pow` appears in a hot loop below, prefer a 512/1024-entry LUT +(matching `AUDIO_WAVEFORM_SIZE` conventions) or the polynomial approximations the +gun code already favors. + +--- + +# Family A — Bass (GM 33–40) + +All eight basses are **plucked/struck strings**, so the foundation is the existing +Karplus-Strong harp generator (audio.c:300), extended per Jaffe & Smith 1983. +Synth basses (39–40) are the exception: subtractive oscillator + filter. + +**Shared KS recipe (extended Karplus-Strong / "EKS"):** + +``` +delay length N = sr / freq (one wavelength) +loss filter y[n] = g·(h0·x[n] + h1·x[n-1]) (2-point averaging, unity-DC) +stretch / decay g = S (S<1 sets T60; high notes need S→1) +pluck excitation seed N samples of (optionally lowpassed) white noise +pluck position comb multiply seed spectrum by (1 - z^-βN), β = pluck point (0..0.5) +dynamics→brightness excitation LPF cutoff ∝ velocity (louder = brighter) +``` + +The pluck-position comb filter (Jaffe & Smith 1983, §"Pluck Position") is the +single most important timbral control for bass: plucking near the bridge (small +β ≈ 0.1) gives a thin, bright, harmonically-rich tone; plucking over the +fingerboard (β ≈ 0.3–0.4) gives the round, dark fundamental-heavy tone. It is +implemented for free as a 1-zero comb on the seed: `seed[n] -= seed[n - round(βN)]`. + +References for the whole family: +- Karplus, K. & Strong, A. (1983). "Digital Synthesis of Plucked-String and Drum Timbres," *Computer Music Journal* 7(2), 43–55. +- Jaffe, D. A. & Smith, J. O. (1983). "Extensions of the Karplus-Strong Plucked-String Algorithm," *CMJ* 7(2), 56–69. +- Smith, J. O. *Physical Audio Signal Processing*, CCRMA — https://ccrma.stanford.edu/~jos/pasp/Karplus_Strong_Algorithm.html +- Karplus–Strong overview — https://en.wikipedia.org/wiki/Karplus%E2%80%93Strong_string_synthesis + +### 33 — Acoustic Bass + +Method: **Karplus-Strong, dark variant.** Upright/double bass is gut/round-wound on +a large resonant body. Use a low-cutoff excitation (pluck the felt finger, not a +pick) and a body resonator. + +- `N = sr/freq`; freq range ~41 Hz (E1) to ~250 Hz. At 192 kHz E1 ⇒ N≈4680 — **exceeds the 2048 bore buffer**; either grow the buffer to 8192 for bass voices or run bass voices at a /4 internal rate (48 kHz) and upsample. Recommend a dedicated 8192-sample buffer for KS bass voices. +- Loss filter: 2-point average `0.5(x[n]+x[n-1])`, stretch `S≈0.996` for a ~1–2 s plucked decay. +- Pluck position β ≈ 0.35 (over fingerboard) → fat fundamental. +- Excitation: white noise through a 1-pole LPF at ~1.5 kHz before seeding → soft thumb attack, very little high-frequency "zing." +- Body resonance: one parallel biquad bandpass at ~90–110 Hz (the big air mode of the corpus), low gain, fed by the string output. Reuse the `gun_body_*` biquad slots. +- Key timbre params: low excitation cutoff, high β, body BP at ~100 Hz, no attack click. + +### 34 — Electric Bass (finger) + +Method: **KS, round-wound flatwound variant** — brighter than upright, with a +short attack thump but a sustained, slightly metallic ring (magnetic pickup ≈ a +fixed comb + gentle high-mid emphasis). + +- Same N/loss as acoustic, but stretch `S≈0.998` (longer sustain, electric strings ring) and pluck position β ≈ 0.25. +- Excitation LPF cutoff ~3 kHz, velocity-scaled (`cutoff = 1500 + 4000·vel`). +- Add a fixed **pickup comb**: tap the delay line at a second point ~10–15% along its length and sum (models the magnetic pickup's position picking up a node-weighted spectrum) — cheap second `whistle_frac_read`. +- Finger attack: a tiny (~3 ms) noise burst LPF'd at ~800 Hz layered at note-on = the fingertip "thp". +- Distinguish from pick (35): softer attack burst, lower excitation cutoff, β larger. + +### 35 — Electric Bass (pick) + +Method: **KS, bright pick variant.** The plectrum injects a sharper, broadband +transient and excites the string nearer the bridge. + +- Pluck position β ≈ 0.12 (near bridge) → bright, hollow, lots of upper harmonics. +- Excitation: *less* lowpassing — cutoff ~6 kHz, plus a 1-sample bipolar "click" prepended to the seed (the pick release transient). Reuse the gun click-layer idea (`gun_click_*`). +- Stretch `S≈0.997`; slightly faster decay than finger. +- Loss filter weighted toward brightness: use `0.6·x[n]+0.4·x[n-1]` (less HF damping than the symmetric average). +- Key timbre params vs finger: small β, sharp transient click, higher excitation cutoff, brighter loss filter. + +### 36 — Fretless Bass + +Method: **KS with continuous-glide pitch + "mwah" lowpass.** Fretless = the same +string model as finger bass, but (a) pitch can glide (no fret quantization) and +(b) the characteristic resonant "mwah" from the string pressing directly on the +fingerboard wood = a *time-varying* lowpass that opens slightly after attack. + +- Base = electric finger (34) with β ≈ 0.3. +- Portamento: smooth `frequency → target_frequency` already exists (audio.c:1514) — slow the slew (use ~0.0008 instead of 0.0003) so legato notes audibly glide. +- "Mwah" filter: a resonant 1-pole/biquad LPF on the *output* whose cutoff envelopes from ~700 Hz up to ~2.5 kHz over ~120 ms at note-on, Q≈3. This is the defining fretless timbre. +- Slightly longer stretch `S≈0.9985` and a touch of slow vibrato (reuse `whistle_vibrato_phase`, ~5 Hz, depth ±0.3%). +- Key timbre params: glide slew, attack-swept resonant LPF, vibrato. + +### 37 — Slap Bass 1 + +Method: **KS + dual excitation (thumb slap + snap).** Slap is two gestures — +*thumb* striking the string against the frets (percussive, broadband, with a +metallic "clack" from string-on-fret) and the string's normal ringing. + +- Base KS with β ≈ 0.2, `S≈0.997`. +- Thumb-slap transient: a short (~5 ms) noise burst **bandpassed at ~2–3 kHz** (string slapping the fret) summed at note-on, plus a sub-ms HF click. This is the "clack." +- Fret-buzz: briefly raise the loss-filter gain toward 1.0 for the first ~20 ms so the string rings against the fret with extra high-harmonic energy, then settle. Equivalent to momentarily reducing damping. +- Strong velocity→brightness coupling (slap is dynamically extreme). +- Key timbre params: prominent bandpassed clack, transient under-damping, β small-to-mid. + +### 38 — Slap Bass 2 + +Method: **KS slap, "pop" variant.** Slap 2 in most GM sets is the *popped* (pulled) +string — the finger yanks the string and lets it snap back against the fingerboard. + +- Same engine as Slap 1 but: pop transient is **brighter and tighter** — noise burst bandpassed higher (~3–5 kHz), shorter (~3 ms), with a hard click. +- Pluck position β ≈ 0.1 (pulled near bridge) → very bright. +- Faster decay (`S≈0.995`) and a more pronounced fret-slap impact (the snap-back). Add a secondary excitation ~8–12 ms after the first (string returning and hitting the wood) — reuse the gun `gun_secondary_trig` mechanism. +- Key timbre params vs Slap 1: higher/tighter pop band, secondary snap-back impact, smaller β. + +### 39 — Synth Bass 1 + +Method: **Subtractive — single/dual saw or square through a resonant 24 dB/oct +lowpass with an envelope-swept cutoff.** This is the classic Minimoog/TB-303 bass, +not a physical model. + +- Oscillator: 1–2 phase-increment sawtooths (reuse `WAVE_SAWTOOTH` math), optionally one detuned −7 cents or one square sub-octave for weight. +- Filter: a resonant lowpass — implement a 2-pole state-variable or ladder-style LPF (cheap: a cascade of two of the existing biquads, or a TPT/Zavalishin 1-pole pair for stability under modulation). Cutoff swept by an envelope: fast attack to ~2–3 kHz, decay to ~300–500 Hz over ~150 ms. Resonance Q≈2–4. +- Slight pitch-envelope (drop ~+2 semitones → 0 over ~10 ms) gives the punchy synth attack. +- Key timbre params: filter envelope amount/time, resonance, sub-oscillator mix. + +### 40 — Synth Bass 2 + +Method: **Subtractive, FM/harder variant.** GM Synth Bass 2 is typically more +aggressive/metallic — either a squarewave-led subtractive patch or a 2-operator +FM bass. + +- Option A (subtractive): square + saw, higher resonance (Q≈4–6), more filter-envelope sweep, faster decay — a "rubbery" acid bass. +- Option B (FM, recommended for contrast): 2-op FM — carrier at f0, modulator at 1× or 2× f0, modulation index enveloped (high at attack → low at sustain). Both operators are phase-increment sines (use a sine LUT). `out = sin(2π(φc + I·sin(2πφm)))`. This gives the bright metallic attack mellowing to a hollow body. Cite Chowning 1973 FM. +- Key timbre params vs Synth Bass 1: squarier/FM-brighter spectrum, faster more aggressive envelope. + +Reference: Chowning, J. (1973). "The Synthesis of Complex Audio Spectra by Means of +Frequency Modulation," *JAES* 21(7). + +--- + +# Family B — Strings & friends (GM 41–48) + +Programs 41–44 (Violin/Viola/Cello/Contrabass) are **solo bowed strings** → bowed +digital waveguide. 45 (Tremolo) is bowed + amplitude LFO. 46 (Pizzicato) and 47 +(Harp) are plucked → KS. 48 (Timpani) is modal membrane. + +## Bowed digital waveguide (shared by 41–45) + +This is the marquee algorithm of the dossier and the one with the most rigorous +literature. The architecture mirrors the existing whistle waveguide: a delay line +loop is the resonator, but the *bow* replaces the *jet*, and a **friction +nonlinearity (the "bow table")** replaces the cubic. + +**Physical basis — Helmholtz motion via the friction curve:** McIntyre, Schumacher +& Woodhouse (1983) showed bowed-string oscillation is a *stick-slip* limit cycle +governed by a memoryless friction characteristic relating the bow-string +*differential velocity* (`bow velocity − string velocity at the bow point`) to the +force on the string. Smith (1986) recast this as a digital waveguide: two delay +lines split at the bow point, and a **scattering junction** at the bow whose +reflection/transmission depends on the instantaneous differential velocity through +the friction curve. + +**STK `Bowed` per-sample tick (the canonical implementation, Cook/Scavone), exact arithmetic:** + +```c +// two delay lines split at the bow point by betaRatio_ (bow position 0..1): +// bridgeDelay_.setDelay(baseDelay * betaRatio_); +// neckDelay_.setDelay (baseDelay * (1 - betaRatio_)); +// baseDelay_ = sr/freq - 4.0; // -4 corrects for filter group delays + +bowVelocity = maxVelocity_ * adsr_.tick(); // bow speed × envelope +bridgeReflection = -stringFilter_.tick(bridgeDelay_.lastOut()); // bridge: 1-pole loss + sign-invert +nutReflection = -neckDelay_.lastOut(); // nut: rigid sign-invert +stringVelocity = bridgeReflection + nutReflection; // velocity AT the bow +deltaV = bowVelocity - stringVelocity; // differential velocity + +newVelocity = deltaV * bowTable_.tick(deltaV); // FRICTION NONLINEARITY +bridgeDelay_.tick(bridgeReflection + newVelocity); // inject into both halves +neckDelay_.tick (nutReflection + newVelocity); + +out = 0.1248 * bodyFilterCascade(bridgeDelay_.lastOut()); // 6-biquad body, scaled +``` + +**The bow table (friction curve), exact STK formula:** + +```c +// BowTable::tick(input == deltaV): +sample = (input + offset_) * slope_; // offset_≈0 default; slope_ ∝ bow force +sample = fabs(sample) + 0.75; +sample = pow(sample, -4.0); // power-law friction falloff +if (sample < minOutput_) sample = minOutput_; // typ. 0.01 +if (sample > maxOutput_) sample = maxOutput_; // typ. 0.98 +return sample; // multiplied by deltaV by caller +``` + +The `^(-4)` power law is the key: near zero differential velocity (stick) the +multiplier saturates high → string locks to the bow; as |deltaV| grows (slip) the +multiplier collapses → string breaks free. That sharp transition *is* the +Helmholtz stick-slip cycle. `slope_` encodes **bow force/pressure** (steeper = +more pressure = wider stick region) and `betaRatio_` encodes **bow position** +(near bridge = brighter/"sul ponticello," near fingerboard = softer/"sul tasto"). + +**Porting into AC's whistle generator:** +- Reuse `whistle_bore_buf` as the combined string delay; maintain two read taps (`bridgeDelay` length and `neckDelay` length) instead of jet+bore. Total tap = `sr/freq`. +- Replace the cubic `pd*(pd*pd-1)` with the bow-table `pow(...,-4)` (precompute as a 256-entry LUT keyed on `slope·deltaV` to avoid `pow` in the hot loop). +- `stringFilter_` (bridge loss) = the existing 1-pole loop LPF (`whistle_lp1`), tuned a touch brighter than flute. +- Body filter: 1–3 parallel biquads (not STK's 6) at the violin's main resonances; reuse the `gun_body_*` slots. Output scale ~0.12. +- Vibrato: reuse `whistle_vibrato_phase` modulating `baseDelay`. +- Attack: ramp `maxVelocity_` (bow speed) over ~30–60 ms — the slow bow onset is what makes it read as bowed and not plucked. + +References: +- McIntyre, M. E., Schumacher, R. T. & Woodhouse, J. (1983). "On the oscillations of musical instruments," *JASA* 74(5), 1325–1345. https://pubs.aip.org/asa/jasa/article-pdf/74/5/1325/11420986/1325_1_online.pdf +- Smith, J. O. "Digital Waveguide Bowed-String," *Physical Audio Signal Processing*, CCRMA. https://ccrma.stanford.edu/~jos/pasp/Digital_Waveguide_Bowed_String.html +- Smith, J. O. "MUS420 Lecture: Digital Waveguide Modeling of Bowed Strings." https://ccrma.stanford.edu/~jos/BowedStrings/BowedStrings.pdf +- Cook, P. R. & Scavone, G. P. STK `Bowed` / `BowTable` classes. https://ccrma.stanford.edu/software/stk/classstk_1_1Bowed.html · source: https://github.com/thestk/stk/blob/master/src/Bowed.cpp +- Välimäki, V. et al. (2006). "Discrete-Time Modelling of Musical Instruments," *Reports on Progress in Physics* 69. (Digital waveguide review.) + +### 41 — Violin + +- Pitch range G3 (196 Hz) – ~E7. `baseDelay = sr/f − 4`; comfortably inside a 2048 buffer above ~94 Hz at 192 kHz. +- `betaRatio_ ≈ 0.13` (bow ~1/7 from bridge — typical violin bow point). +- Bridge loss filter: bright, only gentle HF rolloff (small strings lose little). +- Body resonances: violin "main air" ~280 Hz and "main wood" ~460 Hz → two body biquads. Plus the **bridge-hill** broad peak ~2.5–3 kHz (a wide biquad) for the singing brilliance. +- Vibrato ~5–6 Hz, depth ±0.5–1%. +- Key timbre param vs viola/cello: shortest delay (highest pitch) + highest body resonances. + +### 42 — Viola + +- Range C3 (131 Hz) – ~A6, a fifth below violin. Slightly **larger body** but acoustically "too small" for its range — the famously nasal, dark-but-thin viola voice. +- Same waveguide, `betaRatio_ ≈ 0.12`. +- Body resonances dropped ~25–30%: air ~220 Hz, wood ~350 Hz. Bridge hill lower (~2 kHz) and **less prominent** (the viola's missing brilliance) → reduce that biquad's gain. +- Slightly stronger bridge loss (a touch darker than violin). +- Key timbre param: lower body modes + weakened bridge hill. + +### 43 — Cello + +- Range C2 (65 Hz) – ~A5. At 192 kHz C2 ⇒ delay ≈ 2950 > 2048 → **needs the 8192 bass buffer** (share the same enlarged buffer as acoustic bass). +- `betaRatio_ ≈ 0.10` (bow closer to bridge proportionally on the long strings). +- Body resonances: air ~110 Hz, main wood ~180–200 Hz; the big resonant corpus → use 3 body biquads with higher Q for the woody warmth. +- Stronger low-frequency body gain; bridge hill ~1.2 kHz. +- Vibrato slower/wider (~5 Hz, ±0.8%). +- Key timbre param: long delay (8192 buffer), low high-Q body modes, warm bridge. + +### 44 — Contrabass + +- Range C1/E1 (~33–41 Hz) – ~G4. Definitely the **8192 buffer**, possibly run at a /4 internal rate. +- `betaRatio_ ≈ 0.08`. +- Heaviest bridge loss → dark, fundamental-dominated; very few audible upper partials. +- Body modes ~60 Hz / ~100 Hz, broad. Bow attack noise more prominent (rosin grind audible on big strings) → add a low-level bandpassed noise gated to the bow-onset. +- Key timbre param: lowest delay, darkest loss filter, audible bow-grind transient. + +### 45 — Tremolo Strings + +Method: **Bowed waveguide (violin-family) + fast amplitude LFO + retrigger bow +direction.** Tremolo = rapid back-and-forth bowing; the spectral content is a +section-string tone whose amplitude pulses ~7–12 Hz with a slightly noisy, gritty +attack on each stroke. + +- Base = a *blend* of violin+cello waveguide voices (or a single mid-range bowed voice) but layered/detuned like an ensemble (see Family C) for the section feel. +- Tremolo LFO: amplitude modulation at ~8–10 Hz, depth ~40–60% (not to zero — strokes overlap). Reuse a phase-increment LFO. +- On each LFO trough, briefly re-inject bow-grind noise (bow direction change) → the characteristic "shimmer/grit." +- Key timbre param: LFO rate + depth + per-stroke grit; otherwise inherits the bowed model. + +### 46 — Pizzicato Strings + +Method: **Karplus-Strong, short bright pluck, section-detuned.** Pizzicato = the +bowed string *plucked* — short decay, bright attack, body resonance. + +- KS with `N=sr/f`, β ≈ 0.15 (plucked near the fingerboard end), excitation cutoff ~4 kHz. +- **Short** stretch `S≈0.992` → ~0.3–0.6 s decay (pizz dies fast; the finger damps it). +- Body biquads from the relevant string body (violin-ish, ~280/460 Hz) for the "tock." +- Section feel: run 2–3 slightly detuned/delayed KS voices (±a few cents, ±a few ms onset) so it's a *section* pizz, not a solo. Or apply the ensemble-detune trick to a single voice's excitation. +- Key timbre param: short stretch (fast decay) + body "tock" + section detune. + +### 47 — Orchestral Harp + +Method: **Karplus-Strong, long sustain** — already implemented as `WAVE_HARP` +(audio.c:300). The existing generator *is* the GM harp. + +- Reuse as-is: `S≈0.9985` (T60 ~15 s), 2-point loss filter, noise pluck. +- Refinements for realism: β ≈ 0.2 pluck-position comb; a soundboard body biquad ~150–250 Hz; gentle excitation LPF for the nylon-ish softness already noted in-code. +- Key timbre param: long stretch (let it ring), soft excitation, soundboard resonance. + +### 48 — Timpani + +Method: **Modal synthesis (parallel resonant biquads) + noise strike transient.** +A struck, tuned membrane: not a string. The kettledrum's *principal* modes form a +near-harmonic series because of air loading. + +- **Modal ratios (Rossing):** the preferentially-excited diametric modes (1,1),(2,1),(3,1),(4,1),(5,1) ring at frequency ratios **1 : 1.50 : 1.99 : 2.44 : 2.90** relative to the nominal pitch. (Note the principal three are ≈ 2:3:4, giving timpani their clear pitch.) Map the perceived note to mode (1,1). +- Each mode = one resonant biquad (bandpass) excited by a shared strike impulse; per-mode decay time (lower modes ring longest). Reuse the `gun_body_*` parallel-biquad machinery (extend to ~5 modes). +- Strike transient: a short (~4–8 ms) noise burst (the mallet) LPF'd ~2 kHz, mixed at note-on — the "thwack" before the pitched ring blooms. +- Pitch-bend support: timpani are tuned by pedal; allow `frequency` glide (already supported). +- Key timbre params: modal ratios fixed, per-mode decay, strike-noise amount/hardness. + +References: +- Rossing, T. D. (1982). "The physics of kettledrums," and Rossing et al. modal analyses — modes (1,1)…(5,1) in ratios 1, 1.5, 2, 2.44, 2.9. https://pubs.aip.org/asa/jasa/article/66/S1/S18/731734 · https://www.fistecamb.com/JASA117-2.pdf +- Fletcher, H. & Rossing, T. D. (1998). *The Physics of Musical Instruments*, 2nd ed., Springer (Ch. on membranes/timpani). + +--- + +# Family C — Ensemble (GM 49–56) + +Two sub-types: **bowed-section emulations** done with supersaw + slow attack +(49–52), and **vocal formant** patches (53–55), plus the **orchestra hit** (56). + +## Supersaw (shared by 49–52) + +Section strings and synth-strings are most efficiently and convincingly done not as +N physical models but as a **detuned saw ensemble (supersaw)** plus a slow attack +and a chorused stereo image. Adam Szabo's bachelor thesis reverse-engineered the +Roland JP-8000 supersaw and gives exact, ready-to-port formulas. + +**7 phase-increment sawtooths**, one center (always in tune) + 6 detuned (3 above, +3 below). Relative detune offsets (center = 1.0), from Szabo Table 1: + +``` +osc1: 1 − 0.11002313·d osc5: 1 + 0.01991221·d +osc2: 1 − 0.06288439·d osc6: 1 + 0.06216538·d +osc3: 1 − 0.01952356·d osc7: 1 + 0.10745242·d +osc4: 1.0 (center) +``` + +where `d` is **not** the raw detune knob but the knob passed through an 11th-order +fit (Szabo eq.) so the spread is musical across the range: + +``` +d = 10028.7312891634·x^11 − 50818.8652045924·x^10 + 111363.4808729368·x^9 + − 138150.6761080548·x^8 + 106649.6679158292·x^7 − 53046.9642751875·x^6 + + 17019.9518580080·x^5 − 3425.0836591318·x^4 + 404.2703938388·x^3 + − 24.1878824391·x^2 + 0.6717417634·x + 0.0030115596 +``` + +(`x` = detune knob 0..1; precompute `d` once per note, not per sample.) + +**Mix law (Szabo §3.2)** — center and side gains vs the mix knob `m` (0..1): + +``` +centerGain = −0.55366·m + 0.99785 +sideGain = −0.73764·m² + 1.2841·m + 0.044372 // applied to EACH of the 6 sides +``` + +So at higher "mix" the center drops and the sides swell (parabolically). Finally a +**pitch-tracked highpass** at the fundamental removes the sub-pile-up muddiness +(Szabo: HP at the 1st harmonic, "pitch tracked"). One-pole HP is enough. + +Per sample: sum 7 phase-increment saws with these gains, HP, → slow ADSR. ~7 +adds + 1 filter per voice — trivially cheap. + +References: +- Szabo, A. (2010). "How to Emulate the Super Saw," BSc thesis. https://www.adamszabo.com/internet/adam_szabo_how_to_emulate_the_super_saw.pdf +- "An Analysis of Roland's Super Saw Oscillator…" (A. Shore). https://static1.squarespace.com/static/519a384ee4b0079d49c8a1f2/t/592c9030a5790abc03d9df21/1496092742864/An+Analysis+of+Roland's+Super+Saw+Oscillator+and+its+Relation+to+Pads+within+Trance+Music+-+Research+Project+-+A.+Shore.pdf +- Roland JP-8000 — https://en.wikipedia.org/wiki/Roland_JP-8000 + +### 49 — String Ensemble 1 + +Method: **Supersaw + slow attack + ensemble chorus + lowpass.** The lush +"orchestra strings" pad. + +- 7-saw supersaw, moderate detune (`x≈0.25–0.35`), high mix (`m≈0.7`). +- Slow attack ~120–250 ms, slow release ~300–500 ms (bowed onset). +- Gentle lowpass ~4–6 kHz (sawtooth is too buzzy raw; real strings roll off) + a slight ~3 kHz "bridge hill" bump for sheen. +- Stereo chorus: a slow ~0.3–0.6 Hz LFO modulating a short delay (reuse the engine's `wobble` infrastructure) for the moving-section shimmer. +- Add a slow ~5 Hz collective vibrato at low depth. +- Key timbre param vs Ensemble 2: warmer, lower lowpass, slightly less detune. + +### 50 — String Ensemble 2 + +Method: **Supersaw, brighter/wider variant** (often the "slow strings" or a wider +analog-string sound). + +- More detune (`x≈0.4–0.5`) and higher mix → fatter, wider. +- Slightly slower attack still (the GM "slow strings" reading) ~200–350 ms. +- Higher lowpass (~7 kHz) for more air, more chorus depth. +- Key timbre param vs Ensemble 1: wider detune, brighter, slower swell. + +### 51 — SynthStrings 1 + +Method: **Supersaw, overtly synthetic.** No attempt at acoustic body — the analog +poly-synth string machine (Solina/Juno) sound. + +- Supersaw with `x≈0.3`, mix `m≈0.6`; **no body resonance**, just a resonant lowpass with a slow filter-envelope sweep (cutoff rises over the attack). +- Add the classic ensemble BBD chorus (multi-tap modulated delay) heavily — this is the signature of string machines. +- Faster attack than acoustic ensembles (~60–120 ms). +- Key timbre param: filter-envelope sweep + heavy chorus, no body modes. + +### 52 — SynthStrings 2 + +Method: **Supersaw, brighter/more-resonant synth variant.** + +- Higher resonance on the LPF (Q≈3), more detune, a touch faster attack. +- Optional PWM square layer mixed in for a hollower, reedier synth-string color. +- Key timbre param vs SynthStrings 1: more filter resonance, square layer. + +## Vocal formant synthesis (shared by 53–55) + +Choir/voice patches are **source–filter formant synthesis**: a glottal/buzz source +(rich in harmonics) shaped by 3–5 resonant **formant** bandpass filters at vowel- +specific frequencies. Two implementation schools: + +1. **Parallel formant filters (Klatt-style, recommended for AC):** drive a single + buzz source (saw or impulse train at f0, optionally with a glottal pulse shape) + into 3–4 parallel resonant biquads tuned to F1–F4, summed with the table's + per-formant gains. Cheap (3–4 biquads), maps cleanly onto the engine's biquad + machinery, and the formant frequencies are *absolute* (independent of pitch) — + exactly what makes a vowel a vowel. +2. **FOF / CHANT (Rodet):** generate each formant as a damped sinusoidal grain + ("forme d'onde formantique") fired once per glottal period. Higher quality for + solo voice but more state; overkill for a choir pad. + +**Use real formant tables.** Csound's FOF-synthesis appendix tabulates F1–F5, +gains (dB) and bandwidths for soprano/alto/tenor/bass × a,e,i,o,u. The two GM +vowels we need are **/a/ ("Aah")** and **/u/ → /o/ ("Ooh")**. Bass-voice values +(good for a mixed-choir center; transpose up for higher sections): + +| Voice | Vowel | F1 | F2 | F3 | F4 | F5 | gains dB (F1..F5) | BW (F1..F5) | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +| Bass | a | 600 | 1040 | 2250 | 2450 | 2750 | 0,−7,−9,−9,−20 | 60,70,110,120,130 | +| Bass | o | 400 | 750 | 2400 | 2600 | 2900 | 0,−11,−21,−20,−40 | 40,80,100,120,120 | +| Bass | u | 350 | 600 | 2400 | 2675 | 2950 | 0,−20,−32,−28,−36 | 40,80,100,120,120 | +| Tenor | a | 650 | 1080 | 2650 | 2900 | 3250 | 0,−6,−7,−8,−22 | 80,90,120,130,140 | +| Tenor | o | 400 | 800 | 2600 | 2800 | 3000 | 0,−10,−12,−12,−26 | 40,80,100,120,135 | +| Alto | a | 800 | 1150 | 2800 | 3500 | 4950 | 0,−4,−20,−36,−60 | 50,60,170,180,200 | +| Alto | o | 450 | 800 | 2830 | 3500 | 4950 | 0,−9,−16,−28,−55 | 70,80,100,130,135 | +| Soprano | a | 800 | 1150 | 2900 | 3900 | 4950 | 0,−6,−32,−20,−50 | 80,90,120,130,140 | +| Soprano | u | 325 | 700 | 2700 | 3800 | 4950 | 0,−16,−35,−40,−60 | 50,60,170,180,200 | + +(Full Csound table covers e,i and counter-tenor too; gain in dB → linear = +`10^(dB/20)`; biquad bandpass Q ≈ `Fn / BWn`.) The **singer's formant** — a strong +cluster ~2.8–3.2 kHz that lets voices cut over an orchestra — is visible as the +elevated F3/F4 in tenor/bass and can be boosted slightly for a "trained choir" +sheen. + +**Choir-ness (the multi-voice shimmer):** a real choir = many slightly detuned, +slightly out-of-phase, slightly differently-vibratoed voices. Get it cheaply with: +- 3 detuned source oscillators per voice (±5–12 cents) feeding the *same* formant bank, OR +- per-voice random f0 jitter + independent vibrato phase + a chorus on the output. +- Random slow vibrato (~5–6 Hz, ±1–2%, decorrelated phase per layer) is essential — a static formant source sounds like an organ, not a choir. + +References: +- Klatt, D. H. (1980). "Software for a cascade/parallel formant synthesizer," *JASA* 67(3), 971–995. (Canonical formant-synth architecture, formant freq/BW control.) +- Rodet, X., Potard, Y. & Barrière, J.-B. (1984). "The CHANT Project: From the Synthesis of the Singing Voice to Synthesis in General," *Computer Music Journal* 8(3). FOF/CHANT. http://anasynth.ircam.fr/home/english/media/singing-synthesis-chant-program +- Csound FOF formant tables (soprano/alto/tenor/bass × a,e,i,o,u; freq/dB/BW). https://www.csound-tutorial.net/floss_manual/Release06/Cs_FM_06_ScrapBook/default_026.html +- Praat KlattGrid vowel defaults (e.g. /a/: F1 800/F2 1200/F3 2300/F4 2800, BW 80/80/100/140). https://www.fon.hum.uva.nl/praat/manual/Create_KlattGrid_from_vowel___.html +- Peterson, G. E. & Barney, H. L. (1952). "Control Methods Used in a Study of the Vowels," *JASA* 24(2). (Reference vowel formant data.) + +### 53 — Choir Aahs + +Method: **Formant synthesis, vowel /a/, multi-voice.** The bright, open choir +"aah." + +- Source: bandlimited saw or glottal pulse at f0; 3 detuned layers (±7 cents). +- Formant bank: F1–F4 from the **/a/** rows above (pick voice section by pitch: bass below C3, tenor/alto mid, soprano above C5 — or blend two adjacent rows). F1≈600–800, F2≈1040–1150 — the wide-open /a/. +- Slow attack ~80–150 ms, slow release; collective decorrelated vibrato ~5.5 Hz. +- Light breath noise into the formants for the airy choral texture. +- Key timbre param: /a/ formants, pitch-dependent voice section, choir detune+vibrato. + +### 54 — Voice Oohs + +Method: **Formant synthesis, vowel /u/–/o/, multi-voice.** The rounded, dark +"ooh." + +- Identical engine to Choir Aahs but formant bank from the **/u/** (or /o/) rows: + F1≈325–400 (low), F2≈600–800 (low), high formants weak (−20 to −35 dB). The low, + close F1/F2 is what makes it "ooh." +- Slightly darker/softer; less breath noise; rounder source (more lowpassed). +- Key timbre param: /u/ formants (low close F1/F2), darker source. + +### 55 — Synth Voice + +Method: **Formant synthesis, synthetic/"vocoder" voice** — a single (non-choir) +formant voice, often static vowel between /a/ and a neutral schwa, with a clearly +synthetic edge. + +- One source (saw or pulse), one formant bank — no choir detune (or minimal), so it + reads as a solo synthetic voice. +- A neutral/blended vowel (e.g. between /a/ and /o/: F1≈500, F2≈900, F3≈2500) or + let it morph slightly with a slow LFO between two vowel tables for the "talking" + shimmer. +- Brighter source, optional ring-mod or slight chorus for the synthetic sheen. +- Key timbre param: single voice (no choir), neutral/morphing vowel, synthetic source. + +### 56 — Orchestra Hit + +Method: **Cluster of pitched partials + broadband transient + fast decay** — the +iconic "orch hit" stab (originally a Fairlight/sampled chord-stab, reproducible +algorithmically as a dense detuned cluster). + +- A dense **chord cluster**: stack ~6–10 sawtooth/pulse oscillators across a wide + pitched cluster (root + octave + fifth + a spread of detuned partials) for the + "everyone plays at once" mass. +- **Transient:** a short broadband noise burst + brass-like fast attack at t=0 (the + ensemble impact). Reuse the gun crack/click layers for the front-end snap. +- **Envelope:** near-instant attack, fast decay (~250–500 ms), no sustain — it's a + stab. Add a quick downward pitch-blip (+~1 semitone → 0 over ~15 ms) for the + "punch." +- Optional formant-ish broad bandpass ~1–2 kHz for the choral/brass blend the + original sample has. +- Key timbre param: cluster width/density, transient brightness, short percussive + envelope. + +--- + +# Family D — Brass (GM 57–64) + +Programs 57–62 are **lip-reed brass** → brass digital waveguide. 63–64 +(SynthBrass) are subtractive/analog-brass patches with the classic filter-swell +"brass envelope." + +## Brass digital waveguide (shared by 57–62) + +The brass model is the whistle waveguide's sibling: a bore delay-line resonator +driven by a **pressure-controlled valve** — but where the flute used a jet+cubic, +brass uses a **lip-reed**: a tuned resonant filter (the lip's mechanical +resonance) plus a quadratic pressure nonlinearity. The lip resonance must track +the played note (the player "buzzes" at pitch), which is what makes brass models +playable and gives them their characteristic attack and overblow behavior. + +**STK `Brass` per-sample tick (Cook "TBone"/HosePlayer lineage), exact arithmetic:** + +```c +breathPressure = maxPressure_ * adsr_.tick(); // breath envelope +breathPressure += vibratoGain_ * vibrato_.tick(); // add vibrato + +mouthPressure = 0.3 * breathPressure; // pressure at the lips +borePressure = 0.85 * delayLine_.lastOut(); // returning bore wave +deltaPressure = mouthPressure - borePressure; // pressure across lips + +deltaPressure = lipFilter_.tick(deltaPressure); // LIP RESONANCE (biquad @ f0) +deltaPressure *= deltaPressure; // quadratic NL (pressure→area) +if (deltaPressure > 1.0) deltaPressure = 1.0; // valve can only open so far + +// reed/valve scatter: blend mouth & bore by the (squared) lip opening +out = deltaPressure * mouthPressure + (1.0 - deltaPressure) * borePressure; +out = delayLine_.tick( dcBlock_.tick(out) ); // DC-block, into bore +``` + +with setup: +```c +delayLine_.setDelay( sr/frequency * 2.0 + 3.0 ); // half-wave bore (closed-open ⇒ ×2) +lipFilter_.setResonance( frequency, 0.997 ); // lip biquad tracks the note, near-unit pole +lipFilter_.setGain( 0.03 ); +dcBlock_.setBlockZero(); +``` + +**Why it works:** the lip filter is a high-Q resonator centered on the played pitch +— it preferentially feeds energy back at f0, so the bore locks to that harmonic of +its resonance (brass players "lip" to a partial). The **squared** nonlinearity is +the valve: it's a one-sided pressure-to-flow law (the lips can open but the +quadratic + clip prevents negative opening), which is exactly what sustains the +buzz and generates the bright, harmonic-rich brass spectrum that *brightens with +breath pressure* (more `maxPressure_` → the NL saturates harder → more harmonics = +crescendo gets brassier). That breath→brightness coupling is the signature of real +brass and falls out of the model for free. + +**Porting into AC's whistle generator:** +- Reuse `whistle_bore_buf` as the bore (delay = `sr/f·2+3`, half-wave). Single delay line (no jet). +- Replace the cubic with: `lipFilter` (a biquad tracking f0, near-unit pole 0.997) then `x*x` then `min(1)`. +- The lip biquad = one of the existing biquad slots, `setResonance` recomputed on note + on pitch glide. +- `dcBlock_` = the existing `whistle_hp` 1-pole DC blocker. +- `maxPressure_` = breath envelope target (drives loudness *and* brightness — don't normalize it away). +- Vibrato: reuse `whistle_vibrato_phase`. +- Mutes / variants are mostly a lowpass + pressure/lip-gain change (below). + +References: +- Cook, P. R. (1991). "TBone: An Interactive WaveGuide Brass Instrument Synthesis Workbench for the NeXT Machine," *Proc. ICMC*. https://www.semanticscholar.org/paper/3fe3399da2ef21815debf47dcf2e3e4e81f23d1b +- Cook, P. R. & Scavone, G. P. STK `Brass` class. https://ccrma.stanford.edu/software/stk/classstk_1_1Brass.html · source: https://github.com/thestk/stk/blob/master/src/Brass.cpp +- Smith, J. O. *Physical Audio Signal Processing* (waveguide wind instruments), CCRMA. https://ccrma.stanford.edu/~jos/pasp/ +- Vergez, C. & Rodet, X. (work on trumpet physical models / lip nonlinearity, IRCAM) — companion to the lip-reed approach. + +### 57 — Trumpet + +- Highest, brightest open brass. Pitch ~E3–C6. `delay = sr/f·2+3`. +- High `maxPressure_` → bright, harmonically rich; lip pole 0.997. +- Minimal bore loss (small bright instrument) → keep the loop bright. +- Bell radiation: a gentle high-shelf/zero on the output (brass radiates highs more efficiently) → reuse a 1-zero HPF like the gun `gun_radiation_a`. +- Attack: a short noise "spit" + fast pressure ramp (~20–40 ms). +- Key timbre param vs others: shortest bore, brightest (high pressure + low loss), bell HPF. + +### 58 — Trombone + +- Lower, big-bore brass; pitch ~E2–F4. Longer `delay`; possibly the 8192 buffer for the lowest notes. +- Slightly **darker** than trumpet (bigger bore = more HF loss) → stronger loop LPF. +- Slide → supports continuous pitch *glissando*: slow the frequency slew so legato slides glide (like fretless). +- Slightly slower attack (larger air column). +- Key timbre param: long bore, darker loss, glissando-capable glide. + +### 59 — Tuba + +- Lowest brass; pitch ~D1–F3. **8192 buffer / possibly /4 internal rate.** +- Heavy bore loss → very dark, fundamental-dominated, few upper partials. +- Lower lip pole / lower `maxPressure_` brightness coupling (tuba rarely sounds "brassy/bright"); broad, round tone. +- Slow attack (~50–80 ms — moving a lot of air). +- Key timbre param: longest bore, darkest loss, soft round attack. + +### 60 — Muted Trumpet + +Method: **Trumpet waveguide + mute filter.** A straight/cup mute = a resonant +notch+lowpass on the bell radiation plus a nasal mid-peak. + +- Base = trumpet (57) but: insert a **mute filter** on the output — a lowpass (~3–4 kHz) plus a resonant bandpass peak ~1.5–2 kHz (the mute's nasal "pinch"), and reduce overall level/brightness. +- Reduce bell-radiation HPF (mute kills the highs). +- Slightly increase lip damping (mute loads the bore). +- Key timbre param vs open trumpet: mute LPF + nasal mid-peak, reduced brightness — this is the *only* difference, so a single `muted` flag toggling the output filter covers it. + +### 61 — French Horn + +Method: **Brass waveguide, dark/mellow, conical, hand-in-bell.** The horn is long, +conical, and played with a hand in the bell → famously round, dark, blendy. + +- Long bore; mellow lip pole (slightly lower Q, e.g. 0.995) → rounder, less edgy than trumpet. +- Stronger loop LPF (conical + hand-in-bell rolls off highs) → warm. +- **Soft attack** (~40–70 ms, little spit) — horn entrances are smooth. +- Vibrato minimal (orchestral horn vibrato is subtle). +- Key timbre param: dark loss + mellow lip + soft attack; the "noble warm" brass. + +### 62 — Brass Section + +Method: **Stack of 3–4 detuned brass-waveguide voices** (or a brass waveguide +layered like the supersaw ensemble) — a *section* of trumpets/trombones/horns +hitting together. + +- Run 3 brass voices at the played pitch, detuned ±5–12 cents and onset-staggered ±10–25 ms (sections never attack perfectly together) → the fat unison "stab/swell." +- Shared brighter pressure (sections play loud), light chorus on the output. +- If voice budget is tight: one brass waveguide + the ensemble detune/chorus trick on its output (cheaper, still convincing). +- Key timbre param: detune spread + onset stagger + collective brightness. + +### 63 — SynthBrass 1 + +Method: **Subtractive analog brass** — sawtooth(s) → resonant lowpass with a +fast filter-envelope swell (the classic "brass" patch: cutoff snaps up on attack +then settles), plus PWM and slight oscillator detune. + +- 2 detuned saws (±5–10 cents) + optional PWM square → resonant LPF. +- **Filter envelope (the brass signature):** cutoff fast-attacks to ~5–6 kHz then decays to ~1.5–2 kHz over ~80–150 ms; resonance Q≈2. This filter swell *is* the synth-brass. +- Slight pitch overshoot on attack; light vibrato/PWM motion on sustain. +- Key timbre param: filter-envelope amount/speed, detune, PWM. + +### 64 — SynthBrass 2 + +Method: **Subtractive synth brass, brighter/harder or FM variant** (often the more +aggressive, faster, "analog stab" brass). + +- More resonance, faster/snappier filter envelope, brighter base spectrum; or a + 2-op FM brass (carrier f0, modulator ~1×, index enveloped) for a metallic edge. +- Tighter, more percussive attack than SynthBrass 1. +- Key timbre param vs SynthBrass 1: faster/harder filter env, higher resonance / FM edge. + +--- + +## Implementation roadmap (mapping to audio.c) + +1. **Add WaveTypes** (audio.h enum): `WAVE_BOWED`, `WAVE_BRASS`, `WAVE_VOICE`, `WAVE_SUPERSAW`, `WAVE_MODAL`, plus a `WAVE_KS_BASS` (or reuse `WAVE_HARP` with a `model_variant`). Add a per-voice `uint8_t gm_program` / `model_variant` to select variant params. +2. **Bowed** (41–45): fork `generate_whistle_sample` → `generate_bowed_sample`. Two delay taps off `whistle_bore_buf` (bridge+neck split by `betaRatio`), bow-table LUT (256-entry `pow(x,-4)`), 1-pole bridge loss, 1–3 body biquads (reuse `gun_body_*`). Params per instrument from the violin/viola/cello/contrabass tables above. +3. **Brass** (57–62): fork the whistle → `generate_brass_sample`. Single bore delay (`sr/f·2+3`), lip biquad (one biquad slot, `setResonance(f0, 0.997)`), `x*x` + clip NL, existing DC blocker. Mute (60)/section (62) are output-filter / layering options. +4. **KS bass + pizz + harp** (33–38, 46–47): extend `generate_harp_sample` with: pluck-position comb on the seed, velocity→excitation-cutoff, body biquad, and per-program `(β, S, excitation-cutoff, transient)` params. **Grow the delay buffer to 8192** (or add a low-rate path) for bass/cello/contrabass/tuba. +5. **Supersaw** (49–52): new `generate_supersaw_sample` — 7 phase-increment saws with the Szabo detune offsets + mix law + pitch-tracked HP, then slow ADSR + output chorus (reuse `wobble`). +6. **Formant voice** (53–55): new `generate_voice_sample` — buzz source (saw/glottal) → 3–4 parallel formant biquads from the vowel table (select section by pitch), decorrelated vibrato + chorus for choir-ness. +7. **Modal** (48 timpani, 56 orch-hit): extend the `gun_body_*` parallel-biquad machinery to ~5 modes with the Rossing ratios + a noise/transient strike; orch-hit = detuned cluster + transient + fast decay. +8. **Synth bass/brass** (39–40, 63–64): a small subtractive core (saw/square/FM osc → resonant LPF with a filter envelope) — the only genuinely new primitive needed is a **modulatable resonant lowpass** (a TPT/SVF or cascaded biquads), worth adding once and sharing. + +The only new shared infrastructure: a larger (8192) delay buffer for sub-94 Hz +voices, a 256-entry bow-table LUT, a sine LUT for FM, and a modulatable resonant +SVF/ladder lowpass. Everything else reuses existing whistle/harp/gun primitives. diff --git a/fedac/native/docs/gm-synthesis/03-reed-pipe-synthlead-synthpad.md b/fedac/native/docs/gm-synthesis/03-reed-pipe-synthlead-synthpad.md new file mode 100644 index 0000000000..ba4c6c4e20 --- /dev/null +++ b/fedac/native/docs/gm-synthesis/03-reed-pipe-synthlead-synthpad.md @@ -0,0 +1,477 @@ +# GM Synthesis Dossier 03 — Reed, Pipe, Synth Lead, Synth Pad + +Real-time C synthesis algorithms for 32 General MIDI programs (65–96), targeting +the Aesthetic Computer native audio engine (`fedac/native/src/audio.c`, +`audio.h`). 32 voices, per-sample loop, up to 192 kHz, mono-per-voice with +post pan. + +This dossier covers four GM families: + +- **Reed** (65–72): Soprano/Alto/Tenor/Baritone Sax, Oboe, English Horn, Bassoon, Clarinet +- **Pipe** (73–80): Piccolo, Flute, Recorder, Pan Flute, Blown Bottle, Shakuhachi, Whistle, Ocarina +- **Synth Lead** (81–88): Square, Sawtooth, Calliope, Chiff, Charang, Voice, Fifths, Bass+Lead +- **Synth Pad** (89–96): New Age, Warm, Polysynth, Choir, Bowed, Metallic, Halo, Sweep + +--- + +## 0. The engine substrate — what we build on + +The AC engine already ships a **Perry Cook STK digital-waveguide flute model** +as `WAVE_WHISTLE` (`generate_whistle_sample`, audio.c:177). Its signal flow: + +``` + breath ─►(+)─► jetDelay ─► NL(x·(x²−1)) ─► dcBlock ─►(+)─► boreDelay ─┬─► out + ▲ −jetRefl·temp ▲ +endRefl·temp │ + └──────────── 1-pole loop LPF ◄────────────┴─────────────────┘ +``` + +Key existing state (per `ACVoice`, audio.h:88–104): + +- `whistle_bore_buf[2048]` + `whistle_bore_w` — **bore delay line** (`= SR/freq`), + the primary resonator. Fractional read via `whistle_frac_read()` (audio.c:140). +- `whistle_jet_buf[512]` + `whistle_jet_w` — **jet delay** (`0.32 × bore`), models + air-jet travel across the embouchure. +- `whistle_lp1` — 1-pole loop LPF (bore losses → tone darkens). +- `whistle_hp_x1/y1` — DC blocker after the nonlinearity. +- `whistle_breath`, `whistle_vibrato_phase` — breath envelope + ~5 Hz vibrato. +- `noise_seed` — xorshift32 white noise for breath turbulence. + +This **flute waveguide is the direct basis for the entire Pipe family** (retune +the jet ratio + breath noise) and, with the cubic nonlinearity swapped for a +**reed reflection table**, becomes the **entire Reed family**. Basic oscillators +(`WAVE_SINE/TRIANGLE/SAWTOOTH/SQUARE/NOISE`, audio.c:1469–1496) plus a biquad +LPF (`setup_noise_filter`, audio.c:1528) are the substrate for the **subtractive +Synth Lead and Synth Pad** families. + +Repo convention: **phase-increment oscillators** (`phase += freq/SR`), not +`sin(TAU·f·t)`; one-pole/biquad filters; table or polyBLEP for band-limiting. + +### The reed table — the one new primitive the Reed family needs + +The flute uses a **cubic** nonlinearity `x·(x²−1)` (a limit-cycle generator that +turns DC breath into oscillation). A *reed* instrument instead uses a +**pressure-controlled reflection coefficient** — the McIntyre–Schumacher– +Woodhouse reed model, implemented in STK as a saturating affine map: + +``` +STK ReedTable: reflection = offset + slope · pressureDiff + offset = 0.6 slope = −0.8 (STK defaults) + clamp reflection to [−1, +1] +``` + +The reed nearly closes as mouth-pressure rises (negative slope), then *slams +shut* at the clamp — that hard saturation is what gives reeds their buzzy, +harmonic-rich tone versus the flute's pure jet whistle. Source: STK +`ReedTable.h`; J.O. Smith *PASP* "Single-Reed Theory." The **clarinet's +cylindrical closed bore reflects with inversion → only odd harmonics survive**; +the **conical sax/oboe/bassoon bores reflect without inversion → full harmonic +series**. This single bore-reflection sign is the deepest physical distinction +in the whole Reed family. + +A single shared `generate_reed_sample(v, sr, ReedParams *rp)` with a +`ReedParams` struct (bore ratio, conical flag, reed stiffness/offset/slope, +breath-noise gain, vibrato) covers all 8 reeds. Likewise one +`generate_pipe_sample()` covers all 8 pipes via a `PipeParams` struct. The two +synth families share `generate_subtractive_voice()`. + +--- + +## Family A — Reed (GM 65–72) + +**Method:** digital waveguide bore + STK reed reflection table nonlinearity. +Extend the existing whistle waveguide: keep `whistle_bore_buf`/`whistle_jet_buf`, +**replace the cubic `pd·(pd²−1)` with the reed table** and add a bore-reflection +sign that flips for cylindrical (clarinet) vs conical (sax/oboe/bassoon) bores. + +**Core references (whole family):** + +- M.E. McIntyre, R.T. Schumacher, J. Woodhouse (1983), "On the Oscillations of + Musical Instruments," *J. Acoust. Soc. Am.* **74**(5):1325–1345. The foundational + time-domain reed/bowed/jet oscillator model. +- J.O. Smith, *Physical Audio Signal Processing*, CCRMA Stanford (online): + "Single-Reed Theory," "Clarinet" chapters. +- P.R. Cook & G.P. Scavone, *The Synthesis ToolKit in C++* (STK): `Clarinet`, + `Saxofony`, `BlowHole` classes. , +- G.P. Scavone & P.R. Cook (1998), tonehole/register-hole BlowHole model. +- N.H. Fletcher & T.D. Rossing, *The Physics of Musical Instruments*, 2nd ed., + Springer 1998 (Ch. 13–15, reed instruments; conical vs cylindrical bores). +- V. Välimäki et al., woodwind digital-waveguide synthesis papers (fractional + delay tuning, toneholes). +- Gordon Reid, "Synth Secrets — Synthesizing Wind Instruments," *Sound on Sound*. + Closed pipe (clarinet) = square (odd harmonics); conical bore = sawtooth (full + series). + +**Shared reed-waveguide per-sample sketch:** + +```c +typedef struct { + double bore_ratio; // jet/embouchure unused for reeds; bore = SR/freq + int conical; // 1 = sax/oboe/bassoon (full harmonics), + // 0 = clarinet (odd-only, inverting reflection) + double reed_offset; // STK 0.6 nominal; lower = softer reed + double reed_slope; // STK −0.8 nominal; steeper = brighter/buzzier + double loop_damp; // 1-pole LPF coeff (bore loss; bigger bore = darker) + double noise_gain; // breath turbulence (sax > oboe > clarinet) + double vib_rate, vib_depth; +} ReedParams; + +static inline double generate_reed_sample(ACVoice *v, double sr, const ReedParams *rp) { + double env = compute_envelope(v); + // Breath pressure: DC drives the reed into self-oscillation. + double pTarget = 0.55 + 0.45 * env; // mouth pressure Pm + v->whistle_breath += (pTarget - v->whistle_breath) * 0.01; + // vibrato + turbulent breath noise + v->whistle_vibrato_phase += rp->vib_rate / sr; + if (v->whistle_vibrato_phase >= 1.0) v->whistle_vibrato_phase -= 1.0; + double vib = sin(2*M_PI*v->whistle_vibrato_phase) * rp->vib_depth; + double white = ((double)xorshift32(&v->noise_seed)/UINT32_MAX)*2-1; + double Pm = v->whistle_breath * (1.0 + rp->noise_gain*white + vib); + + double freq = clampd(v->frequency, 30.0, sr*0.20); + double bore_delay = sr / freq; + const int BORE_N = 2048; + if (bore_delay > BORE_N-2) bore_delay = BORE_N-2; + + // Bore round-trip: read, damp (1-pole loop LPF = bore losses) + double bore_out = whistle_frac_read(v->whistle_bore_buf, BORE_N, + v->whistle_bore_w, bore_delay); + v->whistle_lp1 = (1-rp->loop_damp)*bore_out + rp->loop_damp*v->whistle_lp1; + // CLARINET: cylinder reflects with INVERSION → odd harmonics only. + // SAX/OBOE/BASSOON: conical bore reflects WITHOUT inversion → full series. + double refl = rp->conical ? v->whistle_lp1 : -v->whistle_lp1; + + // Pressure difference across the reed, then the STK reed table. + double pDiff = Pm - refl; // Pd = Pm − Pb + double reedRefl = rp->reed_offset + rp->reed_slope * pDiff; // affine + if (reedRefl > 1.0) reedRefl = 1.0; // reed slams shut + if (reedRefl < -1.0) reedRefl = -1.0; + // Reflected pressure wave entering the bore (McIntyre et al.): + double into_bore = refl + reedRefl * pDiff; + + // DC block + write back into the bore loop. + double y = into_bore - v->whistle_hp_x1 + 0.995*v->whistle_hp_y1; + v->whistle_hp_x1 = into_bore; v->whistle_hp_y1 = y; + v->whistle_bore_buf[v->whistle_bore_w] = (float)y; + v->whistle_bore_w = (v->whistle_bore_w + 1) % BORE_N; + return 0.3 * y * env; +} +``` + +**Param-vs-variant table** (the timbre knobs that distinguish all 8): + +| GM | Instrument | bore (octave) | conical | reed_slope | noise_gain | notes | +|----|-----------|---------------|---------|-----------|-----------|-------| +| 65 | Soprano Sax | short (high) | 1 | −0.85 | 0.10 | bright conical, B♭ soprano | +| 66 | Alto Sax | medium | 1 | −0.80 | 0.12 | E♭, classic sax buzz | +| 67 | Tenor Sax | long | 1 | −0.75 | 0.13 | B♭, warmer/breathier | +| 68 | Baritone Sax| longest | 1 | −0.70 | 0.15 | E♭, dark, big noise floor | +| 69 | Oboe | short, narrow | 1 | −0.90 | 0.05 | double reed → very stiff (steep slope), thin nasal | +| 70 | English Horn| medium, narrow | 1 | −0.88 | 0.06 | alto oboe, F, rounder than oboe | +| 71 | Bassoon | long, narrow | 1 | −0.82 | 0.07 | bass double reed, hollow low register | +| 72 | Clarinet | medium | **0** | −0.80 | 0.04 | **cylindrical → odd harmonics**, woody hollow tone | + +### 65–68 Saxophones +Conical bore (full harmonic series), single reed. STK `Saxofony` is the +reference: a "blowed-string" hybrid where excitation position morphs +clarinet↔sax. Bore length is the *only* primary distinguisher across the four +saxes (soprano shortest → baritone longest); set `bore_delay = SR/freq` from the +played pitch and let the longer/larger members get more `noise_gain` (breathier) +and a heavier `loop_damp` (darker). Reed `slope ≈ −0.8`, `offset 0.6`. + +### 69–70 Oboe & English Horn +Conical narrow bore + **double reed** → model as a *stiffer* reed (steeper +`reed_slope ≈ −0.9`, smaller aperture) and lighter breath noise. Their thin, +nasal, harmonically dense timbre comes from the narrow cone emphasizing upper +formants — boost a fixed bandpass "singer's formant" ~1.4 kHz (oboe) / +~1.1 kHz (English horn) on the output for the characteristic reedy edge +(Fletcher & Rossing). English horn = oboe scaled down a fifth (F instrument). + +### 71 Bassoon +Conical, long, narrow double reed. Same model as oboe with a long bore and a +mild low-pass on output (the bassoon's hollow lower register). A formant peak +around 440–500 Hz gives its woody "buzz." + +### 72 Clarinet +**The odd-harmonic case.** Cylindrical closed bore → set `conical = 0` so the +loop reflection **inverts** (`refl = −whistle_lp1`). A closed cylinder resonates +at odd multiples only, producing the clarinet's hollow, square-wave-like spectrum +(Reid; Smith *PASP* Clarinet). This is the most physically distinct reed and +should A/B clearly against the saxes. Subtractive fallback: filtered **square +wave** (odd harmonics) confirms the target spectrum. + +--- + +## Family B — Pipe (GM 73–80) + +**Method:** the **existing Cook flute waveguide unchanged in structure** — +retune jet ratio, breath-noise level, and vibrato per instrument. These are +air-jet (no reed) instruments, so the **cubic nonlinearity stays**. This is the +cheapest family: it is literally `generate_whistle_sample` with a small +`PipeParams` struct. + +**References:** + +- P.R. Cook, STK `Flute` (jet-delay air-reed waveguide). +- J.O. Smith, *PASP*, "Blown Bottles and Pipes" / flute waveguide. +- M.E. McIntyre, R.T. Schumacher, J. Woodhouse (1983), JASA 74:1325 (air-jet + oscillation, flute family). +- Fletcher & Rossing, *Physics of Musical Instruments*, Ch. 16–17 (flutes, + organ flue pipes, jet drive). +- Gordon Reid, "Synthesizing Wind Instruments," *SoS* — open pipe = full + harmonic series (sawtooth/triangle). + +```c +typedef struct { + double jet_ratio; // jet delay / bore delay. Cook flute 0.32; + // pennywhistle 0.45; ocarina ~0.5 (Helmholtz) + double noise_gain; // breath chiff: pan flute/shakuhachi high, whistle low + double vib_rate, vib_depth; + double loop_damp; // brightness: piccolo bright, recorder pure + int helmholtz; // ocarina/bottle: vessel resonator, no overblow +} PipeParams; +``` + +**Param-vs-variant table:** + +| GM | Instrument | jet_ratio | noise_gain | vib | notes | +|----|-----------|-----------|-----------|-----|-------| +| 73 | Piccolo | 0.30 | 0.06 | 5 Hz/0.03 | flute up an octave; short bore, very bright | +| 74 | Flute | 0.32 | 0.08 | 5 Hz/0.03 | the canonical Cook model (current default) | +| 75 | Recorder | 0.32 | 0.04 | minimal | pure, low noise; open pipe, near-pure tone | +| 76 | Pan Flute | 0.40 | **0.22** | 4 Hz/0.05 | strong breathy chiff, prominent onset noise | +| 77 | Blown Bottle | n/a | 0.18 | slow | **Helmholtz** resonator: single-mode, no overblow | +| 78 | Shakuhachi| 0.38 | **0.25** | 6 Hz/0.08 expressive | very breathy, pitch-bend/meri-kari portamento | +| 79 | Whistle | 0.45 | 0.05 | 6 Hz/0.04 | tin/penny whistle; higher jet ratio, bright | +| 80 | Ocarina | 0.50 | 0.06 | gentle | **Helmholtz vessel**; pure, slightly hollow | + +### 73 Piccolo / 74 Flute / 75 Recorder +Same waveguide; piccolo plays an octave higher (shorter `bore_delay`), recorder +drops vibrato and breath noise to near zero for its pure tone. Reid: recorder is +an open pipe → full harmonic series; subtractive fallback is sawtooth/triangle +through a gentle LPF. + +### 76 Pan Flute / 78 Shakuhachi +The **breath-noise instruments**. Raise `noise_gain` to 0.22–0.25 and increase +attack-onset chiff (the engine already scales noise by `onset = 1 − env`, +audio.c:196). Shakuhachi additionally wants expressive vibrato/portamento and a +slightly higher `jet_ratio` for its airy edge — use the existing +`target_frequency` slew (audio.c:1514) for bends. + +### 77 Blown Bottle / 80 Ocarina +**Helmholtz vessel resonators**, not pipes — a single resonant mode, no overblown +octave. Cheapest model: skip the bore delay loop, drive a single resonant +**bandpass biquad** (Q ~10) at the played pitch with breath noise + a touch of +the jet nonlinearity for the "edge tone." `helmholtz=1` branch: + +```c +// Bottle/ocarina: resonant bandpass excited by breath noise + jet edge tone. +double exc = Pm * (1.0 + noise_gain*white); +double y = bp_b0*exc + bp_b1*x1 + bp_b2*x2 - bp_a1*y1 - bp_a2*y2; // biquad BPF +``` + +### 79 Whistle +Tin/penny whistle: `jet_ratio = 0.45` (per the existing code comment, audio.c:201), +bright, low breath noise, lively ~6 Hz vibrato. Essentially the current whistle +default tuned a hair brighter. + +--- + +## Family C — Synth Lead (GM 81–88) + +**Method:** classic **subtractive synthesis** — named waveform(s) → resonant +low-pass filter → amp+filter envelope, with detune/PWM/LFO per variant. No +physical modeling. Many GM leads are explicitly derived from the **Roland D-50 / +MT-32** factory sounds (Calliope = "Living Calliope," Chiff = "Breathy Chiffer"). + +**References:** + +- Gordon Reid, "Synth Secrets" (63-part series), *Sound on Sound* — subtractive + synthesis bible; wind/brass/lead patches. +- *Synth Secrets* on PWM/pulse leads and square-vs-saw harmonic content. +- FreePats GM Synth Lead set (ZynAddSubFX/Surge recipes). +- GM spec / D-50 & MT-32 lineage. +- Minimoog 24 dB/oct ladder filter, detuned-oscillator lead practice. + +**Shared subtractive voice + state-variable filter:** + +```c +typedef struct { + WaveType osc[3]; // up to 3 oscillators + double detune[3]; // cents offset (×2^(c/1200)) + int n_osc; + double pwm_rate, pwm_depth; // square→pulse PWM (lead 1) + double cutoff, res; // SVF / ladder + double env_amount; // filter envelope depth + double fa, fd, fs, fr; // filter ADSR + double lfo_rate, lfo_depth; // pitch/cutoff vibrato + int sub_octave; // bass+lead: extra −1 oct osc +} LeadParams; +``` + +For band-limiting, use **polyBLEP** on saw/square (cheap, one branch per +discontinuity) so high leads don't alias at 192 kHz harmonics — the current +naive `2·phase−1` saw (audio.c:1483) is fine at the engine rate but polyBLEP is +worth it for bright leads. A **2-pole state-variable filter** (cheaper than the +existing RBJ biquad recompute, and trivially modulatable per-sample) gives the +resonant sweep: + +```c +// Chamberlin SVF, per sample (f = 2·sin(π·cutoff/SR), q = 1/res): +lp += f*bp; hp = in - lp - q*bp; bp += f*hp; // lp = low-pass out +``` + +| GM | Lead | oscillators | filter / motion | character | +|----|------|------------|-----------------|-----------| +| 81 | Square | 1× square + **PWM** (LFO ~0.3–6 Hz) | static LPF, mild res | hollow odd-harmonic lead, classic chiptune | +| 82 | Sawtooth | 1–2× saw, detuned ~7 cents | LPF + env sweep | bright buzzy "all-harmonics" lead | +| 83 | Calliope | triangle/sine + soft saw, slow attack | gentle LPF | **woodwind/steam-organ**; near-pure + breath; D-50 "Living Calliope" | +| 84 | Chiff | saw/pulse + **noise burst on attack** | LPF, fast env | **breathy chiff transient** then tone; D-50 "Breathy Chiffer" | +| 85 | Charang | 2× saw, hard detune + **drive/distortion** | resonant LPF | aggressive **guitar-like** lead; run through engine `drive_mix` | +| 86 | Voice | saw/pulse + **formant BPF bank** | vowel-shaped | synth-voice "aah" lead, fast attack | +| 87 | Fifths | **two oscillators a perfect fifth apart** (×1.5 freq) | LPF | parallel-5ths power lead (organum) | +| 88 | Bass+Lead | saw + **sub-octave square** | LPF, keytracked | split: fat bass low, lead high; sub_octave=1 | + +**Variant recipes:** + +- **81 Square:** one square osc; modulate pulse width with a slow LFO (PWM) for + motion. Square = odd harmonics only → the "hollow" lead. Light filter. +- **82 Sawtooth:** one or two slightly detuned saws (≈7 cents) → fatter; LPF with + a moderate envelope sweep. The textbook bright analog lead. +- **83 Calliope:** softer, woodwind-like — triangle or filtered saw, slow-ish + attack, low resonance, gentle vibrato. Add a hint of breath noise (reuse the + pipe `noise_gain`) for the "steam organ" air. +- **84 Chiff:** the defining feature is a **short filtered noise burst at note + onset** (the "chiff") layered before the pitched tone settles — gate a + decaying white-noise→BPF for ~30–50 ms via the envelope, then the saw/pulse + carries the sustain. +- **85 Charang:** two hard-detuned saws through **soft saturation** (the engine's + `audio_set_drive_mix`/tanh stage already exists) + resonant LPF → buzzy + guitar-ish lead. +- **86 Voice:** rich oscillator (saw) shaped by a **3-band formant filter bank** + (see Choir pad below) on an "aah/ooh" vowel; fast attack, slight vibrato. +- **87 Fifths:** run two copies of the lead oscillator, the second at **1.5× + frequency** (a perfect fifth), mixed equally → the parallel-fifths + "organum" lead. +- **88 Bass+Lead:** keyboard-split layering — a sub-octave square/saw for the + bass register plus the saw lead, low notes weighted to bass. Use + `sub_octave` to add a −1-octave oscillator. + +--- + +## Family D — Synth Pad (GM 89–96) + +**Method:** **detuned multi-oscillator** subtractive + **slow filter sweep** + +**chorus** for width, with per-variant character modules (formant for Choir, +FM/ring-mod for Metallic, filter-motion for Sweep/Halo). Pads are defined by +*slow attack/release envelopes* and *evolving motion*, not transients. + +**References:** + +- Gordon Reid, "Synth Secrets" — string-machine / ensemble / PWM-pad + installments; "From Sound on Sound." +- *Synth Secrets* on PWM for lush string pads (synthesizing PWM when absent). +- *Sound on Sound*, "Formant Synthesis" — vowel formant frequencies + bandpass + bank for choir pads. +- Chowning FM (bell/metallic timbres) — inharmonic partials from + non-integer modulator ratios; ring modulation for metallic spectra. + +**Shared pad engine:** `LeadParams` extended with multiple detuned voices + a +**chorus** (3-tap modulated delay; the engine already has a flange/`wobble_buf` +that can be repurposed) and a **slow bipolar LFO** on filter cutoff. + +```c +typedef struct { + int n_detune; // 3–7 stacked detuned saw/PWM oscillators + double spread_cents; // ±detune for "supersaw" width + double attack, release; // slow (0.3–2.0 s) + double cutoff, res, sweep_rate, sweep_depth; // slow filter LFO + double chorus_depth, chorus_rate; + int formant; // choir pad + int ring_mod; // metallic pad + double fm_ratio, fm_index; // metallic/halo inharmonic content +} PadParams; +``` + +| GM | Pad | engine | character | +|----|-----|--------|-----------| +| 89 | New Age | detuned saws + long reverb + slow LPF sweep | shimmery, ethereal, slow swell | +| 90 | Warm | 2–3 detuned saws, **low cutoff**, soft attack | rounded analog warmth, no high edge | +| 91 | Polysynth | bright detuned saws/PWM, medium attack | classic "Jump"-style poly stack | +| 92 | Choir | saw/pulse + **formant BPF bank** ("ooh/aah") + chorus | vocal pad | +| 93 | Bowed | saw + **slow noisy attack + body resonance** | bowed-glass/string swell | +| 94 | Metallic | **FM or ring-mod** inharmonic + slow attack | bell-like shimmering metal | +| 95 | Halo | choir/saw + **bright formant + reverb + slow tremolo** | airy heavenly pad | +| 96 | Sweep | detuned saws + **dramatic resonant LPF sweep LFO** | the signature filter-sweep pad | + +**Variant recipes:** + +- **89 New Age / 90 Warm / 91 Polysynth:** all are the same detuned-saw stack + (3–7 oscillators, `spread_cents` ±5–15) into a resonant LPF; they differ only + in **filter cutoff** (warm = dark/low, polysynth = bright) and attack time. + PWM on the pulses adds the slow internal motion (Reid's string-machine trick). +- **92 Choir:** harmonically rich oscillator → **3 parallel bandpass filters at + vowel formants**, e.g. "ooh" F1=300/F2=870/F3=2250 Hz, "aah" F1=660/F2=1700/ + F3=2400 Hz (SoS formant table), + chorus + slow attack. Pitch-independent + formants are what make it read as a voice. +- **93 Bowed (Bowed Glass):** slow noisy attack (filtered noise ramp) feeding a + high-Q resonant body, then a steady saw/triangle sustain — the "bowed + glass/string" swell. A bowed-string waveguide (friction nonlinearity, MSW + 1983) is the deluxe option, but the subtractive swell is adequate. +- **94 Metallic:** **inharmonic** — either 2-operator **FM** with a non-integer + modulator ratio (e.g. ratio 1.4, moderate index → bell partials, Chowning) or + **ring-modulate** two oscillators (`out = oscA·oscB`) to inject metallic sum/ + difference tones, with a slow attack and long decay. +- **95 Halo:** a bright, airy choir-ish pad — formant-shaped saws (bright vowel), + heavy reverb (engine `room_mix`), and a slow tremolo LFO on amplitude for the + "halo" shimmer. +- **96 Sweep:** the canonical **filter-sweep pad** — detuned saws into a resonant + LPF whose cutoff is driven by a **slow triangle/sine LFO** (sweep_rate ~0.1– + 0.3 Hz) across a wide range with high resonance, so the harmonic content + visibly sweeps up and down. + +--- + +## Implementation roadmap for AC + +1. **Reed family (8):** add `generate_reed_sample()` + `ReedParams` table — reuse + `whistle_bore_buf`/`whistle_jet_buf`/`whistle_lp1`/`whistle_hp_*`. Only new + logic: STK reed table (`offset + slope·Pd`, clamp) replacing the cubic, and + the `conical` reflection-sign flip (clarinet = invert → odd harmonics). +2. **Pipe family (8):** add `generate_pipe_sample()` + `PipeParams` — structurally + identical to the current whistle; per-instrument `jet_ratio`, `noise_gain`, + `vib`, plus a `helmholtz` biquad-resonator branch for bottle/ocarina. +3. **Synth Lead (8) + Synth Pad (16):** add `generate_subtractive_voice()` with a + Chamberlin SVF + polyBLEP saw/square + N detuned oscillators + slow LFOs + + chorus (reuse `wobble_buf`); plus a small **formant BPF bank** shared by + Voice lead, Choir pad, and Halo pad, and an **FM/ring-mod** branch for the + Metallic pad. Extend `WaveType` enum and the `generate_sample` switch + (audio.c:1468) accordingly. + +All four families fit the existing per-voice/per-sample architecture; the reeds +and pipes need **zero new buffers** (they share the whistle/harp delay lines), +and the synth families add only small filter/LFO state. + +--- + +## Consolidated source list + +- McIntyre, Schumacher & Woodhouse (1983), "On the Oscillations of Musical + Instruments," *JASA* 74(5):1325–1345. +- J.O. Smith, *Physical Audio Signal Processing*, CCRMA. +- P.R. Cook & G.P. Scavone, STK (Clarinet/Saxofony/BlowHole/Flute/ReedTable). + · + · + +- N.H. Fletcher & T.D. Rossing, *The Physics of Musical Instruments*, 2nd ed., + Springer 1998. +- Gordon Reid, "Synth Secrets," *Sound on Sound* (esp. "Synthesizing Wind + Instruments"). · + +- "Formant Synthesis," *Sound on Sound*. +- FreePats GM Synth Lead. +- General MIDI (D-50/MT-32 lineage), Wikipedia. +- Synthesis ToolKit `ReedTable` (offset 0.6, slope −0.8): thestk/stk source. + +*Existing AC engine reference: `fedac/native/src/audio.c` (generate_whistle_sample +:177, generate_sample :1466, setup_noise_filter :1528), `audio.h` (ACVoice +waveguide state :88–104).* diff --git a/fedac/native/docs/gm-synthesis/04-synthfx-ethnic-percussive-soundfx.md b/fedac/native/docs/gm-synthesis/04-synthfx-ethnic-percussive-soundfx.md new file mode 100644 index 0000000000..c8bc906119 --- /dev/null +++ b/fedac/native/docs/gm-synthesis/04-synthfx-ethnic-percussive-soundfx.md @@ -0,0 +1,688 @@ +# GM Synthesis Dossier — Part 04 + +## Synth Effects · Ethnic · Percussive · Sound Effects (GM programs 97–128) + +Real-time, algorithmic synthesis methods for the final 32 General MIDI +programs, written for implementation in C inside the Aesthetic Computer +native audio engine (`fedac/native/src/audio.c`). Every recipe is sized +for the existing per-sample voice loop: a `switch` on `WaveType`, one +`ACVoice` worth of state, 32 voices, sample rates up to 192 kHz. + +This part deliberately reuses the engine's existing physical models: + +| Existing model (`audio.c`) | Reused for | +|---------------------------------------|------------| +| `WAVE_HARP` — Karplus-Strong loop (`generate_harp_sample`) | sitar, banjo, shamisen, koto, fret noise | +| `WAVE_GUN` — DWG bore + body modes + Friedlander blast (`generate_gun_*`) | gunshot, taiko/drum transients | +| `WAVE_WHISTLE` — Cook flute waveguide (`generate_whistle_sample`) | bagpipe, shanai reeds, bird tweet | +| `WAVE_NOISE` — biquad-filtered xorshift noise | seashore, applause, breath, helicopter, reverse cymbal | +| `WAVE_PIANO` — modal additive + inharmonic partials | template for *all* modal percussion (steel drum, bell, agogo, woodblock, tom, kalimba) | + +Two new reusable primitives are proposed and used throughout: + +1. **`modal_bank`** — a parallel bank of N two-pole resonators + (biquad band-pass), each tuned to a measured modal frequency with its + own T60 decay. This is the workhorse for *all* pitched/struck metal & + wood percussion and for the kalimba tine. It generalizes the piano's + additive partial loop into a struck-resonator loop. +2. **`phisem`** — Perry Cook's *Physically Informed Stochastic Event + Modeling*: a stochastic shake-energy variable that fires random + collision impulses into one or more resonators. This is the workhorse + for the *particle* sounds: rain, applause, seashore, and the shaker + family. + +--- + +### Shared primitive A — Modal resonator bank + +A single mode is a two-pole resonator (a biquad with the band-pass +numerator). Given a mode frequency `f_k`, a pole radius `R_k` (decay), +and the sample rate `sr`: + +``` +ω_k = 2π·f_k / sr +R_k = exp(−π / (T60_k · sr) · 6.9078 / 6.9078) // see below +a1 = −2·R_k·cos(ω_k) +a2 = R_k·R_k +// band-pass numerator (resonator, unity-ish peak gain): +b0 = (1 − R_k) // simple normalized 2-pole resonator +y[n] = b0·x[n] − a1·y[n−1] − a2·y[n−2] +``` + +**T60 → pole radius.** A mode that loses 60 dB over `T60` seconds has a +per-sample amplitude multiplier `R = 10^(−3/(T60·sr)) = exp(−6.9078/(T60·sr))`. +Equivalently, using the loss factor formulation from Nathan Ho's modal +synthesis notes, `R_k = 1/τ_k` with `τ ≈ T60/4.605` (τ is the 1/e time). +Frequency-dependent damping uses `R_k = b1 + b3·f_k²` so high modes die +faster — exactly the perceptual "bright attack mellowing to a hum" the +piano model already exploits (Weinreich's `damping ∝ ω²`). + +The struck excitation is a 1–3 sample impulse (or a short ~2 ms +raised-cosine burst) injected as `x[n]` into every resonator in parallel; +the modes ring on their own afterward. This is cheaper than additive +sine partials because each mode is 1 multiply-add chain instead of a +`sin()` per sample, and it self-decays without an envelope multiply. + +> **Sources:** Adrien, J.-M. (1991) "The Missing Link: Modal Synthesis," +> in *Representations of Musical Signals*, MIT Press, pp. 269–298. +> Smith, J.O., *Physical Audio Signal Processing* (CCRMA, online), +> "Modal Synthesis" — https://ccrma.stanford.edu/~jos/pasp/ . +> Nathan Ho, "Exploring Modal Synthesis" — +> https://nathan.ho.name/posts/exploring-modal-synthesis/ (loss-factor +> `R_k = b1 + b3·f_k²`; free-free & cantilever beam ratios used below). + +### Shared primitive B — PhISEM particle engine + +Cook's PhISEM models a system of many colliding particles (beans, beads, +coins, raindrops, clapping hands) by a single energy variable that leaks +away and stochastically fires collision events into resonator(s). + +The canonical STK `Shakers` per-sample tick: + +``` +// shakeEnergy is pumped on noteOn / shake; it leaks each sample: +shakeEnergy *= systemDecay; // e.g. 0.999 (maraca) +// probability that one of N particles collides this sample: +if (random_uniform() < (numObjects * shakeEnergy)) { + sndLevel += shakeEnergy; // inject a collision impulse +} +sndLevel *= soundDecay; // e.g. 0.95 — per-collision ring decay +input = sndLevel * white_noise(); // each collision = a noise grain +// excite a 2-pole resonator at the system resonance: +output = input - a1*y1 - a2*y2; // a1=-2R cosω, a2=R² +y2 = y1; y1 = output; +``` + +**Measured STK constants** (Cook / STK `Shakers.cpp`): + +| Instrument | numObjects | systemDecay | soundDecay | resFreq(s) Hz | pole R | gain | +|-------------|-----------:|------------:|-----------:|---------------|-------:|-----:| +| Maraca | 25 | 0.999 | 0.95 | 3200 | 0.96 | 4.0 | +| Sekere | 64 | 0.999 | 0.96 | 5500 | 0.60 | 4.0 | +| Cabasa | 512 | 0.997 | 0.96 | 3000 | 0.70 | 8.0 | +| Tambourine | 32 | 0.9985 | 0.95 | 2300/5600/8100| .96/.99/.99 | .1/.8/1 | +| Sleighbells | 32 | 0.9994 | 0.97 | 2500/5300/6500/8300/9800 | 0.99 | 1/1/1/.5/.3 | +| Guiro | 128 | — | 0.95 | 2500/4000 | 0.97 | 1/1 | + +The same engine, with `numObjects` raised into the hundreds–thousands +and per-grain resonance broadened, becomes rain, applause, frying bacon, +a waterfall, and crowd noise — Cook explicitly reports all of these +emerge from the one rain model by sliding parameters. + +> **Sources:** Cook, P.R. (1997) "Physically Informed Sonic Modeling +> (PhISM): Synthesis of Percussive Sounds," *Computer Music Journal* +> 21(3), pp. 38–49. Cook, P.R. (2002) *Real Sound Synthesis for +> Interactive Applications*, A K Peters — ch. on PhISEM/particle models +> (rain, applause, maracas, windchimes, coins, gravel, ice). +> STK `Shakers` class — https://ccrma.stanford.edu/software/stk/classstk_1_1Shakers.html . +> McGill MUMT-618 PhISEM notes — https://www.music.mcgill.ca/~gary/618/week12/phism.html . + +--- + +# Family 1 — Synth Effects (GM 97–104) + +These are *sound-design recipes*, not single physical instruments. The +GM/Roland Sound Canvas FX patches are layered: a tonal core (FM or +detuned saws), a noise/texture bed, and a time-domain effect +(delay/echo, chorus, ring mod). The engine already has a global +reverb/room, glitch, drive, and wobble/flange; these patches mostly need +a per-voice **delay line**, an **LFO**, and a **ring-mod** multiply, +which are cheap additions to `ACVoice`. + +> **General source for this family:** Reid, G., "Synth Secrets" (Sound On +> Sound, 63-part series, 1999–2004), esp. *Synthesizing Bells* (FM +> enharmonic partials) — https://www.soundonsound.com/techniques/synthesizing-bells — +> and the additive/pad/ring-mod installments: +> https://www.soundonsound.com/series/synth-secrets-sound-sound . + +## 97 — FX 1 (rain) +- **Method:** PhISEM with very high `numObjects` (≈800–2000 droplets), + short `soundDecay` (≈0.92), and a wide band-pass resonance (low R ≈ + 0.6) around 1.5–4 kHz so each drop is a soft tick rather than a pitch. + Layer a slowly-detuned shimmer (two sines a few cents apart, ring-mod'd) + for the "musical FX-1" Roland flavor that distinguishes the GM patch + from a pure rainfall SFX. "Heavier rain" = raise `shakeEnergy` floor + and `numObjects`. +- **C state:** `phisem{energy, sysDecay=0.9995, sndDecay=0.92, num, res biquad}` + + a 2-osc shimmer pair. +- **Key params:** droplet density (energy × num), resonance brightness, + shimmer detune. + +## 98 — FX 2 (soundtrack) +- **Method:** Detuned sawtooth pad (2–3 saws, ±5–12 cents) through a slow + LFO-swept low-pass, plus a slow stereo chorus. This is the classic + "sweeping ensemble" pad. Use 3 phase-increment saws summed, a 1-pole + LP whose cutoff is modulated by a ~0.1–0.3 Hz triangle LFO. +- **C state:** 3 saw phases + LFO phase + LP state. +- **Key params:** detune spread, LFO rate/depth, cutoff range. + +## 99 — FX 3 (crystal) +- **Method:** Bell-like FM (a la Synth Secrets *Synthesizing Bells*): one + modulator FM-ing a carrier at a non-integer ratio (e.g. 1:1.41 or + 1:3.5) to make inharmonic metallic partials, with a fast-decaying + envelope and a bright, glassy attack. Add a short feedback delay + (~80–150 ms) for the twinkling repeats. Two FM pairs detuned makes the + "dense enharmonic fog" Reid describes. +- **C state:** carrier phase + modulator phase + mod index env + delay ring. +- **Key params:** C:M ratio (inharmonicity), mod index decay, delay time/feedback. + +## 100 — FX 4 (atmosphere) +- **Method:** Soft, breathy pad: filtered noise bed (LP noise, slow + amplitude LFO) + a low detuned sine/triangle drone + slow chorus. Think + "guitar-harp-into-pad" Roland patch — a plucked KS attack (reuse + `WAVE_HARP`) crossfading into a sustained noise-pad tail. +- **C state:** KS loop (existing harp) + noise biquad + drone osc + xfade env. +- **Key params:** noise/tone balance, attack pluck brightness, drone pitch. + +## 101 — FX 5 (brightness) +- **Method:** Bright additive/FM pad with a strong high-harmonic content + and a slow filter-open sweep. Stacked saw + a high-ratio FM partial + (C:M ≈ 1:7) with the modulator on an attack-rising envelope so the + timbre brightens *into* the note. PWM (Reid's pad trick) adds motion. +- **C state:** saw phase + FM pair + rising mod-index env + LFO for PWM. +- **Key params:** brightness (top harmonic level), sweep rate, PWM depth. + +## 102 — FX 6 (goblins) +- **Method:** Dark, vocal, evolving pad with ring modulation and slow + random/sample-hold modulation of pitch and filter — the "ominous voice" + patch. Core is a low detuned-saw pad ring-modulated by a low (40–120 Hz) + sine to fracture it into a growl, plus a slow random LFO (sample-and-hold) + bending the cutoff. Formant-ish band-passes give the "ah/oh" vowel. +- **C state:** saw pair + ringmod sine + S&H LFO + 2 formant biquads. +- **Key params:** ringmod freq, S&H rate, formant centers. + +## 103 — FX 7 (echoes) +- **Method:** A bright pluck/bell tone fed into a long, regenerating + multi-tap delay with each tap filtered darker (HF damping) so repeats + fade *and* dull — the canonical "echo drops." Reuse the harp pluck or a + short FM ping as the source. +- **C state:** source voice + delay ring (≥400 ms @ sr) + per-tap 1-pole LP + feedback. +- **Key params:** delay time (often tempo-synced), feedback, HF damping. + +## 104 — FX 8 (sci-fi) +- **Method:** Aggressive sweep: a saw/square swept by a fast envelope + through a resonant filter, with pitch glide and ring/FM noise bursts — + "laser zap / sci-fi" gesture. Reuse the engine's resonant biquad with a + fast downward cutoff sweep + a downward pitch sweep (the gun-classic + pitch-sweep machinery is directly applicable). Add a noise burst at the + attack. +- **C state:** osc phase + pitch-sweep mult + resonant biquad with swept cutoff + noise burst env. +- **Key params:** sweep direction/rate, resonance Q, noise transient level. + +--- + +# Family 2 — Ethnic (GM 105–112) + +Plucked members (sitar, banjo, shamisen, koto) are **extended +Karplus-Strong** — the engine's `WAVE_HARP` is the base; each adds a +characteristic tweak. Kalimba is a **modal tine** (cantilever beam). +Bagpipe and shanai are **reed waveguides + drone** (extend `WAVE_WHISTLE`). +Fiddle is a **bowed waveguide** (friction-driven loop). + +> **Plucked-string foundation (all four):** Karplus, K. & Strong, A. +> (1983) "Digital Synthesis of Plucked-String and Drum Timbres," *CMJ* +> 7(2), pp. 43–55. Jaffe, D. & Smith, J.O. (1983) "Extensions of the +> Karplus-Strong Plucked-String Algorithm," *CMJ* 7(2), pp. 56–69 +> (fractional delay, tunable loop filter, decay-stretch). Smith, +> *Physical Audio Signal Processing* (CCRMA) — +> https://ccrma.stanford.edu/~jos/pasp/ . These are already cited in +> `audio.h` above the harp fields. + +## 105 — Sitar +- **Method:** Karplus-Strong main string **+ sympathetic strings + + jawari (buzzing bridge) nonlinearity.** The jawari is a curved bridge + the string slaps against; it imposes a *position-dependent, dynamic + delay-length modulation* (the string's effective length shortens when it + touches the bridge), generating the buzzing high-overtone shimmer. + Practical real-time recipe (Siddiq): run the main KS string, then pass + the loop signal through a soft nonlinearity / dynamic fractional-delay + modulator keyed by signal amplitude (large excursions "buzz" against the + bridge → add a short, signal-dependent delay perturbation + mild + waveshaping). Add 9–13 detuned sympathetic strings as cheap parallel KS + loops (or a comb bank) tuned to the raga, lightly coupled to the main + string's output — they bloom as the note sustains. +- **C state (extends harp):** main KS loop (existing) + `jawari_thresh`, + `jawari_depth` (dynamic delay perturbation) + array of N sympathetic + KS delay lines with their own `lp1`. +- **Key params:** jawari buzz depth/threshold (the defining timbre), + sympathetic count/tuning/coupling, pluck position (loop-filter + brightness). + +> **Sources:** Siddiq, S. (2012) "A Physical Model of the Nonlinear Sitar +> String," *Archives of Acoustics* 37(1) — +> https://acoustics.ippt.pan.pl/index.php/aa/article/view/129 . +> "The Physical Modelling of a Sitar," ISSTA — +> http://issta.ie/wp-content/uploads/The-Physical-Modelling-of-a-Sitar.pdf +> (jawari as dynamically-changing delay line). Tanpura jvari: same +> bridge-string collision mechanism. + +## 106 — Banjo +- **Method:** Karplus-Strong with a **bright, lightly-damped loop** + (short, fast-decaying — banjo notes ring briefly and plinky) **+ a + resonant drumhead body.** The banjo head is a tensioned membrane: model + it as 2–3 parallel biquad body resonances (≈300 Hz head mode + a couple + higher) excited by the string output, lending the snappy "pop." Keep the + KS loop-filter cutoff high (less HF damping than guitar) for the + characteristic twang; short T60. +- **C state (extends harp):** KS loop + 2–3 body biquads (reuse the gun + `gun_body_*` 3-biquad bank verbatim). +- **Key params:** loop decay (short), loop brightness (high), head + resonance freq/Q, body mix. + +> **Source:** Politis, Bank et al. / "Acoustics of the banjo: measurements +> and sound synthesis," *Acta Acustica* 5 (2021) — +> https://acta-acustica.edpsciences.org/articles/aacus/full_html/2021/01/aacus200055/aacus200055.html +> (head membrane + bridge resonances over the string). + +## 107 — Shamisen +- **Method:** Karplus-Strong **+ sawari buzz** (the shamisen's deliberate + buzzing at the nut on the lowest string — a *lighter cousin of the + jawari*) **+ skin-membrane body.** Use the sitar's jawari nonlinearity + at low depth, a percussive plectrum (bachi) attack (short bright noise + burst before the pluck), and 2 body resonances for the skin. Strong + attack transient, fast decay. +- **C state:** KS loop + low-depth jawari perturbation + attack noise + burst + 2 body biquads. +- **Key params:** sawari buzz (subtle), bachi attack brightness, decay + (short), body resonance. + +## 108 — Koto +- **Method:** Karplus-Strong **+ pitch bend** (the koto's signature is + *oshi-de* / *ato-oshi* pressing the string left of the movable bridge to + bend pitch up a semitone/tone). The existing harp's `target_frequency` + smoothing already provides glide — drive it from a per-note bend + envelope. Mellow, harp-like loop (more HF damping than banjo), medium + decay, with a soft felt-pick attack. +- **C state (extends harp):** KS loop + bend envelope writing + `target_frequency`. +- **Key params:** bend amount/curve (defining gesture), loop decay + (medium-long), attack softness. + +> **Source (koto bend technique):** Koto (instrument), Wikipedia — +> https://en.wikipedia.org/wiki/Koto_(instrument) (ōshi pressure-bending +> raises pitch a half/whole tone). + +## 109 — Kalimba +- **Method:** **Modal tine** — a thumb-piano lamella is a *cantilever + beam* (clamped one end, free the other, intermediate bridge). Its + overtones are strongly inharmonic: measured ratios put the prominent + overtones at **≈5× and ≈14× the fundamental** (Euler-Bernoulli beam). + Use the modal bank with 3 modes at ratios `{1.0, 5.4, 14.7}` (cantilever + `B^cant`: 0.597, 1.494, 2.500 → squared-scaled gives the ~5×, ~14× + family), fast HF decay (top mode T60 ≈ 60 ms, fundamental ≈ 600 ms–1 s), + struck by a short impulse. A soft "thunk" body resonance (the wooden + box) under it. This is far more authentic than KS for the kalimba's + pure, bell-like tine ping. +- **C state:** `modal_bank` with 3 resonators (freqs, T60s, gains) + 1 + body biquad. +- **Key params:** inharmonic mode ratios (≈1 / 5 / 14), per-mode decay, + attack hardness, box resonance. + +> **Sources:** "The tones of the kalimba (African thumb piano)" — and +> "Vibrational frequencies and tuning of the African mbira" (ResearchGate) +> — first three transverse beam modes, prominent overtones ~5× & ~14× +> fundamental: +> https://www.researchgate.net/publication/221780579_The_tones_of_the_kalimba_African_thumb_piano . +> Cantilever-beam ratios from Nathan Ho / modal-synthesis literature +> (`B^cant(1)=0.597, (2)=1.494, (3)=2.500`). + +## 110 — Bag pipe +- **Method:** **Reed waveguide chanter + continuous drones.** Reuse the + whistle/flute waveguide (`WAVE_WHISTLE`) but swap the air-jet excitation + for a *single/double reed* nonlinearity: a pressure-controlled reed + reflection (clipped/saturating reflection coefficient at the bore + entrance) gives the buzzy, constant-pressure timbre. Crucially: **no + envelope silence between notes** (bagpipe is continuously blown — pitch + changes are slurred) and **two or three fixed drone voices** (tonic + + octave) sounding underneath at all times. +- **C state:** whistle bore delay + reed reflection table/saturation + + separate sustained drone voices. +- **Key params:** reed stiffness (buzz), bore length (pitch), drone + pitches, continuous (no-gap) legato. + +> **Source:** woodwind digital-waveguide reed modeling (reed as SHO + +> bore transmission line; clipped reflection) — "Synthesis of woodwind +> instruments sounds using digital waveguide modelling" — +> https://www.academia.edu/8223502/ . Cook flute model already in +> `generate_whistle_sample` (`audio.h:88`) is the structural base. + +## 111 — Fiddle +- **Method:** **Bowed digital waveguide.** A string delay loop driven by a + *friction (bow) excitation*: at the bow point, the string velocity is + pushed by a nonlinear bow-friction curve (stick-slip — a sharply + saturating function of the velocity difference between bow and string). + This is the McIntyre-Schumacher-Woodhouse / Smith bowed-string model: + two delay rails (nut side, bridge side) summed at the bow point through + the friction table; add vibrato LFO on the loop length. Sustained, not + plucked — the bow continuously injects energy. +- **C state:** two delay rails (or one loop + bow-point tap) + friction + lookup/saturation + bow velocity/force params + vibrato LFO + body biquads. +- **Key params:** bow velocity & force (timbre/loudness), bow position + (harmonic content), vibrato depth/rate, body resonance. + +> **Sources:** Smith, J.O., *PASP*, "Bowed Strings" / digital-waveguide +> bowed string — https://ccrma.stanford.edu/~jos/pasp/ . McIntyre, M., +> Schumacher, R., Woodhouse, J. (1983) "On the oscillations of musical +> instruments," *JASA* 74(5). "Empirical physical modeling for bowed +> string instruments" (ResearchGate, friction model). + +## 112 — Shanai (shehnai) +- **Method:** **Double-reed waveguide + drone**, very close to bagpipe but + *with* expressive per-note envelopes and bends (it's a played melodic + oboe-like instrument, not a constant drone instrument — though often + accompanied by a sur/drone). Reuse the bagpipe reed waveguide with a + brighter, more open reflection (richer high harmonics, nasal formant + via a couple of band-passes), strong vibrato, and pitch-bend glides. +- **C state:** whistle bore + double-reed saturation + 2 formant biquads + + vibrato + bend env. +- **Key params:** reed brightness, nasal formant centers, vibrato, + expressive bend. + +--- + +# Family 3 — Percussive (GM 113–120) + +The pitched/struck metal & wood instruments (tinkle bell, agogo, steel +drum, woodblock, melodic tom, synth drum) are **modal synthesis** — the +shared modal bank, tuned to measured frequency ratios. Membrane drums +(taiko, tom) add a 2D-membrane / nonlinear-tension flavor. Reverse cymbal +is the odd one out: **reversed noise envelope.** + +## 113 — Tinkle Bell +- **Method:** Modal bank, **bell ratios.** A small struck bell has + characteristic inharmonic partials. Use 4–5 modes at the classic bell + ratios relative to a nominal strike note: **hum ≈0.5, prime 1.0, tierce + ≈1.2 (minor third), quint ≈1.5, nominal ≈2.0**, plus a high "tink" + partial ~3–4×. High partials decay fast (tinkle = bright, short); a + small bell skips the long hum, so weight the upper modes and keep T60 + ≈0.3–0.8 s. +- **C state:** `modal_bank` 5 modes (freqs from ratios, decreasing T60). +- **Key params:** mode ratios (esp. minor-third tierce → "bell" identity), + brightness, short decay. + +> **Source:** Reid, "Synthesizing Bells" (Sound On Sound) — bell partials +> hum/prime/tierce(minor 3rd)/quint/nominal; FM or additive realizations — +> https://www.soundonsound.com/techniques/synthesizing-bells . +> Fletcher & Rossing, *The Physics of Musical Instruments* (2nd ed., +> Springer 1998), bell-partial chapter. + +## 114 — Agogo +- **Method:** Modal bank, **two-pitch metal bell** (the agogo is a pair of + tuned cowbell-like bells, usually a ~minor-third or fourth apart). + Cowbell-style metal: 2–3 sharply inharmonic modes (a clangy, slightly + detuned pair around ~550 Hz & ~830 Hz for the big bell), very short + bright decay, hard mallet attack (1-sample impulse + tiny click). Pick + bell A vs B by note range. +- **C state:** `modal_bank` 2–3 modes × two presets (hi/lo bell). +- **Key params:** the two bell pitches, inharmonic detune, hard short + decay, metallic brightness. + +## 115 — Steel Drums (steelpan) +- **Method:** Modal bank with the steelpan's **specific inharmonic + modes.** Measured steelpan notes are tuned so the *octave and twelfth + are reinforced*, but the third prominent mode is characteristically + **non-harmonic** — often near an **octave-plus-a-third or + octave-plus-a-fourth** above the fundamental (≈2.5–2.66×) rather than a + pure 2× or 3×. Use ~5 modes at ratios `{1.0, 2.0, ~2.6, 3.0, ~4.2}` with + medium decay and a bright shimmer; the note "blooms" because energy + transfers between modes (nonlinear coupling). A cheap nod to that + nonlinearity: feed a little of the fundamental mode's output, squared, + into the upper modes so they swell after the strike. +- **C state:** `modal_bank` 5 modes + optional nonlinear cross-feed term. +- **Key params:** the non-harmonic 3rd mode ratio (the steelpan + fingerprint), bloom/coupling amount, medium-long ringing decay. + +> **Sources:** Rossing, T.D. et al., steelpan modal studies; Monteil, Touzé +> et al., "Identification of mode couplings in nonlinear vibrations of the +> steelpan," *Applied Acoustics* (2015) — +> https://www.sciencedirect.com/science/article/abs/pii/S0003682X14002151 +> (nonlinear energy exchange between modes). Stockholm Steel Band, "Tone +> generation in steel pans" — third mode often an octave-plus-third/fourth +> above fundamental: +> https://stockholmsteelband.se/pan/tuning/theory20_tone_generation.php . + +## 116 — Woodblock +- **Method:** Modal bank, **free-free bar / wood block** with very few, + high, fast-decaying modes. A wood block is mostly one dominant pitched + "tock" plus a couple of higher partials. Use 2–3 modes (`f`, ~2.7·f, + ~5.4·f from free-free beam ratios `B^free`: 1.506, 2.500, …) with very + short T60 (≈40–120 ms) and a hard impulse excitation. The "tock" comes + from the fast decay + the dominant low mode; wood = quick HF rolloff. +- **C state:** `modal_bank` 2–3 modes, short T60, hard impulse. +- **Key params:** dominant pitch, very short decay, wooden (dull-but-clicky) + brightness. + +> **Source:** free-free beam mode ratios (`B^free(1)=1.506, (2)=2.500, +> (k)=k+0.5`) — xylophone/marimba/wood-bar family — Nathan Ho modal notes; +> Fletcher & Rossing ch. on bars. + +## 117 — Taiko Drum +- **Method:** **Membrane modes + body + pitch drop.** A large drum is a + circular membrane: its modes follow Bessel-zero ratios `{1.0, 1.59, + 2.14, 2.30, 2.65, 2.92, …}` (the `(m,n)` modes `J_(m-1)n`). For a deep + taiko boom, weight the low modes heavily, add a **downward pitch sweep** + (membrane tension/air-load makes the perceived pitch drop right after + the hit — the gun-classic `boom` pitch-sweep machinery is a perfect + fit), plus a big low-frequency body/shell resonance and a noise "thwack" + transient. The existing `WAVE_GUN` boom+tail layering is *directly* + reusable for taiko (low sine boom with exp pitch drop + LPF noise tail). +- **C state:** 3–5 membrane modes (or reuse gun boom-sine + pitch-sweep) + + shell biquad + attack noise burst. +- **Key params:** fundamental (big & low), pitch-drop depth/rate, membrane + vs body balance, attack noise. + +> **Sources:** Fletcher & Rossing, membrane chapter (Bessel-zero mode +> ratios). Cook, *Real Sound Synthesis*, drum/membrane modeling. Reuse of +> the engine's `generate_gun_classic_sample` boom+tail (`audio.c:1451+`, +> documented `audio.h:142+`). + +## 118 — Melodic Tom +- **Method:** Same membrane-mode engine as taiko, **tuned and pitched** + across notes, less boom and more tonal "dooong." Fewer low modes, + emphasize the membrane fundamental, moderate pitch drop, tighter tuning + (tunable drums favor the first membrane mode). Pitch tracks the MIDI + note. Medium decay. +- **C state:** membrane modes scaled to note + moderate pitch sweep + body + biquad. +- **Key params:** note pitch, modest pitch drop, decay length, head + tension (mode spread). + +## 119 — Synth Drum +- **Method:** **Pitched sine/triangle with a fast downward pitch sweep** + (the 808/Simmons "pew" tom) + a click transient. This is the synthetic + cousin of the melodic tom — *not* physically modeled. A single sine, + amplitude AD-envelope (fast attack, exp decay), frequency exponentially + sweeping from ~3–5× start down to the note (reuse `gun_boom` sweep). Add + a tiny noise click. Optionally a touch of FM for the "electro" zap. +- **C state:** 1 sine osc + pitch-sweep mult + amp env + click env. +- **Key params:** start/end pitch ratio (the "dewww" amount), decay, + click level, optional FM index. + +## 120 — Reverse Cymbal +- **Method:** **Reversed-envelope filtered noise.** A cymbal is dense + high-frequency noise; "reverse" means the amplitude envelope *rises* to + a peak and then cuts — a swell. Generate band-passed/high-passed bright + noise (reuse `WAVE_NOISE` with a high cutoff, plus a couple of resonant + metallic band-passes for shimmer) under a **rising attack envelope** + (linear or exp rise over ~0.5–2 s) terminated by a hard cut at the + downbeat. No pitch. +- **C state:** noise biquad(s) + rising-envelope generator + hard cut. +- **Key params:** swell duration, brightness (cutoff), 2–3 metallic + resonances, cut sharpness. + +--- + +# Family 4 — Sound Effects (GM 121–128) + +Noise shaping + filtering + amplitude/pitch modulation. Several map onto +existing engine models almost directly (gunshot → `WAVE_GUN`, bird → +`WAVE_WHISTLE`, seashore/applause/breath/helicopter → `WAVE_NOISE` / +PhISEM). + +## 121 — Guitar Fret Noise +- **Method:** Short **filtered-noise squeak/scrape** — finger sliding on a + wound string. A band-passed noise burst (centered ~1.5–4 kHz) with a + fast pitch/cutoff glide (the squeak rises or falls as the finger moves) + and a very short envelope, optionally a faint plucked KS click at the + end (the finger landing). Reuse `WAVE_NOISE` + a swept band-pass cutoff. +- **C state:** noise biquad with swept center freq + short env. +- **Key params:** scrape brightness, glide direction/rate, duration + (very short). + +## 122 — Breath Noise +- **Method:** **Filtered-noise puff** — band-limited (low-pass ~2–4 kHz, + high-pass ~300 Hz) white noise with a soft attack-decay envelope, gentle + amplitude wobble. This is the same air-noise the whistle/flute model + already injects; standalone it's just shaped noise. Optionally add a + faint formant band-pass for a more "hh" vocal breath. +- **C state:** noise biquad (BP) + AD env + slow amp LFO. +- **Key params:** band center/width, puff length, breathiness. + +## 123 — Seashore +- **Method:** **PhISEM / filtered-noise surf** — broadband noise shaped by + a *slow swell LFO* (waves come and go over ~3–8 s) through a low-pass + whose cutoff opens at the wave's peak (the "shhh" brightens as the wave + breaks). Equivalent to Cook's rain model with huge `numObjects` and a + very slow energy LFO. Layer two or three independent swells at different + rates/phases for a natural, non-repeating shore. +- **C state:** noise biquad + 2–3 slow swell LFOs (random rates) modulating + amplitude *and* cutoff. +- **Key params:** swell rate(s) & depth, brightness at peak, number of + overlapping swells. + +> **Source:** Cook, *Real Sound Synthesis* — the rain/PhISEM model yields +> waterfall/surf by parameter change; filtered-noise-with-swell is the +> standard ocean recipe. + +## 124 — Bird Tweet +- **Method:** **Pitch-swept sine / small-index FM** in the 2–8 kHz bird + range. A bird chirp is a fast frequency glide (often up-then-down) on a + near-sine, with rapid trills (a fast LFO on pitch) and short syllables. + Recipe: a sine whose frequency follows a short swept envelope (e.g. + rise 3→5 kHz over 40 ms), optional shallow FM for the "warble," gated + into 1–3 quick syllables. Reuse the whistle waveguide for a more "tweet" + timbre, or a plain swept sine for the classic GM bird. +- **C state:** sine phase + pitch-sweep envelope + trill LFO + syllable + gating env. +- **Key params:** frequency range/sweep shape, trill rate, syllable count. + +> **Source:** birdsong is dominated by rapid FM in the ~2–8 kHz band — +> Stowell & Plumbley (2014), "Large-scale analysis of frequency modulation +> in birdsong databases," *Methods in Ecology and Evolution* — +> https://besjournals.onlinelibrary.wiley.com/doi/full/10.1111/2041-210X.12223 . + +## 125 — Telephone Ring +- **Method:** **Dual-sine tone, gated.** A North-American ringtone is two + sines (440 Hz + 480 Hz) summed, hard-gated in a **2 s on / 4 s off** + cadence (the classic telephone-ring pattern, conceptually DTMF-like dual + tones). For a "trimphone/electronic" warble, frequency-modulate the + pair. Trivial: 2 phase-increment sines + a square-wave gate envelope. +- **C state:** 2 sine phases + cadence gate counter. +- **Key params:** the two frequencies (440/480 = US; 400/450 = UK), + on/off cadence, warble rate. + +## 126 — Helicopter +- **Method:** **Amplitude-modulated broadband noise (blade chop) + low + rotor thump.** The rotor produces periodic amplitude modulation (PAM) of + broadband aero-noise: take low-pass noise and multiply by a periodic + pulse train at the blade-passage rate (~10–20 Hz: rotor RPM × blade + count), giving the "whump-whump-whump." Add a low body/engine rumble + (LP noise or a low buzzy oscillator) and a high turbine whine. The pulse + shape (sharper = more "slap") sets the character. +- **C state:** noise biquad (broadband) × periodic AM pulse (phase-increment + at blade rate, shaped) + low rumble layer. +- **Key params:** blade-passage rate (chop tempo), pulse sharpness (blade + slap), rumble/whine balance. + +> **Source:** periodic amplitude modulation (PAM) model of rotor +> aero-noise — "An analytical approach to rotor blade modulation," +> *Applied Acoustics* (2021) — +> https://www.sciencedirect.com/science/article/abs/pii/S0165212521000603 . + +## 127 — Applause +- **Method:** **PhISEM crowd model.** Each clap is a short filtered-noise + burst; a crowd is many stochastic clap events. Use the PhISEM engine + with high `numObjects` (hundreds–thousands of clappers), a clap + resonance band-pass (~1–2 kHz), and an "affinity" parameter (Cook's + ClapLab): low affinity → dense random wash (full applause), high + affinity → synchronized rhythmic clapping. Swell the overall energy in + and out. This is literally one of Cook's named PhISEM outputs. +- **C state:** PhISEM (energy, sysDecay, large num, clap biquad) + + affinity/clustering term + global swell env. +- **Key params:** crowd density (num × energy), clap brightness, affinity + (sync vs wash), swell. + +> **Source:** Cook, *Real Sound Synthesis* — ClapLab / applause from +> PhISEM (mean/SD of clap center freq, period, and "affinity" = +> sync tendency, 0 random … 128 unison). + +## 128 — Gunshot +- **Method:** **Use the existing `WAVE_GUN` model directly.** The engine + already implements both a 3-layer classic gunshot (crack BPF burst + + pitched boom with downward sweep + LPF noise tail + sub-ms click) and a + physical DWG model (Friedlander blast-wave excitation → bore resonance + + 3 body-mode biquads + radiation HPF + ground-reflection echo), with 12 + weapon presets. For GM "Gunshot," `GUN_PISTOL` or `GUN_RIFLE` classic is + the canonical short bang. No new code needed — just route program 128 to + `audio_synth_gun(...)`. +- **C state:** existing `gun_*` fields in `ACVoice`. +- **Key params:** preset (pistol/rifle), `pressure_scale` (loudness/size), + model (classic vs physical). + +> **Source:** in-engine model, documented `audio.h:38–179` & +> `audio.c:generate_gun_*`. Friedlander blast-wave reference embedded +> there; classic 3-layer per `audio.h:142+`. + +--- + +## Implementation map — what to add to `audio.c` + +1. **`WAVE_MODAL`** + a `modal_bank` (N≤6 two-pole resonators in + `ACVoice`: `mode_f[6]`, `mode_R[6]`, `mode_g[6]`, `mode_y1[6]`, + `mode_y2[6]`, struck by an impulse on note-on). Serves: kalimba, tinkle + bell, agogo, steel drum, woodblock, melodic tom (and taiko's tonal + part). Tables of `{ratios, T60s, gains}` per instrument, exactly like + `gun_presets[]`. +2. **`WAVE_PHISEM`** + a particle struct (`energy, sysDecay, sndDecay, + numObjects, res biquad(s), gain`). Serves: rain, seashore, applause, + and the shaker family. Constant tables from the STK numbers above. +3. **Reuse**: `WAVE_HARP` (sitar/banjo/shamisen/koto/fret — add a + `jawari` nonlinearity flag + sympathetic-loop array + bend env), + `WAVE_WHISTLE` (bagpipe/shanai/bird — add a reed-saturation reflection + option + drone voices), `WAVE_NOISE` (breath/helicopter/reverse-cymbal + — add swept cutoff + AM-pulse + rising-envelope options), `WAVE_GUN` + (gunshot/taiko boom — already present). +4. **Small new per-voice DSP utilities** shared by Synth FX: a delay-line + ring (echoes/crystal/atmosphere), an LFO phase, a ring-mod multiply. + +Per-sample cost stays low: modal = N×(1 mul-add chain); PhISEM = 1 RNG + +1 resonator; everything else is the existing oscillator/noise/waveguide +loops. All fit comfortably inside 32 voices at 192 kHz. + +--- + +## Consolidated references + +- Karplus, K. & Strong, A. (1983). "Digital Synthesis of Plucked-String + and Drum Timbres." *Computer Music Journal* 7(2): 43–55. +- Jaffe, D.A. & Smith, J.O. (1983). "Extensions of the Karplus-Strong + Plucked-String Algorithm." *CMJ* 7(2): 56–69. +- Smith, J.O. *Physical Audio Signal Processing* (online, CCRMA). https://ccrma.stanford.edu/~jos/pasp/ +- Cook, P.R. (1997). "Physically Informed Sonic Modeling (PhISM)." *CMJ* 21(3): 38–49. https://ccrma.stanford.edu/software/stk/classstk_1_1Shakers.html +- Cook, P.R. (2002). *Real Sound Synthesis for Interactive Applications.* A K Peters. +- McGill MUMT-618 PhISEM notes. https://www.music.mcgill.ca/~gary/618/week12/phism.html +- Adrien, J.-M. (1991). "The Missing Link: Modal Synthesis." In *Representations of Musical Signals*, MIT Press. +- Nathan Ho, "Exploring Modal Synthesis." https://nathan.ho.name/posts/exploring-modal-synthesis/ +- Siddiq, S. (2012). "A Physical Model of the Nonlinear Sitar String." *Archives of Acoustics* 37(1). https://acoustics.ippt.pan.pl/index.php/aa/article/view/129 +- "The Physical Modelling of a Sitar," ISSTA. http://issta.ie/wp-content/uploads/The-Physical-Modelling-of-a-Sitar.pdf +- "Acoustics of the banjo: measurements and sound synthesis." *Acta Acustica* 5 (2021). https://acta-acustica.edpsciences.org/articles/aacus/full_html/2021/01/aacus200055/aacus200055.html +- "The tones of the kalimba (African thumb piano)." https://www.researchgate.net/publication/221780579 +- Monteil, Touzé et al. (2015). "Identification of mode couplings in nonlinear vibrations of the steelpan." *Applied Acoustics*. https://www.sciencedirect.com/science/article/abs/pii/S0003682X14002151 +- Stockholm Steel Band, "Tone generation in steel pans." https://stockholmsteelband.se/pan/tuning/theory20_tone_generation.php +- Fletcher, H. & Rossing, T.D. (1998). *The Physics of Musical Instruments,* 2nd ed., Springer. +- McIntyre, M., Schumacher, R., Woodhouse, J. (1983). "On the oscillations of musical instruments." *JASA* 74(5). +- Reid, G. "Synth Secrets" (Sound On Sound). https://www.soundonsound.com/series/synth-secrets-sound-sound — esp. "Synthesizing Bells" https://www.soundonsound.com/techniques/synthesizing-bells +- Stowell, D. & Plumbley, M. (2014). "Large-scale analysis of frequency modulation in birdsong databases." *Methods in Ecology and Evolution.* https://besjournals.onlinelibrary.wiley.com/doi/full/10.1111/2041-210X.12223 +- "An analytical approach to rotor blade modulation." *Applied Acoustics* (2021). https://www.sciencedirect.com/science/article/abs/pii/S0165212521000603 + +*Documented: 32 / 32 GM programs (97–128). — fedac native audio research dossier, part 04.* diff --git a/fedac/native/pieces/notepat.mjs b/fedac/native/pieces/notepat.mjs index 41891ab549..521bd5f36f 100644 --- a/fedac/native/pieces/notepat.mjs +++ b/fedac/native/pieces/notepat.mjs @@ -2155,10 +2155,12 @@ function udpMidiSendRecency() { function rememberSound(key, entry, system, velocity = 1) { if (!entry) return; - // Capture linger intent (Feature 4) at press time — shift may release - // before the key. lingerCat picks the fade shape: GM-recipe sustained - // families fade smoothly over ~4s; staccato families get a short tail. - if (entry.lingerOnRelease === undefined) entry.lingerOnRelease = shiftHeld; + // Capture linger/latch intent at press time — shift may release before the + // key. A shift-held note now LATCHES: lifting the key keeps it sounding + // (true sustain) by adding it to heldKeys, exactly like the Enter-hold + // latch. lingerCat is retained for the (now rare) fade-tail path. + const latchOnRelease = shiftHeld; + if (entry.lingerOnRelease === undefined) entry.lingerOnRelease = false; if (entry.lingerCat === undefined) { entry.lingerCat = gmProgram !== null ? gmRecipe(gmProgram).linger : "sustain"; } @@ -2168,9 +2170,10 @@ function rememberSound(key, entry, system, velocity = 1) { // live pressure (55% velocity + 45% smoothed pressure). entry.velocity = velocity; sounds[key] = entry; - // While Enter is held, auto-latch new notes into heldKeys so they - // sustain on key-up. Otherwise notes release normally. - if (enterHeld) heldKeys.add(key); + // While Enter is held OR shift was held at press, auto-latch the note into + // heldKeys so it sustains on key-up. Press the same key again (or panic) to + // release. Otherwise notes release normally. + if (enterHeld || latchOnRelease) heldKeys.add(key); system?.usbMidi?.noteOn?.(entry.midiNote, velocityToMidi(velocity), entry.midiChannel); sendUdpMidiEvent(system, "note_on", entry.midiNote, velocityToMidi(velocity), entry.midiChannel); pushUsbMidiRecent(">", entry.note, entry.octave); @@ -2258,7 +2261,9 @@ function releaseTouchNote(pid, sound, system, fade = 0.08) { if (entry.drumHold) { releasePercussionHold(sound, entry.drumHold); } else if (entry.key) { - stopSoundKey(entry.key, sound, system, fade); + // Latched notes (shift-linger / Enter-hold) stay ringing on lift — match + // the keyboard key-up behavior. + if (!heldKeys.has(entry.key)) stopSoundKey(entry.key, sound, system, fade); } delete touchNotes[pid]; } @@ -2911,6 +2916,17 @@ function act({ event: e, sound, wifi, system }) { } // If this key is currently recording in per-key mode, suppress playback if (perKeyRecording === key) return; + // Latch toggle-off: a note that's currently LATCHED (shift-linger or + // Enter-hold) is still ringing in `sounds[key]`. Pressing its key again + // un-latches and stops it — the intuitive "press again to release" gesture + // for held notes. We force a hard stop (clear lingerOnRelease) so it + // doesn't re-enter a fade tail, and align with the existing heldKeys set. + if (noteName && sounds[key] && heldKeys.has(key)) { + heldKeys.delete(key); + if (sounds[key]) sounds[key].lingerOnRelease = false; + stopSoundKey(key, sound, system, quickMode ? 0.02 : 0.08); + return; + } if (noteName && !sounds[key]) { // Re-arm the audio-history ring before this note hits speakers. // Capture stays paused after a space-release so the ring doesn't @@ -3018,8 +3034,10 @@ function act({ event: e, sound, wifi, system }) { system?.usbMidi?.noteOn?.(mn, velocityToMidi(velocity), 0); sendUdpMidiEvent(system, "note_on", mn, velocityToMidi(velocity), 0); } - sounds[key] = { passthroughNotes: midiNotes, voices, note: letter, octave: noteOctave, gridOffset: offset, lingerOnRelease }; - if (enterHeld) heldKeys.add(key); + sounds[key] = { passthroughNotes: midiNotes, voices, note: letter, octave: noteOctave, gridOffset: offset, lingerOnRelease: false }; + // Shift/Enter latch passthrough notes too — the relayed note_off is + // deferred until un-latch so downstream gear holds the note. + if (enterHeld || lingerOnRelease) heldKeys.add(key); trail[key] = { note: letter, octave: noteOctave, brightness: velocity }; return; } @@ -3057,9 +3075,12 @@ function act({ event: e, sound, wifi, system }) { voices, note: letter, octave: noteOctave, baseFreq: freq, gridOffset: offset, baseVol: baseVol * recipeVol, midiNote: rootMidi, midiChannel: 0, chordIvls: ivls.length > 1 ? ivls : null, velocity, - lingerOnRelease, lingerCat: recipe ? recipe.linger : "sustain", + lingerOnRelease: false, lingerCat: recipe ? recipe.linger : "sustain", }; - if (enterHeld) heldKeys.add(key); + // Shift-held → latch the whole chord (every voice lives on this one + // key, so latching the key keeps all chord tones ringing). Enter-hold + // latches too. Release by re-pressing the key or via panic. + if (enterHeld || lingerOnRelease) heldKeys.add(key); pushUsbMidiRecent(">", letter, noteOctave); trail[key] = { note: letter, octave: noteOctave, brightness: velocity }; return; @@ -5732,6 +5753,24 @@ function paint({ wipe, ink, box, line, write, screen, sound, system, trackpad, p // Expose grid layout for touch hit-testing in act() globalThis.__gridInfo = { leftX, rightX, gridTop, btnW, btnH, gap }; + // Chord-tone HUD highlight: gather every MIDI note that's currently + // SOUNDING as part of a held chord, so drawGrid can light up the whole + // chord shape (root + extensions) on the board — not just the physically + // pressed key. `chordRootMidi` holds the pressed roots so we can tint the + // root differently from the +3/+4/+2/+7 extension tones. Tones whose MIDI + // number doesn't land on any on-board pad are simply not lit. + const chordToneMidi = new Set(); // all chord tones (root + extensions) + const chordRootMidi = new Set(); // just the pressed roots + for (const k of Object.keys(sounds)) { + const s = sounds[k]; + if (!s) continue; + const ivls = s.chordIvls; + if (!ivls || ivls.length < 2) continue; // single notes use the normal path + const root = noteToMidiNumber(s.note, s.octave); + chordRootMidi.add(root); + for (const iv of ivls) chordToneMidi.add(Math.max(0, Math.min(127, root + iv))); + } + function drawGrid(grid, startX, octOffset, side) { // Active kit for this side: "off" = melodic notes, "perc" = drums, // "war" = physically-modeled weapons. Colors, labels and notation @@ -5750,7 +5789,14 @@ function paint({ wipe, ink, box, line, write, screen, sound, system, trackpad, p // Use octOffset directly as the side's octave shift const noteOctave = octave + octOffset; const key = NOTE_TO_KEY[noteName]; - const isActive = key && sounds[key] !== undefined; + const directActive = key && sounds[key] !== undefined; + // Chord-tone lighting: this pad lights up if its MIDI note is part of + // a held chord, even when its own key isn't physically pressed. Drum + // sides never participate (chords are melodic-only). + const padMidi = isKit ? -1 : noteToMidiNumber(letter, noteOctave); + const chordTone = !directActive && padMidi >= 0 && chordToneMidi.has(padMidi); + const chordIsRoot = chordTone && chordRootMidi.has(padMidi); + const isActive = directActive || chordTone; const trailInfo = key && trail[key]; const sharp = letter.includes("#"); const drumActive = isKit && kitNames && !!kitNames[letter]; @@ -5761,7 +5807,15 @@ function paint({ wipe, ink, box, line, write, screen, sound, system, trackpad, p const isHovered = hoverX >= x && hoverX < x + btnW && hoverY >= y && hoverY < y + btnH; if (isActive) { - ink(nc[0], nc[1], nc[2]); + if (chordTone && !chordIsRoot) { + // Chord EXTENSION (the +2/+3/+4/+7 tones): dimmer note-color + // fill so the chord shape is visible but the pressed root still + // reads as the brightest pad. + ink(Math.floor(nc[0] * 0.7), Math.floor(nc[1] * 0.7), Math.floor(nc[2] * 0.7)); + } else { + // Directly-pressed pad or chord root: full note color. + ink(nc[0], nc[1], nc[2]); + } box(x, y, btnW, btnH, true); ink(255, 255, 255); } else if (trailInfo && trailInfo.brightness > 0.05) { @@ -6196,6 +6250,58 @@ function paint({ wipe, ink, box, line, write, screen, sound, system, trackpad, p gmNotice = null; } + // Persistent INSTRUMENT readout — always-on status line so the player can + // see the current melodic voice (GM program, MIDI passthru, or legacy + // wave) at a glance, and watch the program number FORM as digits are + // typed. Sits just under the aux-pad row (same lane the transient gmNotice + // flash uses) so it never collides with the grid, sliders or DJ strip. The + // transient flash above takes visual priority while it's alive; this line + // fills the rest of the time. + if (!(gmNotice && frame < gmNotice.until)) { + const dark2 = isDark(); + // Mid-entry? gmDigitBuffer holds the partial number and gmDigitLastMs is + // within the 0.8s commit window. Show it forming with a blinking caret. + const typingActive = gmDigitBuffer.length > 0 && (Date.now() - gmDigitLastMs) <= 800; + let label, accent; + if (typingActive) { + const caret = (frame % 30) < 15 ? "_" : " "; // ~0.5s blink + label = "prog " + gmDigitBuffer + caret; // e.g. "prog 12_" + accent = true; + } else if (gmPassthrough) { + label = "MIDI PASSTHRU"; + accent = false; + } else if (gmProgram !== null) { + const num = String(gmProgram + 1).padStart(3, "0"); + label = num + " " + (GM_PROGRAM_NAMES[gmProgram] || ""); + accent = false; + } else { + // No GM: fall back to the legacy freely-cycled wave name. + label = "wave " + (wave || "—"); + accent = false; + } + const tw = Math.max(1, label.length) * 6 + 12; + const tx = Math.floor((w - tw) / 2); + const ty = gridTop + gridH + 18; + ink(0, 0, 0, 140); + box(tx - 1, ty - 1, tw + 2, 14, true); + if (accent) { + // Live-typing: warm-highlight fill + bright outline so it reads as + // "entry in progress" rather than a committed selection. + ink(dark2 ? 70 : 250, dark2 ? 55 : 235, dark2 ? 20 : 180, 220); + box(tx, ty, tw, 12, true); + ink(dark2 ? 255 : 200, dark2 ? 230 : 140, dark2 ? 90 : 40); + box(tx, ty, tw, 12, "outline"); + ink(dark2 ? 255 : 30, dark2 ? 245 : 30, dark2 ? 170 : 70); + } else { + ink(dark2 ? 28 : 235, dark2 ? 32 : 238, dark2 ? 42 : 244, 200); + box(tx, ty, tw, 12, true); + ink(dark2 ? 70 : 200, dark2 ? 90 : 205, dark2 ? 110 : 210); + box(tx, ty, tw, 12, "outline"); + ink(dark2 ? 170 : 60, dark2 ? 190 : 70, dark2 ? 210 : 90); + } + write(label, { x: tx + 4, y: ty + 2, size: 1, font: "font_1" }); + } + // === SLIDERS: fx mix, echo, pitch, bitcrush + per-effect X/Y routing + reset === const settingsY = topBarH; const sliderH = 12; -- 2.51.2