From 6f37bb7b5d6fb59a75bce3a6ba68b0f973d2d6f3 Mon Sep 17 00:00:00 2001 From: "prompt.ac/@jeffrey" Date: Thu, 16 Jul 2026 23:21:46 -0700 Subject: [PATCH] Add instant spoken digit samples to Menu Band --- .../Sources/MenuBand/MenuBandController.swift | 7 + .../MenuBand/MenuBandSpeechVoice.swift | 132 ++++++++++++++++-- .../Sources/MenuBand/MenuBandSynth.swift | 6 + 3 files changed, 137 insertions(+), 8 deletions(-) diff --git a/slab/menuband/Sources/MenuBand/MenuBandController.swift b/slab/menuband/Sources/MenuBand/MenuBandController.swift index b5b297cf0c..edaba4683a 100644 --- a/slab/menuband/Sources/MenuBand/MenuBandController.swift +++ b/slab/menuband/Sources/MenuBand/MenuBandController.swift @@ -2889,6 +2889,13 @@ final class MenuBandController { } } if isDown && !isRepeat { + // Audio feedback is a true key-down sample: the speech voice + // rendered 0–9 at startup, so this call only schedules cached + // PCM and does not wait for AVSpeech to synthesize the growing + // voice-slot number. + DispatchQueue.main.async { [weak self] in + self?.synth.playSpokenDigit(digit) + } // Start fresh if the buffer hit its cap OR enough time // passed since the last digit that this is plainly a // new pick, not the next digit of a longer number. diff --git a/slab/menuband/Sources/MenuBand/MenuBandSpeechVoice.swift b/slab/menuband/Sources/MenuBand/MenuBandSpeechVoice.swift index 518f6c365f..719b9d2416 100644 --- a/slab/menuband/Sources/MenuBand/MenuBandSpeechVoice.swift +++ b/slab/menuband/Sources/MenuBand/MenuBandSpeechVoice.swift @@ -16,7 +16,14 @@ import AVFoundation /// just renders silence until buffers arrive. final class MenuBandSpeechVoice { private let synthesizer = AVSpeechSynthesizer() + private let sampleRenderer = AVSpeechSynthesizer() private let player = AVAudioPlayerNode() + /// Digit feedback is polyphonic: rapid number entry takes a fresh player + /// instead of cutting off the word already sounding. + private let digitPlayers = (0..<8).map { _ in AVAudioPlayerNode() } + private let digitPitch = AVAudioUnitTimePitch() + private let digitMixer = AVAudioMixerNode() + private var nextDigitPlayer = 0 /// Pitch-only shift (rate stays 1.0) so the trackpad bend slides the /// spoken voice without speeding it up — like the radio backend. The /// reverb/echo inserts already sit downstream on the fx bus; pitch has @@ -25,6 +32,7 @@ final class MenuBandSpeechVoice { private let mixer = AVAudioMixerNode() private weak var engine: AVAudioEngine? private var attached = false + private var warmedUp = false /// Fixed format the player is connected with; speech buffers are /// converted into this before scheduling. @@ -34,6 +42,11 @@ final class MenuBandSpeechVoice { /// Reused converter; rebuilt if the synthesizer's output format changes. private var converter: AVAudioConverter? private var converterInputFormat: AVAudioFormat? + /// Startup-rendered digit clips. Each value is the sequence of PCM chunks + /// AVSpeech produced for that word; scheduling those chunks directly makes + /// number-key feedback start on key-down without a live TTS render. + private var digitSamples: [Int: [AVAudioPCMBuffer]] = [:] + private var digitSampleStarted: Set = [] func attach(to engine: AVAudioEngine, output: AVAudioNode) { guard !attached else { return } @@ -41,11 +54,106 @@ final class MenuBandSpeechVoice { engine.attach(player) engine.attach(pitch) engine.attach(mixer) + engine.attach(digitPitch) + engine.attach(digitMixer) + for digitPlayer in digitPlayers { + engine.attach(digitPlayer) + engine.connect(digitPlayer, to: digitMixer, format: renderFormat) + } engine.connect(player, to: pitch, format: renderFormat) engine.connect(pitch, to: mixer, format: renderFormat) engine.connect(mixer, to: output, format: nil) + engine.connect(digitMixer, to: digitPitch, format: renderFormat) + engine.connect(digitPitch, to: output, format: nil) mixer.outputVolume = 1.0 + digitMixer.outputVolume = 1.0 attached = true + prewarm() + prepareDigitSamples() + } + + /// Prime AVSpeech's renderer while the rest of Menu Band is starting. + /// The first real `write` otherwise pays Apple's lazy voice-loading cost + /// after the number key is already down, which makes a sampled gesture + /// feel noticeably late. A whitespace utterance produces no scheduled + /// audio but loads the selected English voice and render machinery ahead + /// of the player's first digit entry. + private func prewarm() { + guard !warmedUp else { return } + warmedUp = true + let utterance = AVSpeechUtterance(string: " ") + utterance.voice = Self.bestVoice(for: "en") + utterance.rate = AVSpeechUtteranceDefaultSpeechRate * 0.92 + synthesizer.write(utterance) { _ in } + } + + /// Render the ten spoken digits once, off the interaction path. AVSpeech + /// may return a phrase in several buffers, so retain every non-empty chunk + /// and schedule the sequence as one monophonic sample when its key lands. + private func prepareDigitSamples() { + let words = ["zero", "one", "two", "three", "four", + "five", "six", "seven", "eight", "nine"] + for (digit, word) in words.enumerated() { + let utterance = AVSpeechUtterance(string: word) + utterance.voice = Self.bestVoice(for: "en") + utterance.rate = AVSpeechUtteranceDefaultSpeechRate * 0.92 + sampleRenderer.write(utterance) { [weak self] buffer in + guard let self = self, + let pcm = buffer as? AVAudioPCMBuffer, + pcm.frameLength > 0 else { return } + DispatchQueue.main.async { + guard let converted = self.convertedBuffer(pcm) else { return } + if self.digitSampleStarted.contains(digit) { + self.digitSamples[digit, default: []].append(converted) + } else if let trimmed = self.trimmingLeadingSilence(converted) { + self.digitSampleStarted.insert(digit) + self.digitSamples[digit, default: []].append(trimmed) + } + } + } + } + } + + /// Remove AVSpeech's leading render pad so the consonant begins almost at + /// key-down. Keep 64 frames (~1.5 ms) before the first audible sample to + /// preserve the attack and avoid introducing a hard-zero click. + private func trimmingLeadingSilence(_ buffer: AVAudioPCMBuffer) -> AVAudioPCMBuffer? { + guard let source = buffer.floatChannelData?[0] else { return buffer } + let count = Int(buffer.frameLength) + let threshold: Float = 0.0015 + guard let audible = (0..= threshold }) + else { return nil } + let start = max(0, audible - 64) + let remaining = count - start + guard let out = AVAudioPCMBuffer(pcmFormat: buffer.format, + frameCapacity: AVAudioFrameCount(remaining)), + let destination = out.floatChannelData?[0] else { return buffer } + for frame in 0.. AVAudioPCMBuffer? { if converterInputFormat != pcm.format { converter = AVAudioConverter(from: pcm.format, to: renderFormat) converterInputFormat = pcm.format } - guard let converter = converter else { return } + guard let converter = converter else { return nil } let ratio = renderFormat.sampleRate / pcm.format.sampleRate let capacity = AVAudioFrameCount(Double(pcm.frameLength) * ratio) + 1_024 guard let out = AVAudioPCMBuffer(pcmFormat: renderFormat, - frameCapacity: capacity) else { return } + frameCapacity: capacity) else { return nil } var fed = false var error: NSError? converter.convert(to: out, error: &error) { _, status in @@ -95,10 +213,8 @@ final class MenuBandSpeechVoice { status.pointee = .haveData return pcm } - guard error == nil, out.frameLength > 0 else { return } - if !engine.isRunning { try? engine.start() } - player.scheduleBuffer(out, completionHandler: nil) - if !player.isPlaying { player.play() } + guard error == nil, out.frameLength > 0 else { return nil } + return out } /// BCP-47 locale for each of our short language codes, used to pick a diff --git a/slab/menuband/Sources/MenuBand/MenuBandSynth.swift b/slab/menuband/Sources/MenuBand/MenuBandSynth.swift index 8351d95f41..afcaa6ed27 100644 --- a/slab/menuband/Sources/MenuBand/MenuBandSynth.swift +++ b/slab/menuband/Sources/MenuBand/MenuBandSynth.swift @@ -874,6 +874,12 @@ final class MenuBandSynth { speechVoice.say(text, languageCode: languageCode) } + /// Immediate number-row feedback from the speech voice's pre-rendered + /// digit bank. Unlike `speak`, this does no synthesis on key-down. + func playSpokenDigit(_ digit: Int) { + speechVoice.playDigit(digit) + } + /// Spacebar reverse-replay: play the master-output tape backwards from /// the session's reverse cursor (first press anchors "now"; later /// presses resume where the last one stopped). Returns false if there -- 2.51.2