From 371c04b1adf4be0765d3fa46bc768cb5664faa3d Mon Sep 17 00:00:00 2001 From: Jeffrey Alan Scudder Date: Mon, 30 Mar 2026 14:13:30 -0700 Subject: [PATCH] Add gpt-4o-mini-tts support with instructions for emotional TTS (scream mode) Upgrades the say API to use OpenAI's gpt-4o-mini-tts model when instructions are provided, enabling emotional/style control. Adds say:scream colon option for blood-curdling screams. Falls back to tts-1 when no instructions given. Co-Authored-By: Claude Opus 4.6 (1M context) --- output/.gitignore | 2 + system/netlify/functions/say.js | 55 +++++++++++-------- .../public/aesthetic.computer/disks/say.mjs | 13 ++++- .../public/aesthetic.computer/lib/speech.mjs | 5 +- 4 files changed, 49 insertions(+), 26 deletions(-) create mode 100644 output/.gitignore diff --git a/output/.gitignore b/output/.gitignore new file mode 100644 index 0000000000..d6b7ef32c8 --- /dev/null +++ b/output/.gitignore @@ -0,0 +1,2 @@ +* +!.gitignore diff --git a/system/netlify/functions/say.js b/system/netlify/functions/say.js index 9fb3b7a939..fd81c5fba9 100644 --- a/system/netlify/functions/say.js +++ b/system/netlify/functions/say.js @@ -6,11 +6,12 @@ const tts = require("@google-cloud/text-to-speech"); const crypto = require("crypto"); const { S3Client, HeadObjectCommand, PutObjectCommand } = require("@aws-sdk/client-s3"); -// OpenAI voice mapping: alloy, echo, fable, onyx, nova, shimmer +// OpenAI voice mapping +// gpt-4o-mini-tts voices: alloy, ash, ballad, coral, echo, fable, nova, onyx, sage, shimmer, verse const OPENAI_VOICES = { - male: ["onyx", "echo", "alloy"], - female: ["nova", "shimmer", "fable"], - neutral: ["alloy", "fable", "echo"], + male: ["onyx", "echo", "ash", "alloy"], + female: ["nova", "shimmer", "fable", "coral"], + neutral: ["alloy", "fable", "echo", "sage", "verse", "ballad"], }; // Initialize S3 client for Digital Ocean Spaces @@ -27,9 +28,10 @@ const BUCKET = process.env.ART_SPACE_NAME; const CDN_URL = "https://art.aesthetic.computer"; const CACHE_PREFIX = "tts-cache/"; -// Generate cache key from provider + voice + text -function getCacheKey(provider, voiceId, text) { - const hash = crypto.createHash("sha256").update(`${provider}:${voiceId}:${text}`).digest("hex"); +// Generate cache key from provider + voice + text + instructions +function getCacheKey(provider, voiceId, text, instructions) { + const parts = `${provider}:${voiceId}:${text}${instructions ? `:${instructions}` : ""}`; + const hash = crypto.createHash("sha256").update(parts).digest("hex"); return `${CACHE_PREFIX}${hash}.mp3`; } @@ -67,18 +69,24 @@ async function saveToCache(key, audioBuffer) { } // Generate audio with OpenAI TTS -async function generateOpenAI(text, gender, set) { +// Uses gpt-4o-mini-tts when instructions are provided (supports emotional/style control), +// falls back to tts-1 otherwise. +async function generateOpenAI(text, gender, set, instructions) { const voiceList = OPENAI_VOICES[gender] || OPENAI_VOICES.neutral; const voice = voiceList[set % voiceList.length]; - + const openai = new OpenAI({ apiKey: process.env.OPENAI_API_KEY }); - - const mp3Response = await openai.audio.speech.create({ - model: "tts-1", + + const params = { + model: instructions ? "gpt-4o-mini-tts" : "tts-1", voice: voice, input: text, response_format: "mp3", - }); + }; + + if (instructions) params.instructions = instructions; + + const mp3Response = await openai.audio.speech.create(params); return { buffer: Buffer.from(await mp3Response.arrayBuffer()), @@ -154,17 +162,20 @@ exports.handler = async (event) => { const utterance = body.from || "aesthetic.computer"; const set = parseInt(body.voice?.split(":")[1]) || 0; const gender = body.voice?.split(":")[0]?.toLowerCase() || "neutral"; - + // Provider: "openai" (default), "google" // Can be set via body.provider or defaults to openai const provider = body.provider || "openai"; - + + // Instructions for gpt-4o-mini-tts emotional/style control (OpenAI only) + const instructions = provider === "openai" ? (body.instructions || null) : null; + // Cache bust: if true, skip cache lookup and regenerate const bustCache = body.bust === true; // Check for SSML (only Google supports it) const isSSML = utterance.indexOf("") !== -1; - + // Strip SSML tags for OpenAI (it doesn't support them) let text = utterance; if (isSSML && provider === "openai") { @@ -173,13 +184,13 @@ exports.handler = async (event) => { // Build voice identifier for cache key const voiceSpec = `${provider}-${gender}-${set}`; - const cacheKey = getCacheKey(provider, voiceSpec, text); + const cacheKey = getCacheKey(provider, voiceSpec, text, instructions); try { // Check cache first - return redirect to CDN if cached (unless bust=true) if (!bustCache) { const cachedUrl = await checkCache(cacheKey); - + if (cachedUrl) { console.log(`🎯 TTS cache hit: ${cachedUrl}`); return { @@ -197,13 +208,13 @@ exports.handler = async (event) => { } // Cache miss (or bust) - generate with selected provider - console.log(`🔄 TTS ${bustCache ? 'regenerating' : 'cache miss'} (${provider}): ${text.substring(0, 50)}...`); - + console.log(`🔄 TTS ${bustCache ? "regenerating" : "cache miss"} (${provider}): ${text.substring(0, 50)}...`); + let result; if (provider === "google") { result = await generateGoogle(text, gender, set, isSSML); } else { - result = await generateOpenAI(text, gender, set); + result = await generateOpenAI(text, gender, set, instructions); } const { buffer: audioBuffer, voiceId } = result; @@ -220,7 +231,7 @@ exports.handler = async (event) => { // Cache for next time const cdnUrl = await saveToCache(cacheKey, audioBuffer); - + if (cdnUrl) { return { statusCode: 302, diff --git a/system/public/aesthetic.computer/disks/say.mjs b/system/public/aesthetic.computer/disks/say.mjs index 706abea7e4..b83929a999 100644 --- a/system/public/aesthetic.computer/disks/say.mjs +++ b/system/public/aesthetic.computer/disks/say.mjs @@ -16,6 +16,7 @@ let lastSpoken = ""; let status = "idle"; // idle, speaking, error let provider = "openai"; // "openai" or "google" let gender = "neutral"; +let instructions = null; // 🥾 Boot function boot({ params, colon }) { @@ -33,6 +34,10 @@ function boot({ params, colon }) { else if (part === "openai") provider = "openai"; else if (part === "male") gender = "male"; else if (part === "female") gender = "female"; + else if (part === "scream") { + provider = "openai"; + instructions = "Deliver this as a blood-curdling scream. Shriek at the absolute top of your lungs with your voice cracking. Pure primal rage. Do NOT speak normally — only scream, raw and unhinged."; + } } console.log(`Provider: ${provider}, Gender: ${gender}`); } @@ -45,8 +50,9 @@ function paint({ wipe, ink, write, screen }) { // Note: Top-left corner is reserved for prompt HUD label // Provider indicator (below HUD area) - const providerColor = provider === "google" ? "cyan" : "lime"; - ink(providerColor).write(`[${provider}]`, { x: 6, y: 18 }); + const providerColor = instructions ? "red" : provider === "google" ? "cyan" : "lime"; + const providerLabel = instructions ? `[${provider} SCREAM]` : `[${provider}]`; + ink(providerColor).write(providerLabel, { x: 6, y: 18 }); // Instructions ink("gray").write("say ", { x: 6, y: 32 }); @@ -80,10 +86,11 @@ function act({ event: e, speak }) { const voice = `${gender}:0`; - console.log(`🗣️ Speaking: "${text}" with ${provider}, voice: ${voice}`); + console.log(`🗣️ Speaking: "${text}" with ${provider}, voice: ${voice}${instructions ? " [SCREAM]" : ""}`); speak(text, voice, "cloud", { volume: 1, provider: provider, + instructions, }); } diff --git a/system/public/aesthetic.computer/lib/speech.mjs b/system/public/aesthetic.computer/lib/speech.mjs index 66274b59eb..9c68a8e13f 100644 --- a/system/public/aesthetic.computer/lib/speech.mjs +++ b/system/public/aesthetic.computer/lib/speech.mjs @@ -68,7 +68,8 @@ function speak(words, voice, mode = "local", opts = {}) { synth.speak(utterance); } else if (mode === "cloud") { - const label = `speech:${voice}:${opts.provider || "openai"} - ${words}`; + const instTag = opts.instructions ? `:${opts.instructions.slice(0, 32)}` : ""; + const label = `speech:${voice}:${opts.provider || "openai"}${instTag} - ${words}`; // For preload-only mode, return a promise that resolves when cached let preloadResolve = null; @@ -199,6 +200,8 @@ function speak(words, voice, mode = "local", opts = {}) { bust: needsBust, // Force regenerate on server if marked }; + if (opts.instructions) payload.instructions = opts.instructions; + // Create a promise that resolves when the fetch completes let fetchResolve; const fetchPromise = new Promise(resolve => { fetchResolve = resolve; }); -- 2.51.2