diff --git a/help/relay/README.md b/help/relay/README.md index 7d7795a204..c685c3e717 100644 --- a/help/relay/README.md +++ b/help/relay/README.md @@ -35,6 +35,21 @@ Every private request needs `Authorization: Bearer `. - `GET /api/aesel/sessions/:id?after=`: events (up to 500), pending approvals, request statuses. - `POST /api/aesel/sessions/:id/respond`: `{id, result}` matching an outstanding engine request. - `POST /api/aesel/sessions/:id/interrupt`: `{}`. +- `POST /api/aesel/transcription-session`: a 60-second OpenAI client secret for + `gpt-live-transcribe`, 24 kHz mono PCM, with manual commit. The phone streams + microphone chunks immediately and displays deltas during recording. +- `POST /api/aesel/transcribe`: `{audio:}`, at most + 46 seconds. Whisper returns final text and measured word timestamps after a + performance or when on-device recognition has no words. Audio is transient + here; the original recording remains on the phone. + +Speech uses the separate `WHISTLEGRAPH_TRANSCRIPTION_KEY` server credential +and the OpenAI API, independently of personal Claude/Codex subscriptions. +The key is captured in the speech handlers and removed from the environment +before any provider CLI can inherit it. Neither API keys nor recordings are +stored in speech session logs. Both speech routes use the same verified-owner +gate. Live text is provisional; network arrival times are never used as word +timestamps. Validate with `node --test help/relay/transcription.test.mjs`. Submit each attempt with a stable request UUID. Retries with the same UUID return its status without executing again. The full input, including image diff --git a/help/relay/service.mjs b/help/relay/service.mjs index 7b4a685d6c..4bd3979be9 100644 --- a/help/relay/service.mjs +++ b/help/relay/service.mjs @@ -6,6 +6,7 @@ import {join} from 'node:path'; import {fileURLToPath, pathToFileURL} from 'node:url'; import {ClaudeServer} from '../../aesel/src/claude-server.mjs'; import {AppServer} from '../../aesel/src/app-server.mjs'; +import {createTranscriber,createSpeechSession} from './transcription.mjs'; const uuid = /^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$/i; const fail = (status, message) => Object.assign(new Error(message), {status}); @@ -33,7 +34,7 @@ export function ownerAuth({adminSub, domain='aesthetic.us.auth0.com', fetch=glob }; } -export function createRelay({root, authorize, factory, maxActive=2}={}) { +export function createRelay({root, authorize, factory, transcribe, speechSession, maxActive=2}={}) { if (!root || !authorize) throw Error('State directory and authorization are required'); mkdirSync(root,{recursive:true,mode:0o700}); const sessions=new Map(); @@ -136,6 +137,14 @@ export function createRelay({root, authorize, factory, maxActive=2}={}) { toolReplies.set(id,{s,res,timer});res.on('close',()=>{clearTimeout(timer);toolReplies.delete(id);s.pending.delete(id);});return; } await authorize(req.headers.authorization); + if(url.pathname==='/api/aesel/transcription-session' && req.method==='POST') { + if(!speechSession)throw fail(503,'Live speech is unavailable'); + return json(res,200,await speechSession()); + } + if(url.pathname==='/api/aesel/transcribe' && req.method==='POST') { + if(!transcribe)throw fail(503,'Speech transcription is unavailable'); + return json(res,200,await transcribe(await body(req))); + } if(url.pathname==='/api/aesel/sessions' && req.method==='POST') { const input=await body(req); if(!['claude','codex'].includes(input.provider))throw fail(400,'Choose claude or codex'); @@ -205,9 +214,12 @@ export function createRelay({root, authorize, factory, maxActive=2}={}) { } if(process.argv[1] && import.meta.url===pathToFileURL(process.argv[1]).href) { + const transcribe=createTranscriber({apiKey:process.env.WHISTLEGRAPH_TRANSCRIPTION_KEY}); + const speechSession=createSpeechSession({apiKey:process.env.WHISTLEGRAPH_TRANSCRIPTION_KEY}); + delete process.env.WHISTLEGRAPH_TRANSCRIPTION_KEY; // A private subscription relay must never silently fall through to paid API keys. for(const key of ['ANTHROPIC_API_KEY','ANTHROPIC_AUTH_TOKEN','OPENAI_API_KEY','OPENROUTER_API_KEY'])delete process.env[key]; - const server=createRelay({root:process.env.AESEL_RELAY_STATE||'/var/lib/aesel-relay',authorize:ownerAuth({adminSub:process.env.ADMIN_SUB,domain:process.env.AUTH0_DOMAIN})}); + const server=createRelay({root:process.env.AESEL_RELAY_STATE||'/var/lib/aesel-relay',transcribe,speechSession,authorize:ownerAuth({adminSub:process.env.ADMIN_SUB,domain:process.env.AUTH0_DOMAIN})}); server.listen(Number(process.env.AESEL_RELAY_PORT||3006),'127.0.0.1',()=>console.log('Aesel relay listening on loopback')); for(const signal of ['SIGTERM','SIGINT'])process.on(signal,()=>{server.close();setTimeout(()=>process.exit(0),1000).unref();}); } diff --git a/help/relay/transcription.mjs b/help/relay/transcription.mjs new file mode 100644 index 0000000000..b88cf88d77 --- /dev/null +++ b/help/relay/transcription.mjs @@ -0,0 +1,53 @@ +// Dedicated paid speech path. Keys and audio never enter a provider CLI session. +const fail=(status,message)=>Object.assign(Error(message),{status}); +export const liveSession={type:'transcription',audio:{input:{format:{type:'audio/pcm',rate:24000},transcription:{model:'gpt-live-transcribe',languages:['en'],delay:'low'},turn_detection:null}}}; +export function createSpeechSession({apiKey,fetch=globalThis.fetch}={}) { + return async()=>{ + if(!apiKey)throw fail(503,'Live speech is unavailable'); + try { + const response=await fetch('https://api.openai.com/v1/realtime/client_secrets',{method:'POST',headers:{Authorization:'Bearer '+apiKey,'content-type':'application/json'}, + body:JSON.stringify({expires_after:{anchor:'created_at',seconds:60},session:liveSession}),signal:AbortSignal.timeout(8000)}); + if(!response.ok)throw fail(502,'Could not open live speech'); + const value=await response.json(); + if(typeof value.value!=='string'||!Number.isFinite(value.expires_at))throw fail(502,'Invalid live speech session'); + return {value:value.value,expiresAt:value.expires_at,session:liveSession}; + }catch(error){if(error.status)throw error;throw fail(502,'Live speech unavailable');} + }; +} +export function wavInput(input) { + if(typeof input?.audio!=='string'||input.audio.length>2_000_000||input.audio.length%4||!/^[A-Za-z0-9+/]*={0,2}$/.test(input.audio))throw fail(400,'Invalid recording'); + const b=Buffer.from(input.audio,'base64'); + if(b.length<46||b.toString('ascii',0,4)!=='RIFF'||b.toString('ascii',8,12)!=='WAVE'|| + b.toString('ascii',12,16)!=='fmt '||b.readUInt32LE(16)!==16||b.readUInt16LE(20)!==1|| + b.readUInt16LE(22)!==1||b.readUInt32LE(24)!==16000||b.readUInt32LE(28)!==32000|| + b.readUInt16LE(32)!==2||b.readUInt16LE(34)!==16||b.toString('ascii',36,40)!=='data'|| + b.readUInt32LE(4)!==b.length-8||b.readUInt32LE(40)!==b.length-44||(b.length-44)%2)throw fail(400,'Expected mono 16 kHz PCM WAV'); + const durationMs=(b.length-44)/32; + if(durationMs<100||durationMs>46000)throw fail(400,'Recording must be under 46 seconds'); + return {audio:b,durationMs}; +} +export function createTranscriber({apiKey,fetch=globalThis.fetch}={}) { + let active=0; + return async input=>{ + if(!apiKey)throw fail(503,'Speech transcription is unavailable'); + const {audio,durationMs}=wavInput(input); + if(active>=2)throw fail(429,'Speech transcription is busy'); + active++; + try { + const form=new FormData();form.set('file',new Blob([audio],{type:'audio/wav'}),'recording.wav'); + form.set('model','whisper-1');form.set('language','en');form.set('response_format','verbose_json'); + form.append('timestamp_granularities[]','word'); + const start=performance.now(); + const response=await fetch('https://api.openai.com/v1/audio/transcriptions',{method:'POST',headers:{Authorization:'Bearer '+apiKey},body:form,signal:AbortSignal.timeout(12000)}); + if(!response.ok)throw fail(502,'Speech transcription failed'); + const result=await response.json(); + if(typeof result.text!=='string'||result.text.length>12000||!Array.isArray(result.words)||result.words.length>256)throw fail(502,'Invalid speech response'); + const words=result.words.map(w=>{ + if(typeof w.word!=='string'||w.word.length>2000||!Number.isFinite(w.start)||!Number.isFinite(w.end)||w.start<0||w.enddurationMs+250)throw fail(502,'Invalid word timing'); + return {text:w.word,atMs:Math.round(w.start*1000),durationMs:Math.round((w.end-w.start)*1000)}; + }); + return {transcript:result.text.trim(),words,provider:'openai',model:'whisper-1',elapsedMs:Math.round(performance.now()-start)}; + }catch(error){if(error.status)throw error;throw fail(502,'Speech transcription unavailable');} + finally{active--;} + }; +} diff --git a/help/relay/transcription.test.mjs b/help/relay/transcription.test.mjs new file mode 100644 index 0000000000..e482ed000b --- /dev/null +++ b/help/relay/transcription.test.mjs @@ -0,0 +1,36 @@ +import test from 'node:test'; +import assert from 'node:assert/strict'; +import {once} from 'node:events'; +import {mkdtempSync,rmSync,readdirSync} from 'node:fs'; +import {tmpdir} from 'node:os'; +import {join} from 'node:path'; +import {createRelay} from './service.mjs'; +import {wavInput,createTranscriber,createSpeechSession,liveSession} from './transcription.mjs'; +function wav(seconds=1){const b=Buffer.alloc(44+32000*seconds);b.write('RIFF');b.writeUInt32LE(b.length-8,4);b.write('WAVEfmt ',8);b.writeUInt32LE(16,16);b.writeUInt16LE(1,20);b.writeUInt16LE(1,22);b.writeUInt32LE(16000,24);b.writeUInt32LE(32000,28);b.writeUInt16LE(2,32);b.writeUInt16LE(16,34);b.write('data',36);b.writeUInt32LE(b.length-44,40);return {audio:b.toString('base64')};} +test('speech upload validates actual PCM format and bounded recording length',()=>{ + assert.equal(wavInput(wav(46)).durationMs,46000); + for(const input of [wav(47),{audio:'not an audio file'},{audio:Buffer.alloc(60).toString('base64')}])assert.throws(()=>wavInput(input),{status:400}); + const broken=Buffer.from(wav().audio,'base64');broken.writeUInt32LE(44100,24);assert.throws(()=>wavInput({audio:broken.toString('base64')}),{status:400}); +}); +test('Whisper response preserves measured word times without accepting impossible boundaries',async()=>{ + let payload={text:'Hello',words:[{word:'Hello',start:0.1,end:0.4}]}; + const transcribe=createTranscriber({apiKey:'test',fetch:async(url,options)=>{assert.equal(options.body.get('model'),'whisper-1');assert.equal(options.body.get('timestamp_granularities[]'),'word');return {ok:true,json:async()=>payload};}}); + const result=await transcribe(wav());assert.deepEqual(result.words,[{text:'Hello',atMs:100,durationMs:300}]); + payload.words[0].end=7;await assert.rejects(transcribe(wav()),{status:502}); + await assert.rejects(createTranscriber({})(wav()),{status:503}); +}); +test('ephemeral sessions use live transcription with manual end-of-take commit',async()=>{ + const mint=createSpeechSession({apiKey:'test',fetch:async(url,o)=>{assert.deepEqual(JSON.parse(o.body),{expires_after:{anchor:'created_at',seconds:60},session:liveSession});return {ok:true,json:async()=>({value:'ephemeral',expires_at:123})};}}); + assert.equal((await mint()).value,'ephemeral');assert.equal(liveSession.audio.input.turn_detection,null); +}); +test('both speech routes require owner auth and leave no recording/session files',async t=>{ + const root=mkdtempSync(join(tmpdir(),'speech-'));let calls=0; + const server=createRelay({root,authorize:async h=>{if(h!=='Bearer owner')throw Object.assign(Error('Private'),{status:403});},speechSession:async()=>{calls++;return {value:'ephemeral'}},transcribe:async()=>{calls++;return {transcript:'hello',words:[]}}}); + server.listen(0,'127.0.0.1');await once(server,'listening');t.after(()=>{server.closeAllConnections();server.close();rmSync(root,{recursive:true,force:true});}); + for(const route of ['transcribe','transcription-session']){ + const url=`http://127.0.0.1:${server.address().port}/api/aesel/${route}`; + assert.equal((await fetch(url,{method:'POST'})).status,403); + assert.equal((await fetch(url,{method:'POST',headers:{authorization:'Bearer owner'},body:'{}'})).status,200); + } + assert.equal(calls,2);assert.deepEqual(readdirSync(root),[]); +});