diff --git a/e2e/harness/net/http.ts b/e2e/harness/net/http.ts index 1a52a484..0f3d4bbe 100644 --- a/e2e/harness/net/http.ts +++ b/e2e/harness/net/http.ts @@ -214,23 +214,41 @@ export const mockDufflepud = (context: BrowserContext, fixtures: DufflepudFixtur return route.fallback() }) -// The rate the endpoint documents for raw pcm, at 16 bits a sample, which is what the app assumes -// when it writes a wav header for one. +// The rate the harness writes its silence at. Nothing downstream reads it: the app decodes what it +// is answered and writes its own header from the result. const SPEECH_RATE = 24000 -// Headerless silence. Nothing in it says how long it is, so the duration a spec reads off the -// player is the one the app computed. -const silence = (seconds: number) => Buffer.alloc(seconds * SPEECH_RATE * 2) +// Silence as a wav, built rather than inlined so a spec can ask for a length and then assert the +// duration the player reads off it. +const silence = (seconds: number) => { + const bytes = seconds * SPEECH_RATE * 2 + const wav = Buffer.alloc(44 + bytes) -// mp3 is the other format the real endpoint encodes, and the harness has no encoder for it, so -// asking for one here is a mistake rather than a case to serve. -const SPEECH_FORMAT = "pcm" + wav.write("RIFF", 0) + wav.writeUInt32LE(36 + bytes, 4) + wav.write("WAVEfmt ", 8) + wav.writeUInt32LE(16, 16) + wav.writeUInt16LE(1, 20) + wav.writeUInt16LE(1, 22) + wav.writeUInt32LE(SPEECH_RATE, 24) + wav.writeUInt32LE(SPEECH_RATE * 2, 28) + wav.writeUInt16LE(2, 32) + wav.writeUInt16LE(16, 34) + wav.write("data", 36) + wav.writeUInt32LE(bytes, 40) + + return wav +} + +// The format the app asks the real endpoint for. The harness has no mp3 encoder, so it answers a +// wav instead — the app decodes either, and only an answer a browser can decode exercises that. +const SPEECH_FORMAT = "mp3" /** * OpenRouter's text to speech, answering every request with the same silence. The array it returns - * collects what the app asked to have read, in the order it asked. Answering a decodable wav to a - * request for some other format would prove nothing about the container the app builds, so anything - * but raw pcm is refused. + * collects what the app asked to have read, in the order it asked. Asking for raw pcm is refused + * the way the real endpoint refuses an unknown format, since pcm is the one answer whose sample + * rate and sample format the app would have to guess at. */ export const mockOpenRouterSpeech = async (context: BrowserContext, seconds = 3) => { const spoken: string[] = [] @@ -241,10 +259,7 @@ export const mockOpenRouterSpeech = async (context: BrowserContext, seconds = 3) if (response_format === SPEECH_FORMAT) { spoken.push(input) - return route.fulfill({ - contentType: `audio/pcm;rate=${SPEECH_RATE};channels=1`, - body: silence(seconds), - }) + return route.fulfill({contentType: "audio/wav", body: silence(seconds)}) } return route.fulfill({ diff --git a/e2e/specs/rooms.spec.ts b/e2e/specs/rooms.spec.ts index 24a9e731..44bcc985 100644 --- a/e2e/specs/rooms.spec.ts +++ b/e2e/specs/rooms.spec.ts @@ -986,8 +986,8 @@ test("US-119 have a message read out loud", async ({seed, as}) => { "another message\n\nheads up Alice Anchor, the notice is at a link to harbor.example", ]) - // The mock answers headerless pcm, so the duration is only right if the wav header the app put - // in front of it is, which is what makes the whole clip scrubbable. + // The app decodes what it is answered and rebuilds the container around the samples, so the + // duration is only right if that round trip kept every one of them. await expect(alice.getByText("/ 0:10")).toBeVisible() // The clip carries autoplay and the button follows the audio element's own play event, so it diff --git a/src/app/speech.ts b/src/app/speech.ts index 367cd0dc..93924cf7 100644 --- a/src/app/speech.ts +++ b/src/app/speech.ts @@ -15,14 +15,13 @@ const SPEECH_MODEL = "hexgrad/kokoro-82m" const SPEECH_VOICE = "af_bella" -// The endpoint encodes mp3 and raw pcm, and its mp3 carries a xing header naming a fraction of the -// frames it holds, so a browser reads a fifth more audio than is there and the scrubber never -// reaches the end. Raw pcm claims no length at all, so the wav header below is the only one. -const SPEECH_FORMAT = "pcm" +// The endpoint encodes mp3 and raw pcm. Raw pcm carries no header, so playing it means assuming a +// sample rate and a sample format the response never states, and either assumption wrong is static +// rather than an error. Decoding the mp3 reads both off the audio. +const SPEECH_FORMAT = "mp3" -const SPEECH_RATE = 24000 - -const SPEECH_CHANNELS = 1 +// Decoding resamples to the context's rate, so this is the rate the wav ends up at. +const SPEECH_RATE = 48000 const SPEECH_BIT_DEPTH = 16 @@ -36,9 +35,15 @@ export type Speech = { export const speech = writable>(undefined) -const toWav = (pcm: ArrayBuffer) => { - const bytesPerFrame = (SPEECH_CHANNELS * SPEECH_BIT_DEPTH) / 8 - const wav = new ArrayBuffer(WAV_HEADER_LENGTH + pcm.byteLength) +// An offline context never reaches for the speakers. +const decode = (data: ArrayBuffer) => + new OfflineAudioContext(1, 1, SPEECH_RATE).decodeAudioData(data) + +const toWav = (audio: AudioBuffer) => { + const {numberOfChannels, sampleRate, length} = audio + const bytesPerFrame = (numberOfChannels * SPEECH_BIT_DEPTH) / 8 + const dataLength = length * bytesPerFrame + const wav = new ArrayBuffer(WAV_HEADER_LENGTH + dataLength) const view = new DataView(wav) const ascii = (offset: number, value: string) => { for (let i = 0; i < value.length; i++) { @@ -47,19 +52,33 @@ const toWav = (pcm: ArrayBuffer) => { } ascii(0, "RIFF") - view.setUint32(4, 36 + pcm.byteLength, true) + view.setUint32(4, 36 + dataLength, true) ascii(8, "WAVEfmt ") view.setUint32(16, 16, true) view.setUint16(20, 1, true) - view.setUint16(22, SPEECH_CHANNELS, true) - view.setUint32(24, SPEECH_RATE, true) - view.setUint32(28, SPEECH_RATE * bytesPerFrame, true) + view.setUint16(22, numberOfChannels, true) + view.setUint32(24, sampleRate, true) + view.setUint32(28, sampleRate * bytesPerFrame, true) view.setUint16(32, bytesPerFrame, true) view.setUint16(34, SPEECH_BIT_DEPTH, true) ascii(36, "data") - view.setUint32(40, pcm.byteLength, true) + view.setUint32(40, dataLength, true) - new Uint8Array(wav, WAV_HEADER_LENGTH).set(new Uint8Array(pcm)) + const channels = Array.from({length: numberOfChannels}, (_, i) => audio.getChannelData(i)) + + let offset = WAV_HEADER_LENGTH + + for (let frame = 0; frame < length; frame++) { + for (const samples of channels) { + // A decoded sample runs from -1 to 1, and the two ends of a signed 16 bit range are not the + // same size, so each end scales by its own bound. + const sample = Math.max(-1, Math.min(1, samples[frame])) + + view.setInt16(offset, Math.round(sample * (sample < 0 ? 0x8000 : 0x7fff)), true) + + offset += 2 + } + } return new Blob([wav], {type: "audio/wav"}) } @@ -87,7 +106,7 @@ export const synthesize = async (text: string) => { throw new Error(error?.message || `OpenRouter returned a ${response.status}.`) } - return toWav(await response.arrayBuffer()) + return toWav(await decode(await response.arrayBuffer())) } export const stopSpeech = () =>