Read a message aloud from audio the browser decoded rather than bytes the app guessed at (#611)

This commit is contained in:
Coracle-Bot 2026-09-22 15:40:08 +00:00 committed by hodlbod
parent f2473d67b4
commit 8cc970906d
3 changed files with 68 additions and 34 deletions

View file

@ -214,23 +214,41 @@ export const mockDufflepud = (context: BrowserContext, fixtures: DufflepudFixtur
return route.fallback()
})
// The rate the endpoint documents for raw pcm, at 16 bits a sample, which is what the app assumes
// when it writes a wav header for one.
// The rate the harness writes its silence at. Nothing downstream reads it: the app decodes what it
// is answered and writes its own header from the result.
const SPEECH_RATE = 24000
// Headerless silence. Nothing in it says how long it is, so the duration a spec reads off the
// player is the one the app computed.
const silence = (seconds: number) => Buffer.alloc(seconds * SPEECH_RATE * 2)
// Silence as a wav, built rather than inlined so a spec can ask for a length and then assert the
// duration the player reads off it.
const silence = (seconds: number) => {
const bytes = seconds * SPEECH_RATE * 2
const wav = Buffer.alloc(44 + bytes)
// mp3 is the other format the real endpoint encodes, and the harness has no encoder for it, so
// asking for one here is a mistake rather than a case to serve.
const SPEECH_FORMAT = "pcm"
wav.write("RIFF", 0)
wav.writeUInt32LE(36 + bytes, 4)
wav.write("WAVEfmt ", 8)
wav.writeUInt32LE(16, 16)
wav.writeUInt16LE(1, 20)
wav.writeUInt16LE(1, 22)
wav.writeUInt32LE(SPEECH_RATE, 24)
wav.writeUInt32LE(SPEECH_RATE * 2, 28)
wav.writeUInt16LE(2, 32)
wav.writeUInt16LE(16, 34)
wav.write("data", 36)
wav.writeUInt32LE(bytes, 40)
return wav
}
// The format the app asks the real endpoint for. The harness has no mp3 encoder, so it answers a
// wav instead — the app decodes either, and only an answer a browser can decode exercises that.
const SPEECH_FORMAT = "mp3"
/**
* OpenRouter's text to speech, answering every request with the same silence. The array it returns
* collects what the app asked to have read, in the order it asked. Answering a decodable wav to a
* request for some other format would prove nothing about the container the app builds, so anything
* but raw pcm is refused.
* collects what the app asked to have read, in the order it asked. Asking for raw pcm is refused
* the way the real endpoint refuses an unknown format, since pcm is the one answer whose sample
* rate and sample format the app would have to guess at.
*/
export const mockOpenRouterSpeech = async (context: BrowserContext, seconds = 3) => {
const spoken: string[] = []
@ -241,10 +259,7 @@ export const mockOpenRouterSpeech = async (context: BrowserContext, seconds = 3)
if (response_format === SPEECH_FORMAT) {
spoken.push(input)
return route.fulfill({
contentType: `audio/pcm;rate=${SPEECH_RATE};channels=1`,
body: silence(seconds),
})
return route.fulfill({contentType: "audio/wav", body: silence(seconds)})
}
return route.fulfill({

View file

@ -986,8 +986,8 @@ test("US-119 have a message read out loud", async ({seed, as}) => {
"another message\n\nheads up Alice Anchor, the notice is at a link to harbor.example",
])
// The mock answers headerless pcm, so the duration is only right if the wav header the app put
// in front of it is, which is what makes the whole clip scrubbable.
// The app decodes what it is answered and rebuilds the container around the samples, so the
// duration is only right if that round trip kept every one of them.
await expect(alice.getByText("/ 0:10")).toBeVisible()
// The clip carries autoplay and the button follows the audio element's own play event, so it

View file

@ -15,14 +15,13 @@ const SPEECH_MODEL = "hexgrad/kokoro-82m"
const SPEECH_VOICE = "af_bella"
// The endpoint encodes mp3 and raw pcm, and its mp3 carries a xing header naming a fraction of the
// frames it holds, so a browser reads a fifth more audio than is there and the scrubber never
// reaches the end. Raw pcm claims no length at all, so the wav header below is the only one.
const SPEECH_FORMAT = "pcm"
// The endpoint encodes mp3 and raw pcm. Raw pcm carries no header, so playing it means assuming a
// sample rate and a sample format the response never states, and either assumption wrong is static
// rather than an error. Decoding the mp3 reads both off the audio.
const SPEECH_FORMAT = "mp3"
const SPEECH_RATE = 24000
const SPEECH_CHANNELS = 1
// Decoding resamples to the context's rate, so this is the rate the wav ends up at.
const SPEECH_RATE = 48000
const SPEECH_BIT_DEPTH = 16
@ -36,9 +35,15 @@ export type Speech = {
export const speech = writable<Maybe<Speech>>(undefined)
const toWav = (pcm: ArrayBuffer) => {
const bytesPerFrame = (SPEECH_CHANNELS * SPEECH_BIT_DEPTH) / 8
const wav = new ArrayBuffer(WAV_HEADER_LENGTH + pcm.byteLength)
// An offline context never reaches for the speakers.
const decode = (data: ArrayBuffer) =>
new OfflineAudioContext(1, 1, SPEECH_RATE).decodeAudioData(data)
const toWav = (audio: AudioBuffer) => {
const {numberOfChannels, sampleRate, length} = audio
const bytesPerFrame = (numberOfChannels * SPEECH_BIT_DEPTH) / 8
const dataLength = length * bytesPerFrame
const wav = new ArrayBuffer(WAV_HEADER_LENGTH + dataLength)
const view = new DataView(wav)
const ascii = (offset: number, value: string) => {
for (let i = 0; i < value.length; i++) {
@ -47,19 +52,33 @@ const toWav = (pcm: ArrayBuffer) => {
}
ascii(0, "RIFF")
view.setUint32(4, 36 + pcm.byteLength, true)
view.setUint32(4, 36 + dataLength, true)
ascii(8, "WAVEfmt ")
view.setUint32(16, 16, true)
view.setUint16(20, 1, true)
view.setUint16(22, SPEECH_CHANNELS, true)
view.setUint32(24, SPEECH_RATE, true)
view.setUint32(28, SPEECH_RATE * bytesPerFrame, true)
view.setUint16(22, numberOfChannels, true)
view.setUint32(24, sampleRate, true)
view.setUint32(28, sampleRate * bytesPerFrame, true)
view.setUint16(32, bytesPerFrame, true)
view.setUint16(34, SPEECH_BIT_DEPTH, true)
ascii(36, "data")
view.setUint32(40, pcm.byteLength, true)
view.setUint32(40, dataLength, true)
new Uint8Array(wav, WAV_HEADER_LENGTH).set(new Uint8Array(pcm))
const channels = Array.from({length: numberOfChannels}, (_, i) => audio.getChannelData(i))
let offset = WAV_HEADER_LENGTH
for (let frame = 0; frame < length; frame++) {
for (const samples of channels) {
// A decoded sample runs from -1 to 1, and the two ends of a signed 16 bit range are not the
// same size, so each end scales by its own bound.
const sample = Math.max(-1, Math.min(1, samples[frame]))
view.setInt16(offset, Math.round(sample * (sample < 0 ? 0x8000 : 0x7fff)), true)
offset += 2
}
}
return new Blob([wav], {type: "audio/wav"})
}
@ -87,7 +106,7 @@ export const synthesize = async (text: string) => {
throw new Error(error?.message || `OpenRouter returned a ${response.status}.`)
}
return toWav(await response.arrayBuffer())
return toWav(await decode(await response.arrayBuffer()))
}
export const stopSpeech = () =>