Read a message aloud from audio the browser decoded rather than bytes the app guessed at (#611)
This commit is contained in:
parent
f2473d67b4
commit
8cc970906d
3 changed files with 68 additions and 34 deletions
|
|
@ -214,23 +214,41 @@ export const mockDufflepud = (context: BrowserContext, fixtures: DufflepudFixtur
|
|||
return route.fallback()
|
||||
})
|
||||
|
||||
// The rate the endpoint documents for raw pcm, at 16 bits a sample, which is what the app assumes
|
||||
// when it writes a wav header for one.
|
||||
// The rate the harness writes its silence at. Nothing downstream reads it: the app decodes what it
|
||||
// is answered and writes its own header from the result.
|
||||
const SPEECH_RATE = 24000
|
||||
|
||||
// Headerless silence. Nothing in it says how long it is, so the duration a spec reads off the
|
||||
// player is the one the app computed.
|
||||
const silence = (seconds: number) => Buffer.alloc(seconds * SPEECH_RATE * 2)
|
||||
// Silence as a wav, built rather than inlined so a spec can ask for a length and then assert the
|
||||
// duration the player reads off it.
|
||||
const silence = (seconds: number) => {
|
||||
const bytes = seconds * SPEECH_RATE * 2
|
||||
const wav = Buffer.alloc(44 + bytes)
|
||||
|
||||
// mp3 is the other format the real endpoint encodes, and the harness has no encoder for it, so
|
||||
// asking for one here is a mistake rather than a case to serve.
|
||||
const SPEECH_FORMAT = "pcm"
|
||||
wav.write("RIFF", 0)
|
||||
wav.writeUInt32LE(36 + bytes, 4)
|
||||
wav.write("WAVEfmt ", 8)
|
||||
wav.writeUInt32LE(16, 16)
|
||||
wav.writeUInt16LE(1, 20)
|
||||
wav.writeUInt16LE(1, 22)
|
||||
wav.writeUInt32LE(SPEECH_RATE, 24)
|
||||
wav.writeUInt32LE(SPEECH_RATE * 2, 28)
|
||||
wav.writeUInt16LE(2, 32)
|
||||
wav.writeUInt16LE(16, 34)
|
||||
wav.write("data", 36)
|
||||
wav.writeUInt32LE(bytes, 40)
|
||||
|
||||
return wav
|
||||
}
|
||||
|
||||
// The format the app asks the real endpoint for. The harness has no mp3 encoder, so it answers a
|
||||
// wav instead — the app decodes either, and only an answer a browser can decode exercises that.
|
||||
const SPEECH_FORMAT = "mp3"
|
||||
|
||||
/**
|
||||
* OpenRouter's text to speech, answering every request with the same silence. The array it returns
|
||||
* collects what the app asked to have read, in the order it asked. Answering a decodable wav to a
|
||||
* request for some other format would prove nothing about the container the app builds, so anything
|
||||
* but raw pcm is refused.
|
||||
* collects what the app asked to have read, in the order it asked. Asking for raw pcm is refused
|
||||
* the way the real endpoint refuses an unknown format, since pcm is the one answer whose sample
|
||||
* rate and sample format the app would have to guess at.
|
||||
*/
|
||||
export const mockOpenRouterSpeech = async (context: BrowserContext, seconds = 3) => {
|
||||
const spoken: string[] = []
|
||||
|
|
@ -241,10 +259,7 @@ export const mockOpenRouterSpeech = async (context: BrowserContext, seconds = 3)
|
|||
if (response_format === SPEECH_FORMAT) {
|
||||
spoken.push(input)
|
||||
|
||||
return route.fulfill({
|
||||
contentType: `audio/pcm;rate=${SPEECH_RATE};channels=1`,
|
||||
body: silence(seconds),
|
||||
})
|
||||
return route.fulfill({contentType: "audio/wav", body: silence(seconds)})
|
||||
}
|
||||
|
||||
return route.fulfill({
|
||||
|
|
|
|||
|
|
@ -986,8 +986,8 @@ test("US-119 have a message read out loud", async ({seed, as}) => {
|
|||
"another message\n\nheads up Alice Anchor, the notice is at a link to harbor.example",
|
||||
])
|
||||
|
||||
// The mock answers headerless pcm, so the duration is only right if the wav header the app put
|
||||
// in front of it is, which is what makes the whole clip scrubbable.
|
||||
// The app decodes what it is answered and rebuilds the container around the samples, so the
|
||||
// duration is only right if that round trip kept every one of them.
|
||||
await expect(alice.getByText("/ 0:10")).toBeVisible()
|
||||
|
||||
// The clip carries autoplay and the button follows the audio element's own play event, so it
|
||||
|
|
|
|||
|
|
@ -15,14 +15,13 @@ const SPEECH_MODEL = "hexgrad/kokoro-82m"
|
|||
|
||||
const SPEECH_VOICE = "af_bella"
|
||||
|
||||
// The endpoint encodes mp3 and raw pcm, and its mp3 carries a xing header naming a fraction of the
|
||||
// frames it holds, so a browser reads a fifth more audio than is there and the scrubber never
|
||||
// reaches the end. Raw pcm claims no length at all, so the wav header below is the only one.
|
||||
const SPEECH_FORMAT = "pcm"
|
||||
// The endpoint encodes mp3 and raw pcm. Raw pcm carries no header, so playing it means assuming a
|
||||
// sample rate and a sample format the response never states, and either assumption wrong is static
|
||||
// rather than an error. Decoding the mp3 reads both off the audio.
|
||||
const SPEECH_FORMAT = "mp3"
|
||||
|
||||
const SPEECH_RATE = 24000
|
||||
|
||||
const SPEECH_CHANNELS = 1
|
||||
// Decoding resamples to the context's rate, so this is the rate the wav ends up at.
|
||||
const SPEECH_RATE = 48000
|
||||
|
||||
const SPEECH_BIT_DEPTH = 16
|
||||
|
||||
|
|
@ -36,9 +35,15 @@ export type Speech = {
|
|||
|
||||
export const speech = writable<Maybe<Speech>>(undefined)
|
||||
|
||||
const toWav = (pcm: ArrayBuffer) => {
|
||||
const bytesPerFrame = (SPEECH_CHANNELS * SPEECH_BIT_DEPTH) / 8
|
||||
const wav = new ArrayBuffer(WAV_HEADER_LENGTH + pcm.byteLength)
|
||||
// An offline context never reaches for the speakers.
|
||||
const decode = (data: ArrayBuffer) =>
|
||||
new OfflineAudioContext(1, 1, SPEECH_RATE).decodeAudioData(data)
|
||||
|
||||
const toWav = (audio: AudioBuffer) => {
|
||||
const {numberOfChannels, sampleRate, length} = audio
|
||||
const bytesPerFrame = (numberOfChannels * SPEECH_BIT_DEPTH) / 8
|
||||
const dataLength = length * bytesPerFrame
|
||||
const wav = new ArrayBuffer(WAV_HEADER_LENGTH + dataLength)
|
||||
const view = new DataView(wav)
|
||||
const ascii = (offset: number, value: string) => {
|
||||
for (let i = 0; i < value.length; i++) {
|
||||
|
|
@ -47,19 +52,33 @@ const toWav = (pcm: ArrayBuffer) => {
|
|||
}
|
||||
|
||||
ascii(0, "RIFF")
|
||||
view.setUint32(4, 36 + pcm.byteLength, true)
|
||||
view.setUint32(4, 36 + dataLength, true)
|
||||
ascii(8, "WAVEfmt ")
|
||||
view.setUint32(16, 16, true)
|
||||
view.setUint16(20, 1, true)
|
||||
view.setUint16(22, SPEECH_CHANNELS, true)
|
||||
view.setUint32(24, SPEECH_RATE, true)
|
||||
view.setUint32(28, SPEECH_RATE * bytesPerFrame, true)
|
||||
view.setUint16(22, numberOfChannels, true)
|
||||
view.setUint32(24, sampleRate, true)
|
||||
view.setUint32(28, sampleRate * bytesPerFrame, true)
|
||||
view.setUint16(32, bytesPerFrame, true)
|
||||
view.setUint16(34, SPEECH_BIT_DEPTH, true)
|
||||
ascii(36, "data")
|
||||
view.setUint32(40, pcm.byteLength, true)
|
||||
view.setUint32(40, dataLength, true)
|
||||
|
||||
new Uint8Array(wav, WAV_HEADER_LENGTH).set(new Uint8Array(pcm))
|
||||
const channels = Array.from({length: numberOfChannels}, (_, i) => audio.getChannelData(i))
|
||||
|
||||
let offset = WAV_HEADER_LENGTH
|
||||
|
||||
for (let frame = 0; frame < length; frame++) {
|
||||
for (const samples of channels) {
|
||||
// A decoded sample runs from -1 to 1, and the two ends of a signed 16 bit range are not the
|
||||
// same size, so each end scales by its own bound.
|
||||
const sample = Math.max(-1, Math.min(1, samples[frame]))
|
||||
|
||||
view.setInt16(offset, Math.round(sample * (sample < 0 ? 0x8000 : 0x7fff)), true)
|
||||
|
||||
offset += 2
|
||||
}
|
||||
}
|
||||
|
||||
return new Blob([wav], {type: "audio/wav"})
|
||||
}
|
||||
|
|
@ -87,7 +106,7 @@ export const synthesize = async (text: string) => {
|
|||
throw new Error(error?.message || `OpenRouter returned a ${response.status}.`)
|
||||
}
|
||||
|
||||
return toWav(await response.arrayBuffer())
|
||||
return toWav(await decode(await response.arrayBuffer()))
|
||||
}
|
||||
|
||||
export const stopSpeech = () =>
|
||||
|
|
|
|||
Loading…
Reference in a new issue