diff --git a/e2e/harness/net/http.ts b/e2e/harness/net/http.ts index 28bee34e..f424cd81 100644 --- a/e2e/harness/net/http.ts +++ b/e2e/harness/net/http.ts @@ -236,17 +236,31 @@ const silence = (seconds: number) => { return wav } +// The only two the real endpoint encodes; anything else comes back a 400 naming the pair. +const SPEECH_FORMATS = ["mp3", "pcm"] + /** * OpenRouter's text to speech, answering every request with the same silence. The array it returns - * collects what the app asked to have read, in the order it asked. + * collects what the app asked to have read, in the order it asked. A request for a format the real + * endpoint does not encode is refused the way it refuses one, since a mock that plays anything back + * cannot tell whether the app asked for audio a browser can decode. */ export const mockOpenRouterSpeech = async (context: BrowserContext, seconds = 3) => { const spoken: string[] = [] await context.route(`${OPENROUTER_ORIGIN}/api/v1/audio/speech`, route => { - spoken.push(JSON.parse(route.request().postData() ?? "{}").input) + const {input, response_format} = JSON.parse(route.request().postData() ?? "{}") - return route.fulfill({contentType: "audio/wav", body: silence(seconds)}) + if (SPEECH_FORMATS.includes(response_format)) { + spoken.push(input) + + return route.fulfill({contentType: "audio/wav", body: silence(seconds)}) + } + + return route.fulfill({ + status: 400, + json: {error: {message: `Invalid option: expected one of ${SPEECH_FORMATS.join("|")}`}}, + }) }) return spoken diff --git a/src/app/speech.ts b/src/app/speech.ts index de10b33f..659c3824 100644 --- a/src/app/speech.ts +++ b/src/app/speech.ts @@ -11,9 +11,11 @@ import {pushModal} from "@app/modal" import {getSetting} from "@app/settings" import {pushToast} from "@app/toast" -const SPEECH_MODEL = "openai/gpt-audio-mini" +const SPEECH_MODEL = "hexgrad/kokoro-82m" -const SPEECH_VOICE = "alloy" +const SPEECH_VOICE = "af_bella" + +const SPEECH_FORMAT = "mp3" export type Speech = { id: string @@ -30,7 +32,12 @@ export const synthesize = async (text: string) => { Authorization: `Bearer ${getSetting("openrouter_key")}`, "Content-Type": "application/json", }, - body: JSON.stringify({model: SPEECH_MODEL, voice: SPEECH_VOICE, input: text}), + body: JSON.stringify({ + model: SPEECH_MODEL, + voice: SPEECH_VOICE, + input: text, + response_format: SPEECH_FORMAT, + }), }) // A successful response is audio rather than json, so the error body is only worth reading once