Synthesize speech with a model OpenRouter serves, as playable mp3

This commit is contained in:
Coracle-Bot 2026-09-09 15:17:09 +00:00 committed by hodlbod
parent e66dde1ea0
commit 89cc1b337c
2 changed files with 27 additions and 6 deletions

View file

@ -236,17 +236,31 @@ const silence = (seconds: number) => {
return wav return wav
} }
// The only two the real endpoint encodes; anything else comes back a 400 naming the pair.
const SPEECH_FORMATS = ["mp3", "pcm"]
/** /**
* OpenRouter's text to speech, answering every request with the same silence. The array it returns * OpenRouter's text to speech, answering every request with the same silence. The array it returns
* collects what the app asked to have read, in the order it asked. * collects what the app asked to have read, in the order it asked. A request for a format the real
* endpoint does not encode is refused the way it refuses one, since a mock that plays anything back
* cannot tell whether the app asked for audio a browser can decode.
*/ */
export const mockOpenRouterSpeech = async (context: BrowserContext, seconds = 3) => { export const mockOpenRouterSpeech = async (context: BrowserContext, seconds = 3) => {
const spoken: string[] = [] const spoken: string[] = []
await context.route(`${OPENROUTER_ORIGIN}/api/v1/audio/speech`, route => { await context.route(`${OPENROUTER_ORIGIN}/api/v1/audio/speech`, route => {
spoken.push(JSON.parse(route.request().postData() ?? "{}").input) const {input, response_format} = JSON.parse(route.request().postData() ?? "{}")
return route.fulfill({contentType: "audio/wav", body: silence(seconds)}) if (SPEECH_FORMATS.includes(response_format)) {
spoken.push(input)
return route.fulfill({contentType: "audio/wav", body: silence(seconds)})
}
return route.fulfill({
status: 400,
json: {error: {message: `Invalid option: expected one of ${SPEECH_FORMATS.join("|")}`}},
})
}) })
return spoken return spoken

View file

@ -11,9 +11,11 @@ import {pushModal} from "@app/modal"
import {getSetting} from "@app/settings" import {getSetting} from "@app/settings"
import {pushToast} from "@app/toast" import {pushToast} from "@app/toast"
const SPEECH_MODEL = "openai/gpt-audio-mini" const SPEECH_MODEL = "hexgrad/kokoro-82m"
const SPEECH_VOICE = "alloy" const SPEECH_VOICE = "af_bella"
const SPEECH_FORMAT = "mp3"
export type Speech = { export type Speech = {
id: string id: string
@ -30,7 +32,12 @@ export const synthesize = async (text: string) => {
Authorization: `Bearer ${getSetting("openrouter_key")}`, Authorization: `Bearer ${getSetting("openrouter_key")}`,
"Content-Type": "application/json", "Content-Type": "application/json",
}, },
body: JSON.stringify({model: SPEECH_MODEL, voice: SPEECH_VOICE, input: text}), body: JSON.stringify({
model: SPEECH_MODEL,
voice: SPEECH_VOICE,
input: text,
response_format: SPEECH_FORMAT,
}),
}) })
// A successful response is audio rather than json, so the error body is only worth reading once // A successful response is audio rather than json, so the error body is only worth reading once