Synthesize speech with a model OpenRouter serves, as playable mp3
This commit is contained in:
parent
e66dde1ea0
commit
89cc1b337c
2 changed files with 27 additions and 6 deletions
|
|
@ -236,17 +236,31 @@ const silence = (seconds: number) => {
|
|||
return wav
|
||||
}
|
||||
|
||||
// The only two the real endpoint encodes; anything else comes back a 400 naming the pair.
|
||||
const SPEECH_FORMATS = ["mp3", "pcm"]
|
||||
|
||||
/**
|
||||
* OpenRouter's text to speech, answering every request with the same silence. The array it returns
|
||||
* collects what the app asked to have read, in the order it asked.
|
||||
* collects what the app asked to have read, in the order it asked. A request for a format the real
|
||||
* endpoint does not encode is refused the way it refuses one, since a mock that plays anything back
|
||||
* cannot tell whether the app asked for audio a browser can decode.
|
||||
*/
|
||||
export const mockOpenRouterSpeech = async (context: BrowserContext, seconds = 3) => {
|
||||
const spoken: string[] = []
|
||||
|
||||
await context.route(`${OPENROUTER_ORIGIN}/api/v1/audio/speech`, route => {
|
||||
spoken.push(JSON.parse(route.request().postData() ?? "{}").input)
|
||||
const {input, response_format} = JSON.parse(route.request().postData() ?? "{}")
|
||||
|
||||
return route.fulfill({contentType: "audio/wav", body: silence(seconds)})
|
||||
if (SPEECH_FORMATS.includes(response_format)) {
|
||||
spoken.push(input)
|
||||
|
||||
return route.fulfill({contentType: "audio/wav", body: silence(seconds)})
|
||||
}
|
||||
|
||||
return route.fulfill({
|
||||
status: 400,
|
||||
json: {error: {message: `Invalid option: expected one of ${SPEECH_FORMATS.join("|")}`}},
|
||||
})
|
||||
})
|
||||
|
||||
return spoken
|
||||
|
|
|
|||
|
|
@ -11,9 +11,11 @@ import {pushModal} from "@app/modal"
|
|||
import {getSetting} from "@app/settings"
|
||||
import {pushToast} from "@app/toast"
|
||||
|
||||
const SPEECH_MODEL = "openai/gpt-audio-mini"
|
||||
const SPEECH_MODEL = "hexgrad/kokoro-82m"
|
||||
|
||||
const SPEECH_VOICE = "alloy"
|
||||
const SPEECH_VOICE = "af_bella"
|
||||
|
||||
const SPEECH_FORMAT = "mp3"
|
||||
|
||||
export type Speech = {
|
||||
id: string
|
||||
|
|
@ -30,7 +32,12 @@ export const synthesize = async (text: string) => {
|
|||
Authorization: `Bearer ${getSetting("openrouter_key")}`,
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
body: JSON.stringify({model: SPEECH_MODEL, voice: SPEECH_VOICE, input: text}),
|
||||
body: JSON.stringify({
|
||||
model: SPEECH_MODEL,
|
||||
voice: SPEECH_VOICE,
|
||||
input: text,
|
||||
response_format: SPEECH_FORMAT,
|
||||
}),
|
||||
})
|
||||
|
||||
// A successful response is audio rather than json, so the error body is only worth reading once
|
||||
|
|
|
|||
Loading…
Reference in a new issue