Read entities aloud by name, and give the audio a length the player can trust
This commit is contained in:
parent
89cc1b337c
commit
78472f58b0
4 changed files with 129 additions and 41 deletions
|
|
@ -449,6 +449,7 @@ Acceptance:
|
|||
same prompt dictation uses.
|
||||
- Once a key is saved, the same menu item puts a player at the bottom of the
|
||||
app naming whose message is being read.
|
||||
- A quote, a mention or a url in the message is named rather than spelled out.
|
||||
- The player plays, pauses, scrubs, and closes, and closing it takes it away.
|
||||
|
||||
### US-115 — Connect a wallet while sending a zap
|
||||
|
|
|
|||
|
|
@ -213,37 +213,23 @@ export const mockDufflepud = (context: BrowserContext, fixtures: DufflepudFixtur
|
|||
return route.fallback()
|
||||
})
|
||||
|
||||
// Silence as a wav, built rather than inlined so a spec can ask for a length and then assert the
|
||||
// duration the player reads off it. 16 bit mono pcm is the shortest header a browser will decode.
|
||||
const silence = (seconds: number) => {
|
||||
const rate = 8000
|
||||
const bytes = seconds * rate * 2
|
||||
const wav = Buffer.alloc(44 + bytes)
|
||||
// The rate the endpoint documents for raw pcm, at 16 bits a sample, which is what the app assumes
|
||||
// when it writes a wav header for one.
|
||||
const SPEECH_RATE = 24000
|
||||
|
||||
wav.write("RIFF", 0)
|
||||
wav.writeUInt32LE(36 + bytes, 4)
|
||||
wav.write("WAVEfmt ", 8)
|
||||
wav.writeUInt32LE(16, 16)
|
||||
wav.writeUInt16LE(1, 20)
|
||||
wav.writeUInt16LE(1, 22)
|
||||
wav.writeUInt32LE(rate, 24)
|
||||
wav.writeUInt32LE(rate * 2, 28)
|
||||
wav.writeUInt16LE(2, 32)
|
||||
wav.writeUInt16LE(16, 34)
|
||||
wav.write("data", 36)
|
||||
wav.writeUInt32LE(bytes, 40)
|
||||
// Headerless silence. Nothing in it says how long it is, so the duration a spec reads off the
|
||||
// player is the one the app computed.
|
||||
const silence = (seconds: number) => Buffer.alloc(seconds * SPEECH_RATE * 2)
|
||||
|
||||
return wav
|
||||
}
|
||||
|
||||
// The only two the real endpoint encodes; anything else comes back a 400 naming the pair.
|
||||
const SPEECH_FORMATS = ["mp3", "pcm"]
|
||||
// mp3 is the other format the real endpoint encodes, and the harness has no encoder for it, so
|
||||
// asking for one here is a mistake rather than a case to serve.
|
||||
const SPEECH_FORMAT = "pcm"
|
||||
|
||||
/**
|
||||
* OpenRouter's text to speech, answering every request with the same silence. The array it returns
|
||||
* collects what the app asked to have read, in the order it asked. A request for a format the real
|
||||
* endpoint does not encode is refused the way it refuses one, since a mock that plays anything back
|
||||
* cannot tell whether the app asked for audio a browser can decode.
|
||||
* collects what the app asked to have read, in the order it asked. Answering a decodable wav to a
|
||||
* request for some other format would prove nothing about the container the app builds, so anything
|
||||
* but raw pcm is refused.
|
||||
*/
|
||||
export const mockOpenRouterSpeech = async (context: BrowserContext, seconds = 3) => {
|
||||
const spoken: string[] = []
|
||||
|
|
@ -251,15 +237,18 @@ export const mockOpenRouterSpeech = async (context: BrowserContext, seconds = 3)
|
|||
await context.route(`${OPENROUTER_ORIGIN}/api/v1/audio/speech`, route => {
|
||||
const {input, response_format} = JSON.parse(route.request().postData() ?? "{}")
|
||||
|
||||
if (SPEECH_FORMATS.includes(response_format)) {
|
||||
if (response_format === SPEECH_FORMAT) {
|
||||
spoken.push(input)
|
||||
|
||||
return route.fulfill({contentType: "audio/wav", body: silence(seconds)})
|
||||
return route.fulfill({
|
||||
contentType: `audio/pcm;rate=${SPEECH_RATE};channels=1`,
|
||||
body: silence(seconds),
|
||||
})
|
||||
}
|
||||
|
||||
return route.fulfill({
|
||||
status: 400,
|
||||
json: {error: {message: `Invalid option: expected one of ${SPEECH_FORMATS.join("|")}`}},
|
||||
json: {error: {message: `Invalid option: expected ${SPEECH_FORMAT}`}},
|
||||
})
|
||||
})
|
||||
|
||||
|
|
|
|||
|
|
@ -1,3 +1,4 @@
|
|||
import {npubEncode} from "nostr-tools/nip19"
|
||||
import {DAY, HOUR, MINUTE, WEEK, bech32ToHex} from "@welshman/lib"
|
||||
import {getLnUrl} from "@welshman/util"
|
||||
import {MessagingRelayList, Profile, RelayList, displayPubkey} from "@welshman/domain"
|
||||
|
|
@ -871,8 +872,17 @@ test("US-119 have a message read out loud", async ({seed, as}) => {
|
|||
space.room("general", {name: "General"})
|
||||
space.join(user.alice, "general")
|
||||
space.join(user.bob, "general")
|
||||
space.profile(user.alice, {name: "Alice Anchor"})
|
||||
space.profile(user.bob, {name: "Bob Barnacle"})
|
||||
space.message(user.bob, "general", "the dock is closed on sunday", at(2, HOUR))
|
||||
|
||||
const notice = space.message(user.bob, "general", "the dock is closed on sunday", at(2, HOUR))
|
||||
|
||||
space.reply(
|
||||
user.bob,
|
||||
notice,
|
||||
`heads up nostr:${npubEncode(user.alice.pubkey)}, the notice is at https://harbor.example/dock?ref=1`,
|
||||
at(1, HOUR),
|
||||
)
|
||||
|
||||
seedChatter(space, user.alice)
|
||||
})
|
||||
|
|
@ -881,10 +891,11 @@ test("US-119 have a message read out loud", async ({seed, as}) => {
|
|||
const alice = await as(users.alice, roomPath(url, "general"))
|
||||
const spoken = await mockOpenRouterSpeech(alice.context())
|
||||
|
||||
await expect(message(alice, "the dock is closed on sunday")).toBeVisible()
|
||||
// The mention has to have resolved on screen before it can be expected in what was spoken.
|
||||
await expect(message(alice, "heads up")).toContainText("@Alice Anchor")
|
||||
|
||||
// With no key saved, reading a message asks for one the way dictation does.
|
||||
await openMessageMenu(alice, "the dock is closed on sunday")
|
||||
await openMessageMenu(alice, "heads up")
|
||||
await alice.getByRole("button", {name: "Read Out Loud"}).click()
|
||||
|
||||
const enable = dialog(alice, "Enable read out loud?")
|
||||
|
|
@ -894,16 +905,19 @@ test("US-119 have a message read out loud", async ({seed, as}) => {
|
|||
|
||||
await expect(alice.getByRole("alert")).toContainText("Read out loud is ready to use!")
|
||||
|
||||
await openMessageMenu(alice, "the dock is closed on sunday")
|
||||
await openMessageMenu(alice, "heads up")
|
||||
await alice.getByRole("button", {name: "Read Out Loud"}).click()
|
||||
|
||||
await expect(alice.getByText("a message from Bob Barnacle")).toBeVisible()
|
||||
|
||||
// Only what the message says is sent, so the nostr uri wrapping bob's mention never is.
|
||||
expect(spoken).toEqual(["the dock is closed on sunday"])
|
||||
// The quote, the mention and the url are each named rather than spelled out, since none of them
|
||||
// is intelligible read a character at a time.
|
||||
expect(spoken).toEqual([
|
||||
"another message\n\nheads up Alice Anchor, the notice is at a link to harbor.example",
|
||||
])
|
||||
|
||||
// The duration is the mock's, which is what proves the player is on audio it decoded rather
|
||||
// than on an element that failed to load.
|
||||
// The mock answers headerless pcm, so the duration is only right if the wav header the app put
|
||||
// in front of it is, which is what makes the whole clip scrubbable.
|
||||
await expect(alice.getByText("/ 0:03")).toBeVisible()
|
||||
|
||||
// Chromium decides for itself whether the autoplay is allowed, so the control is read for
|
||||
|
|
|
|||
|
|
@ -1,5 +1,6 @@
|
|||
import {get, writable} from "svelte/store"
|
||||
import {parse, renderAsText} from "@welshman/content"
|
||||
import {ParsedType, isImage, parse} from "@welshman/content"
|
||||
import type {Parsed} from "@welshman/content"
|
||||
import type {Maybe} from "@welshman/lib"
|
||||
import type {TrustedEvent} from "@welshman/util"
|
||||
import {errorMessage} from "@lib/util"
|
||||
|
|
@ -15,7 +16,18 @@ const SPEECH_MODEL = "hexgrad/kokoro-82m"
|
|||
|
||||
const SPEECH_VOICE = "af_bella"
|
||||
|
||||
const SPEECH_FORMAT = "mp3"
|
||||
// The endpoint encodes mp3 and raw pcm, and its mp3 carries a xing header naming a fraction of the
|
||||
// frames it holds, so a browser reads a fifth more audio than is there and the scrubber never
|
||||
// reaches the end. Raw pcm claims no length at all, so the wav header below is the only one.
|
||||
const SPEECH_FORMAT = "pcm"
|
||||
|
||||
const SPEECH_RATE = 24000
|
||||
|
||||
const SPEECH_CHANNELS = 1
|
||||
|
||||
const SPEECH_BIT_DEPTH = 16
|
||||
|
||||
const WAV_HEADER_LENGTH = 44
|
||||
|
||||
export type Speech = {
|
||||
id: string
|
||||
|
|
@ -25,6 +37,78 @@ export type Speech = {
|
|||
|
||||
export const speech = writable<Maybe<Speech>>(undefined)
|
||||
|
||||
const toWav = (pcm: ArrayBuffer) => {
|
||||
const bytesPerFrame = (SPEECH_CHANNELS * SPEECH_BIT_DEPTH) / 8
|
||||
const wav = new ArrayBuffer(WAV_HEADER_LENGTH + pcm.byteLength)
|
||||
const view = new DataView(wav)
|
||||
const ascii = (offset: number, value: string) => {
|
||||
for (let i = 0; i < value.length; i++) {
|
||||
view.setUint8(offset + i, value.charCodeAt(i))
|
||||
}
|
||||
}
|
||||
|
||||
ascii(0, "RIFF")
|
||||
view.setUint32(4, 36 + pcm.byteLength, true)
|
||||
ascii(8, "WAVEfmt ")
|
||||
view.setUint32(16, 16, true)
|
||||
view.setUint16(20, 1, true)
|
||||
view.setUint16(22, SPEECH_CHANNELS, true)
|
||||
view.setUint32(24, SPEECH_RATE, true)
|
||||
view.setUint32(28, SPEECH_RATE * bytesPerFrame, true)
|
||||
view.setUint16(32, bytesPerFrame, true)
|
||||
view.setUint16(34, SPEECH_BIT_DEPTH, true)
|
||||
ascii(36, "data")
|
||||
view.setUint32(40, pcm.byteLength, true)
|
||||
|
||||
new Uint8Array(wav, WAV_HEADER_LENGTH).set(new Uint8Array(pcm))
|
||||
|
||||
return new Blob([wav], {type: "audio/wav"})
|
||||
}
|
||||
|
||||
// An entity or a url read a character at a time is unintelligible, so anything that is not prose
|
||||
// is named rather than spelled out.
|
||||
const speakOne = (parsed: Parsed): string => {
|
||||
switch (parsed.type) {
|
||||
case ParsedType.Address:
|
||||
return "another post"
|
||||
case ParsedType.Cashu:
|
||||
return "a cashu token"
|
||||
case ParsedType.Code:
|
||||
return parsed.value
|
||||
case ParsedType.Command:
|
||||
return parsed.raw
|
||||
case ParsedType.Ellipsis:
|
||||
return "\u2026"
|
||||
case ParsedType.Email:
|
||||
return parsed.value
|
||||
case ParsedType.Emoji:
|
||||
return parsed.value.name
|
||||
case ParsedType.Event:
|
||||
return "another message"
|
||||
case ParsedType.Invoice:
|
||||
return "a lightning invoice"
|
||||
case ParsedType.Link: {
|
||||
const {host} = parsed.value.url
|
||||
|
||||
return isImage(parsed) ? "an image" : `a link to ${host}`
|
||||
}
|
||||
case ParsedType.LinkGrid:
|
||||
return "some images"
|
||||
case ParsedType.Newline:
|
||||
return parsed.value
|
||||
case ParsedType.Profile:
|
||||
return profiles.get().display(parsed.value.pubkey).get()
|
||||
case ParsedType.Room:
|
||||
return parsed.value.room
|
||||
case ParsedType.Text:
|
||||
return parsed.value
|
||||
case ParsedType.Topic:
|
||||
return parsed.value.slice(1)
|
||||
}
|
||||
}
|
||||
|
||||
const speakable = (event: TrustedEvent) => parse(event).map(speakOne).join("").trim()
|
||||
|
||||
export const synthesize = async (text: string) => {
|
||||
const response = await fetch("https://openrouter.ai/api/v1/audio/speech", {
|
||||
method: "POST",
|
||||
|
|
@ -48,7 +132,7 @@ export const synthesize = async (text: string) => {
|
|||
throw new Error(error?.message || `OpenRouter returned a ${response.status}.`)
|
||||
}
|
||||
|
||||
return response.blob()
|
||||
return toWav(await response.arrayBuffer())
|
||||
}
|
||||
|
||||
export const stopSpeech = () =>
|
||||
|
|
@ -87,7 +171,7 @@ const play = async (event: TrustedEvent, text: string) => {
|
|||
}
|
||||
|
||||
export const readAloud = (event: TrustedEvent) => {
|
||||
const text = renderAsText(parse(event)).toString().trim()
|
||||
const text = speakable(event)
|
||||
|
||||
if (getSetting("openrouter_key")) {
|
||||
if (get(isCallActive)) {
|
||||
|
|
|
|||
Loading…
Reference in a new issue