Read entities aloud by name, and give the audio a length the player can trust

This commit is contained in:
Coracle-Bot 2026-09-09 16:38:45 +00:00 committed by hodlbod
parent 89cc1b337c
commit 78472f58b0
4 changed files with 129 additions and 41 deletions

View file

@ -449,6 +449,7 @@ Acceptance:
same prompt dictation uses. same prompt dictation uses.
- Once a key is saved, the same menu item puts a player at the bottom of the - Once a key is saved, the same menu item puts a player at the bottom of the
app naming whose message is being read. app naming whose message is being read.
- A quote, a mention or a url in the message is named rather than spelled out.
- The player plays, pauses, scrubs, and closes, and closing it takes it away. - The player plays, pauses, scrubs, and closes, and closing it takes it away.
### US-115 — Connect a wallet while sending a zap ### US-115 — Connect a wallet while sending a zap

View file

@ -213,37 +213,23 @@ export const mockDufflepud = (context: BrowserContext, fixtures: DufflepudFixtur
return route.fallback() return route.fallback()
}) })
// Silence as a wav, built rather than inlined so a spec can ask for a length and then assert the // The rate the endpoint documents for raw pcm, at 16 bits a sample, which is what the app assumes
// duration the player reads off it. 16 bit mono pcm is the shortest header a browser will decode. // when it writes a wav header for one.
const silence = (seconds: number) => { const SPEECH_RATE = 24000
const rate = 8000
const bytes = seconds * rate * 2
const wav = Buffer.alloc(44 + bytes)
wav.write("RIFF", 0) // Headerless silence. Nothing in it says how long it is, so the duration a spec reads off the
wav.writeUInt32LE(36 + bytes, 4) // player is the one the app computed.
wav.write("WAVEfmt ", 8) const silence = (seconds: number) => Buffer.alloc(seconds * SPEECH_RATE * 2)
wav.writeUInt32LE(16, 16)
wav.writeUInt16LE(1, 20)
wav.writeUInt16LE(1, 22)
wav.writeUInt32LE(rate, 24)
wav.writeUInt32LE(rate * 2, 28)
wav.writeUInt16LE(2, 32)
wav.writeUInt16LE(16, 34)
wav.write("data", 36)
wav.writeUInt32LE(bytes, 40)
return wav // mp3 is the other format the real endpoint encodes, and the harness has no encoder for it, so
} // asking for one here is a mistake rather than a case to serve.
const SPEECH_FORMAT = "pcm"
// The only two the real endpoint encodes; anything else comes back a 400 naming the pair.
const SPEECH_FORMATS = ["mp3", "pcm"]
/** /**
* OpenRouter's text to speech, answering every request with the same silence. The array it returns * OpenRouter's text to speech, answering every request with the same silence. The array it returns
* collects what the app asked to have read, in the order it asked. A request for a format the real * collects what the app asked to have read, in the order it asked. Answering a decodable wav to a
* endpoint does not encode is refused the way it refuses one, since a mock that plays anything back * request for some other format would prove nothing about the container the app builds, so anything
* cannot tell whether the app asked for audio a browser can decode. * but raw pcm is refused.
*/ */
export const mockOpenRouterSpeech = async (context: BrowserContext, seconds = 3) => { export const mockOpenRouterSpeech = async (context: BrowserContext, seconds = 3) => {
const spoken: string[] = [] const spoken: string[] = []
@ -251,15 +237,18 @@ export const mockOpenRouterSpeech = async (context: BrowserContext, seconds = 3)
await context.route(`${OPENROUTER_ORIGIN}/api/v1/audio/speech`, route => { await context.route(`${OPENROUTER_ORIGIN}/api/v1/audio/speech`, route => {
const {input, response_format} = JSON.parse(route.request().postData() ?? "{}") const {input, response_format} = JSON.parse(route.request().postData() ?? "{}")
if (SPEECH_FORMATS.includes(response_format)) { if (response_format === SPEECH_FORMAT) {
spoken.push(input) spoken.push(input)
return route.fulfill({contentType: "audio/wav", body: silence(seconds)}) return route.fulfill({
contentType: `audio/pcm;rate=${SPEECH_RATE};channels=1`,
body: silence(seconds),
})
} }
return route.fulfill({ return route.fulfill({
status: 400, status: 400,
json: {error: {message: `Invalid option: expected one of ${SPEECH_FORMATS.join("|")}`}}, json: {error: {message: `Invalid option: expected ${SPEECH_FORMAT}`}},
}) })
}) })

View file

@ -1,3 +1,4 @@
import {npubEncode} from "nostr-tools/nip19"
import {DAY, HOUR, MINUTE, WEEK, bech32ToHex} from "@welshman/lib" import {DAY, HOUR, MINUTE, WEEK, bech32ToHex} from "@welshman/lib"
import {getLnUrl} from "@welshman/util" import {getLnUrl} from "@welshman/util"
import {MessagingRelayList, Profile, RelayList, displayPubkey} from "@welshman/domain" import {MessagingRelayList, Profile, RelayList, displayPubkey} from "@welshman/domain"
@ -871,8 +872,17 @@ test("US-119 have a message read out loud", async ({seed, as}) => {
space.room("general", {name: "General"}) space.room("general", {name: "General"})
space.join(user.alice, "general") space.join(user.alice, "general")
space.join(user.bob, "general") space.join(user.bob, "general")
space.profile(user.alice, {name: "Alice Anchor"})
space.profile(user.bob, {name: "Bob Barnacle"}) space.profile(user.bob, {name: "Bob Barnacle"})
space.message(user.bob, "general", "the dock is closed on sunday", at(2, HOUR))
const notice = space.message(user.bob, "general", "the dock is closed on sunday", at(2, HOUR))
space.reply(
user.bob,
notice,
`heads up nostr:${npubEncode(user.alice.pubkey)}, the notice is at https://harbor.example/dock?ref=1`,
at(1, HOUR),
)
seedChatter(space, user.alice) seedChatter(space, user.alice)
}) })
@ -881,10 +891,11 @@ test("US-119 have a message read out loud", async ({seed, as}) => {
const alice = await as(users.alice, roomPath(url, "general")) const alice = await as(users.alice, roomPath(url, "general"))
const spoken = await mockOpenRouterSpeech(alice.context()) const spoken = await mockOpenRouterSpeech(alice.context())
await expect(message(alice, "the dock is closed on sunday")).toBeVisible() // The mention has to have resolved on screen before it can be expected in what was spoken.
await expect(message(alice, "heads up")).toContainText("@Alice Anchor")
// With no key saved, reading a message asks for one the way dictation does. // With no key saved, reading a message asks for one the way dictation does.
await openMessageMenu(alice, "the dock is closed on sunday") await openMessageMenu(alice, "heads up")
await alice.getByRole("button", {name: "Read Out Loud"}).click() await alice.getByRole("button", {name: "Read Out Loud"}).click()
const enable = dialog(alice, "Enable read out loud?") const enable = dialog(alice, "Enable read out loud?")
@ -894,16 +905,19 @@ test("US-119 have a message read out loud", async ({seed, as}) => {
await expect(alice.getByRole("alert")).toContainText("Read out loud is ready to use!") await expect(alice.getByRole("alert")).toContainText("Read out loud is ready to use!")
await openMessageMenu(alice, "the dock is closed on sunday") await openMessageMenu(alice, "heads up")
await alice.getByRole("button", {name: "Read Out Loud"}).click() await alice.getByRole("button", {name: "Read Out Loud"}).click()
await expect(alice.getByText("a message from Bob Barnacle")).toBeVisible() await expect(alice.getByText("a message from Bob Barnacle")).toBeVisible()
// Only what the message says is sent, so the nostr uri wrapping bob's mention never is. // The quote, the mention and the url are each named rather than spelled out, since none of them
expect(spoken).toEqual(["the dock is closed on sunday"]) // is intelligible read a character at a time.
expect(spoken).toEqual([
"another message\n\nheads up Alice Anchor, the notice is at a link to harbor.example",
])
// The duration is the mock's, which is what proves the player is on audio it decoded rather // The mock answers headerless pcm, so the duration is only right if the wav header the app put
// than on an element that failed to load. // in front of it is, which is what makes the whole clip scrubbable.
await expect(alice.getByText("/ 0:03")).toBeVisible() await expect(alice.getByText("/ 0:03")).toBeVisible()
// Chromium decides for itself whether the autoplay is allowed, so the control is read for // Chromium decides for itself whether the autoplay is allowed, so the control is read for

View file

@ -1,5 +1,6 @@
import {get, writable} from "svelte/store" import {get, writable} from "svelte/store"
import {parse, renderAsText} from "@welshman/content" import {ParsedType, isImage, parse} from "@welshman/content"
import type {Parsed} from "@welshman/content"
import type {Maybe} from "@welshman/lib" import type {Maybe} from "@welshman/lib"
import type {TrustedEvent} from "@welshman/util" import type {TrustedEvent} from "@welshman/util"
import {errorMessage} from "@lib/util" import {errorMessage} from "@lib/util"
@ -15,7 +16,18 @@ const SPEECH_MODEL = "hexgrad/kokoro-82m"
const SPEECH_VOICE = "af_bella" const SPEECH_VOICE = "af_bella"
const SPEECH_FORMAT = "mp3" // The endpoint encodes mp3 and raw pcm, and its mp3 carries a xing header naming a fraction of the
// frames it holds, so a browser reads a fifth more audio than is there and the scrubber never
// reaches the end. Raw pcm claims no length at all, so the wav header below is the only one.
const SPEECH_FORMAT = "pcm"
const SPEECH_RATE = 24000
const SPEECH_CHANNELS = 1
const SPEECH_BIT_DEPTH = 16
const WAV_HEADER_LENGTH = 44
export type Speech = { export type Speech = {
id: string id: string
@ -25,6 +37,78 @@ export type Speech = {
export const speech = writable<Maybe<Speech>>(undefined) export const speech = writable<Maybe<Speech>>(undefined)
const toWav = (pcm: ArrayBuffer) => {
const bytesPerFrame = (SPEECH_CHANNELS * SPEECH_BIT_DEPTH) / 8
const wav = new ArrayBuffer(WAV_HEADER_LENGTH + pcm.byteLength)
const view = new DataView(wav)
const ascii = (offset: number, value: string) => {
for (let i = 0; i < value.length; i++) {
view.setUint8(offset + i, value.charCodeAt(i))
}
}
ascii(0, "RIFF")
view.setUint32(4, 36 + pcm.byteLength, true)
ascii(8, "WAVEfmt ")
view.setUint32(16, 16, true)
view.setUint16(20, 1, true)
view.setUint16(22, SPEECH_CHANNELS, true)
view.setUint32(24, SPEECH_RATE, true)
view.setUint32(28, SPEECH_RATE * bytesPerFrame, true)
view.setUint16(32, bytesPerFrame, true)
view.setUint16(34, SPEECH_BIT_DEPTH, true)
ascii(36, "data")
view.setUint32(40, pcm.byteLength, true)
new Uint8Array(wav, WAV_HEADER_LENGTH).set(new Uint8Array(pcm))
return new Blob([wav], {type: "audio/wav"})
}
// An entity or a url read a character at a time is unintelligible, so anything that is not prose
// is named rather than spelled out.
const speakOne = (parsed: Parsed): string => {
switch (parsed.type) {
case ParsedType.Address:
return "another post"
case ParsedType.Cashu:
return "a cashu token"
case ParsedType.Code:
return parsed.value
case ParsedType.Command:
return parsed.raw
case ParsedType.Ellipsis:
return "\u2026"
case ParsedType.Email:
return parsed.value
case ParsedType.Emoji:
return parsed.value.name
case ParsedType.Event:
return "another message"
case ParsedType.Invoice:
return "a lightning invoice"
case ParsedType.Link: {
const {host} = parsed.value.url
return isImage(parsed) ? "an image" : `a link to ${host}`
}
case ParsedType.LinkGrid:
return "some images"
case ParsedType.Newline:
return parsed.value
case ParsedType.Profile:
return profiles.get().display(parsed.value.pubkey).get()
case ParsedType.Room:
return parsed.value.room
case ParsedType.Text:
return parsed.value
case ParsedType.Topic:
return parsed.value.slice(1)
}
}
const speakable = (event: TrustedEvent) => parse(event).map(speakOne).join("").trim()
export const synthesize = async (text: string) => { export const synthesize = async (text: string) => {
const response = await fetch("https://openrouter.ai/api/v1/audio/speech", { const response = await fetch("https://openrouter.ai/api/v1/audio/speech", {
method: "POST", method: "POST",
@ -48,7 +132,7 @@ export const synthesize = async (text: string) => {
throw new Error(error?.message || `OpenRouter returned a ${response.status}.`) throw new Error(error?.message || `OpenRouter returned a ${response.status}.`)
} }
return response.blob() return toWav(await response.arrayBuffer())
} }
export const stopSpeech = () => export const stopSpeech = () =>
@ -87,7 +171,7 @@ const play = async (event: TrustedEvent, text: string) => {
} }
export const readAloud = (event: TrustedEvent) => { export const readAloud = (event: TrustedEvent) => {
const text = renderAsText(parse(event)).toString().trim() const text = speakable(event)
if (getSetting("openrouter_key")) { if (getSetting("openrouter_key")) {
if (get(isCallActive)) { if (get(isCallActive)) {