diff --git a/e2e/USER_STORIES.md b/e2e/USER_STORIES.md index 19a5b082..43f0d070 100644 --- a/e2e/USER_STORIES.md +++ b/e2e/USER_STORIES.md @@ -449,6 +449,7 @@ Acceptance: same prompt dictation uses. - Once a key is saved, the same menu item puts a player at the bottom of the app naming whose message is being read. +- A quote, a mention or a url in the message is named rather than spelled out. - The player plays, pauses, scrubs, and closes, and closing it takes it away. ### US-115 — Connect a wallet while sending a zap diff --git a/e2e/harness/net/http.ts b/e2e/harness/net/http.ts index f424cd81..43c4b7ff 100644 --- a/e2e/harness/net/http.ts +++ b/e2e/harness/net/http.ts @@ -213,37 +213,23 @@ export const mockDufflepud = (context: BrowserContext, fixtures: DufflepudFixtur return route.fallback() }) -// Silence as a wav, built rather than inlined so a spec can ask for a length and then assert the -// duration the player reads off it. 16 bit mono pcm is the shortest header a browser will decode. -const silence = (seconds: number) => { - const rate = 8000 - const bytes = seconds * rate * 2 - const wav = Buffer.alloc(44 + bytes) +// The rate the endpoint documents for raw pcm, at 16 bits a sample, which is what the app assumes +// when it writes a wav header for one. +const SPEECH_RATE = 24000 - wav.write("RIFF", 0) - wav.writeUInt32LE(36 + bytes, 4) - wav.write("WAVEfmt ", 8) - wav.writeUInt32LE(16, 16) - wav.writeUInt16LE(1, 20) - wav.writeUInt16LE(1, 22) - wav.writeUInt32LE(rate, 24) - wav.writeUInt32LE(rate * 2, 28) - wav.writeUInt16LE(2, 32) - wav.writeUInt16LE(16, 34) - wav.write("data", 36) - wav.writeUInt32LE(bytes, 40) +// Headerless silence. Nothing in it says how long it is, so the duration a spec reads off the +// player is the one the app computed. +const silence = (seconds: number) => Buffer.alloc(seconds * SPEECH_RATE * 2) - return wav -} - -// The only two the real endpoint encodes; anything else comes back a 400 naming the pair. -const SPEECH_FORMATS = ["mp3", "pcm"] +// mp3 is the other format the real endpoint encodes, and the harness has no encoder for it, so +// asking for one here is a mistake rather than a case to serve. +const SPEECH_FORMAT = "pcm" /** * OpenRouter's text to speech, answering every request with the same silence. The array it returns - * collects what the app asked to have read, in the order it asked. A request for a format the real - * endpoint does not encode is refused the way it refuses one, since a mock that plays anything back - * cannot tell whether the app asked for audio a browser can decode. + * collects what the app asked to have read, in the order it asked. Answering a decodable wav to a + * request for some other format would prove nothing about the container the app builds, so anything + * but raw pcm is refused. */ export const mockOpenRouterSpeech = async (context: BrowserContext, seconds = 3) => { const spoken: string[] = [] @@ -251,15 +237,18 @@ export const mockOpenRouterSpeech = async (context: BrowserContext, seconds = 3) await context.route(`${OPENROUTER_ORIGIN}/api/v1/audio/speech`, route => { const {input, response_format} = JSON.parse(route.request().postData() ?? "{}") - if (SPEECH_FORMATS.includes(response_format)) { + if (response_format === SPEECH_FORMAT) { spoken.push(input) - return route.fulfill({contentType: "audio/wav", body: silence(seconds)}) + return route.fulfill({ + contentType: `audio/pcm;rate=${SPEECH_RATE};channels=1`, + body: silence(seconds), + }) } return route.fulfill({ status: 400, - json: {error: {message: `Invalid option: expected one of ${SPEECH_FORMATS.join("|")}`}}, + json: {error: {message: `Invalid option: expected ${SPEECH_FORMAT}`}}, }) }) diff --git a/e2e/specs/rooms.spec.ts b/e2e/specs/rooms.spec.ts index f94e8281..9b2b1a49 100644 --- a/e2e/specs/rooms.spec.ts +++ b/e2e/specs/rooms.spec.ts @@ -1,3 +1,4 @@ +import {npubEncode} from "nostr-tools/nip19" import {DAY, HOUR, MINUTE, WEEK, bech32ToHex} from "@welshman/lib" import {getLnUrl} from "@welshman/util" import {MessagingRelayList, Profile, RelayList, displayPubkey} from "@welshman/domain" @@ -871,8 +872,17 @@ test("US-119 have a message read out loud", async ({seed, as}) => { space.room("general", {name: "General"}) space.join(user.alice, "general") space.join(user.bob, "general") + space.profile(user.alice, {name: "Alice Anchor"}) space.profile(user.bob, {name: "Bob Barnacle"}) - space.message(user.bob, "general", "the dock is closed on sunday", at(2, HOUR)) + + const notice = space.message(user.bob, "general", "the dock is closed on sunday", at(2, HOUR)) + + space.reply( + user.bob, + notice, + `heads up nostr:${npubEncode(user.alice.pubkey)}, the notice is at https://harbor.example/dock?ref=1`, + at(1, HOUR), + ) seedChatter(space, user.alice) }) @@ -881,10 +891,11 @@ test("US-119 have a message read out loud", async ({seed, as}) => { const alice = await as(users.alice, roomPath(url, "general")) const spoken = await mockOpenRouterSpeech(alice.context()) - await expect(message(alice, "the dock is closed on sunday")).toBeVisible() + // The mention has to have resolved on screen before it can be expected in what was spoken. + await expect(message(alice, "heads up")).toContainText("@Alice Anchor") // With no key saved, reading a message asks for one the way dictation does. - await openMessageMenu(alice, "the dock is closed on sunday") + await openMessageMenu(alice, "heads up") await alice.getByRole("button", {name: "Read Out Loud"}).click() const enable = dialog(alice, "Enable read out loud?") @@ -894,16 +905,19 @@ test("US-119 have a message read out loud", async ({seed, as}) => { await expect(alice.getByRole("alert")).toContainText("Read out loud is ready to use!") - await openMessageMenu(alice, "the dock is closed on sunday") + await openMessageMenu(alice, "heads up") await alice.getByRole("button", {name: "Read Out Loud"}).click() await expect(alice.getByText("a message from Bob Barnacle")).toBeVisible() - // Only what the message says is sent, so the nostr uri wrapping bob's mention never is. - expect(spoken).toEqual(["the dock is closed on sunday"]) + // The quote, the mention and the url are each named rather than spelled out, since none of them + // is intelligible read a character at a time. + expect(spoken).toEqual([ + "another message\n\nheads up Alice Anchor, the notice is at a link to harbor.example", + ]) - // The duration is the mock's, which is what proves the player is on audio it decoded rather - // than on an element that failed to load. + // The mock answers headerless pcm, so the duration is only right if the wav header the app put + // in front of it is, which is what makes the whole clip scrubbable. await expect(alice.getByText("/ 0:03")).toBeVisible() // Chromium decides for itself whether the autoplay is allowed, so the control is read for diff --git a/src/app/speech.ts b/src/app/speech.ts index 659c3824..c747f38c 100644 --- a/src/app/speech.ts +++ b/src/app/speech.ts @@ -1,5 +1,6 @@ import {get, writable} from "svelte/store" -import {parse, renderAsText} from "@welshman/content" +import {ParsedType, isImage, parse} from "@welshman/content" +import type {Parsed} from "@welshman/content" import type {Maybe} from "@welshman/lib" import type {TrustedEvent} from "@welshman/util" import {errorMessage} from "@lib/util" @@ -15,7 +16,18 @@ const SPEECH_MODEL = "hexgrad/kokoro-82m" const SPEECH_VOICE = "af_bella" -const SPEECH_FORMAT = "mp3" +// The endpoint encodes mp3 and raw pcm, and its mp3 carries a xing header naming a fraction of the +// frames it holds, so a browser reads a fifth more audio than is there and the scrubber never +// reaches the end. Raw pcm claims no length at all, so the wav header below is the only one. +const SPEECH_FORMAT = "pcm" + +const SPEECH_RATE = 24000 + +const SPEECH_CHANNELS = 1 + +const SPEECH_BIT_DEPTH = 16 + +const WAV_HEADER_LENGTH = 44 export type Speech = { id: string @@ -25,6 +37,78 @@ export type Speech = { export const speech = writable>(undefined) +const toWav = (pcm: ArrayBuffer) => { + const bytesPerFrame = (SPEECH_CHANNELS * SPEECH_BIT_DEPTH) / 8 + const wav = new ArrayBuffer(WAV_HEADER_LENGTH + pcm.byteLength) + const view = new DataView(wav) + const ascii = (offset: number, value: string) => { + for (let i = 0; i < value.length; i++) { + view.setUint8(offset + i, value.charCodeAt(i)) + } + } + + ascii(0, "RIFF") + view.setUint32(4, 36 + pcm.byteLength, true) + ascii(8, "WAVEfmt ") + view.setUint32(16, 16, true) + view.setUint16(20, 1, true) + view.setUint16(22, SPEECH_CHANNELS, true) + view.setUint32(24, SPEECH_RATE, true) + view.setUint32(28, SPEECH_RATE * bytesPerFrame, true) + view.setUint16(32, bytesPerFrame, true) + view.setUint16(34, SPEECH_BIT_DEPTH, true) + ascii(36, "data") + view.setUint32(40, pcm.byteLength, true) + + new Uint8Array(wav, WAV_HEADER_LENGTH).set(new Uint8Array(pcm)) + + return new Blob([wav], {type: "audio/wav"}) +} + +// An entity or a url read a character at a time is unintelligible, so anything that is not prose +// is named rather than spelled out. +const speakOne = (parsed: Parsed): string => { + switch (parsed.type) { + case ParsedType.Address: + return "another post" + case ParsedType.Cashu: + return "a cashu token" + case ParsedType.Code: + return parsed.value + case ParsedType.Command: + return parsed.raw + case ParsedType.Ellipsis: + return "\u2026" + case ParsedType.Email: + return parsed.value + case ParsedType.Emoji: + return parsed.value.name + case ParsedType.Event: + return "another message" + case ParsedType.Invoice: + return "a lightning invoice" + case ParsedType.Link: { + const {host} = parsed.value.url + + return isImage(parsed) ? "an image" : `a link to ${host}` + } + case ParsedType.LinkGrid: + return "some images" + case ParsedType.Newline: + return parsed.value + case ParsedType.Profile: + return profiles.get().display(parsed.value.pubkey).get() + case ParsedType.Room: + return parsed.value.room + case ParsedType.Text: + return parsed.value + case ParsedType.Topic: + return parsed.value.slice(1) + } +} + +const speakable = (event: TrustedEvent) => parse(event).map(speakOne).join("").trim() + export const synthesize = async (text: string) => { const response = await fetch("https://openrouter.ai/api/v1/audio/speech", { method: "POST", @@ -48,7 +132,7 @@ export const synthesize = async (text: string) => { throw new Error(error?.message || `OpenRouter returned a ${response.status}.`) } - return response.blob() + return toWav(await response.arrayBuffer()) } export const stopSpeech = () => @@ -87,7 +171,7 @@ const play = async (event: TrustedEvent, text: string) => { } export const readAloud = (event: TrustedEvent) => { - const text = renderAsText(parse(event)).toString().trim() + const text = speakable(event) if (getSetting("openrouter_key")) { if (get(isCallActive)) {