Read entities aloud by name, and give the audio a length the player can trust
This commit is contained in:
parent
89cc1b337c
commit
78472f58b0
4 changed files with 129 additions and 41 deletions
|
|
@ -449,6 +449,7 @@ Acceptance:
|
||||||
same prompt dictation uses.
|
same prompt dictation uses.
|
||||||
- Once a key is saved, the same menu item puts a player at the bottom of the
|
- Once a key is saved, the same menu item puts a player at the bottom of the
|
||||||
app naming whose message is being read.
|
app naming whose message is being read.
|
||||||
|
- A quote, a mention or a url in the message is named rather than spelled out.
|
||||||
- The player plays, pauses, scrubs, and closes, and closing it takes it away.
|
- The player plays, pauses, scrubs, and closes, and closing it takes it away.
|
||||||
|
|
||||||
### US-115 — Connect a wallet while sending a zap
|
### US-115 — Connect a wallet while sending a zap
|
||||||
|
|
|
||||||
|
|
@ -213,37 +213,23 @@ export const mockDufflepud = (context: BrowserContext, fixtures: DufflepudFixtur
|
||||||
return route.fallback()
|
return route.fallback()
|
||||||
})
|
})
|
||||||
|
|
||||||
// Silence as a wav, built rather than inlined so a spec can ask for a length and then assert the
|
// The rate the endpoint documents for raw pcm, at 16 bits a sample, which is what the app assumes
|
||||||
// duration the player reads off it. 16 bit mono pcm is the shortest header a browser will decode.
|
// when it writes a wav header for one.
|
||||||
const silence = (seconds: number) => {
|
const SPEECH_RATE = 24000
|
||||||
const rate = 8000
|
|
||||||
const bytes = seconds * rate * 2
|
|
||||||
const wav = Buffer.alloc(44 + bytes)
|
|
||||||
|
|
||||||
wav.write("RIFF", 0)
|
// Headerless silence. Nothing in it says how long it is, so the duration a spec reads off the
|
||||||
wav.writeUInt32LE(36 + bytes, 4)
|
// player is the one the app computed.
|
||||||
wav.write("WAVEfmt ", 8)
|
const silence = (seconds: number) => Buffer.alloc(seconds * SPEECH_RATE * 2)
|
||||||
wav.writeUInt32LE(16, 16)
|
|
||||||
wav.writeUInt16LE(1, 20)
|
|
||||||
wav.writeUInt16LE(1, 22)
|
|
||||||
wav.writeUInt32LE(rate, 24)
|
|
||||||
wav.writeUInt32LE(rate * 2, 28)
|
|
||||||
wav.writeUInt16LE(2, 32)
|
|
||||||
wav.writeUInt16LE(16, 34)
|
|
||||||
wav.write("data", 36)
|
|
||||||
wav.writeUInt32LE(bytes, 40)
|
|
||||||
|
|
||||||
return wav
|
// mp3 is the other format the real endpoint encodes, and the harness has no encoder for it, so
|
||||||
}
|
// asking for one here is a mistake rather than a case to serve.
|
||||||
|
const SPEECH_FORMAT = "pcm"
|
||||||
// The only two the real endpoint encodes; anything else comes back a 400 naming the pair.
|
|
||||||
const SPEECH_FORMATS = ["mp3", "pcm"]
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* OpenRouter's text to speech, answering every request with the same silence. The array it returns
|
* OpenRouter's text to speech, answering every request with the same silence. The array it returns
|
||||||
* collects what the app asked to have read, in the order it asked. A request for a format the real
|
* collects what the app asked to have read, in the order it asked. Answering a decodable wav to a
|
||||||
* endpoint does not encode is refused the way it refuses one, since a mock that plays anything back
|
* request for some other format would prove nothing about the container the app builds, so anything
|
||||||
* cannot tell whether the app asked for audio a browser can decode.
|
* but raw pcm is refused.
|
||||||
*/
|
*/
|
||||||
export const mockOpenRouterSpeech = async (context: BrowserContext, seconds = 3) => {
|
export const mockOpenRouterSpeech = async (context: BrowserContext, seconds = 3) => {
|
||||||
const spoken: string[] = []
|
const spoken: string[] = []
|
||||||
|
|
@ -251,15 +237,18 @@ export const mockOpenRouterSpeech = async (context: BrowserContext, seconds = 3)
|
||||||
await context.route(`${OPENROUTER_ORIGIN}/api/v1/audio/speech`, route => {
|
await context.route(`${OPENROUTER_ORIGIN}/api/v1/audio/speech`, route => {
|
||||||
const {input, response_format} = JSON.parse(route.request().postData() ?? "{}")
|
const {input, response_format} = JSON.parse(route.request().postData() ?? "{}")
|
||||||
|
|
||||||
if (SPEECH_FORMATS.includes(response_format)) {
|
if (response_format === SPEECH_FORMAT) {
|
||||||
spoken.push(input)
|
spoken.push(input)
|
||||||
|
|
||||||
return route.fulfill({contentType: "audio/wav", body: silence(seconds)})
|
return route.fulfill({
|
||||||
|
contentType: `audio/pcm;rate=${SPEECH_RATE};channels=1`,
|
||||||
|
body: silence(seconds),
|
||||||
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
return route.fulfill({
|
return route.fulfill({
|
||||||
status: 400,
|
status: 400,
|
||||||
json: {error: {message: `Invalid option: expected one of ${SPEECH_FORMATS.join("|")}`}},
|
json: {error: {message: `Invalid option: expected ${SPEECH_FORMAT}`}},
|
||||||
})
|
})
|
||||||
})
|
})
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -1,3 +1,4 @@
|
||||||
|
import {npubEncode} from "nostr-tools/nip19"
|
||||||
import {DAY, HOUR, MINUTE, WEEK, bech32ToHex} from "@welshman/lib"
|
import {DAY, HOUR, MINUTE, WEEK, bech32ToHex} from "@welshman/lib"
|
||||||
import {getLnUrl} from "@welshman/util"
|
import {getLnUrl} from "@welshman/util"
|
||||||
import {MessagingRelayList, Profile, RelayList, displayPubkey} from "@welshman/domain"
|
import {MessagingRelayList, Profile, RelayList, displayPubkey} from "@welshman/domain"
|
||||||
|
|
@ -871,8 +872,17 @@ test("US-119 have a message read out loud", async ({seed, as}) => {
|
||||||
space.room("general", {name: "General"})
|
space.room("general", {name: "General"})
|
||||||
space.join(user.alice, "general")
|
space.join(user.alice, "general")
|
||||||
space.join(user.bob, "general")
|
space.join(user.bob, "general")
|
||||||
|
space.profile(user.alice, {name: "Alice Anchor"})
|
||||||
space.profile(user.bob, {name: "Bob Barnacle"})
|
space.profile(user.bob, {name: "Bob Barnacle"})
|
||||||
space.message(user.bob, "general", "the dock is closed on sunday", at(2, HOUR))
|
|
||||||
|
const notice = space.message(user.bob, "general", "the dock is closed on sunday", at(2, HOUR))
|
||||||
|
|
||||||
|
space.reply(
|
||||||
|
user.bob,
|
||||||
|
notice,
|
||||||
|
`heads up nostr:${npubEncode(user.alice.pubkey)}, the notice is at https://harbor.example/dock?ref=1`,
|
||||||
|
at(1, HOUR),
|
||||||
|
)
|
||||||
|
|
||||||
seedChatter(space, user.alice)
|
seedChatter(space, user.alice)
|
||||||
})
|
})
|
||||||
|
|
@ -881,10 +891,11 @@ test("US-119 have a message read out loud", async ({seed, as}) => {
|
||||||
const alice = await as(users.alice, roomPath(url, "general"))
|
const alice = await as(users.alice, roomPath(url, "general"))
|
||||||
const spoken = await mockOpenRouterSpeech(alice.context())
|
const spoken = await mockOpenRouterSpeech(alice.context())
|
||||||
|
|
||||||
await expect(message(alice, "the dock is closed on sunday")).toBeVisible()
|
// The mention has to have resolved on screen before it can be expected in what was spoken.
|
||||||
|
await expect(message(alice, "heads up")).toContainText("@Alice Anchor")
|
||||||
|
|
||||||
// With no key saved, reading a message asks for one the way dictation does.
|
// With no key saved, reading a message asks for one the way dictation does.
|
||||||
await openMessageMenu(alice, "the dock is closed on sunday")
|
await openMessageMenu(alice, "heads up")
|
||||||
await alice.getByRole("button", {name: "Read Out Loud"}).click()
|
await alice.getByRole("button", {name: "Read Out Loud"}).click()
|
||||||
|
|
||||||
const enable = dialog(alice, "Enable read out loud?")
|
const enable = dialog(alice, "Enable read out loud?")
|
||||||
|
|
@ -894,16 +905,19 @@ test("US-119 have a message read out loud", async ({seed, as}) => {
|
||||||
|
|
||||||
await expect(alice.getByRole("alert")).toContainText("Read out loud is ready to use!")
|
await expect(alice.getByRole("alert")).toContainText("Read out loud is ready to use!")
|
||||||
|
|
||||||
await openMessageMenu(alice, "the dock is closed on sunday")
|
await openMessageMenu(alice, "heads up")
|
||||||
await alice.getByRole("button", {name: "Read Out Loud"}).click()
|
await alice.getByRole("button", {name: "Read Out Loud"}).click()
|
||||||
|
|
||||||
await expect(alice.getByText("a message from Bob Barnacle")).toBeVisible()
|
await expect(alice.getByText("a message from Bob Barnacle")).toBeVisible()
|
||||||
|
|
||||||
// Only what the message says is sent, so the nostr uri wrapping bob's mention never is.
|
// The quote, the mention and the url are each named rather than spelled out, since none of them
|
||||||
expect(spoken).toEqual(["the dock is closed on sunday"])
|
// is intelligible read a character at a time.
|
||||||
|
expect(spoken).toEqual([
|
||||||
|
"another message\n\nheads up Alice Anchor, the notice is at a link to harbor.example",
|
||||||
|
])
|
||||||
|
|
||||||
// The duration is the mock's, which is what proves the player is on audio it decoded rather
|
// The mock answers headerless pcm, so the duration is only right if the wav header the app put
|
||||||
// than on an element that failed to load.
|
// in front of it is, which is what makes the whole clip scrubbable.
|
||||||
await expect(alice.getByText("/ 0:03")).toBeVisible()
|
await expect(alice.getByText("/ 0:03")).toBeVisible()
|
||||||
|
|
||||||
// Chromium decides for itself whether the autoplay is allowed, so the control is read for
|
// Chromium decides for itself whether the autoplay is allowed, so the control is read for
|
||||||
|
|
|
||||||
|
|
@ -1,5 +1,6 @@
|
||||||
import {get, writable} from "svelte/store"
|
import {get, writable} from "svelte/store"
|
||||||
import {parse, renderAsText} from "@welshman/content"
|
import {ParsedType, isImage, parse} from "@welshman/content"
|
||||||
|
import type {Parsed} from "@welshman/content"
|
||||||
import type {Maybe} from "@welshman/lib"
|
import type {Maybe} from "@welshman/lib"
|
||||||
import type {TrustedEvent} from "@welshman/util"
|
import type {TrustedEvent} from "@welshman/util"
|
||||||
import {errorMessage} from "@lib/util"
|
import {errorMessage} from "@lib/util"
|
||||||
|
|
@ -15,7 +16,18 @@ const SPEECH_MODEL = "hexgrad/kokoro-82m"
|
||||||
|
|
||||||
const SPEECH_VOICE = "af_bella"
|
const SPEECH_VOICE = "af_bella"
|
||||||
|
|
||||||
const SPEECH_FORMAT = "mp3"
|
// The endpoint encodes mp3 and raw pcm, and its mp3 carries a xing header naming a fraction of the
|
||||||
|
// frames it holds, so a browser reads a fifth more audio than is there and the scrubber never
|
||||||
|
// reaches the end. Raw pcm claims no length at all, so the wav header below is the only one.
|
||||||
|
const SPEECH_FORMAT = "pcm"
|
||||||
|
|
||||||
|
const SPEECH_RATE = 24000
|
||||||
|
|
||||||
|
const SPEECH_CHANNELS = 1
|
||||||
|
|
||||||
|
const SPEECH_BIT_DEPTH = 16
|
||||||
|
|
||||||
|
const WAV_HEADER_LENGTH = 44
|
||||||
|
|
||||||
export type Speech = {
|
export type Speech = {
|
||||||
id: string
|
id: string
|
||||||
|
|
@ -25,6 +37,78 @@ export type Speech = {
|
||||||
|
|
||||||
export const speech = writable<Maybe<Speech>>(undefined)
|
export const speech = writable<Maybe<Speech>>(undefined)
|
||||||
|
|
||||||
|
const toWav = (pcm: ArrayBuffer) => {
|
||||||
|
const bytesPerFrame = (SPEECH_CHANNELS * SPEECH_BIT_DEPTH) / 8
|
||||||
|
const wav = new ArrayBuffer(WAV_HEADER_LENGTH + pcm.byteLength)
|
||||||
|
const view = new DataView(wav)
|
||||||
|
const ascii = (offset: number, value: string) => {
|
||||||
|
for (let i = 0; i < value.length; i++) {
|
||||||
|
view.setUint8(offset + i, value.charCodeAt(i))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
ascii(0, "RIFF")
|
||||||
|
view.setUint32(4, 36 + pcm.byteLength, true)
|
||||||
|
ascii(8, "WAVEfmt ")
|
||||||
|
view.setUint32(16, 16, true)
|
||||||
|
view.setUint16(20, 1, true)
|
||||||
|
view.setUint16(22, SPEECH_CHANNELS, true)
|
||||||
|
view.setUint32(24, SPEECH_RATE, true)
|
||||||
|
view.setUint32(28, SPEECH_RATE * bytesPerFrame, true)
|
||||||
|
view.setUint16(32, bytesPerFrame, true)
|
||||||
|
view.setUint16(34, SPEECH_BIT_DEPTH, true)
|
||||||
|
ascii(36, "data")
|
||||||
|
view.setUint32(40, pcm.byteLength, true)
|
||||||
|
|
||||||
|
new Uint8Array(wav, WAV_HEADER_LENGTH).set(new Uint8Array(pcm))
|
||||||
|
|
||||||
|
return new Blob([wav], {type: "audio/wav"})
|
||||||
|
}
|
||||||
|
|
||||||
|
// An entity or a url read a character at a time is unintelligible, so anything that is not prose
|
||||||
|
// is named rather than spelled out.
|
||||||
|
const speakOne = (parsed: Parsed): string => {
|
||||||
|
switch (parsed.type) {
|
||||||
|
case ParsedType.Address:
|
||||||
|
return "another post"
|
||||||
|
case ParsedType.Cashu:
|
||||||
|
return "a cashu token"
|
||||||
|
case ParsedType.Code:
|
||||||
|
return parsed.value
|
||||||
|
case ParsedType.Command:
|
||||||
|
return parsed.raw
|
||||||
|
case ParsedType.Ellipsis:
|
||||||
|
return "\u2026"
|
||||||
|
case ParsedType.Email:
|
||||||
|
return parsed.value
|
||||||
|
case ParsedType.Emoji:
|
||||||
|
return parsed.value.name
|
||||||
|
case ParsedType.Event:
|
||||||
|
return "another message"
|
||||||
|
case ParsedType.Invoice:
|
||||||
|
return "a lightning invoice"
|
||||||
|
case ParsedType.Link: {
|
||||||
|
const {host} = parsed.value.url
|
||||||
|
|
||||||
|
return isImage(parsed) ? "an image" : `a link to ${host}`
|
||||||
|
}
|
||||||
|
case ParsedType.LinkGrid:
|
||||||
|
return "some images"
|
||||||
|
case ParsedType.Newline:
|
||||||
|
return parsed.value
|
||||||
|
case ParsedType.Profile:
|
||||||
|
return profiles.get().display(parsed.value.pubkey).get()
|
||||||
|
case ParsedType.Room:
|
||||||
|
return parsed.value.room
|
||||||
|
case ParsedType.Text:
|
||||||
|
return parsed.value
|
||||||
|
case ParsedType.Topic:
|
||||||
|
return parsed.value.slice(1)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
const speakable = (event: TrustedEvent) => parse(event).map(speakOne).join("").trim()
|
||||||
|
|
||||||
export const synthesize = async (text: string) => {
|
export const synthesize = async (text: string) => {
|
||||||
const response = await fetch("https://openrouter.ai/api/v1/audio/speech", {
|
const response = await fetch("https://openrouter.ai/api/v1/audio/speech", {
|
||||||
method: "POST",
|
method: "POST",
|
||||||
|
|
@ -48,7 +132,7 @@ export const synthesize = async (text: string) => {
|
||||||
throw new Error(error?.message || `OpenRouter returned a ${response.status}.`)
|
throw new Error(error?.message || `OpenRouter returned a ${response.status}.`)
|
||||||
}
|
}
|
||||||
|
|
||||||
return response.blob()
|
return toWav(await response.arrayBuffer())
|
||||||
}
|
}
|
||||||
|
|
||||||
export const stopSpeech = () =>
|
export const stopSpeech = () =>
|
||||||
|
|
@ -87,7 +171,7 @@ const play = async (event: TrustedEvent, text: string) => {
|
||||||
}
|
}
|
||||||
|
|
||||||
export const readAloud = (event: TrustedEvent) => {
|
export const readAloud = (event: TrustedEvent) => {
|
||||||
const text = renderAsText(parse(event)).toString().trim()
|
const text = speakable(event)
|
||||||
|
|
||||||
if (getSetting("openrouter_key")) {
|
if (getSetting("openrouter_key")) {
|
||||||
if (get(isCallActive)) {
|
if (get(isCallActive)) {
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue