Cover dictation in the e2e suite (#545)

This commit is contained in:
Coracle-Bot 2026-09-15 17:59:36 +00:00 committed by hodlbod
parent d1040932a5
commit e6f0b6ec8a
5 changed files with 153 additions and 2 deletions

View file

@ -935,6 +935,19 @@ Acceptance:
- The composer returns to composing a new message with its previous draft
intact.
### US-125 — Dictate a message
As alice, I want to speak a message instead of typing it, so that I can write
one without my hands.
Acceptance:
- The dictation button with no OpenRouter key saved asks for one, the same
prompt reading a message out loud uses.
- Recording and stopping puts the transcript in the composer, ready to send.
- Leaving the room while a transcription is still out does not lose it: the
transcript lands in the composer that is there when it comes back.
## Rich content & media rendering
### US-060 — Reveal a flagged sensitive message

View file

@ -82,8 +82,15 @@ export {
mockDufflepud,
mockLivekit,
mockOpenRouterSpeech,
mockOpenRouterTranscription,
} from "./net/http"
export type {
DufflepudFixtures,
HostingFixtures,
HostingHandle,
HostingRecord,
Transcription,
} from "./net/http"
export type {DufflepudFixtures, HostingFixtures, HostingHandle, HostingRecord} from "./net/http"
export type {WebLnInfo} from "./app/webln"
// Mirrors encodeRelay in src/app/relays.ts. Importing it reaches the app's module graph, and with

View file

@ -1,6 +1,7 @@
import {createHash} from "node:crypto"
import type {BrowserContext} from "@playwright/test"
import {HOUR, int, now, omit} from "@welshman/lib"
import type {Maybe} from "@welshman/lib"
import type {Handle} from "@welshman/util"
import type {ZapperValues} from "@welshman/domain"
import {tenantByUrl} from "../zooid/config"
@ -255,6 +256,52 @@ export const mockOpenRouterSpeech = async (context: BrowserContext, seconds = 3)
return spoken
}
export type Transcription = {
// The name each recording was uploaded under, in the order it was uploaded. OpenRouter picks a
// decoder off the extension, so this is how a spec reads what the recorder produced.
uploads: string[]
// Stops answering. A request that arrives while this is on is held open until `release`, which
// is what a spec needs to watch a transcription that is still out.
hold(): void
release(): void
}
/**
* OpenRouter's speech to text, answering every recording with the same transcript.
*/
export const mockOpenRouterTranscription = async (
context: BrowserContext,
text: string,
): Promise<Transcription> => {
const uploads: string[] = []
let held: Maybe<Promise<void>>
let release: Maybe<() => void>
await context.route(`${OPENROUTER_ORIGIN}/api/v1/audio/transcriptions`, async route => {
const body = route.request().postData() ?? ""
uploads.push(body.match(/filename="([^"]+)"/)?.[1] ?? "")
await held
return route.fulfill({json: {text}})
})
return {
uploads,
hold: () => {
held = new Promise<void>(resolve => {
release = resolve
})
},
release: () => {
release?.()
held = undefined
},
}
}
// Where an upload lands when nothing else is configured. getBlossomServer probes the space's own
// origin first — blossom is off in every tenant's toml, so that probe is meant to fail — then the
// user's kind-10063 list, and VITE_DEFAULT_BLOSSOM_SERVERS is what is left.

View file

@ -7,11 +7,14 @@ import {
chooseFile,
composer,
composerEnabled,
dialog,
expect,
gifFile,
makeTestUser,
messageActions,
mockBlossom,
mockOpenRouterTranscription,
roomLink,
roomPath,
sendButton,
test,
@ -416,3 +419,70 @@ test("US-059 cancel a reply or an edit in progress", async ({seed, as}) => {
await expect(timeline(page).getByText("my first message")).toHaveCount(1)
await expect(timeline(page).getByText("half-written thought")).toHaveCount(0)
})
// Dictation's two buttons are one button in two states, so the label is what says which.
const dictateButton = (page: Page) => page.getByRole("button", {name: "Start dictation"})
const stopButton = (page: Page) => page.getByRole("button", {name: "Stop recording"})
test("US-125 dictate a message", async ({seed, as}) => {
const scenario = await seed(({relay, user}) => {
const space = relay("space")
space.room("general", {name: "General"})
space.room("lounge", {name: "Lounge"})
space.join(user.alice, "general")
space.join(user.alice, "lounge")
space.join(user.bob, "general")
})
const {url} = scenario.space("space")
const alice = await as(users.alice, roomPath(url, "general"))
const transcription = await mockOpenRouterTranscription(alice.context(), "the tide turns at six")
// With no key saved, dictation asks for one, the same prompt reading a message out loud uses.
await dictateButton(alice).click()
const enable = dialog(alice, "Enable voice input?")
await enable.locator('input[name="flotilla-openrouter-key"]').fill("sk-or-test")
await enable.getByRole("button", {name: "Enable voice input"}).click()
await expect(alice.getByRole("alert")).toContainText("Voice input is ready to use!")
await dictateButton(alice).click()
await expect(stopButton(alice)).toBeVisible()
await stopButton(alice).click()
await expect(composer(alice)).toContainText("the tide turns at six")
// OpenRouter picks its decoder off the extension, so what the recorder produced has to reach it
// under a name that names the format.
expect(transcription.uploads).toEqual([expect.stringMatching(/^dictation\.\w+$/)])
await composer(alice).press("Enter")
await expect(timeline(alice)).toContainText("the tide turns at six")
// A transcription outlives the composer that asked for it: the room it was recorded in can be
// left while the request is still out, and the transcript waits for whichever composer is next.
transcription.hold()
await dictateButton(alice).click()
await expect(stopButton(alice)).toBeVisible()
await stopButton(alice).click()
// In-app rather than a fresh load: a dictation is held by the app rather than by the composer
// that started one, so reloading the page is losing it rather than leaving it.
await roomLink(alice, "Lounge").click()
await expect(composer(alice)).toBeVisible()
await roomLink(alice, "General").click()
transcription.release()
await expect(composer(alice)).toContainText("the tide turns at six")
})

View file

@ -9,7 +9,17 @@ const deviceNames: Record<string, string> = {
webkit: "Desktop Safari",
}
const device = devices[deviceNames[process.env.E2E_BROWSER ?? "chromium"]]
const browserName = process.env.E2E_BROWSER ?? "chromium"
const device = devices[deviceNames[browserName]]
// Dictation records from a microphone, and headless chromium has none. The fake device answers
// getUserMedia with a generated tone and the fake ui grants it without a prompt. Both are launch
// arguments rather than context options, so they belong to the whole run.
const launchOptions =
browserName === "chromium"
? {args: ["--use-fake-device-for-media-stream", "--use-fake-ui-for-media-stream"]}
: {}
// vite.config.ts's port, and what the harness lets past its block-all. Overridable so a run can
// stand up its own dev server next to one that is already holding the default port.
@ -37,6 +47,10 @@ export default defineConfig({
use: {
baseURL: `http://localhost:${port}`,
trace: "on-first-retry",
launchOptions,
// Granted for every context: the fake device is the only one there is, so nothing is reachable
// through it that a spec has not asked for.
permissions: ["microphone"],
},
// Boots the SvelteKit dev server before the suite and reuses one if already running locally. The
// app resolves its VITE_ values against a key the harness injects per browser context, so any