From e6f0b6ec8acbd9f924e8e4d57a5312523f2322b0 Mon Sep 17 00:00:00 2001 From: Coracle-Bot Date: Tue, 15 Sep 2026 17:59:36 +0000 Subject: [PATCH] Cover dictation in the e2e suite (#545) --- e2e/USER_STORIES.md | 13 +++++++ e2e/harness/index.ts | 9 ++++- e2e/harness/net/http.ts | 47 +++++++++++++++++++++++++ e2e/specs/composer.spec.ts | 70 ++++++++++++++++++++++++++++++++++++++ playwright.config.ts | 16 ++++++++- 5 files changed, 153 insertions(+), 2 deletions(-) diff --git a/e2e/USER_STORIES.md b/e2e/USER_STORIES.md index 2ad9a9a0..f30571b9 100644 --- a/e2e/USER_STORIES.md +++ b/e2e/USER_STORIES.md @@ -935,6 +935,19 @@ Acceptance: - The composer returns to composing a new message with its previous draft intact. +### US-125 — Dictate a message + +As alice, I want to speak a message instead of typing it, so that I can write +one without my hands. + +Acceptance: + +- The dictation button with no OpenRouter key saved asks for one, the same + prompt reading a message out loud uses. +- Recording and stopping puts the transcript in the composer, ready to send. +- Leaving the room while a transcription is still out does not lose it: the + transcript lands in the composer that is there when it comes back. + ## Rich content & media rendering ### US-060 — Reveal a flagged sensitive message diff --git a/e2e/harness/index.ts b/e2e/harness/index.ts index 35a41b74..c6cf93f8 100644 --- a/e2e/harness/index.ts +++ b/e2e/harness/index.ts @@ -82,8 +82,15 @@ export { mockDufflepud, mockLivekit, mockOpenRouterSpeech, + mockOpenRouterTranscription, +} from "./net/http" +export type { + DufflepudFixtures, + HostingFixtures, + HostingHandle, + HostingRecord, + Transcription, } from "./net/http" -export type {DufflepudFixtures, HostingFixtures, HostingHandle, HostingRecord} from "./net/http" export type {WebLnInfo} from "./app/webln" // Mirrors encodeRelay in src/app/relays.ts. Importing it reaches the app's module graph, and with diff --git a/e2e/harness/net/http.ts b/e2e/harness/net/http.ts index 6e7be392..1a52a484 100644 --- a/e2e/harness/net/http.ts +++ b/e2e/harness/net/http.ts @@ -1,6 +1,7 @@ import {createHash} from "node:crypto" import type {BrowserContext} from "@playwright/test" import {HOUR, int, now, omit} from "@welshman/lib" +import type {Maybe} from "@welshman/lib" import type {Handle} from "@welshman/util" import type {ZapperValues} from "@welshman/domain" import {tenantByUrl} from "../zooid/config" @@ -255,6 +256,52 @@ export const mockOpenRouterSpeech = async (context: BrowserContext, seconds = 3) return spoken } +export type Transcription = { + // The name each recording was uploaded under, in the order it was uploaded. OpenRouter picks a + // decoder off the extension, so this is how a spec reads what the recorder produced. + uploads: string[] + // Stops answering. A request that arrives while this is on is held open until `release`, which + // is what a spec needs to watch a transcription that is still out. + hold(): void + release(): void +} + +/** + * OpenRouter's speech to text, answering every recording with the same transcript. + */ +export const mockOpenRouterTranscription = async ( + context: BrowserContext, + text: string, +): Promise => { + const uploads: string[] = [] + + let held: Maybe> + let release: Maybe<() => void> + + await context.route(`${OPENROUTER_ORIGIN}/api/v1/audio/transcriptions`, async route => { + const body = route.request().postData() ?? "" + + uploads.push(body.match(/filename="([^"]+)"/)?.[1] ?? "") + + await held + + return route.fulfill({json: {text}}) + }) + + return { + uploads, + hold: () => { + held = new Promise(resolve => { + release = resolve + }) + }, + release: () => { + release?.() + held = undefined + }, + } +} + // Where an upload lands when nothing else is configured. getBlossomServer probes the space's own // origin first — blossom is off in every tenant's toml, so that probe is meant to fail — then the // user's kind-10063 list, and VITE_DEFAULT_BLOSSOM_SERVERS is what is left. diff --git a/e2e/specs/composer.spec.ts b/e2e/specs/composer.spec.ts index 822deaf8..26a1bb42 100644 --- a/e2e/specs/composer.spec.ts +++ b/e2e/specs/composer.spec.ts @@ -7,11 +7,14 @@ import { chooseFile, composer, composerEnabled, + dialog, expect, gifFile, makeTestUser, messageActions, mockBlossom, + mockOpenRouterTranscription, + roomLink, roomPath, sendButton, test, @@ -416,3 +419,70 @@ test("US-059 cancel a reply or an edit in progress", async ({seed, as}) => { await expect(timeline(page).getByText("my first message")).toHaveCount(1) await expect(timeline(page).getByText("half-written thought")).toHaveCount(0) }) + +// Dictation's two buttons are one button in two states, so the label is what says which. +const dictateButton = (page: Page) => page.getByRole("button", {name: "Start dictation"}) + +const stopButton = (page: Page) => page.getByRole("button", {name: "Stop recording"}) + +test("US-125 dictate a message", async ({seed, as}) => { + const scenario = await seed(({relay, user}) => { + const space = relay("space") + + space.room("general", {name: "General"}) + space.room("lounge", {name: "Lounge"}) + space.join(user.alice, "general") + space.join(user.alice, "lounge") + space.join(user.bob, "general") + }) + + const {url} = scenario.space("space") + const alice = await as(users.alice, roomPath(url, "general")) + const transcription = await mockOpenRouterTranscription(alice.context(), "the tide turns at six") + + // With no key saved, dictation asks for one, the same prompt reading a message out loud uses. + await dictateButton(alice).click() + + const enable = dialog(alice, "Enable voice input?") + + await enable.locator('input[name="flotilla-openrouter-key"]').fill("sk-or-test") + await enable.getByRole("button", {name: "Enable voice input"}).click() + + await expect(alice.getByRole("alert")).toContainText("Voice input is ready to use!") + + await dictateButton(alice).click() + await expect(stopButton(alice)).toBeVisible() + + await stopButton(alice).click() + + await expect(composer(alice)).toContainText("the tide turns at six") + + // OpenRouter picks its decoder off the extension, so what the recorder produced has to reach it + // under a name that names the format. + expect(transcription.uploads).toEqual([expect.stringMatching(/^dictation\.\w+$/)]) + + await composer(alice).press("Enter") + + await expect(timeline(alice)).toContainText("the tide turns at six") + + // A transcription outlives the composer that asked for it: the room it was recorded in can be + // left while the request is still out, and the transcript waits for whichever composer is next. + transcription.hold() + + await dictateButton(alice).click() + await expect(stopButton(alice)).toBeVisible() + + await stopButton(alice).click() + + // In-app rather than a fresh load: a dictation is held by the app rather than by the composer + // that started one, so reloading the page is losing it rather than leaving it. + await roomLink(alice, "Lounge").click() + + await expect(composer(alice)).toBeVisible() + + await roomLink(alice, "General").click() + + transcription.release() + + await expect(composer(alice)).toContainText("the tide turns at six") +}) diff --git a/playwright.config.ts b/playwright.config.ts index 66769b91..fd20bf6a 100644 --- a/playwright.config.ts +++ b/playwright.config.ts @@ -9,7 +9,17 @@ const deviceNames: Record = { webkit: "Desktop Safari", } -const device = devices[deviceNames[process.env.E2E_BROWSER ?? "chromium"]] +const browserName = process.env.E2E_BROWSER ?? "chromium" + +const device = devices[deviceNames[browserName]] + +// Dictation records from a microphone, and headless chromium has none. The fake device answers +// getUserMedia with a generated tone and the fake ui grants it without a prompt. Both are launch +// arguments rather than context options, so they belong to the whole run. +const launchOptions = + browserName === "chromium" + ? {args: ["--use-fake-device-for-media-stream", "--use-fake-ui-for-media-stream"]} + : {} // vite.config.ts's port, and what the harness lets past its block-all. Overridable so a run can // stand up its own dev server next to one that is already holding the default port. @@ -37,6 +47,10 @@ export default defineConfig({ use: { baseURL: `http://localhost:${port}`, trace: "on-first-retry", + launchOptions, + // Granted for every context: the fake device is the only one there is, so nothing is reachable + // through it that a spec has not asked for. + permissions: ["microphone"], }, // Boots the SvelteKit dev server before the suite and reuses one if already running locally. The // app resolves its VITE_ values against a key the harness injects per browser context, so any