From 10350dd622d925c65abfebbec61e40940416f849 Mon Sep 17 00:00:00 2001 From: Coracle-Bot Date: Wed, 16 Sep 2026 00:27:54 +0000 Subject: [PATCH] Ask whether to transcribe a recording or send it as a voice note --- e2e/USER_STORIES.md | 18 +++- e2e/specs/composer.spec.ts | 59 ++++++++++--- src/app/components/ChatCompose.svelte | 14 ++- src/app/components/DictationAction.svelte | 101 ++++++++++++++++++++++ src/app/components/DictationButton.svelte | 96 ++++++++++++++------ src/app/components/RoomCompose.svelte | 14 ++- src/app/dictation.ts | 43 +++++---- src/routes/settings/content/+page.svelte | 4 +- 8 files changed, 290 insertions(+), 59 deletions(-) create mode 100644 src/app/components/DictationAction.svelte diff --git a/e2e/USER_STORIES.md b/e2e/USER_STORIES.md index 7ae976a6..0b4f8575 100644 --- a/e2e/USER_STORIES.md +++ b/e2e/USER_STORIES.md @@ -944,10 +944,26 @@ Acceptance: - The dictation button with no OpenRouter key saved asks for one, the same prompt reading a message out loud uses. -- Recording and stopping puts the transcript in the composer, ready to send. +- Stopping a recording asks whether to transcribe it or send it as a voice + note. +- Choosing to transcribe puts the transcript in the composer, ready to send. - Leaving the room while a transcription is still out does not lose it: the transcript lands in the composer that is there when it comes back. +### US-126 — Send a voice note + +As alice, I want to send a recording as it is, so that the message carries my +voice rather than a transcript of it. + +Acceptance: + +- Choosing to send a voice note uploads the recording and attaches it to the + composer. +- Sending it publishes the message with the audio, which renders as a player in + the timeline. +- Discarding the recording instead leaves the composer empty and uploads + nothing. + ## Rich content & media rendering ### US-060 — Reveal a flagged sensitive message diff --git a/e2e/specs/composer.spec.ts b/e2e/specs/composer.spec.ts index 26a1bb42..a25751f7 100644 --- a/e2e/specs/composer.spec.ts +++ b/e2e/specs/composer.spec.ts @@ -425,6 +425,15 @@ const dictateButton = (page: Page) => page.getByRole("button", {name: "Start dic const stopButton = (page: Page) => page.getByRole("button", {name: "Stop recording"}) +// Every recording ends at the same question, so each spec starts from the answer it is about. +const record = async (page: Page) => { + await dictateButton(page).click() + await expect(stopButton(page)).toBeVisible() + await stopButton(page).click() + + return dialog(page, "Transcribe or send?") +} + test("US-125 dictate a message", async ({seed, as}) => { const scenario = await seed(({relay, user}) => { const space = relay("space") @@ -440,8 +449,11 @@ test("US-125 dictate a message", async ({seed, as}) => { const alice = await as(users.alice, roomPath(url, "general")) const transcription = await mockOpenRouterTranscription(alice.context(), "the tide turns at six") - // With no key saved, dictation asks for one, the same prompt reading a message out loud uses. - await dictateButton(alice).click() + // Recording asks for nothing. Asking for a transcript with no key saved asks for one, the same + // prompt reading a message out loud uses. + const action = await record(alice) + + await action.getByRole("button", {name: "Transcribe it"}).click() const enable = dialog(alice, "Enable voice input?") @@ -450,10 +462,8 @@ test("US-125 dictate a message", async ({seed, as}) => { await expect(alice.getByRole("alert")).toContainText("Voice input is ready to use!") - await dictateButton(alice).click() - await expect(stopButton(alice)).toBeVisible() - - await stopButton(alice).click() + // The recording is still waiting behind that prompt for the answer it asked for. + await action.getByRole("button", {name: "Transcribe it"}).click() await expect(composer(alice)).toContainText("the tide turns at six") @@ -469,10 +479,7 @@ test("US-125 dictate a message", async ({seed, as}) => { // left while the request is still out, and the transcript waits for whichever composer is next. transcription.hold() - await dictateButton(alice).click() - await expect(stopButton(alice)).toBeVisible() - - await stopButton(alice).click() + await (await record(alice)).getByRole("button", {name: "Transcribe it"}).click() // In-app rather than a fresh load: a dictation is held by the app rather than by the composer // that started one, so reloading the page is losing it rather than leaving it. @@ -486,3 +493,35 @@ test("US-125 dictate a message", async ({seed, as}) => { await expect(composer(alice)).toContainText("the tide turns at six") }) + +test("US-126 send a voice note", async ({seed, as}) => { + const scenario = await seed(({relay, user}) => { + const space = relay("space") + + space.room("general", {name: "General"}) + space.join(user.alice, "general") + }) + + const {url} = scenario.space("space") + const alice = await as(users.alice, roomPath(url, "general")) + + await mockBlossom(alice.context(), {server: DEFAULT_BLOSSOM_ORIGIN}) + + // Leaving the question unanswered throws the recording away, so nothing is uploaded and the + // composer is where it was. + await (await record(alice)).getByRole("button", {name: "Discard"}).click() + + await expect(dialog(alice, "Transcribe or send?")).toHaveCount(0) + await expect(composer(alice)).toHaveText("") + + // Sending the recording as it is needs no OpenRouter key, only somewhere to upload it. + await (await record(alice)).getByRole("button", {name: "Send a voice note"}).click() + + await expect(composer(alice)).toContainText(DEFAULT_BLOSSOM_ORIGIN) + + await composer(alice).press("Enter") + + // The imeta on the message says the upload is audio, which is what gives it a player rather + // than a link. + await expect(timeline(alice).locator(`audio[src^="${DEFAULT_BLOSSOM_ORIGIN}/"]`)).toBeVisible() +}) diff --git a/src/app/components/ChatCompose.svelte b/src/app/components/ChatCompose.svelte index 69d29586..07de8e3d 100644 --- a/src/app/components/ChatCompose.svelte +++ b/src/app/components/ChatCompose.svelte @@ -73,6 +73,14 @@ ed.chain().focus().insertContent(escapeHtml(text)).run() } + const attachVoiceNote = async (audio: File) => { + const ed = await editor + + ed.chain() + .addFile(audio, ed.state.selection.from + 1) + .run() + } + const submit = async () => { if ($uploading || disabled) return @@ -156,7 +164,11 @@ {#if dictating || ($empty && !disabled)} - + {:else} + + + + + + diff --git a/src/app/components/DictationButton.svelte b/src/app/components/DictationButton.svelte index 118682c9..382097a9 100644 --- a/src/app/components/DictationButton.svelte +++ b/src/app/components/DictationButton.svelte @@ -8,9 +8,8 @@ import Button from "@lib/components/Button.svelte" import Spinner from "@lib/components/Spinner.svelte" import {errorMessage} from "@lib/util" - import OpenRouterEnable from "@app/components/OpenRouterEnable.svelte" - import {clearDictation, getDictation, startDictation} from "@app/dictation" - import {getSetting} from "@app/settings" + import DictationAction from "@app/components/DictationAction.svelte" + import {clearDictation, getDictation, startDictation, transcribeDictation} from "@app/dictation" import {pushModal} from "@app/modal" import {pushToast} from "@app/toast" @@ -18,9 +17,10 @@ key: string dictating?: boolean onTranscript: (text: string) => MaybeAsync + onVoiceNote: (audio: File) => MaybeAsync } - let {key, dictating = $bindable(false), onTranscript}: Props = $props() + let {key, dictating = $bindable(false), onTranscript, onVoiceNote}: Props = $props() // Room noise idles just below this, so the pulse follows quiet speech too. const onLevel = (level: number) => { @@ -28,25 +28,18 @@ } const start = async () => { - if (getSetting("openrouter_key")) { - // Granting microphone access can sit on a permission prompt for a while, so hold the button - // until it resolves — a second click would open a stream nothing is left holding on to. - loading = true + // Granting microphone access can sit on a permission prompt for a while, so hold the button + // until it resolves — a second click would open a stream nothing is left holding on to. + loading = true - try { - await startDictation(key, onLevel) - recording = true - } catch (error) { - console.error(error) - pushToast({theme: "error", message: "Failed to access your microphone."}) - } finally { - loading = false - } - } else { - pushModal(OpenRouterEnable, { - feature: "Voice input", - subtitle: "Dictate your messages instead of typing them.", - }) + try { + await startDictation(key, onLevel) + recording = true + } catch (error) { + console.error(error) + pushToast({theme: "error", message: "Failed to access your microphone."}) + } finally { + loading = false } } @@ -55,14 +48,52 @@ recording = false loud = false + // Hold the button until the recording has been spent, rather than offering a second one + // behind the modal. + loading = true - deliver() + choose() + } + + const choose = () => + pushModal(DictationAction, { + onTranscribe: transcribe, + onVoiceNote: attach, + onDiscard: discard, + }) + + const transcribe = () => { + const dictation = getDictation(key) + + if (dictation) { + transcribeDictation(dictation) + deliver() + } + } + + const attach = async () => { + const dictation = getDictation(key) + + if (dictation) { + const audio = await dictation.audio + + clearDictation(key) + loading = false + + await onVoiceNote(audio) + } + } + + const discard = () => { + clearDictation(key) + + loading = false } const deliver = async () => { const dictation = getDictation(key) - if (dictation) { + if (dictation?.finished) { loading = true await dictation.finished @@ -92,7 +123,7 @@ let destroyed = false let recording = $state(false) // Starts out in flight when a dictation is already waiting to be picked up, so that the composer - // keeps rendering this button until its transcript has been handed over. + // keeps rendering this button until the recording has been spent. let loading = $state(Boolean(getDictation(key))) let loud = $state(false) @@ -108,8 +139,17 @@ dictating = recording || loading }) - // Pick up a dictation an earlier composer left running. - onMount(deliver) + // Pick up a dictation an earlier composer left running, either where it was already transcribing + // or at the choice the speaker never got to make. + onMount(() => { + const dictation = getDictation(key) + + if (dictation?.finished) { + deliver() + } else if (dictation) { + choose() + } + }) onDestroy(() => { destroyed = true @@ -124,7 +164,7 @@