Ask whether to transcribe a recording or send it as a voice note

This commit is contained in:
Coracle-Bot 2026-09-16 00:27:54 +00:00
parent e6f277eeca
commit 10350dd622
8 changed files with 290 additions and 59 deletions

View file

@ -944,10 +944,26 @@ Acceptance:
- The dictation button with no OpenRouter key saved asks for one, the same
prompt reading a message out loud uses.
- Recording and stopping puts the transcript in the composer, ready to send.
- Stopping a recording asks whether to transcribe it or send it as a voice
note.
- Choosing to transcribe puts the transcript in the composer, ready to send.
- Leaving the room while a transcription is still out does not lose it: the
transcript lands in the composer that is there when it comes back.
### US-126 — Send a voice note
As alice, I want to send a recording as it is, so that the message carries my
voice rather than a transcript of it.
Acceptance:
- Choosing to send a voice note uploads the recording and attaches it to the
composer.
- Sending it publishes the message with the audio, which renders as a player in
the timeline.
- Discarding the recording instead leaves the composer empty and uploads
nothing.
## Rich content & media rendering
### US-060 — Reveal a flagged sensitive message

View file

@ -425,6 +425,15 @@ const dictateButton = (page: Page) => page.getByRole("button", {name: "Start dic
const stopButton = (page: Page) => page.getByRole("button", {name: "Stop recording"})
// Every recording ends at the same question, so each spec starts from the answer it is about.
const record = async (page: Page) => {
await dictateButton(page).click()
await expect(stopButton(page)).toBeVisible()
await stopButton(page).click()
return dialog(page, "Transcribe or send?")
}
test("US-125 dictate a message", async ({seed, as}) => {
const scenario = await seed(({relay, user}) => {
const space = relay("space")
@ -440,8 +449,11 @@ test("US-125 dictate a message", async ({seed, as}) => {
const alice = await as(users.alice, roomPath(url, "general"))
const transcription = await mockOpenRouterTranscription(alice.context(), "the tide turns at six")
// With no key saved, dictation asks for one, the same prompt reading a message out loud uses.
await dictateButton(alice).click()
// Recording asks for nothing. Asking for a transcript with no key saved asks for one, the same
// prompt reading a message out loud uses.
const action = await record(alice)
await action.getByRole("button", {name: "Transcribe it"}).click()
const enable = dialog(alice, "Enable voice input?")
@ -450,10 +462,8 @@ test("US-125 dictate a message", async ({seed, as}) => {
await expect(alice.getByRole("alert")).toContainText("Voice input is ready to use!")
await dictateButton(alice).click()
await expect(stopButton(alice)).toBeVisible()
await stopButton(alice).click()
// The recording is still waiting behind that prompt for the answer it asked for.
await action.getByRole("button", {name: "Transcribe it"}).click()
await expect(composer(alice)).toContainText("the tide turns at six")
@ -469,10 +479,7 @@ test("US-125 dictate a message", async ({seed, as}) => {
// left while the request is still out, and the transcript waits for whichever composer is next.
transcription.hold()
await dictateButton(alice).click()
await expect(stopButton(alice)).toBeVisible()
await stopButton(alice).click()
await (await record(alice)).getByRole("button", {name: "Transcribe it"}).click()
// In-app rather than a fresh load: a dictation is held by the app rather than by the composer
// that started one, so reloading the page is losing it rather than leaving it.
@ -486,3 +493,35 @@ test("US-125 dictate a message", async ({seed, as}) => {
await expect(composer(alice)).toContainText("the tide turns at six")
})
test("US-126 send a voice note", async ({seed, as}) => {
const scenario = await seed(({relay, user}) => {
const space = relay("space")
space.room("general", {name: "General"})
space.join(user.alice, "general")
})
const {url} = scenario.space("space")
const alice = await as(users.alice, roomPath(url, "general"))
await mockBlossom(alice.context(), {server: DEFAULT_BLOSSOM_ORIGIN})
// Leaving the question unanswered throws the recording away, so nothing is uploaded and the
// composer is where it was.
await (await record(alice)).getByRole("button", {name: "Discard"}).click()
await expect(dialog(alice, "Transcribe or send?")).toHaveCount(0)
await expect(composer(alice)).toHaveText("")
// Sending the recording as it is needs no OpenRouter key, only somewhere to upload it.
await (await record(alice)).getByRole("button", {name: "Send a voice note"}).click()
await expect(composer(alice)).toContainText(DEFAULT_BLOSSOM_ORIGIN)
await composer(alice).press("Enter")
// The imeta on the message says the upload is audio, which is what gives it a player rather
// than a link.
await expect(timeline(alice).locator(`audio[src^="${DEFAULT_BLOSSOM_ORIGIN}/"]`)).toBeVisible()
})

View file

@ -73,6 +73,14 @@
ed.chain().focus().insertContent(escapeHtml(text)).run()
}
const attachVoiceNote = async (audio: File) => {
const ed = await editor
ed.chain()
.addFile(audio, ed.state.selection.from + 1)
.run()
}
const submit = async () => {
if ($uploading || disabled) return
@ -156,7 +164,11 @@
<EditorContent {autofocus} {editor} />
</div>
{#if dictating || ($empty && !disabled)}
<DictationButton key={dictationKey} bind:dictating onTranscript={insertTranscript} />
<DictationButton
key={dictationKey}
bind:dictating
onTranscript={insertTranscript}
onVoiceNote={attachVoiceNote} />
{:else}
<Button
data-tip="{window.navigator.platform.includes('Mac') ? 'cmd' : 'ctrl'}+enter to send"

View file

@ -0,0 +1,101 @@
<script lang="ts">
import {onDestroy} from "svelte"
import DocumentText from "@assets/icons/document-text.svg?dataurl"
import Soundwave from "@assets/icons/soundwave.svg?dataurl"
import TrashBin from "@assets/icons/trash-bin-minimalistic.svg?dataurl"
import Icon from "@lib/components/Icon.svelte"
import Button from "@lib/components/Button.svelte"
import CardButton from "@lib/components/CardButton.svelte"
import Modal from "@lib/components/Modal.svelte"
import ModalBody from "@lib/components/ModalBody.svelte"
import ModalHeader from "@lib/components/ModalHeader.svelte"
import ModalTitle from "@lib/components/ModalTitle.svelte"
import ModalSubtitle from "@lib/components/ModalSubtitle.svelte"
import ModalFooter from "@lib/components/ModalFooter.svelte"
import OpenRouterEnable from "@app/components/OpenRouterEnable.svelte"
import {getSetting} from "@app/settings"
import {popModal, pushModal} from "@app/modal"
type Props = {
onTranscribe: () => void
onVoiceNote: () => void
onDiscard: () => void
}
const {onTranscribe, onVoiceNote, onDiscard}: Props = $props()
const transcribe = () => {
if (getSetting("openrouter_key")) {
pending = false
popModal()
onTranscribe()
} else {
// Nested, so that the recording is still here to transcribe once a key has been saved.
pushModal(
OpenRouterEnable,
{
feature: "Voice input",
subtitle: "Dictate your messages instead of typing them.",
},
{nested: true},
)
}
}
const sendVoiceNote = () => {
pending = false
popModal()
onVoiceNote()
}
let pending = true
// Leaving without choosing throws the recording away, whether that was the discard button, the
// escape key or a navigation out of the conversation.
onDestroy(() => {
if (pending) {
onDiscard()
}
})
</script>
<Modal>
<ModalBody>
<ModalHeader>
<ModalTitle>Transcribe or send?</ModalTitle>
<ModalSubtitle>Your recording can become text, or go as it is.</ModalSubtitle>
</ModalHeader>
<Button onclick={transcribe}>
<CardButton primary>
{#snippet icon()}
<div><Icon icon={DocumentText} size={7} /></div>
{/snippet}
{#snippet title()}
<div>Transcribe it</div>
{/snippet}
{#snippet info()}
<div>Turn what you said into text you can edit before sending.</div>
{/snippet}
</CardButton>
</Button>
<Button onclick={sendVoiceNote}>
<CardButton>
{#snippet icon()}
<div><Icon icon={Soundwave} size={7} /></div>
{/snippet}
{#snippet title()}
<div>Send a voice note</div>
{/snippet}
{#snippet info()}
<div>Attach the recording itself to your message.</div>
{/snippet}
</CardButton>
</Button>
</ModalBody>
<ModalFooter>
<Button class="button button-link" onclick={popModal}>
<Icon icon={TrashBin} />
Discard
</Button>
</ModalFooter>
</Modal>

View file

@ -8,9 +8,8 @@
import Button from "@lib/components/Button.svelte"
import Spinner from "@lib/components/Spinner.svelte"
import {errorMessage} from "@lib/util"
import OpenRouterEnable from "@app/components/OpenRouterEnable.svelte"
import {clearDictation, getDictation, startDictation} from "@app/dictation"
import {getSetting} from "@app/settings"
import DictationAction from "@app/components/DictationAction.svelte"
import {clearDictation, getDictation, startDictation, transcribeDictation} from "@app/dictation"
import {pushModal} from "@app/modal"
import {pushToast} from "@app/toast"
@ -18,9 +17,10 @@
key: string
dictating?: boolean
onTranscript: (text: string) => MaybeAsync<void>
onVoiceNote: (audio: File) => MaybeAsync<void>
}
let {key, dictating = $bindable(false), onTranscript}: Props = $props()
let {key, dictating = $bindable(false), onTranscript, onVoiceNote}: Props = $props()
// Room noise idles just below this, so the pulse follows quiet speech too.
const onLevel = (level: number) => {
@ -28,25 +28,18 @@
}
const start = async () => {
if (getSetting("openrouter_key")) {
// Granting microphone access can sit on a permission prompt for a while, so hold the button
// until it resolves — a second click would open a stream nothing is left holding on to.
loading = true
// Granting microphone access can sit on a permission prompt for a while, so hold the button
// until it resolves — a second click would open a stream nothing is left holding on to.
loading = true
try {
await startDictation(key, onLevel)
recording = true
} catch (error) {
console.error(error)
pushToast({theme: "error", message: "Failed to access your microphone."})
} finally {
loading = false
}
} else {
pushModal(OpenRouterEnable, {
feature: "Voice input",
subtitle: "Dictate your messages instead of typing them.",
})
try {
await startDictation(key, onLevel)
recording = true
} catch (error) {
console.error(error)
pushToast({theme: "error", message: "Failed to access your microphone."})
} finally {
loading = false
}
}
@ -55,14 +48,52 @@
recording = false
loud = false
// Hold the button until the recording has been spent, rather than offering a second one
// behind the modal.
loading = true
deliver()
choose()
}
const choose = () =>
pushModal(DictationAction, {
onTranscribe: transcribe,
onVoiceNote: attach,
onDiscard: discard,
})
const transcribe = () => {
const dictation = getDictation(key)
if (dictation) {
transcribeDictation(dictation)
deliver()
}
}
const attach = async () => {
const dictation = getDictation(key)
if (dictation) {
const audio = await dictation.audio
clearDictation(key)
loading = false
await onVoiceNote(audio)
}
}
const discard = () => {
clearDictation(key)
loading = false
}
const deliver = async () => {
const dictation = getDictation(key)
if (dictation) {
if (dictation?.finished) {
loading = true
await dictation.finished
@ -92,7 +123,7 @@
let destroyed = false
let recording = $state(false)
// Starts out in flight when a dictation is already waiting to be picked up, so that the composer
// keeps rendering this button until its transcript has been handed over.
// keeps rendering this button until the recording has been spent.
let loading = $state(Boolean(getDictation(key)))
let loud = $state(false)
@ -108,8 +139,17 @@
dictating = recording || loading
})
// Pick up a dictation an earlier composer left running.
onMount(deliver)
// Pick up a dictation an earlier composer left running, either where it was already transcribing
// or at the choice the speaker never got to make.
onMount(() => {
const dictation = getDictation(key)
if (dictation?.finished) {
deliver()
} else if (dictation) {
choose()
}
})
onDestroy(() => {
destroyed = true
@ -124,7 +164,7 @@
<Button
class={buttonClass}
data-tip={recording ? "Stop and transcribe" : "Record a message"}
data-tip={recording ? "Stop recording" : "Record a message"}
aria-label={recording ? "Stop recording" : "Start dictation"}
disabled={loading}
onclick={toggle}>

View file

@ -76,6 +76,14 @@
ed.chain().focus().insertContent(escapeHtml(transcript)).run()
}
const attachVoiceNote = async (audio: File) => {
const ed = await editor
ed.chain()
.addFile(audio, ed.state.selection.from + 1)
.run()
}
// Argument tokens are whitespace-delimited, so separate one from whatever precedes it.
const insertCommandToken = async (token: string) => {
const ed = await editor
@ -183,7 +191,11 @@
<EditorContent {autofocus} {editor} />
</div>
{#if dictating || $empty}
<DictationButton {key} bind:dictating onTranscript={insertTranscript} />
<DictationButton
{key}
bind:dictating
onTranscript={insertTranscript}
onVoiceNote={attachVoiceNote} />
{:else}
<Button
data-tip="{window.navigator.platform.includes('Mac') ? 'cmd' : 'ctrl'}+enter to send"

View file

@ -12,12 +12,11 @@ const EXTENSIONS_BY_MIME_TYPE: Record<string, string> = {
"audio/wav": "wav",
}
const transcribe = async (audio: Blob) => {
const [mimeType] = audio.type.split(";")
const transcribe = async (audio: File) => {
const body = new FormData()
body.append("model", TRANSCRIPTION_MODEL)
body.append("file", audio, `dictation.${EXTENSIONS_BY_MIME_TYPE[mimeType] || "webm"}`)
body.append("file", audio, audio.name)
const response = await fetch("https://openrouter.ai/api/v1/audio/transcriptions", {
method: "POST",
@ -37,9 +36,12 @@ const transcribe = async (audio: Blob) => {
export type Dictation = {
recording: boolean
stop: () => void
// Resolves once the transcript or the error is on the dictation, so that awaiting it never takes
// the result out of the registry — whoever is still around when it lands reads it from there.
finished: Promise<void>
// Resolves once recording has stopped, with the audio named for the format the recorder chose.
audio: Promise<File>
// Set when the speaker asks for a transcript, and resolves once that or the error is on the
// dictation, so that awaiting it never takes the result out of the registry — whoever is still
// around when it lands reads it from there.
finished?: Promise<void>
transcript?: string
error?: unknown
}
@ -53,6 +55,17 @@ export const getDictation = (key: string) => dictations.get(key)
export const clearDictation = (key: string) => dictations.delete(key)
export const transcribeDictation = (dictation: Dictation) => {
dictation.finished = dictation.audio.then(transcribe).then(
transcript => {
dictation.transcript = transcript
},
error => {
dictation.error = error
},
)
}
// Reports how loud the microphone is once per frame, so the caller can show the speaker that we're
// hearing them. Levels follow the waveform's envelope — jumping to each peak, then decaying — since
// the raw root mean square drops to nothing in the gaps between words.
@ -93,7 +106,7 @@ export const startDictation = async (key: string, onLevel: (level: number) => vo
let frame = requestAnimationFrame(measure)
const audio = new Promise<Blob>(resolve => {
const audio = new Promise<File>(resolve => {
recorder.addEventListener("stop", () => {
cancelAnimationFrame(frame)
context.close()
@ -102,7 +115,12 @@ export const startDictation = async (key: string, onLevel: (level: number) => vo
track.stop()
}
resolve(new Blob(chunks, {type: recorder.mimeType}))
// The recorder names its codec alongside the container, which is more than the imeta on a
// voice note or the extension on a blossom url can carry, so keep the container alone.
const [type] = recorder.mimeType.split(";")
const extension = EXTENSIONS_BY_MIME_TYPE[type] || "webm"
resolve(new File(chunks, `dictation.${extension}`, {type}))
})
})
@ -112,14 +130,7 @@ export const startDictation = async (key: string, onLevel: (level: number) => vo
dictation.recording = false
recorder.stop()
},
finished: audio.then(transcribe).then(
transcript => {
dictation.transcript = transcript
},
error => {
dictation.error = error
},
),
audio,
}
dictations.set(key, dictation)

View file

@ -159,8 +159,8 @@
{#snippet info()}
<p>
Add an <Link external href="https://openrouter.ai/settings/keys" class="text-primary"
>OpenRouter API key</Link> to dictate messages using the microphone button in your composer,
and to have messages read out loud to you.
>OpenRouter API key</Link> to transcribe what you record with the microphone button in your
composer, and to have messages read out loud to you.
</p>
{/snippet}
</Field>