Ask whether to transcribe a recording or send it as a voice note
This commit is contained in:
parent
e6f277eeca
commit
10350dd622
8 changed files with 290 additions and 59 deletions
|
|
@ -944,10 +944,26 @@ Acceptance:
|
|||
|
||||
- The dictation button with no OpenRouter key saved asks for one, the same
|
||||
prompt reading a message out loud uses.
|
||||
- Recording and stopping puts the transcript in the composer, ready to send.
|
||||
- Stopping a recording asks whether to transcribe it or send it as a voice
|
||||
note.
|
||||
- Choosing to transcribe puts the transcript in the composer, ready to send.
|
||||
- Leaving the room while a transcription is still out does not lose it: the
|
||||
transcript lands in the composer that is there when it comes back.
|
||||
|
||||
### US-126 — Send a voice note
|
||||
|
||||
As alice, I want to send a recording as it is, so that the message carries my
|
||||
voice rather than a transcript of it.
|
||||
|
||||
Acceptance:
|
||||
|
||||
- Choosing to send a voice note uploads the recording and attaches it to the
|
||||
composer.
|
||||
- Sending it publishes the message with the audio, which renders as a player in
|
||||
the timeline.
|
||||
- Discarding the recording instead leaves the composer empty and uploads
|
||||
nothing.
|
||||
|
||||
## Rich content & media rendering
|
||||
|
||||
### US-060 — Reveal a flagged sensitive message
|
||||
|
|
|
|||
|
|
@ -425,6 +425,15 @@ const dictateButton = (page: Page) => page.getByRole("button", {name: "Start dic
|
|||
|
||||
const stopButton = (page: Page) => page.getByRole("button", {name: "Stop recording"})
|
||||
|
||||
// Every recording ends at the same question, so each spec starts from the answer it is about.
|
||||
const record = async (page: Page) => {
|
||||
await dictateButton(page).click()
|
||||
await expect(stopButton(page)).toBeVisible()
|
||||
await stopButton(page).click()
|
||||
|
||||
return dialog(page, "Transcribe or send?")
|
||||
}
|
||||
|
||||
test("US-125 dictate a message", async ({seed, as}) => {
|
||||
const scenario = await seed(({relay, user}) => {
|
||||
const space = relay("space")
|
||||
|
|
@ -440,8 +449,11 @@ test("US-125 dictate a message", async ({seed, as}) => {
|
|||
const alice = await as(users.alice, roomPath(url, "general"))
|
||||
const transcription = await mockOpenRouterTranscription(alice.context(), "the tide turns at six")
|
||||
|
||||
// With no key saved, dictation asks for one, the same prompt reading a message out loud uses.
|
||||
await dictateButton(alice).click()
|
||||
// Recording asks for nothing. Asking for a transcript with no key saved asks for one, the same
|
||||
// prompt reading a message out loud uses.
|
||||
const action = await record(alice)
|
||||
|
||||
await action.getByRole("button", {name: "Transcribe it"}).click()
|
||||
|
||||
const enable = dialog(alice, "Enable voice input?")
|
||||
|
||||
|
|
@ -450,10 +462,8 @@ test("US-125 dictate a message", async ({seed, as}) => {
|
|||
|
||||
await expect(alice.getByRole("alert")).toContainText("Voice input is ready to use!")
|
||||
|
||||
await dictateButton(alice).click()
|
||||
await expect(stopButton(alice)).toBeVisible()
|
||||
|
||||
await stopButton(alice).click()
|
||||
// The recording is still waiting behind that prompt for the answer it asked for.
|
||||
await action.getByRole("button", {name: "Transcribe it"}).click()
|
||||
|
||||
await expect(composer(alice)).toContainText("the tide turns at six")
|
||||
|
||||
|
|
@ -469,10 +479,7 @@ test("US-125 dictate a message", async ({seed, as}) => {
|
|||
// left while the request is still out, and the transcript waits for whichever composer is next.
|
||||
transcription.hold()
|
||||
|
||||
await dictateButton(alice).click()
|
||||
await expect(stopButton(alice)).toBeVisible()
|
||||
|
||||
await stopButton(alice).click()
|
||||
await (await record(alice)).getByRole("button", {name: "Transcribe it"}).click()
|
||||
|
||||
// In-app rather than a fresh load: a dictation is held by the app rather than by the composer
|
||||
// that started one, so reloading the page is losing it rather than leaving it.
|
||||
|
|
@ -486,3 +493,35 @@ test("US-125 dictate a message", async ({seed, as}) => {
|
|||
|
||||
await expect(composer(alice)).toContainText("the tide turns at six")
|
||||
})
|
||||
|
||||
test("US-126 send a voice note", async ({seed, as}) => {
|
||||
const scenario = await seed(({relay, user}) => {
|
||||
const space = relay("space")
|
||||
|
||||
space.room("general", {name: "General"})
|
||||
space.join(user.alice, "general")
|
||||
})
|
||||
|
||||
const {url} = scenario.space("space")
|
||||
const alice = await as(users.alice, roomPath(url, "general"))
|
||||
|
||||
await mockBlossom(alice.context(), {server: DEFAULT_BLOSSOM_ORIGIN})
|
||||
|
||||
// Leaving the question unanswered throws the recording away, so nothing is uploaded and the
|
||||
// composer is where it was.
|
||||
await (await record(alice)).getByRole("button", {name: "Discard"}).click()
|
||||
|
||||
await expect(dialog(alice, "Transcribe or send?")).toHaveCount(0)
|
||||
await expect(composer(alice)).toHaveText("")
|
||||
|
||||
// Sending the recording as it is needs no OpenRouter key, only somewhere to upload it.
|
||||
await (await record(alice)).getByRole("button", {name: "Send a voice note"}).click()
|
||||
|
||||
await expect(composer(alice)).toContainText(DEFAULT_BLOSSOM_ORIGIN)
|
||||
|
||||
await composer(alice).press("Enter")
|
||||
|
||||
// The imeta on the message says the upload is audio, which is what gives it a player rather
|
||||
// than a link.
|
||||
await expect(timeline(alice).locator(`audio[src^="${DEFAULT_BLOSSOM_ORIGIN}/"]`)).toBeVisible()
|
||||
})
|
||||
|
|
|
|||
|
|
@ -73,6 +73,14 @@
|
|||
ed.chain().focus().insertContent(escapeHtml(text)).run()
|
||||
}
|
||||
|
||||
const attachVoiceNote = async (audio: File) => {
|
||||
const ed = await editor
|
||||
|
||||
ed.chain()
|
||||
.addFile(audio, ed.state.selection.from + 1)
|
||||
.run()
|
||||
}
|
||||
|
||||
const submit = async () => {
|
||||
if ($uploading || disabled) return
|
||||
|
||||
|
|
@ -156,7 +164,11 @@
|
|||
<EditorContent {autofocus} {editor} />
|
||||
</div>
|
||||
{#if dictating || ($empty && !disabled)}
|
||||
<DictationButton key={dictationKey} bind:dictating onTranscript={insertTranscript} />
|
||||
<DictationButton
|
||||
key={dictationKey}
|
||||
bind:dictating
|
||||
onTranscript={insertTranscript}
|
||||
onVoiceNote={attachVoiceNote} />
|
||||
{:else}
|
||||
<Button
|
||||
data-tip="{window.navigator.platform.includes('Mac') ? 'cmd' : 'ctrl'}+enter to send"
|
||||
|
|
|
|||
101
src/app/components/DictationAction.svelte
Normal file
101
src/app/components/DictationAction.svelte
Normal file
|
|
@ -0,0 +1,101 @@
|
|||
<script lang="ts">
|
||||
import {onDestroy} from "svelte"
|
||||
import DocumentText from "@assets/icons/document-text.svg?dataurl"
|
||||
import Soundwave from "@assets/icons/soundwave.svg?dataurl"
|
||||
import TrashBin from "@assets/icons/trash-bin-minimalistic.svg?dataurl"
|
||||
import Icon from "@lib/components/Icon.svelte"
|
||||
import Button from "@lib/components/Button.svelte"
|
||||
import CardButton from "@lib/components/CardButton.svelte"
|
||||
import Modal from "@lib/components/Modal.svelte"
|
||||
import ModalBody from "@lib/components/ModalBody.svelte"
|
||||
import ModalHeader from "@lib/components/ModalHeader.svelte"
|
||||
import ModalTitle from "@lib/components/ModalTitle.svelte"
|
||||
import ModalSubtitle from "@lib/components/ModalSubtitle.svelte"
|
||||
import ModalFooter from "@lib/components/ModalFooter.svelte"
|
||||
import OpenRouterEnable from "@app/components/OpenRouterEnable.svelte"
|
||||
import {getSetting} from "@app/settings"
|
||||
import {popModal, pushModal} from "@app/modal"
|
||||
|
||||
type Props = {
|
||||
onTranscribe: () => void
|
||||
onVoiceNote: () => void
|
||||
onDiscard: () => void
|
||||
}
|
||||
|
||||
const {onTranscribe, onVoiceNote, onDiscard}: Props = $props()
|
||||
|
||||
const transcribe = () => {
|
||||
if (getSetting("openrouter_key")) {
|
||||
pending = false
|
||||
popModal()
|
||||
onTranscribe()
|
||||
} else {
|
||||
// Nested, so that the recording is still here to transcribe once a key has been saved.
|
||||
pushModal(
|
||||
OpenRouterEnable,
|
||||
{
|
||||
feature: "Voice input",
|
||||
subtitle: "Dictate your messages instead of typing them.",
|
||||
},
|
||||
{nested: true},
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
const sendVoiceNote = () => {
|
||||
pending = false
|
||||
popModal()
|
||||
onVoiceNote()
|
||||
}
|
||||
|
||||
let pending = true
|
||||
|
||||
// Leaving without choosing throws the recording away, whether that was the discard button, the
|
||||
// escape key or a navigation out of the conversation.
|
||||
onDestroy(() => {
|
||||
if (pending) {
|
||||
onDiscard()
|
||||
}
|
||||
})
|
||||
</script>
|
||||
|
||||
<Modal>
|
||||
<ModalBody>
|
||||
<ModalHeader>
|
||||
<ModalTitle>Transcribe or send?</ModalTitle>
|
||||
<ModalSubtitle>Your recording can become text, or go as it is.</ModalSubtitle>
|
||||
</ModalHeader>
|
||||
<Button onclick={transcribe}>
|
||||
<CardButton primary>
|
||||
{#snippet icon()}
|
||||
<div><Icon icon={DocumentText} size={7} /></div>
|
||||
{/snippet}
|
||||
{#snippet title()}
|
||||
<div>Transcribe it</div>
|
||||
{/snippet}
|
||||
{#snippet info()}
|
||||
<div>Turn what you said into text you can edit before sending.</div>
|
||||
{/snippet}
|
||||
</CardButton>
|
||||
</Button>
|
||||
<Button onclick={sendVoiceNote}>
|
||||
<CardButton>
|
||||
{#snippet icon()}
|
||||
<div><Icon icon={Soundwave} size={7} /></div>
|
||||
{/snippet}
|
||||
{#snippet title()}
|
||||
<div>Send a voice note</div>
|
||||
{/snippet}
|
||||
{#snippet info()}
|
||||
<div>Attach the recording itself to your message.</div>
|
||||
{/snippet}
|
||||
</CardButton>
|
||||
</Button>
|
||||
</ModalBody>
|
||||
<ModalFooter>
|
||||
<Button class="button button-link" onclick={popModal}>
|
||||
<Icon icon={TrashBin} />
|
||||
Discard
|
||||
</Button>
|
||||
</ModalFooter>
|
||||
</Modal>
|
||||
|
|
@ -8,9 +8,8 @@
|
|||
import Button from "@lib/components/Button.svelte"
|
||||
import Spinner from "@lib/components/Spinner.svelte"
|
||||
import {errorMessage} from "@lib/util"
|
||||
import OpenRouterEnable from "@app/components/OpenRouterEnable.svelte"
|
||||
import {clearDictation, getDictation, startDictation} from "@app/dictation"
|
||||
import {getSetting} from "@app/settings"
|
||||
import DictationAction from "@app/components/DictationAction.svelte"
|
||||
import {clearDictation, getDictation, startDictation, transcribeDictation} from "@app/dictation"
|
||||
import {pushModal} from "@app/modal"
|
||||
import {pushToast} from "@app/toast"
|
||||
|
||||
|
|
@ -18,9 +17,10 @@
|
|||
key: string
|
||||
dictating?: boolean
|
||||
onTranscript: (text: string) => MaybeAsync<void>
|
||||
onVoiceNote: (audio: File) => MaybeAsync<void>
|
||||
}
|
||||
|
||||
let {key, dictating = $bindable(false), onTranscript}: Props = $props()
|
||||
let {key, dictating = $bindable(false), onTranscript, onVoiceNote}: Props = $props()
|
||||
|
||||
// Room noise idles just below this, so the pulse follows quiet speech too.
|
||||
const onLevel = (level: number) => {
|
||||
|
|
@ -28,25 +28,18 @@
|
|||
}
|
||||
|
||||
const start = async () => {
|
||||
if (getSetting("openrouter_key")) {
|
||||
// Granting microphone access can sit on a permission prompt for a while, so hold the button
|
||||
// until it resolves — a second click would open a stream nothing is left holding on to.
|
||||
loading = true
|
||||
// Granting microphone access can sit on a permission prompt for a while, so hold the button
|
||||
// until it resolves — a second click would open a stream nothing is left holding on to.
|
||||
loading = true
|
||||
|
||||
try {
|
||||
await startDictation(key, onLevel)
|
||||
recording = true
|
||||
} catch (error) {
|
||||
console.error(error)
|
||||
pushToast({theme: "error", message: "Failed to access your microphone."})
|
||||
} finally {
|
||||
loading = false
|
||||
}
|
||||
} else {
|
||||
pushModal(OpenRouterEnable, {
|
||||
feature: "Voice input",
|
||||
subtitle: "Dictate your messages instead of typing them.",
|
||||
})
|
||||
try {
|
||||
await startDictation(key, onLevel)
|
||||
recording = true
|
||||
} catch (error) {
|
||||
console.error(error)
|
||||
pushToast({theme: "error", message: "Failed to access your microphone."})
|
||||
} finally {
|
||||
loading = false
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -55,14 +48,52 @@
|
|||
|
||||
recording = false
|
||||
loud = false
|
||||
// Hold the button until the recording has been spent, rather than offering a second one
|
||||
// behind the modal.
|
||||
loading = true
|
||||
|
||||
deliver()
|
||||
choose()
|
||||
}
|
||||
|
||||
const choose = () =>
|
||||
pushModal(DictationAction, {
|
||||
onTranscribe: transcribe,
|
||||
onVoiceNote: attach,
|
||||
onDiscard: discard,
|
||||
})
|
||||
|
||||
const transcribe = () => {
|
||||
const dictation = getDictation(key)
|
||||
|
||||
if (dictation) {
|
||||
transcribeDictation(dictation)
|
||||
deliver()
|
||||
}
|
||||
}
|
||||
|
||||
const attach = async () => {
|
||||
const dictation = getDictation(key)
|
||||
|
||||
if (dictation) {
|
||||
const audio = await dictation.audio
|
||||
|
||||
clearDictation(key)
|
||||
loading = false
|
||||
|
||||
await onVoiceNote(audio)
|
||||
}
|
||||
}
|
||||
|
||||
const discard = () => {
|
||||
clearDictation(key)
|
||||
|
||||
loading = false
|
||||
}
|
||||
|
||||
const deliver = async () => {
|
||||
const dictation = getDictation(key)
|
||||
|
||||
if (dictation) {
|
||||
if (dictation?.finished) {
|
||||
loading = true
|
||||
|
||||
await dictation.finished
|
||||
|
|
@ -92,7 +123,7 @@
|
|||
let destroyed = false
|
||||
let recording = $state(false)
|
||||
// Starts out in flight when a dictation is already waiting to be picked up, so that the composer
|
||||
// keeps rendering this button until its transcript has been handed over.
|
||||
// keeps rendering this button until the recording has been spent.
|
||||
let loading = $state(Boolean(getDictation(key)))
|
||||
let loud = $state(false)
|
||||
|
||||
|
|
@ -108,8 +139,17 @@
|
|||
dictating = recording || loading
|
||||
})
|
||||
|
||||
// Pick up a dictation an earlier composer left running.
|
||||
onMount(deliver)
|
||||
// Pick up a dictation an earlier composer left running, either where it was already transcribing
|
||||
// or at the choice the speaker never got to make.
|
||||
onMount(() => {
|
||||
const dictation = getDictation(key)
|
||||
|
||||
if (dictation?.finished) {
|
||||
deliver()
|
||||
} else if (dictation) {
|
||||
choose()
|
||||
}
|
||||
})
|
||||
|
||||
onDestroy(() => {
|
||||
destroyed = true
|
||||
|
|
@ -124,7 +164,7 @@
|
|||
|
||||
<Button
|
||||
class={buttonClass}
|
||||
data-tip={recording ? "Stop and transcribe" : "Record a message"}
|
||||
data-tip={recording ? "Stop recording" : "Record a message"}
|
||||
aria-label={recording ? "Stop recording" : "Start dictation"}
|
||||
disabled={loading}
|
||||
onclick={toggle}>
|
||||
|
|
|
|||
|
|
@ -76,6 +76,14 @@
|
|||
ed.chain().focus().insertContent(escapeHtml(transcript)).run()
|
||||
}
|
||||
|
||||
const attachVoiceNote = async (audio: File) => {
|
||||
const ed = await editor
|
||||
|
||||
ed.chain()
|
||||
.addFile(audio, ed.state.selection.from + 1)
|
||||
.run()
|
||||
}
|
||||
|
||||
// Argument tokens are whitespace-delimited, so separate one from whatever precedes it.
|
||||
const insertCommandToken = async (token: string) => {
|
||||
const ed = await editor
|
||||
|
|
@ -183,7 +191,11 @@
|
|||
<EditorContent {autofocus} {editor} />
|
||||
</div>
|
||||
{#if dictating || $empty}
|
||||
<DictationButton {key} bind:dictating onTranscript={insertTranscript} />
|
||||
<DictationButton
|
||||
{key}
|
||||
bind:dictating
|
||||
onTranscript={insertTranscript}
|
||||
onVoiceNote={attachVoiceNote} />
|
||||
{:else}
|
||||
<Button
|
||||
data-tip="{window.navigator.platform.includes('Mac') ? 'cmd' : 'ctrl'}+enter to send"
|
||||
|
|
|
|||
|
|
@ -12,12 +12,11 @@ const EXTENSIONS_BY_MIME_TYPE: Record<string, string> = {
|
|||
"audio/wav": "wav",
|
||||
}
|
||||
|
||||
const transcribe = async (audio: Blob) => {
|
||||
const [mimeType] = audio.type.split(";")
|
||||
const transcribe = async (audio: File) => {
|
||||
const body = new FormData()
|
||||
|
||||
body.append("model", TRANSCRIPTION_MODEL)
|
||||
body.append("file", audio, `dictation.${EXTENSIONS_BY_MIME_TYPE[mimeType] || "webm"}`)
|
||||
body.append("file", audio, audio.name)
|
||||
|
||||
const response = await fetch("https://openrouter.ai/api/v1/audio/transcriptions", {
|
||||
method: "POST",
|
||||
|
|
@ -37,9 +36,12 @@ const transcribe = async (audio: Blob) => {
|
|||
export type Dictation = {
|
||||
recording: boolean
|
||||
stop: () => void
|
||||
// Resolves once the transcript or the error is on the dictation, so that awaiting it never takes
|
||||
// the result out of the registry — whoever is still around when it lands reads it from there.
|
||||
finished: Promise<void>
|
||||
// Resolves once recording has stopped, with the audio named for the format the recorder chose.
|
||||
audio: Promise<File>
|
||||
// Set when the speaker asks for a transcript, and resolves once that or the error is on the
|
||||
// dictation, so that awaiting it never takes the result out of the registry — whoever is still
|
||||
// around when it lands reads it from there.
|
||||
finished?: Promise<void>
|
||||
transcript?: string
|
||||
error?: unknown
|
||||
}
|
||||
|
|
@ -53,6 +55,17 @@ export const getDictation = (key: string) => dictations.get(key)
|
|||
|
||||
export const clearDictation = (key: string) => dictations.delete(key)
|
||||
|
||||
export const transcribeDictation = (dictation: Dictation) => {
|
||||
dictation.finished = dictation.audio.then(transcribe).then(
|
||||
transcript => {
|
||||
dictation.transcript = transcript
|
||||
},
|
||||
error => {
|
||||
dictation.error = error
|
||||
},
|
||||
)
|
||||
}
|
||||
|
||||
// Reports how loud the microphone is once per frame, so the caller can show the speaker that we're
|
||||
// hearing them. Levels follow the waveform's envelope — jumping to each peak, then decaying — since
|
||||
// the raw root mean square drops to nothing in the gaps between words.
|
||||
|
|
@ -93,7 +106,7 @@ export const startDictation = async (key: string, onLevel: (level: number) => vo
|
|||
|
||||
let frame = requestAnimationFrame(measure)
|
||||
|
||||
const audio = new Promise<Blob>(resolve => {
|
||||
const audio = new Promise<File>(resolve => {
|
||||
recorder.addEventListener("stop", () => {
|
||||
cancelAnimationFrame(frame)
|
||||
context.close()
|
||||
|
|
@ -102,7 +115,12 @@ export const startDictation = async (key: string, onLevel: (level: number) => vo
|
|||
track.stop()
|
||||
}
|
||||
|
||||
resolve(new Blob(chunks, {type: recorder.mimeType}))
|
||||
// The recorder names its codec alongside the container, which is more than the imeta on a
|
||||
// voice note or the extension on a blossom url can carry, so keep the container alone.
|
||||
const [type] = recorder.mimeType.split(";")
|
||||
const extension = EXTENSIONS_BY_MIME_TYPE[type] || "webm"
|
||||
|
||||
resolve(new File(chunks, `dictation.${extension}`, {type}))
|
||||
})
|
||||
})
|
||||
|
||||
|
|
@ -112,14 +130,7 @@ export const startDictation = async (key: string, onLevel: (level: number) => vo
|
|||
dictation.recording = false
|
||||
recorder.stop()
|
||||
},
|
||||
finished: audio.then(transcribe).then(
|
||||
transcript => {
|
||||
dictation.transcript = transcript
|
||||
},
|
||||
error => {
|
||||
dictation.error = error
|
||||
},
|
||||
),
|
||||
audio,
|
||||
}
|
||||
|
||||
dictations.set(key, dictation)
|
||||
|
|
|
|||
|
|
@ -159,8 +159,8 @@
|
|||
{#snippet info()}
|
||||
<p>
|
||||
Add an <Link external href="https://openrouter.ai/settings/keys" class="text-primary"
|
||||
>OpenRouter API key</Link> to dictate messages using the microphone button in your composer,
|
||||
and to have messages read out loud to you.
|
||||
>OpenRouter API key</Link> to transcribe what you record with the microphone button in your
|
||||
composer, and to have messages read out loud to you.
|
||||
</p>
|
||||
{/snippet}
|
||||
</Field>
|
||||
|
|
|
|||
Loading…
Reference in a new issue