Keep a dictation running when the composer that started it goes away (#500)
This commit is contained in:
parent
5520b851de
commit
e8a1f0bba7
6 changed files with 135 additions and 65 deletions
|
|
@ -353,6 +353,7 @@
|
|||
{onEscape}
|
||||
{onEditPrevious}
|
||||
{initialValues}
|
||||
dictationKey={draftKey.key}
|
||||
draftKey={eventToEdit ? undefined : draftKey}
|
||||
disabled={Boolean(missingRelayLists.length)} />
|
||||
{/key}
|
||||
|
|
|
|||
|
|
@ -13,12 +13,14 @@
|
|||
import DictationButton from "@app/components/DictationButton.svelte"
|
||||
import EditorContent from "@app/editor/EditorContent.svelte"
|
||||
import {makeEditor} from "@app/editor"
|
||||
import {getDictation} from "@app/dictation"
|
||||
import {type DraftKey, type Draft} from "@app/drafts"
|
||||
import {pushToast} from "@app/toast"
|
||||
import type {Share} from "@app/share"
|
||||
|
||||
type Props = {
|
||||
disabled?: boolean
|
||||
dictationKey: string
|
||||
draftKey?: DraftKey<Draft>
|
||||
onEscape?: () => void
|
||||
onEditPrevious?: () => void
|
||||
|
|
@ -29,6 +31,7 @@
|
|||
const {
|
||||
initialValues,
|
||||
disabled = false,
|
||||
dictationKey,
|
||||
draftKey,
|
||||
onEscape,
|
||||
onEditPrevious,
|
||||
|
|
@ -94,7 +97,7 @@
|
|||
let content = $state(
|
||||
initialValues?.type === "text" ? initialValues.value : (draftKey?.get()?.content ?? ""),
|
||||
)
|
||||
let recording = $state(false)
|
||||
let dictating = $state(Boolean(getDictation(dictationKey)))
|
||||
|
||||
const onChange = (json: object) => {
|
||||
content = json
|
||||
|
|
@ -152,8 +155,8 @@
|
|||
<div class={editorClass} aria-disabled={disabled}>
|
||||
<EditorContent {autofocus} {editor} />
|
||||
</div>
|
||||
{#if recording || ($empty && !disabled)}
|
||||
<DictationButton bind:recording onTranscript={insertTranscript} />
|
||||
{#if dictating || ($empty && !disabled)}
|
||||
<DictationButton key={dictationKey} bind:dictating onTranscript={insertTranscript} />
|
||||
{:else}
|
||||
<Button
|
||||
data-tip="{window.navigator.platform.includes('Mac') ? 'cmd' : 'ctrl'}+enter to send"
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
<script lang="ts">
|
||||
import {onDestroy} from "svelte"
|
||||
import {onDestroy, onMount} from "svelte"
|
||||
import cx from "classnames"
|
||||
import type {Maybe, MaybeAsync} from "@welshman/lib"
|
||||
import type {MaybeAsync} from "@welshman/lib"
|
||||
import Microphone from "@assets/icons/microphone.svg?dataurl"
|
||||
import Stop from "@assets/icons/record.svg?dataurl"
|
||||
import Icon from "@lib/components/Icon.svelte"
|
||||
|
|
@ -9,17 +9,18 @@
|
|||
import Spinner from "@lib/components/Spinner.svelte"
|
||||
import {errorMessage} from "@lib/util"
|
||||
import OpenRouterEnable from "@app/components/OpenRouterEnable.svelte"
|
||||
import {startDictation, transcribe} from "@app/dictation"
|
||||
import {clearDictation, getDictation, startDictation} from "@app/dictation"
|
||||
import {getSetting} from "@app/settings"
|
||||
import {pushModal} from "@app/modal"
|
||||
import {pushToast} from "@app/toast"
|
||||
|
||||
type Props = {
|
||||
recording?: boolean
|
||||
key: string
|
||||
dictating?: boolean
|
||||
onTranscript: (text: string) => MaybeAsync<void>
|
||||
}
|
||||
|
||||
let {recording = $bindable(false), onTranscript}: Props = $props()
|
||||
let {key, dictating = $bindable(false), onTranscript}: Props = $props()
|
||||
|
||||
// Room noise idles just below this, so the pulse follows quiet speech too.
|
||||
const onLevel = (level: number) => {
|
||||
|
|
@ -33,7 +34,7 @@
|
|||
loading = true
|
||||
|
||||
try {
|
||||
finish = await startDictation(onLevel)
|
||||
await startDictation(key, onLevel)
|
||||
recording = true
|
||||
} catch (error) {
|
||||
console.error(error)
|
||||
|
|
@ -49,34 +50,50 @@
|
|||
}
|
||||
}
|
||||
|
||||
const stop = async () => {
|
||||
if (finish) {
|
||||
const audio = finish()
|
||||
const stop = () => {
|
||||
getDictation(key)?.stop()
|
||||
|
||||
finish = undefined
|
||||
recording = false
|
||||
recording = false
|
||||
loud = false
|
||||
|
||||
deliver()
|
||||
}
|
||||
|
||||
const deliver = async () => {
|
||||
const dictation = getDictation(key)
|
||||
|
||||
if (dictation) {
|
||||
loading = true
|
||||
loud = false
|
||||
|
||||
try {
|
||||
const text = await transcribe(await audio)
|
||||
await dictation.finished
|
||||
|
||||
await onTranscript(text)
|
||||
} catch (error) {
|
||||
console.error(error)
|
||||
pushToast({theme: "error", message: `Failed to transcribe: ${errorMessage(error)}`})
|
||||
} finally {
|
||||
loading = false
|
||||
// A composer that has gone away has nowhere to put the transcript, so leave the dictation
|
||||
// where it is for whichever one mounts next.
|
||||
if (!destroyed) {
|
||||
if (dictation.error) {
|
||||
console.error(dictation.error)
|
||||
pushToast({
|
||||
theme: "error",
|
||||
message: `Failed to transcribe: ${errorMessage(dictation.error)}`,
|
||||
})
|
||||
} else {
|
||||
await onTranscript(dictation.transcript ?? "")
|
||||
}
|
||||
|
||||
clearDictation(key)
|
||||
}
|
||||
|
||||
loading = false
|
||||
}
|
||||
}
|
||||
|
||||
const toggle = () => (recording ? stop() : start())
|
||||
|
||||
// Held outside of state because only the recording flag drives the markup, and clearing this
|
||||
// before the recorder has finished flushing is what keeps a second stop from re-entering.
|
||||
let finish: Maybe<() => Promise<Blob>>
|
||||
let loading = $state(false)
|
||||
let destroyed = false
|
||||
let recording = $state(false)
|
||||
// Starts out in flight when a dictation is already waiting to be picked up, so that the composer
|
||||
// keeps rendering this button until its transcript has been handed over.
|
||||
let loading = $state(Boolean(getDictation(key)))
|
||||
let loud = $state(false)
|
||||
|
||||
const buttonClass = $derived(
|
||||
|
|
@ -87,8 +104,21 @@
|
|||
),
|
||||
)
|
||||
|
||||
$effect(() => {
|
||||
dictating = recording || loading
|
||||
})
|
||||
|
||||
// Pick up a dictation an earlier composer left running.
|
||||
onMount(deliver)
|
||||
|
||||
onDestroy(() => {
|
||||
finish?.()
|
||||
destroyed = true
|
||||
|
||||
const dictation = getDictation(key)
|
||||
|
||||
if (dictation?.recording) {
|
||||
dictation.stop()
|
||||
}
|
||||
})
|
||||
</script>
|
||||
|
||||
|
|
|
|||
|
|
@ -18,6 +18,7 @@
|
|||
import EditorContent from "@app/editor/EditorContent.svelte"
|
||||
import {makeEditor} from "@app/editor"
|
||||
import {app} from "@app/core"
|
||||
import {getDictation} from "@app/dictation"
|
||||
import {DraftKey, type Draft} from "@app/drafts"
|
||||
import type {Share} from "@app/share"
|
||||
import {onDestroy, onMount} from "svelte"
|
||||
|
|
@ -33,8 +34,9 @@
|
|||
|
||||
const {url, h, initialValues, onEscape, onEditPrevious, onSubmit}: Props = $props()
|
||||
|
||||
const draftKey =
|
||||
(url || h) && !initialValues ? new DraftKey<Draft>(`room:${url ?? ""}:${h ?? ""}`) : undefined
|
||||
const key = `room:${url ?? ""}:${h ?? ""}`
|
||||
|
||||
const draftKey = (url || h) && !initialValues ? new DraftKey<Draft>(key) : undefined
|
||||
|
||||
const autofocus = !isMobile
|
||||
|
||||
|
|
@ -106,7 +108,7 @@
|
|||
let content = $state(
|
||||
initialValues?.type === "text" ? initialValues.value : (draftKey?.get()?.content ?? ""),
|
||||
)
|
||||
let recording = $state(false)
|
||||
let dictating = $state(Boolean(getDictation(key)))
|
||||
|
||||
const onChange = (json: object) => {
|
||||
content = json
|
||||
|
|
@ -180,8 +182,8 @@
|
|||
<div class="chat-editor grow overflow-hidden">
|
||||
<EditorContent {autofocus} {editor} />
|
||||
</div>
|
||||
{#if recording || $empty}
|
||||
<DictationButton bind:recording onTranscript={insertTranscript} />
|
||||
{#if dictating || $empty}
|
||||
<DictationButton {key} bind:dictating onTranscript={insertTranscript} />
|
||||
{:else}
|
||||
<Button
|
||||
data-tip="{window.navigator.platform.includes('Mac') ? 'cmd' : 'ctrl'}+enter to send"
|
||||
|
|
|
|||
|
|
@ -12,10 +12,51 @@ const EXTENSIONS_BY_MIME_TYPE: Record<string, string> = {
|
|||
"audio/wav": "wav",
|
||||
}
|
||||
|
||||
const transcribe = async (audio: Blob) => {
|
||||
const [mimeType] = audio.type.split(";")
|
||||
const body = new FormData()
|
||||
|
||||
body.append("model", TRANSCRIPTION_MODEL)
|
||||
body.append("file", audio, `dictation.${EXTENSIONS_BY_MIME_TYPE[mimeType] || "webm"}`)
|
||||
|
||||
const response = await fetch("https://openrouter.ai/api/v1/audio/transcriptions", {
|
||||
method: "POST",
|
||||
headers: {Authorization: `Bearer ${getSetting("openrouter_key")}`},
|
||||
body,
|
||||
})
|
||||
|
||||
const {text, error}: {text?: string; error?: {message?: string}} = await response.json()
|
||||
|
||||
if (!response.ok) {
|
||||
throw new Error(error?.message || `OpenRouter returned a ${response.status}.`)
|
||||
}
|
||||
|
||||
return text?.trim() ?? ""
|
||||
}
|
||||
|
||||
export type Dictation = {
|
||||
recording: boolean
|
||||
stop: () => void
|
||||
// Resolves once the transcript or the error is on the dictation, so that awaiting it never takes
|
||||
// the result out of the registry — whoever is still around when it lands reads it from there.
|
||||
finished: Promise<void>
|
||||
transcript?: string
|
||||
error?: unknown
|
||||
}
|
||||
|
||||
// Dictations are held here rather than by the composer that started one, so navigating away from a
|
||||
// conversation transcribes in the background and leaves the transcript for the next composer the
|
||||
// way a draft is left.
|
||||
const dictations = new Map<string, Dictation>()
|
||||
|
||||
export const getDictation = (key: string) => dictations.get(key)
|
||||
|
||||
export const clearDictation = (key: string) => dictations.delete(key)
|
||||
|
||||
// Reports how loud the microphone is once per frame, so the caller can show the speaker that we're
|
||||
// hearing them. Levels follow the waveform's envelope — jumping to each peak, then decaying — since
|
||||
// the raw root mean square drops to nothing in the gaps between words.
|
||||
export const startDictation = async (onLevel: (level: number) => void) => {
|
||||
export const startDictation = async (key: string, onLevel: (level: number) => void) => {
|
||||
const stream = await navigator.mediaDevices.getUserMedia({audio: true})
|
||||
const recorder = new MediaRecorder(stream)
|
||||
const chunks: Blob[] = []
|
||||
|
|
@ -52,41 +93,34 @@ export const startDictation = async (onLevel: (level: number) => void) => {
|
|||
|
||||
let frame = requestAnimationFrame(measure)
|
||||
|
||||
return () =>
|
||||
new Promise<Blob>(resolve => {
|
||||
recorder.addEventListener("stop", () => {
|
||||
cancelAnimationFrame(frame)
|
||||
context.close()
|
||||
const audio = new Promise<Blob>(resolve => {
|
||||
recorder.addEventListener("stop", () => {
|
||||
cancelAnimationFrame(frame)
|
||||
context.close()
|
||||
|
||||
for (const track of stream.getTracks()) {
|
||||
track.stop()
|
||||
}
|
||||
for (const track of stream.getTracks()) {
|
||||
track.stop()
|
||||
}
|
||||
|
||||
resolve(new Blob(chunks, {type: recorder.mimeType}))
|
||||
})
|
||||
|
||||
recorder.stop()
|
||||
resolve(new Blob(chunks, {type: recorder.mimeType}))
|
||||
})
|
||||
}
|
||||
|
||||
export const transcribe = async (audio: Blob) => {
|
||||
const [mimeType] = audio.type.split(";")
|
||||
const body = new FormData()
|
||||
|
||||
body.append("model", TRANSCRIPTION_MODEL)
|
||||
body.append("file", audio, `dictation.${EXTENSIONS_BY_MIME_TYPE[mimeType] || "webm"}`)
|
||||
|
||||
const response = await fetch("https://openrouter.ai/api/v1/audio/transcriptions", {
|
||||
method: "POST",
|
||||
headers: {Authorization: `Bearer ${getSetting("openrouter_key")}`},
|
||||
body,
|
||||
})
|
||||
|
||||
const {text, error}: {text?: string; error?: {message?: string}} = await response.json()
|
||||
|
||||
if (!response.ok) {
|
||||
throw new Error(error?.message || `OpenRouter returned a ${response.status}.`)
|
||||
const dictation: Dictation = {
|
||||
recording: true,
|
||||
stop: () => {
|
||||
dictation.recording = false
|
||||
recorder.stop()
|
||||
},
|
||||
finished: audio.then(transcribe).then(
|
||||
transcript => {
|
||||
dictation.transcript = transcript
|
||||
},
|
||||
error => {
|
||||
dictation.error = error
|
||||
},
|
||||
),
|
||||
}
|
||||
|
||||
return text?.trim() ?? ""
|
||||
dictations.set(key, dictation)
|
||||
}
|
||||
|
|
|
|||
|
|
@ -5,7 +5,7 @@ export type Draft = {
|
|||
}
|
||||
|
||||
export class DraftKey<T> {
|
||||
constructor(private key: string) {}
|
||||
constructor(readonly key: string) {}
|
||||
|
||||
get(): T | undefined {
|
||||
return store.get(this.key) as T | undefined
|
||||
|
|
|
|||
Loading…
Reference in a new issue