flotilla/src/app/dictation.ts

100 lines
2.8 KiB
TypeScript
Raw Normal View History

import {watchAudioLevel} from "@lib/audio"
import {postOpenRouter} from "@app/openrouter"
2026-09-01 04:56:37 +00:00
const TRANSCRIPTION_MODEL = "openai/whisper-large-v3-turbo"
// OpenRouter chooses a decoder from the file extension, and MediaRecorder's format varies by engine.
2026-09-01 04:56:37 +00:00
const EXTENSIONS_BY_MIME_TYPE: Record<string, string> = {
"audio/webm": "webm",
"audio/ogg": "ogg",
"audio/mp4": "m4a",
"audio/mpeg": "mp3",
"audio/wav": "wav",
}
const transcribe = async (audio: File) => {
const body = new FormData()
body.append("model", TRANSCRIPTION_MODEL)
body.append("file", audio, audio.name)
const response = await postOpenRouter("audio/transcriptions", body)
const {text}: {text?: string} = await response.json()
return text?.trim() ?? ""
}
export type Dictation = {
recording: boolean
stop: () => void
// Resolves once recording has stopped, with the audio named for the format the recorder chose.
audio: Promise<File>
// Resolves once the transcript or the error is on the dictation, which stays in the registry.
finished?: Promise<void>
transcript?: string
error?: unknown
}
// Held here rather than by the composer, so navigating away transcribes in the background.
const dictations = new Map<string, Dictation>()
export const getDictation = (key: string) => dictations.get(key)
export const clearDictation = (key: string) => dictations.delete(key)
export const transcribeDictation = (dictation: Dictation) => {
dictation.finished = dictation.audio.then(transcribe).then(
transcript => {
dictation.transcript = transcript
},
error => {
dictation.error = error
},
)
}
// Levels follow the waveform's envelope, since raw root mean square drops to nothing between words.
export const startDictation = async (key: string, onLevel: (level: number) => void) => {
2026-09-01 04:56:37 +00:00
const stream = await navigator.mediaDevices.getUserMedia({audio: true})
const recorder = new MediaRecorder(stream)
const chunks: Blob[] = []
recorder.addEventListener("dataavailable", event => chunks.push(event.data))
recorder.start()
let level = 0
const stopMeter = watchAudioLevel(stream, rms => {
level = Math.max(rms, level * 0.92)
2026-09-01 04:56:37 +00:00
onLevel(level)
})
2026-09-01 04:56:37 +00:00
const audio = new Promise<File>(resolve => {
recorder.addEventListener("stop", () => {
stopMeter()
2026-09-01 04:56:37 +00:00
for (const track of stream.getTracks()) {
track.stop()
}
2026-09-01 04:56:37 +00:00
// The recorder names its codec alongside the container, which an imeta or a blossom url can't carry.
const [type] = recorder.mimeType.split(";")
const extension = EXTENSIONS_BY_MIME_TYPE[type] || "webm"
resolve(new File(chunks, `dictation.${extension}`, {type}))
2026-09-01 04:56:37 +00:00
})
})
const dictation: Dictation = {
recording: true,
stop: () => {
dictation.recording = false
recorder.stop()
},
audio,
2026-09-01 04:56:37 +00:00
}
dictations.set(key, dictation)
2026-09-01 04:56:37 +00:00
}