Wire mic through Whisper-compatible transcriptions on LLM_BASE_URL. Photos upload to MinIO and reach the vision model as base64 image_url parts.
130 lines
4.1 KiB
TypeScript
130 lines
4.1 KiB
TypeScript
"use client";
|
|
|
|
import { useCallback, useEffect, useRef, useState } from "react";
|
|
|
|
export type VoiceInputState = "idle" | "recording" | "transcribing";
|
|
|
|
function pickRecorderFormat(): { mimeType: string; extension: string } {
|
|
const candidates = [
|
|
{ mimeType: "audio/wav", extension: "wav" },
|
|
{ mimeType: "audio/webm;codecs=opus", extension: "webm" },
|
|
{ mimeType: "audio/webm", extension: "webm" },
|
|
{ mimeType: "audio/mp4", extension: "m4a" },
|
|
];
|
|
|
|
for (const candidate of candidates) {
|
|
if (typeof MediaRecorder !== "undefined" && MediaRecorder.isTypeSupported(candidate.mimeType)) {
|
|
return candidate;
|
|
}
|
|
}
|
|
|
|
return { mimeType: "", extension: "webm" };
|
|
}
|
|
|
|
export function useVoiceInput(options: {
|
|
onTranscript: (text: string) => void;
|
|
onError: (message: string) => void;
|
|
disabled?: boolean;
|
|
}) {
|
|
const [state, setState] = useState<VoiceInputState>("idle");
|
|
const recorderRef = useRef<MediaRecorder | null>(null);
|
|
const chunksRef = useRef<Blob[]>([]);
|
|
const streamRef = useRef<MediaStream | null>(null);
|
|
const formatRef = useRef(pickRecorderFormat());
|
|
const onTranscriptRef = useRef(options.onTranscript);
|
|
const onErrorRef = useRef(options.onError);
|
|
|
|
useEffect(() => {
|
|
onTranscriptRef.current = options.onTranscript;
|
|
onErrorRef.current = options.onError;
|
|
}, [options.onTranscript, options.onError]);
|
|
|
|
const stopStream = useCallback(() => {
|
|
for (const track of streamRef.current?.getTracks() ?? []) {
|
|
track.stop();
|
|
}
|
|
streamRef.current = null;
|
|
}, []);
|
|
|
|
const stopRecording = useCallback(async () => {
|
|
const recorder = recorderRef.current;
|
|
if (!recorder || recorder.state === "inactive") return;
|
|
|
|
await new Promise<void>((resolve) => {
|
|
recorder.addEventListener("stop", () => resolve(), { once: true });
|
|
recorder.stop();
|
|
});
|
|
|
|
recorderRef.current = null;
|
|
stopStream();
|
|
|
|
const blob = new Blob(chunksRef.current, {
|
|
type: formatRef.current.mimeType || chunksRef.current[0]?.type || "audio/webm",
|
|
});
|
|
chunksRef.current = [];
|
|
|
|
if (blob.size === 0) {
|
|
setState("idle");
|
|
onErrorRef.current("No audio captured");
|
|
return;
|
|
}
|
|
|
|
setState("transcribing");
|
|
try {
|
|
const formData = new FormData();
|
|
formData.append("file", blob, `recording.${formatRef.current.extension}`);
|
|
const response = await fetch("/api/agent/transcribe", { method: "POST", body: formData });
|
|
if (!response.ok) {
|
|
const payload = (await response.json().catch(() => null)) as { error?: string } | null;
|
|
throw new Error(payload?.error ?? "Transcription failed");
|
|
}
|
|
const payload = (await response.json()) as { text: string };
|
|
onTranscriptRef.current(payload.text);
|
|
} catch (err) {
|
|
onErrorRef.current(err instanceof Error ? err.message : "Transcription failed");
|
|
} finally {
|
|
setState("idle");
|
|
}
|
|
}, [stopStream]);
|
|
|
|
const startRecording = useCallback(async () => {
|
|
if (options.disabled || state !== "idle") return;
|
|
if (typeof navigator === "undefined" || !navigator.mediaDevices?.getUserMedia) {
|
|
onErrorRef.current("Microphone not available in this browser");
|
|
return;
|
|
}
|
|
|
|
formatRef.current = pickRecorderFormat();
|
|
|
|
try {
|
|
const stream = await navigator.mediaDevices.getUserMedia({ audio: true });
|
|
streamRef.current = stream;
|
|
const recorder = formatRef.current.mimeType
|
|
? new MediaRecorder(stream, { mimeType: formatRef.current.mimeType })
|
|
: new MediaRecorder(stream);
|
|
chunksRef.current = [];
|
|
recorder.addEventListener("dataavailable", (event) => {
|
|
if (event.data.size > 0) chunksRef.current.push(event.data);
|
|
});
|
|
recorder.start();
|
|
recorderRef.current = recorder;
|
|
setState("recording");
|
|
} catch {
|
|
stopStream();
|
|
onErrorRef.current("Microphone permission denied");
|
|
}
|
|
}, [options.disabled, state, stopStream]);
|
|
|
|
const toggleRecording = useCallback(async () => {
|
|
if (state === "recording") {
|
|
await stopRecording();
|
|
return;
|
|
}
|
|
if (state === "idle") {
|
|
await startRecording();
|
|
}
|
|
}, [startRecording, state, stopRecording]);
|
|
|
|
return { state, toggleRecording };
|
|
}
|