feat(agent): add voice input and photo attachments to assistant

Wire mic through Whisper-compatible transcriptions on LLM_BASE_URL.

Photos upload to MinIO and reach the vision model as base64 image_url parts.
This commit is contained in:
ginnoir
2026-07-05 02:01:48 -05:00
parent 876a283d47
commit c8db5475d3
20 changed files with 636 additions and 65 deletions
+20 -6
View File
@@ -1,9 +1,12 @@
import type { ClientChatMessage } from "./messages";
export type AssistantChatMessage = {
role: "user" | "assistant";
content: string;
imageUrl?: string;
};
const STORAGE_VERSION = "v1";
const STORAGE_VERSION = "v2";
const MAX_MESSAGES = 40;
function storageKey(userId: string) {
@@ -13,11 +16,22 @@ function storageKey(userId: string) {
function isValidMessage(value: unknown): value is AssistantChatMessage {
if (!value || typeof value !== "object") return false;
const row = value as Record<string, unknown>;
return (
(row.role === "user" || row.role === "assistant") &&
typeof row.content === "string" &&
row.content.trim().length > 0
);
if (row.role !== "user" && row.role !== "assistant") return false;
if (typeof row.content !== "string" || row.content.trim().length === 0) return false;
if (row.imageUrl !== undefined && typeof row.imageUrl !== "string") return false;
return true;
}
export function toClientChatMessage(message: AssistantChatMessage): ClientChatMessage {
if (!message.imageUrl) {
return { role: message.role, content: message.content };
}
return {
role: message.role,
content: message.content,
attachments: [{ type: "image", url: message.imageUrl }],
};
}
export function loadAssistantChat(userId: string): AssistantChatMessage[] {
+155 -14
View File
@@ -1,7 +1,7 @@
"use client";
import { useEffect, useRef, useState } from "react";
import { Loader2, Send } from "lucide-react";
import { ImagePlus, Loader2, Mic, Send, Square } from "lucide-react";
import { Button } from "@/components/ui/button";
import { Input } from "@/components/ui/input";
import { consumeAgentChatStream } from "../assistant-chat-stream";
@@ -9,8 +9,10 @@ import {
clearAssistantChat,
loadAssistantChat,
saveAssistantChat,
toClientChatMessage,
type AssistantChatMessage,
} from "../assistant-chat-storage";
import { useVoiceInput } from "./use-voice-input";
type Props = {
configured: boolean;
@@ -18,14 +20,42 @@ type Props = {
assistantName: string;
};
type PendingImage = {
url: string;
};
async function uploadAssistantImage(file: File): Promise<string> {
const formData = new FormData();
formData.append("file", file);
const response = await fetch("/api/uploads?scope=assistant", { method: "POST", body: formData });
if (!response.ok) {
const payload = (await response.json().catch(() => null)) as { error?: string } | null;
throw new Error(payload?.error ?? "Image upload failed");
}
const payload = (await response.json()) as { url: string };
return payload.url;
}
export function AssistantPanel({ configured, userId, assistantName }: Props) {
const [messages, setMessages] = useState<AssistantChatMessage[]>(() => loadAssistantChat(userId));
const [input, setInput] = useState("");
const [pendingImage, setPendingImage] = useState<PendingImage | null>(null);
const [uploadingImage, setUploadingImage] = useState(false);
const [error, setError] = useState<string | null>(null);
const [isPending, setIsPending] = useState(false);
const [activityLabel, setActivityLabel] = useState<string | null>(null);
const listRef = useRef<HTMLDivElement>(null);
const abortRef = useRef<AbortController | null>(null);
const imageInputRef = useRef<HTMLInputElement>(null);
const { state: voiceState, toggleRecording } = useVoiceInput({
disabled: isPending || uploadingImage,
onTranscript: (text) => {
setInput((current) => (current.trim() ? `${current.trim()} ${text}` : text));
setError(null);
},
onError: (message) => setError(message),
});
useEffect(() => {
saveAssistantChat(userId, messages);
@@ -47,22 +77,50 @@ export function AssistantPanel({ configured, userId, assistantName }: Props) {
function clearChat() {
abortRef.current?.abort();
setMessages([]);
setInput("");
setPendingImage(null);
setError(null);
setActivityLabel(null);
setIsPending(false);
clearAssistantChat(userId);
}
async function handleImageSelect(event: React.ChangeEvent<HTMLInputElement>) {
const file = event.target.files?.[0];
event.target.value = "";
if (!file || isPending || uploadingImage) return;
setUploadingImage(true);
setError(null);
try {
const url = await uploadAssistantImage(file);
setPendingImage({ url });
} catch (err) {
setError(err instanceof Error ? err.message : "Image upload failed");
} finally {
setUploadingImage(false);
}
}
async function sendMessage() {
const text = input.trim();
if (!text || isPending) return;
const hasImage = pendingImage !== null;
if ((!text && !hasImage) || isPending || voiceState !== "idle") return;
const nextMessages: AssistantChatMessage[] = [...messages, { role: "user", content: text }];
const content = text || "Help me with this image.";
const userMessage: AssistantChatMessage = {
role: "user",
content,
...(pendingImage ? { imageUrl: pendingImage.url } : {}),
};
const nextMessages: AssistantChatMessage[] = [...messages, userMessage];
setInput("");
setPendingImage(null);
setError(null);
setMessages(nextMessages);
setIsPending(true);
setActivityLabel("Understanding your request…");
setActivityLabel(hasImage ? "Reading your photo…" : "Understanding your request…");
scrollToBottom();
abortRef.current?.abort();
@@ -73,7 +131,10 @@ export function AssistantPanel({ configured, userId, assistantName }: Props) {
const response = await fetch("/api/agent/chat", {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({ messages: nextMessages, stream: true }),
body: JSON.stringify({
messages: nextMessages.map(toClientChatMessage),
stream: true,
}),
signal: controller.signal,
});
@@ -88,7 +149,10 @@ export function AssistantPanel({ configured, userId, assistantName }: Props) {
throw new Error("Assistant returned an empty response");
}
setMessages((current) => [...current, result.message]);
setMessages((current) => [
...current,
{ role: "assistant", content: result.message.content },
]);
scrollToBottom();
} catch (err) {
if (err instanceof Error && err.name === "AbortError") return;
@@ -101,13 +165,19 @@ export function AssistantPanel({ configured, userId, assistantName }: Props) {
}
const showEmptyState = messages.length === 0 && !isPending;
const inputDisabled = isPending || voiceState === "transcribing" || uploadingImage;
const canSend =
!isPending &&
voiceState === "idle" &&
!uploadingImage &&
(input.trim().length > 0 || pendingImage !== null);
return (
<div className="flex min-h-0 flex-1 flex-col gap-3">
<div className="flex items-start justify-between gap-3">
<p className="muted min-w-0 text-[12px] leading-relaxed">
{configured
? "Ask me to update lists, calendar, notes, or journal."
? "Type, talk, or send a photo — I can update lists, calendar, notes, and more."
: "Mock provider active — set LLM_BASE_URL for your homelab model."}
</p>
{messages.length > 0 ? (
@@ -129,8 +199,8 @@ export function AssistantPanel({ configured, userId, assistantName }: Props) {
>
{showEmptyState ? (
<p className="muted text-[12px]">
Try &quot;add milk to the shopping list&quot; or &quot;what&apos;s on the calendar this
week?&quot;
Try &quot;add milk to the shopping list&quot;, tap the mic, or attach a photo of an
appointment card.
</p>
) : (
<div className="grid gap-2">
@@ -147,6 +217,14 @@ export function AssistantPanel({ configured, userId, assistantName }: Props) {
<div className="eyebrow mb-0.5 text-[10px]">
{message.role === "user" ? "You" : assistantName}
</div>
{message.imageUrl ? (
// User-uploaded assistant attachment preview
<img
src={message.imageUrl}
alt=""
className="mb-2 max-h-40 w-full rounded-md object-contain"
/>
) : null}
{message.content}
</div>
))}
@@ -169,24 +247,87 @@ export function AssistantPanel({ configured, userId, assistantName }: Props) {
)}
</div>
{pendingImage ? (
<div className="flex items-center gap-2 rounded-[var(--r-md)] border-[0.5px] bg-[var(--shade)] p-2">
<img
src={pendingImage.url}
alt=""
className="max-h-16 max-w-[40%] rounded-md object-contain"
/>
<div className="min-w-0 flex-1 text-[11px] text-muted-foreground">Photo attached</div>
<button
type="button"
className="text-[11px] text-muted-foreground hover:text-foreground"
onClick={() => setPendingImage(null)}
disabled={inputDisabled}
>
Remove
</button>
</div>
) : null}
{error ? <p className="text-[12px] text-destructive">{error}</p> : null}
<form
className="flex gap-2"
onSubmit={(event) => {
event.preventDefault();
sendMessage();
void sendMessage();
}}
>
<input
ref={imageInputRef}
type="file"
accept="image/*"
className="hidden"
onChange={(event) => void handleImageSelect(event)}
/>
<Button
type="button"
size="sm"
variant="outline"
disabled={inputDisabled}
aria-label="Attach photo"
onClick={() => imageInputRef.current?.click()}
>
{uploadingImage ? (
<Loader2 className="size-4 animate-spin" />
) : (
<ImagePlus className="size-4" />
)}
</Button>
<Button
type="button"
size="sm"
variant={voiceState === "recording" ? "destructive" : "outline"}
disabled={inputDisabled}
aria-label={voiceState === "recording" ? "Stop recording" : "Record voice message"}
aria-pressed={voiceState === "recording"}
onClick={() => void toggleRecording()}
>
{voiceState === "transcribing" ? (
<Loader2 className="size-4 animate-spin" />
) : voiceState === "recording" ? (
<Square className="size-4" />
) : (
<Mic className="size-4" />
)}
</Button>
<Input
value={input}
onChange={(event) => setInput(event.target.value)}
placeholder={`Ask ${assistantName}`}
disabled={isPending}
placeholder={
voiceState === "recording"
? "Listening… tap mic to stop"
: voiceState === "transcribing"
? "Transcribing…"
: `Ask ${assistantName}`
}
disabled={inputDisabled}
aria-label={`Message for ${assistantName}`}
className="h-9"
className="h-9 min-w-0 flex-1"
/>
<Button type="submit" size="sm" disabled={isPending || !input.trim()} aria-label="Send">
<Button type="submit" size="sm" disabled={!canSend} aria-label="Send">
{isPending ? <Loader2 className="size-4 animate-spin" /> : <Send className="size-4" />}
</Button>
</form>
@@ -0,0 +1,129 @@
"use client";
import { useCallback, useEffect, useRef, useState } from "react";
export type VoiceInputState = "idle" | "recording" | "transcribing";
function pickRecorderFormat(): { mimeType: string; extension: string } {
const candidates = [
{ mimeType: "audio/wav", extension: "wav" },
{ mimeType: "audio/webm;codecs=opus", extension: "webm" },
{ mimeType: "audio/webm", extension: "webm" },
{ mimeType: "audio/mp4", extension: "m4a" },
];
for (const candidate of candidates) {
if (typeof MediaRecorder !== "undefined" && MediaRecorder.isTypeSupported(candidate.mimeType)) {
return candidate;
}
}
return { mimeType: "", extension: "webm" };
}
export function useVoiceInput(options: {
onTranscript: (text: string) => void;
onError: (message: string) => void;
disabled?: boolean;
}) {
const [state, setState] = useState<VoiceInputState>("idle");
const recorderRef = useRef<MediaRecorder | null>(null);
const chunksRef = useRef<Blob[]>([]);
const streamRef = useRef<MediaStream | null>(null);
const formatRef = useRef(pickRecorderFormat());
const onTranscriptRef = useRef(options.onTranscript);
const onErrorRef = useRef(options.onError);
useEffect(() => {
onTranscriptRef.current = options.onTranscript;
onErrorRef.current = options.onError;
}, [options.onTranscript, options.onError]);
const stopStream = useCallback(() => {
for (const track of streamRef.current?.getTracks() ?? []) {
track.stop();
}
streamRef.current = null;
}, []);
const stopRecording = useCallback(async () => {
const recorder = recorderRef.current;
if (!recorder || recorder.state === "inactive") return;
await new Promise<void>((resolve) => {
recorder.addEventListener("stop", () => resolve(), { once: true });
recorder.stop();
});
recorderRef.current = null;
stopStream();
const blob = new Blob(chunksRef.current, {
type: formatRef.current.mimeType || chunksRef.current[0]?.type || "audio/webm",
});
chunksRef.current = [];
if (blob.size === 0) {
setState("idle");
onErrorRef.current("No audio captured");
return;
}
setState("transcribing");
try {
const formData = new FormData();
formData.append("file", blob, `recording.${formatRef.current.extension}`);
const response = await fetch("/api/agent/transcribe", { method: "POST", body: formData });
if (!response.ok) {
const payload = (await response.json().catch(() => null)) as { error?: string } | null;
throw new Error(payload?.error ?? "Transcription failed");
}
const payload = (await response.json()) as { text: string };
onTranscriptRef.current(payload.text);
} catch (err) {
onErrorRef.current(err instanceof Error ? err.message : "Transcription failed");
} finally {
setState("idle");
}
}, [stopStream]);
const startRecording = useCallback(async () => {
if (options.disabled || state !== "idle") return;
if (typeof navigator === "undefined" || !navigator.mediaDevices?.getUserMedia) {
onErrorRef.current("Microphone not available in this browser");
return;
}
formatRef.current = pickRecorderFormat();
try {
const stream = await navigator.mediaDevices.getUserMedia({ audio: true });
streamRef.current = stream;
const recorder = formatRef.current.mimeType
? new MediaRecorder(stream, { mimeType: formatRef.current.mimeType })
: new MediaRecorder(stream);
chunksRef.current = [];
recorder.addEventListener("dataavailable", (event) => {
if (event.data.size > 0) chunksRef.current.push(event.data);
});
recorder.start();
recorderRef.current = recorder;
setState("recording");
} catch {
stopStream();
onErrorRef.current("Microphone permission denied");
}
}, [options.disabled, state, stopStream]);
const toggleRecording = useCallback(async () => {
if (state === "recording") {
await stopRecording();
return;
}
if (state === "idle") {
await startRecording();
}
}, [startRecording, state, stopRecording]);
return { state, toggleRecording };
}
+20
View File
@@ -0,0 +1,20 @@
import { z } from "zod";
export const clientChatAttachmentSchema = z.object({
type: z.literal("image"),
url: z.string().trim().min(1).max(2048),
});
export const clientChatMessageSchema = z.object({
role: z.enum(["user", "assistant"]),
content: z.string().trim().min(1).max(8000),
attachments: z.array(clientChatAttachmentSchema).max(3).optional(),
});
export const clientChatInputSchema = z.object({
stream: z.boolean().optional(),
messages: z.array(clientChatMessageSchema).min(1).max(40),
});
export type ClientChatAttachment = z.infer<typeof clientChatAttachmentSchema>;
export type ClientChatMessage = z.infer<typeof clientChatMessageSchema>;
@@ -0,0 +1,35 @@
import { minioClient, MINIO_BUCKET } from "@/lib/minio";
const UPLOAD_PATH_PREFIX = "/api/uploads/";
function uploadKeyFromUrl(url: string): string | null {
if (!url.startsWith(UPLOAD_PATH_PREFIX)) return null;
const key = url.slice(UPLOAD_PATH_PREFIX.length);
if (!key || key.includes("..")) return null;
return key;
}
export async function resolveAssistantImageDataUrl(url: string): Promise<string> {
const key = uploadKeyFromUrl(url);
if (!key) {
throw new Error("Unsupported image URL");
}
const stream = await minioClient.getObject(MINIO_BUCKET, key);
const chunks: Buffer[] = [];
for await (const chunk of stream) {
chunks.push(Buffer.isBuffer(chunk) ? chunk : Buffer.from(chunk as Uint8Array));
}
const buffer = Buffer.concat(chunks);
const stat = await minioClient.statObject(MINIO_BUCKET, key);
const metaData = stat.metaData as Record<string, string> | undefined;
const contentType =
metaData?.["content-type"] ?? metaData?.["Content-Type"] ?? "application/octet-stream";
return `data:${contentType};base64,${buffer.toString("base64")}`;
}
export async function resolveAssistantImageDataUrls(urls: string[]): Promise<string[]> {
return Promise.all(urls.map((url) => resolveAssistantImageDataUrl(url)));
}
+31 -15
View File
@@ -1,16 +1,14 @@
import { createLlmClient, type ChatMessage, type LlmClient } from "@/lib/llm";
import { buildVisionContentParts, textFromMessageContent } from "@/lib/llm/content";
import { AGENT_SYSTEM_PROMPT, AGENT_TOOLS } from "../tools";
import type { ClientChatMessage } from "../messages";
import { describeToolActivity } from "../tool-labels";
import { createApiToolExecutor, type ToolExecutor } from "../tool-executor";
import { resolveAssistantImageDataUrls } from "./resolve-images";
import { thinkingLabel, type AgentProgressEvent } from "./progress";
const MAX_TOOL_ROUNDS = 8;
export type ClientChatMessage = {
role: "user" | "assistant";
content: string;
};
export type AgentToolCallSummary = {
name: string;
status: number;
@@ -23,6 +21,26 @@ export type AgentChatResult = {
export type AgentProgressHandler = (event: AgentProgressEvent) => void;
async function toLlmUserMessage(message: ClientChatMessage): Promise<ChatMessage> {
if (message.role === "assistant") {
return { role: "assistant", content: message.content };
}
const imageUrls =
message.attachments?.filter((attachment) => attachment.type === "image").map((a) => a.url) ??
[];
if (imageUrls.length === 0) {
return { role: "user", content: message.content };
}
const dataUrls = await resolveAssistantImageDataUrls(imageUrls);
return {
role: "user",
content: buildVisionContentParts(message.content, dataUrls),
};
}
export async function runAgentChat(options: {
messages: ClientChatMessage[];
request: Request;
@@ -36,15 +54,11 @@ export async function runAgentChat(options: {
const onProgress = options.onProgress;
const systemPrompt = options.systemPrompt ?? AGENT_SYSTEM_PROMPT;
const transcript: ChatMessage[] = [
{ role: "system", content: systemPrompt },
...options.messages.map(
(message): ChatMessage => ({
role: message.role,
content: message.content,
}),
),
];
const userMessages = await Promise.all(
options.messages.map((message) => toLlmUserMessage(message)),
);
const transcript: ChatMessage[] = [{ role: "system", content: systemPrompt }, ...userMessages];
const toolCalls: AgentToolCallSummary[] = [];
@@ -64,7 +78,9 @@ export async function runAgentChat(options: {
return {
message: {
role: "assistant",
content: assistantMessage.content?.trim() || "I couldn't generate a response.",
content:
textFromMessageContent(assistantMessage.content).trim() ||
"I couldn't generate a response.",
},
toolCalls,
};
+2
View File
@@ -766,6 +766,8 @@ When no dedicated tool fits, or you are unsure how to do something:
1. Call get_api_docs with a relevant search term to find the right /api/v1/* endpoint.
2. Call call_api with the documented method, path, query, and body.
When the user sends a photo, read dates, times, locations, and action items from it, then use tools to act.
Lists: resolve list ids via list_lists. To complete items, list_list_items then update_list_item with done: true.
Journal: per-user private entries. Valid mood ids: ${JOURNAL_MOOD_IDS}. stress is 1-10. pillsTaken is boolean.