From c8db5475d314b239246e4786e362bb044992d660 Mon Sep 17 00:00:00 2001 From: ginnoir Date: Sun, 5 Jul 2026 02:01:48 -0500 Subject: [PATCH] feat(agent): add voice input and photo attachments to assistant Wire mic through Whisper-compatible transcriptions on LLM_BASE_URL. Photos upload to MinIO and reach the vision model as base64 image_url parts. --- .env.example | 2 + src/app/api/agent/chat/route.ts | 17 +- src/app/api/agent/transcribe/route.ts | 51 ++++++ src/app/api/uploads/route.ts | 4 +- src/lib/llm/content.ts | 28 +++ src/lib/llm/mock.ts | 15 +- src/lib/llm/openai-compatible.ts | 4 +- src/lib/llm/transcribe.ts | 47 +++++ src/lib/llm/types.ts | 16 +- src/modules/agent/assistant-chat-storage.ts | 26 ++- .../agent/components/assistant-panel.tsx | 169 ++++++++++++++++-- .../agent/components/use-voice-input.ts | 129 +++++++++++++ src/modules/agent/messages.ts | 20 +++ src/modules/agent/server/resolve-images.ts | 35 ++++ src/modules/agent/server/run.ts | 46 +++-- src/modules/agent/tools.ts | 2 + tests/unit/agent-messages.test.ts | 28 +++ tests/unit/agent-transcribe.test.ts | 23 +++ tests/unit/assistant-chat-storage.test.ts | 12 ++ tests/unit/llm-content.test.ts | 27 +++ 20 files changed, 636 insertions(+), 65 deletions(-) create mode 100644 src/app/api/agent/transcribe/route.ts create mode 100644 src/lib/llm/content.ts create mode 100644 src/lib/llm/transcribe.ts create mode 100644 src/modules/agent/components/use-voice-input.ts create mode 100644 src/modules/agent/messages.ts create mode 100644 src/modules/agent/server/resolve-images.ts create mode 100644 tests/unit/agent-messages.test.ts create mode 100644 tests/unit/agent-transcribe.test.ts create mode 100644 tests/unit/llm-content.test.ts diff --git a/.env.example b/.env.example index a577999..36a21bc 100644 --- a/.env.example +++ b/.env.example @@ -49,6 +49,8 @@ OPENPLANTBOOK_CLIENT_SECRET= # LLM assistant (OpenAI-compatible — Ollama, vLLM, LiteLLM, etc.) # Leave LLM_BASE_URL unset to use the built-in mock provider (CI / local without a model). +# Voice input uses POST {LLM_BASE_URL}/audio/transcriptions (Whisper-compatible). +# Photo messages use vision via the same chat/completions endpoint. LLM_PROVIDER=openai LLM_BASE_URL= LLM_API_KEY= diff --git a/src/app/api/agent/chat/route.ts b/src/app/api/agent/chat/route.ts index 287c62a..1c8c335 100644 --- a/src/app/api/agent/chat/route.ts +++ b/src/app/api/agent/chat/route.ts @@ -1,24 +1,11 @@ -import { z } from "zod"; import { apiError, apiJson } from "@/lib/api-handler"; import { resolveApiAuth } from "@/lib/api-auth"; import { getAssistantPreferences, resolveAssistantSystemPrompt } from "@/lib/assistant-preference"; import { isLlmConfigured } from "@/lib/llm"; +import { clientChatInputSchema } from "@/modules/agent/messages"; import { encodeSseEvent } from "@/modules/agent/server/progress"; import { runAgentChat } from "@/modules/agent/server/run"; -const chatInput = z.object({ - stream: z.boolean().optional(), - messages: z - .array( - z.object({ - role: z.enum(["user", "assistant"]), - content: z.string().trim().min(1).max(8000), - }), - ) - .min(1) - .max(40), -}); - export async function POST(request: Request) { const auth = await resolveApiAuth(request); if (!auth?.userId) { @@ -39,7 +26,7 @@ export async function POST(request: Request) { return apiError("Invalid JSON body", 400); } - const parsed = chatInput.safeParse(body); + const parsed = clientChatInputSchema.safeParse(body); if (!parsed.success) { return apiError(parsed.error.issues[0]?.message ?? "Validation error", 400); } diff --git a/src/app/api/agent/transcribe/route.ts b/src/app/api/agent/transcribe/route.ts new file mode 100644 index 0000000..237553b --- /dev/null +++ b/src/app/api/agent/transcribe/route.ts @@ -0,0 +1,51 @@ +import { apiError, apiJson } from "@/lib/api-handler"; +import { resolveApiAuth } from "@/lib/api-auth"; +import { getAssistantPreferences } from "@/lib/assistant-preference"; +import { transcribeAudioFile } from "@/lib/llm/transcribe"; + +export const runtime = "nodejs"; +export const maxDuration = 120; + +const MAX_AUDIO_BYTES = 25 * 1024 * 1024; + +export async function POST(request: Request) { + const auth = await resolveApiAuth(request); + if (!auth?.userId) { + return apiError("Unauthorized", 401); + } + + const assistant = await getAssistantPreferences(auth.userId); + if (!assistant.enabled) { + return apiError("Assistant not enabled", 403); + } + + let formData: FormData; + try { + formData = await request.formData(); + } catch { + return apiError("Invalid multipart body", 400); + } + + const file = formData.get("file"); + if (!(file instanceof File)) { + return apiError("No audio file in request", 400); + } + + if (!file.type.startsWith("audio/") && file.type !== "video/webm") { + return apiError("Only audio recordings are allowed", 415); + } + + if (file.size > MAX_AUDIO_BYTES) { + return apiError("Recording exceeds 25 MB limit", 413); + } + + const filename = file.name.trim() || "recording.wav"; + + try { + const text = await transcribeAudioFile(file, filename); + return apiJson({ text }); + } catch (err) { + const message = err instanceof Error ? err.message : "Transcription failed"; + return apiError(message, 502); + } +} diff --git a/src/app/api/uploads/route.ts b/src/app/api/uploads/route.ts index b4ba19a..68fc921 100644 --- a/src/app/api/uploads/route.ts +++ b/src/app/api/uploads/route.ts @@ -33,7 +33,9 @@ export async function POST(request: Request) { } const url = new URL(request.url); - const scope = url.searchParams.get("scope") === "notes" ? "notes" : "garden"; + const scopeParam = url.searchParams.get("scope"); + const scope = + scopeParam === "notes" ? "notes" : scopeParam === "assistant" ? "assistant" : "garden"; if (!file.type.startsWith("image/")) { return NextResponse.json({ error: "Only image files are allowed" }, { status: 415 }); diff --git a/src/lib/llm/content.ts b/src/lib/llm/content.ts new file mode 100644 index 0000000..98e811d --- /dev/null +++ b/src/lib/llm/content.ts @@ -0,0 +1,28 @@ +import type { ChatContentPart, ChatMessage } from "./types"; + +export function textFromMessageContent(content: ChatMessage["content"]): string { + if (!content) return ""; + if (typeof content === "string") return content; + return content + .filter((part): part is Extract => part.type === "text") + .map((part) => part.text) + .join("\n") + .trim(); +} + +export function buildVisionContentParts(text: string, imageDataUrls: string[]): ChatContentPart[] { + const parts: ChatContentPart[] = []; + const trimmed = text.trim(); + if (trimmed) { + parts.push({ type: "text", text: trimmed }); + } else if (imageDataUrls.length > 0) { + parts.push({ + type: "text", + text: "Read this image and help with what the user needs.", + }); + } + for (const url of imageDataUrls) { + parts.push({ type: "image_url", image_url: { url } }); + } + return parts; +} diff --git a/src/lib/llm/mock.ts b/src/lib/llm/mock.ts index beb104f..732222f 100644 --- a/src/lib/llm/mock.ts +++ b/src/lib/llm/mock.ts @@ -1,14 +1,5 @@ import type { ChatCompletionRequest, ChatCompletionResult, LlmClient } from "./types"; - -function lastUserText(messages: ChatCompletionRequest["messages"]): string { - for (let i = messages.length - 1; i >= 0; i -= 1) { - const message = messages[i]; - if (message?.role === "user" && message.content) { - return message.content.toLowerCase(); - } - } - return ""; -} +import { textFromMessageContent } from "./content"; function hadToolResults(messages: ChatCompletionRequest["messages"]): boolean { return messages.some((message) => message.role === "tool"); @@ -17,7 +8,9 @@ function hadToolResults(messages: ChatCompletionRequest["messages"]): boolean { export function createMockLlmClient(): LlmClient { return { async chatCompletion(request: ChatCompletionRequest): Promise { - const userText = lastUserText(request.messages); + const userText = textFromMessageContent( + [...request.messages].reverse().find((message) => message.role === "user")?.content ?? "", + ).toLowerCase(); if (hadToolResults(request.messages)) { return { diff --git a/src/lib/llm/openai-compatible.ts b/src/lib/llm/openai-compatible.ts index cb95f86..289783c 100644 --- a/src/lib/llm/openai-compatible.ts +++ b/src/lib/llm/openai-compatible.ts @@ -1,8 +1,8 @@ -import type { ChatCompletionRequest, ChatCompletionResult, LlmClient } from "./types"; +import type { ChatCompletionRequest, ChatCompletionResult, ChatMessage, LlmClient } from "./types"; type OpenAiMessage = { role: string; - content: string | null; + content: ChatMessage["content"]; tool_calls?: Array<{ id: string; type: "function"; diff --git a/src/lib/llm/transcribe.ts b/src/lib/llm/transcribe.ts new file mode 100644 index 0000000..de6b02b --- /dev/null +++ b/src/lib/llm/transcribe.ts @@ -0,0 +1,47 @@ +import { getLlmConfig } from "./config"; + +const MAX_AUDIO_BYTES = 25 * 1024 * 1024; + +export async function transcribeAudioFile(file: File | Blob, filename: string): Promise { + if (file.size > MAX_AUDIO_BYTES) { + throw new Error("Recording exceeds 25 MB limit"); + } + + const config = getLlmConfig(); + if (config.provider === "mock" || !config.baseUrl) { + return "add milk to the shopping list"; + } + + const formData = new FormData(); + formData.append("file", file, filename); + formData.append("model", "whisper-1"); + + const url = `${config.baseUrl.replace(/\/$/, "")}/audio/transcriptions`; + const headers: Record = {}; + if (config.apiKey) { + headers.Authorization = `Bearer ${config.apiKey}`; + } + + const response = await fetch(url, { + method: "POST", + headers, + body: formData, + }); + + if (!response.ok) { + const detail = await response.text(); + throw new Error(`Transcription failed (${response.status}): ${detail.slice(0, 400)}`); + } + + const contentType = response.headers.get("content-type") ?? ""; + if (contentType.includes("application/json")) { + const payload = (await response.json()) as { text?: string }; + const text = payload.text?.trim(); + if (!text) throw new Error("Transcription returned empty text"); + return text; + } + + const text = (await response.text()).trim(); + if (!text) throw new Error("Transcription returned empty text"); + return text; +} diff --git a/src/lib/llm/types.ts b/src/lib/llm/types.ts index d01e893..201cbc8 100644 --- a/src/lib/llm/types.ts +++ b/src/lib/llm/types.ts @@ -1,5 +1,19 @@ export type ChatRole = "system" | "user" | "assistant" | "tool"; +export type ChatTextPart = { + type: "text"; + text: string; +}; + +export type ChatImagePart = { + type: "image_url"; + image_url: { + url: string; + }; +}; + +export type ChatContentPart = ChatTextPart | ChatImagePart; + export type ChatToolCall = { id: string; type: "function"; @@ -11,7 +25,7 @@ export type ChatToolCall = { export type ChatMessage = { role: ChatRole; - content: string | null; + content: string | ChatContentPart[] | null; tool_calls?: ChatToolCall[]; tool_call_id?: string; name?: string; diff --git a/src/modules/agent/assistant-chat-storage.ts b/src/modules/agent/assistant-chat-storage.ts index b9ddc70..96fc657 100644 --- a/src/modules/agent/assistant-chat-storage.ts +++ b/src/modules/agent/assistant-chat-storage.ts @@ -1,9 +1,12 @@ +import type { ClientChatMessage } from "./messages"; + export type AssistantChatMessage = { role: "user" | "assistant"; content: string; + imageUrl?: string; }; -const STORAGE_VERSION = "v1"; +const STORAGE_VERSION = "v2"; const MAX_MESSAGES = 40; function storageKey(userId: string) { @@ -13,11 +16,22 @@ function storageKey(userId: string) { function isValidMessage(value: unknown): value is AssistantChatMessage { if (!value || typeof value !== "object") return false; const row = value as Record; - return ( - (row.role === "user" || row.role === "assistant") && - typeof row.content === "string" && - row.content.trim().length > 0 - ); + if (row.role !== "user" && row.role !== "assistant") return false; + if (typeof row.content !== "string" || row.content.trim().length === 0) return false; + if (row.imageUrl !== undefined && typeof row.imageUrl !== "string") return false; + return true; +} + +export function toClientChatMessage(message: AssistantChatMessage): ClientChatMessage { + if (!message.imageUrl) { + return { role: message.role, content: message.content }; + } + + return { + role: message.role, + content: message.content, + attachments: [{ type: "image", url: message.imageUrl }], + }; } export function loadAssistantChat(userId: string): AssistantChatMessage[] { diff --git a/src/modules/agent/components/assistant-panel.tsx b/src/modules/agent/components/assistant-panel.tsx index ecde6e6..155cf7a 100644 --- a/src/modules/agent/components/assistant-panel.tsx +++ b/src/modules/agent/components/assistant-panel.tsx @@ -1,7 +1,7 @@ "use client"; import { useEffect, useRef, useState } from "react"; -import { Loader2, Send } from "lucide-react"; +import { ImagePlus, Loader2, Mic, Send, Square } from "lucide-react"; import { Button } from "@/components/ui/button"; import { Input } from "@/components/ui/input"; import { consumeAgentChatStream } from "../assistant-chat-stream"; @@ -9,8 +9,10 @@ import { clearAssistantChat, loadAssistantChat, saveAssistantChat, + toClientChatMessage, type AssistantChatMessage, } from "../assistant-chat-storage"; +import { useVoiceInput } from "./use-voice-input"; type Props = { configured: boolean; @@ -18,14 +20,42 @@ type Props = { assistantName: string; }; +type PendingImage = { + url: string; +}; + +async function uploadAssistantImage(file: File): Promise { + const formData = new FormData(); + formData.append("file", file); + const response = await fetch("/api/uploads?scope=assistant", { method: "POST", body: formData }); + if (!response.ok) { + const payload = (await response.json().catch(() => null)) as { error?: string } | null; + throw new Error(payload?.error ?? "Image upload failed"); + } + const payload = (await response.json()) as { url: string }; + return payload.url; +} + export function AssistantPanel({ configured, userId, assistantName }: Props) { const [messages, setMessages] = useState(() => loadAssistantChat(userId)); const [input, setInput] = useState(""); + const [pendingImage, setPendingImage] = useState(null); + const [uploadingImage, setUploadingImage] = useState(false); const [error, setError] = useState(null); const [isPending, setIsPending] = useState(false); const [activityLabel, setActivityLabel] = useState(null); const listRef = useRef(null); const abortRef = useRef(null); + const imageInputRef = useRef(null); + + const { state: voiceState, toggleRecording } = useVoiceInput({ + disabled: isPending || uploadingImage, + onTranscript: (text) => { + setInput((current) => (current.trim() ? `${current.trim()} ${text}` : text)); + setError(null); + }, + onError: (message) => setError(message), + }); useEffect(() => { saveAssistantChat(userId, messages); @@ -47,22 +77,50 @@ export function AssistantPanel({ configured, userId, assistantName }: Props) { function clearChat() { abortRef.current?.abort(); setMessages([]); + setInput(""); + setPendingImage(null); setError(null); setActivityLabel(null); setIsPending(false); clearAssistantChat(userId); } + async function handleImageSelect(event: React.ChangeEvent) { + const file = event.target.files?.[0]; + event.target.value = ""; + if (!file || isPending || uploadingImage) return; + + setUploadingImage(true); + setError(null); + try { + const url = await uploadAssistantImage(file); + setPendingImage({ url }); + } catch (err) { + setError(err instanceof Error ? err.message : "Image upload failed"); + } finally { + setUploadingImage(false); + } + } + async function sendMessage() { const text = input.trim(); - if (!text || isPending) return; + const hasImage = pendingImage !== null; + if ((!text && !hasImage) || isPending || voiceState !== "idle") return; - const nextMessages: AssistantChatMessage[] = [...messages, { role: "user", content: text }]; + const content = text || "Help me with this image."; + const userMessage: AssistantChatMessage = { + role: "user", + content, + ...(pendingImage ? { imageUrl: pendingImage.url } : {}), + }; + + const nextMessages: AssistantChatMessage[] = [...messages, userMessage]; setInput(""); + setPendingImage(null); setError(null); setMessages(nextMessages); setIsPending(true); - setActivityLabel("Understanding your request…"); + setActivityLabel(hasImage ? "Reading your photo…" : "Understanding your request…"); scrollToBottom(); abortRef.current?.abort(); @@ -73,7 +131,10 @@ export function AssistantPanel({ configured, userId, assistantName }: Props) { const response = await fetch("/api/agent/chat", { method: "POST", headers: { "Content-Type": "application/json" }, - body: JSON.stringify({ messages: nextMessages, stream: true }), + body: JSON.stringify({ + messages: nextMessages.map(toClientChatMessage), + stream: true, + }), signal: controller.signal, }); @@ -88,7 +149,10 @@ export function AssistantPanel({ configured, userId, assistantName }: Props) { throw new Error("Assistant returned an empty response"); } - setMessages((current) => [...current, result.message]); + setMessages((current) => [ + ...current, + { role: "assistant", content: result.message.content }, + ]); scrollToBottom(); } catch (err) { if (err instanceof Error && err.name === "AbortError") return; @@ -101,13 +165,19 @@ export function AssistantPanel({ configured, userId, assistantName }: Props) { } const showEmptyState = messages.length === 0 && !isPending; + const inputDisabled = isPending || voiceState === "transcribing" || uploadingImage; + const canSend = + !isPending && + voiceState === "idle" && + !uploadingImage && + (input.trim().length > 0 || pendingImage !== null); return (

{configured - ? "Ask me to update lists, calendar, notes, or journal." + ? "Type, talk, or send a photo — I can update lists, calendar, notes, and more." : "Mock provider active — set LLM_BASE_URL for your homelab model."}

{messages.length > 0 ? ( @@ -129,8 +199,8 @@ export function AssistantPanel({ configured, userId, assistantName }: Props) { > {showEmptyState ? (

- Try "add milk to the shopping list" or "what's on the calendar this - week?" + Try "add milk to the shopping list", tap the mic, or attach a photo of an + appointment card.

) : (
@@ -147,6 +217,14 @@ export function AssistantPanel({ configured, userId, assistantName }: Props) {
{message.role === "user" ? "You" : assistantName}
+ {message.imageUrl ? ( + // User-uploaded assistant attachment preview + + ) : null} {message.content}
))} @@ -169,24 +247,87 @@ export function AssistantPanel({ configured, userId, assistantName }: Props) { )}
+ {pendingImage ? ( +
+ +
Photo attached
+ +
+ ) : null} + {error ?

{error}

: null}
{ event.preventDefault(); - sendMessage(); + void sendMessage(); }} > + void handleImageSelect(event)} + /> + + setInput(event.target.value)} - placeholder={`Ask ${assistantName}…`} - disabled={isPending} + placeholder={ + voiceState === "recording" + ? "Listening… tap mic to stop" + : voiceState === "transcribing" + ? "Transcribing…" + : `Ask ${assistantName}…` + } + disabled={inputDisabled} aria-label={`Message for ${assistantName}`} - className="h-9" + className="h-9 min-w-0 flex-1" /> -
diff --git a/src/modules/agent/components/use-voice-input.ts b/src/modules/agent/components/use-voice-input.ts new file mode 100644 index 0000000..cdac5b5 --- /dev/null +++ b/src/modules/agent/components/use-voice-input.ts @@ -0,0 +1,129 @@ +"use client"; + +import { useCallback, useEffect, useRef, useState } from "react"; + +export type VoiceInputState = "idle" | "recording" | "transcribing"; + +function pickRecorderFormat(): { mimeType: string; extension: string } { + const candidates = [ + { mimeType: "audio/wav", extension: "wav" }, + { mimeType: "audio/webm;codecs=opus", extension: "webm" }, + { mimeType: "audio/webm", extension: "webm" }, + { mimeType: "audio/mp4", extension: "m4a" }, + ]; + + for (const candidate of candidates) { + if (typeof MediaRecorder !== "undefined" && MediaRecorder.isTypeSupported(candidate.mimeType)) { + return candidate; + } + } + + return { mimeType: "", extension: "webm" }; +} + +export function useVoiceInput(options: { + onTranscript: (text: string) => void; + onError: (message: string) => void; + disabled?: boolean; +}) { + const [state, setState] = useState("idle"); + const recorderRef = useRef(null); + const chunksRef = useRef([]); + const streamRef = useRef(null); + const formatRef = useRef(pickRecorderFormat()); + const onTranscriptRef = useRef(options.onTranscript); + const onErrorRef = useRef(options.onError); + + useEffect(() => { + onTranscriptRef.current = options.onTranscript; + onErrorRef.current = options.onError; + }, [options.onTranscript, options.onError]); + + const stopStream = useCallback(() => { + for (const track of streamRef.current?.getTracks() ?? []) { + track.stop(); + } + streamRef.current = null; + }, []); + + const stopRecording = useCallback(async () => { + const recorder = recorderRef.current; + if (!recorder || recorder.state === "inactive") return; + + await new Promise((resolve) => { + recorder.addEventListener("stop", () => resolve(), { once: true }); + recorder.stop(); + }); + + recorderRef.current = null; + stopStream(); + + const blob = new Blob(chunksRef.current, { + type: formatRef.current.mimeType || chunksRef.current[0]?.type || "audio/webm", + }); + chunksRef.current = []; + + if (blob.size === 0) { + setState("idle"); + onErrorRef.current("No audio captured"); + return; + } + + setState("transcribing"); + try { + const formData = new FormData(); + formData.append("file", blob, `recording.${formatRef.current.extension}`); + const response = await fetch("/api/agent/transcribe", { method: "POST", body: formData }); + if (!response.ok) { + const payload = (await response.json().catch(() => null)) as { error?: string } | null; + throw new Error(payload?.error ?? "Transcription failed"); + } + const payload = (await response.json()) as { text: string }; + onTranscriptRef.current(payload.text); + } catch (err) { + onErrorRef.current(err instanceof Error ? err.message : "Transcription failed"); + } finally { + setState("idle"); + } + }, [stopStream]); + + const startRecording = useCallback(async () => { + if (options.disabled || state !== "idle") return; + if (typeof navigator === "undefined" || !navigator.mediaDevices?.getUserMedia) { + onErrorRef.current("Microphone not available in this browser"); + return; + } + + formatRef.current = pickRecorderFormat(); + + try { + const stream = await navigator.mediaDevices.getUserMedia({ audio: true }); + streamRef.current = stream; + const recorder = formatRef.current.mimeType + ? new MediaRecorder(stream, { mimeType: formatRef.current.mimeType }) + : new MediaRecorder(stream); + chunksRef.current = []; + recorder.addEventListener("dataavailable", (event) => { + if (event.data.size > 0) chunksRef.current.push(event.data); + }); + recorder.start(); + recorderRef.current = recorder; + setState("recording"); + } catch { + stopStream(); + onErrorRef.current("Microphone permission denied"); + } + }, [options.disabled, state, stopStream]); + + const toggleRecording = useCallback(async () => { + if (state === "recording") { + await stopRecording(); + return; + } + if (state === "idle") { + await startRecording(); + } + }, [startRecording, state, stopRecording]); + + return { state, toggleRecording }; +} diff --git a/src/modules/agent/messages.ts b/src/modules/agent/messages.ts new file mode 100644 index 0000000..e148acd --- /dev/null +++ b/src/modules/agent/messages.ts @@ -0,0 +1,20 @@ +import { z } from "zod"; + +export const clientChatAttachmentSchema = z.object({ + type: z.literal("image"), + url: z.string().trim().min(1).max(2048), +}); + +export const clientChatMessageSchema = z.object({ + role: z.enum(["user", "assistant"]), + content: z.string().trim().min(1).max(8000), + attachments: z.array(clientChatAttachmentSchema).max(3).optional(), +}); + +export const clientChatInputSchema = z.object({ + stream: z.boolean().optional(), + messages: z.array(clientChatMessageSchema).min(1).max(40), +}); + +export type ClientChatAttachment = z.infer; +export type ClientChatMessage = z.infer; diff --git a/src/modules/agent/server/resolve-images.ts b/src/modules/agent/server/resolve-images.ts new file mode 100644 index 0000000..820ac53 --- /dev/null +++ b/src/modules/agent/server/resolve-images.ts @@ -0,0 +1,35 @@ +import { minioClient, MINIO_BUCKET } from "@/lib/minio"; + +const UPLOAD_PATH_PREFIX = "/api/uploads/"; + +function uploadKeyFromUrl(url: string): string | null { + if (!url.startsWith(UPLOAD_PATH_PREFIX)) return null; + const key = url.slice(UPLOAD_PATH_PREFIX.length); + if (!key || key.includes("..")) return null; + return key; +} + +export async function resolveAssistantImageDataUrl(url: string): Promise { + const key = uploadKeyFromUrl(url); + if (!key) { + throw new Error("Unsupported image URL"); + } + + const stream = await minioClient.getObject(MINIO_BUCKET, key); + const chunks: Buffer[] = []; + for await (const chunk of stream) { + chunks.push(Buffer.isBuffer(chunk) ? chunk : Buffer.from(chunk as Uint8Array)); + } + const buffer = Buffer.concat(chunks); + + const stat = await minioClient.statObject(MINIO_BUCKET, key); + const metaData = stat.metaData as Record | undefined; + const contentType = + metaData?.["content-type"] ?? metaData?.["Content-Type"] ?? "application/octet-stream"; + + return `data:${contentType};base64,${buffer.toString("base64")}`; +} + +export async function resolveAssistantImageDataUrls(urls: string[]): Promise { + return Promise.all(urls.map((url) => resolveAssistantImageDataUrl(url))); +} diff --git a/src/modules/agent/server/run.ts b/src/modules/agent/server/run.ts index 514e6dc..341e267 100644 --- a/src/modules/agent/server/run.ts +++ b/src/modules/agent/server/run.ts @@ -1,16 +1,14 @@ import { createLlmClient, type ChatMessage, type LlmClient } from "@/lib/llm"; +import { buildVisionContentParts, textFromMessageContent } from "@/lib/llm/content"; import { AGENT_SYSTEM_PROMPT, AGENT_TOOLS } from "../tools"; +import type { ClientChatMessage } from "../messages"; import { describeToolActivity } from "../tool-labels"; import { createApiToolExecutor, type ToolExecutor } from "../tool-executor"; +import { resolveAssistantImageDataUrls } from "./resolve-images"; import { thinkingLabel, type AgentProgressEvent } from "./progress"; const MAX_TOOL_ROUNDS = 8; -export type ClientChatMessage = { - role: "user" | "assistant"; - content: string; -}; - export type AgentToolCallSummary = { name: string; status: number; @@ -23,6 +21,26 @@ export type AgentChatResult = { export type AgentProgressHandler = (event: AgentProgressEvent) => void; +async function toLlmUserMessage(message: ClientChatMessage): Promise { + if (message.role === "assistant") { + return { role: "assistant", content: message.content }; + } + + const imageUrls = + message.attachments?.filter((attachment) => attachment.type === "image").map((a) => a.url) ?? + []; + + if (imageUrls.length === 0) { + return { role: "user", content: message.content }; + } + + const dataUrls = await resolveAssistantImageDataUrls(imageUrls); + return { + role: "user", + content: buildVisionContentParts(message.content, dataUrls), + }; +} + export async function runAgentChat(options: { messages: ClientChatMessage[]; request: Request; @@ -36,15 +54,11 @@ export async function runAgentChat(options: { const onProgress = options.onProgress; const systemPrompt = options.systemPrompt ?? AGENT_SYSTEM_PROMPT; - const transcript: ChatMessage[] = [ - { role: "system", content: systemPrompt }, - ...options.messages.map( - (message): ChatMessage => ({ - role: message.role, - content: message.content, - }), - ), - ]; + const userMessages = await Promise.all( + options.messages.map((message) => toLlmUserMessage(message)), + ); + + const transcript: ChatMessage[] = [{ role: "system", content: systemPrompt }, ...userMessages]; const toolCalls: AgentToolCallSummary[] = []; @@ -64,7 +78,9 @@ export async function runAgentChat(options: { return { message: { role: "assistant", - content: assistantMessage.content?.trim() || "I couldn't generate a response.", + content: + textFromMessageContent(assistantMessage.content).trim() || + "I couldn't generate a response.", }, toolCalls, }; diff --git a/src/modules/agent/tools.ts b/src/modules/agent/tools.ts index 578245c..5678cf7 100644 --- a/src/modules/agent/tools.ts +++ b/src/modules/agent/tools.ts @@ -766,6 +766,8 @@ When no dedicated tool fits, or you are unsure how to do something: 1. Call get_api_docs with a relevant search term to find the right /api/v1/* endpoint. 2. Call call_api with the documented method, path, query, and body. +When the user sends a photo, read dates, times, locations, and action items from it, then use tools to act. + Lists: resolve list ids via list_lists. To complete items, list_list_items then update_list_item with done: true. Journal: per-user private entries. Valid mood ids: ${JOURNAL_MOOD_IDS}. stress is 1-10. pillsTaken is boolean. diff --git a/tests/unit/agent-messages.test.ts b/tests/unit/agent-messages.test.ts new file mode 100644 index 0000000..9815d11 --- /dev/null +++ b/tests/unit/agent-messages.test.ts @@ -0,0 +1,28 @@ +import assert from "node:assert/strict"; +import { describe, it } from "node:test"; +import { clientChatInputSchema } from "../../src/modules/agent/messages"; + +describe("clientChatInputSchema", () => { + it("accepts messages with image attachments", () => { + const parsed = clientChatInputSchema.safeParse({ + stream: true, + messages: [ + { + role: "user", + content: "Add this appointment to the calendar", + attachments: [{ type: "image", url: "/api/uploads/assistant/house-1/photo.jpg" }], + }, + ], + }); + + assert.equal(parsed.success, true); + }); + + it("rejects empty message content", () => { + const parsed = clientChatInputSchema.safeParse({ + messages: [{ role: "user", content: " " }], + }); + + assert.equal(parsed.success, false); + }); +}); diff --git a/tests/unit/agent-transcribe.test.ts b/tests/unit/agent-transcribe.test.ts new file mode 100644 index 0000000..9a4e83a --- /dev/null +++ b/tests/unit/agent-transcribe.test.ts @@ -0,0 +1,23 @@ +import assert from "node:assert/strict"; +import { describe, it } from "node:test"; +import { transcribeAudioFile } from "../../src/lib/llm/transcribe"; + +describe("transcribeAudioFile", () => { + it("returns mock text when llm base url is unset", async () => { + const original = process.env.LLM_BASE_URL; + const originalProvider = process.env.LLM_PROVIDER; + delete process.env.LLM_BASE_URL; + delete process.env.LLM_PROVIDER; + + const text = await transcribeAudioFile( + new Blob(["audio"], { type: "audio/wav" }), + "recording.wav", + ); + assert.match(text, /milk/i); + + if (original === undefined) delete process.env.LLM_BASE_URL; + else process.env.LLM_BASE_URL = original; + if (originalProvider === undefined) delete process.env.LLM_PROVIDER; + else process.env.LLM_PROVIDER = originalProvider; + }); +}); diff --git a/tests/unit/assistant-chat-storage.test.ts b/tests/unit/assistant-chat-storage.test.ts index 182dbfd..1dbbe80 100644 --- a/tests/unit/assistant-chat-storage.test.ts +++ b/tests/unit/assistant-chat-storage.test.ts @@ -4,6 +4,7 @@ import { clearAssistantChat, loadAssistantChat, saveAssistantChat, + toClientChatMessage, } from "../../src/modules/agent/assistant-chat-storage"; const storage = new Map(); @@ -49,4 +50,15 @@ describe("assistant chat storage", () => { clearAssistantChat("user-a"); assert.deepEqual(loadAssistantChat("user-a"), []); }); + + it("maps stored image messages to client attachments", () => { + const message = toClientChatMessage({ + role: "user", + content: "read this", + imageUrl: "/api/uploads/assistant/home/photo.jpg", + }); + assert.deepEqual(message.attachments, [ + { type: "image", url: "/api/uploads/assistant/home/photo.jpg" }, + ]); + }); }); diff --git a/tests/unit/llm-content.test.ts b/tests/unit/llm-content.test.ts new file mode 100644 index 0000000..864b692 --- /dev/null +++ b/tests/unit/llm-content.test.ts @@ -0,0 +1,27 @@ +import assert from "node:assert/strict"; +import { describe, it } from "node:test"; +import { buildVisionContentParts, textFromMessageContent } from "../../src/lib/llm/content"; + +describe("llm content helpers", () => { + it("reads plain string content", () => { + assert.equal(textFromMessageContent("hello"), "hello"); + }); + + it("joins text parts from multimodal content", () => { + assert.equal( + textFromMessageContent([ + { type: "text", text: "first" }, + { type: "image_url", image_url: { url: "data:image/png;base64,abc" } }, + { type: "text", text: "second" }, + ]), + "first\nsecond", + ); + }); + + it("builds vision parts with fallback prompt when text is empty", () => { + const parts = buildVisionContentParts("", ["data:image/jpeg;base64,abc"]); + assert.equal(parts.length, 2); + assert.equal(parts[0]?.type, "text"); + assert.equal(parts[1]?.type, "image_url"); + }); +});