feat(agent): add voice input and photo attachments to assistant

Wire mic through Whisper-compatible transcriptions on LLM_BASE_URL.

Photos upload to MinIO and reach the vision model as base64 image_url parts.
This commit is contained in:
ginnoir
2026-07-05 02:01:48 -05:00
parent 876a283d47
commit c8db5475d3
20 changed files with 636 additions and 65 deletions
+2 -15
View File
@@ -1,24 +1,11 @@
import { z } from "zod";
import { apiError, apiJson } from "@/lib/api-handler";
import { resolveApiAuth } from "@/lib/api-auth";
import { getAssistantPreferences, resolveAssistantSystemPrompt } from "@/lib/assistant-preference";
import { isLlmConfigured } from "@/lib/llm";
import { clientChatInputSchema } from "@/modules/agent/messages";
import { encodeSseEvent } from "@/modules/agent/server/progress";
import { runAgentChat } from "@/modules/agent/server/run";
const chatInput = z.object({
stream: z.boolean().optional(),
messages: z
.array(
z.object({
role: z.enum(["user", "assistant"]),
content: z.string().trim().min(1).max(8000),
}),
)
.min(1)
.max(40),
});
export async function POST(request: Request) {
const auth = await resolveApiAuth(request);
if (!auth?.userId) {
@@ -39,7 +26,7 @@ export async function POST(request: Request) {
return apiError("Invalid JSON body", 400);
}
const parsed = chatInput.safeParse(body);
const parsed = clientChatInputSchema.safeParse(body);
if (!parsed.success) {
return apiError(parsed.error.issues[0]?.message ?? "Validation error", 400);
}