feat(agent): add voice input and photo attachments to assistant
Wire mic through Whisper-compatible transcriptions on LLM_BASE_URL. Photos upload to MinIO and reach the vision model as base64 image_url parts.
This commit is contained in:
@@ -0,0 +1,28 @@
|
||||
import type { ChatContentPart, ChatMessage } from "./types";
|
||||
|
||||
export function textFromMessageContent(content: ChatMessage["content"]): string {
|
||||
if (!content) return "";
|
||||
if (typeof content === "string") return content;
|
||||
return content
|
||||
.filter((part): part is Extract<ChatContentPart, { type: "text" }> => part.type === "text")
|
||||
.map((part) => part.text)
|
||||
.join("\n")
|
||||
.trim();
|
||||
}
|
||||
|
||||
export function buildVisionContentParts(text: string, imageDataUrls: string[]): ChatContentPart[] {
|
||||
const parts: ChatContentPart[] = [];
|
||||
const trimmed = text.trim();
|
||||
if (trimmed) {
|
||||
parts.push({ type: "text", text: trimmed });
|
||||
} else if (imageDataUrls.length > 0) {
|
||||
parts.push({
|
||||
type: "text",
|
||||
text: "Read this image and help with what the user needs.",
|
||||
});
|
||||
}
|
||||
for (const url of imageDataUrls) {
|
||||
parts.push({ type: "image_url", image_url: { url } });
|
||||
}
|
||||
return parts;
|
||||
}
|
||||
+4
-11
@@ -1,14 +1,5 @@
|
||||
import type { ChatCompletionRequest, ChatCompletionResult, LlmClient } from "./types";
|
||||
|
||||
function lastUserText(messages: ChatCompletionRequest["messages"]): string {
|
||||
for (let i = messages.length - 1; i >= 0; i -= 1) {
|
||||
const message = messages[i];
|
||||
if (message?.role === "user" && message.content) {
|
||||
return message.content.toLowerCase();
|
||||
}
|
||||
}
|
||||
return "";
|
||||
}
|
||||
import { textFromMessageContent } from "./content";
|
||||
|
||||
function hadToolResults(messages: ChatCompletionRequest["messages"]): boolean {
|
||||
return messages.some((message) => message.role === "tool");
|
||||
@@ -17,7 +8,9 @@ function hadToolResults(messages: ChatCompletionRequest["messages"]): boolean {
|
||||
export function createMockLlmClient(): LlmClient {
|
||||
return {
|
||||
async chatCompletion(request: ChatCompletionRequest): Promise<ChatCompletionResult> {
|
||||
const userText = lastUserText(request.messages);
|
||||
const userText = textFromMessageContent(
|
||||
[...request.messages].reverse().find((message) => message.role === "user")?.content ?? "",
|
||||
).toLowerCase();
|
||||
|
||||
if (hadToolResults(request.messages)) {
|
||||
return {
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
import type { ChatCompletionRequest, ChatCompletionResult, LlmClient } from "./types";
|
||||
import type { ChatCompletionRequest, ChatCompletionResult, ChatMessage, LlmClient } from "./types";
|
||||
|
||||
type OpenAiMessage = {
|
||||
role: string;
|
||||
content: string | null;
|
||||
content: ChatMessage["content"];
|
||||
tool_calls?: Array<{
|
||||
id: string;
|
||||
type: "function";
|
||||
|
||||
@@ -0,0 +1,47 @@
|
||||
import { getLlmConfig } from "./config";
|
||||
|
||||
const MAX_AUDIO_BYTES = 25 * 1024 * 1024;
|
||||
|
||||
export async function transcribeAudioFile(file: File | Blob, filename: string): Promise<string> {
|
||||
if (file.size > MAX_AUDIO_BYTES) {
|
||||
throw new Error("Recording exceeds 25 MB limit");
|
||||
}
|
||||
|
||||
const config = getLlmConfig();
|
||||
if (config.provider === "mock" || !config.baseUrl) {
|
||||
return "add milk to the shopping list";
|
||||
}
|
||||
|
||||
const formData = new FormData();
|
||||
formData.append("file", file, filename);
|
||||
formData.append("model", "whisper-1");
|
||||
|
||||
const url = `${config.baseUrl.replace(/\/$/, "")}/audio/transcriptions`;
|
||||
const headers: Record<string, string> = {};
|
||||
if (config.apiKey) {
|
||||
headers.Authorization = `Bearer ${config.apiKey}`;
|
||||
}
|
||||
|
||||
const response = await fetch(url, {
|
||||
method: "POST",
|
||||
headers,
|
||||
body: formData,
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
const detail = await response.text();
|
||||
throw new Error(`Transcription failed (${response.status}): ${detail.slice(0, 400)}`);
|
||||
}
|
||||
|
||||
const contentType = response.headers.get("content-type") ?? "";
|
||||
if (contentType.includes("application/json")) {
|
||||
const payload = (await response.json()) as { text?: string };
|
||||
const text = payload.text?.trim();
|
||||
if (!text) throw new Error("Transcription returned empty text");
|
||||
return text;
|
||||
}
|
||||
|
||||
const text = (await response.text()).trim();
|
||||
if (!text) throw new Error("Transcription returned empty text");
|
||||
return text;
|
||||
}
|
||||
+15
-1
@@ -1,5 +1,19 @@
|
||||
export type ChatRole = "system" | "user" | "assistant" | "tool";
|
||||
|
||||
export type ChatTextPart = {
|
||||
type: "text";
|
||||
text: string;
|
||||
};
|
||||
|
||||
export type ChatImagePart = {
|
||||
type: "image_url";
|
||||
image_url: {
|
||||
url: string;
|
||||
};
|
||||
};
|
||||
|
||||
export type ChatContentPart = ChatTextPart | ChatImagePart;
|
||||
|
||||
export type ChatToolCall = {
|
||||
id: string;
|
||||
type: "function";
|
||||
@@ -11,7 +25,7 @@ export type ChatToolCall = {
|
||||
|
||||
export type ChatMessage = {
|
||||
role: ChatRole;
|
||||
content: string | null;
|
||||
content: string | ChatContentPart[] | null;
|
||||
tool_calls?: ChatToolCall[];
|
||||
tool_call_id?: string;
|
||||
name?: string;
|
||||
|
||||
Reference in New Issue
Block a user