From 929de8e7bdf06ab26787b22c6e29ec79aaac5bae Mon Sep 17 00:00:00 2001 From: npmrun <1549469775@qq.com> Date: Thu, 13 Aug 2026 00:41:22 +0800 Subject: [PATCH] =?UTF-8?q?feat:=20=E6=B7=BB=E5=8A=A0=E5=A4=9A=E6=A8=A1?= =?UTF-8?q?=E6=80=81=E8=BE=93=E5=85=A5=E6=94=AF=E6=8C=81=EF=BC=8C=E5=85=81?= =?UTF-8?q?=E8=AE=B8=E6=96=87=E6=9C=AC=E4=B8=8E=E5=9B=BE=E7=89=87=E6=B7=B7?= =?UTF-8?q?=E5=90=88=E5=8F=91=E9=80=81?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- SPEC-multimodal-input.md | 517 ------------------------------------------ docs/SPEC-multimodal-input.md | 517 ++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 517 insertions(+), 517 deletions(-) delete mode 100644 SPEC-multimodal-input.md create mode 100644 docs/SPEC-multimodal-input.md diff --git a/SPEC-multimodal-input.md b/SPEC-multimodal-input.md deleted file mode 100644 index 46ac03d..0000000 --- a/SPEC-multimodal-input.md +++ /dev/null @@ -1,517 +0,0 @@ -# 多模态输入支持 — 设计规格 - -## 1. 概述 - -为 Agent 对话输入框增加图片上传能力,使消息支持「文本 + 图片」混合内容。后端将图片以 URL 引用方式传递给 LLM(Vercel AI SDK 的 `image` content part),实现多模态对话。 - -### 范围 - -- **支持模态**:仅图片(PNG / JPG / WebP) -- **传输方式**:先上传到服务器再引用 URL(复用已有 `server/api/file/upload.post.ts`) -- **模型能力检测**:前端根据 `llmModels.type` 字段(`text | vision | multimodal`)拦截,非 vision/multimodal 模型时禁用图片上传 -- **限制**:单条消息最多 4 张图片,单张 ≤ 5MB(已有上传限制) - -### 不在范围内 - -- 音频、视频、文件附件 -- Base64 内嵌传输 -- 新建上传 API(复用已有) -- 重连/刷新恢复机制改造(现有 `stream.get.ts` + stream buffer 已够用) -- 数据库 schema 迁移(`agentMessages.content` + `agentMessages.parts` JSON 字段已足够) - ---- - -## 2. 核心数据结构 - -### 2.1 `ContentPart` — 统一多模态消息内容 - -```typescript -// app/types/chat.ts (新增) - -export interface ContentTextPart { - type: "text"; - text: string; -} - -export interface ContentImagePart { - type: "image"; - image: string; // 服务器返回的 URL,如 /static/upload/xxx.png - mimeType: string; // image/png | image/jpeg | image/webp - name?: string; // 原始文件名(用于编辑时展示) -} - -export type ContentPart = ContentTextPart | ContentImagePart; -``` - -### 2.2 `MessagePart` 扩展 - -```typescript -// app/types/chat.ts (修改) - -export type MessagePartType = - | "text" - | "reasoning" - | "tool-call" - | "tool-result" - | "tool-approval" - | "image"; // ← 新增 - -export interface MessagePart { - id: string; - type: MessagePartType; - text?: string; - toolName?: string; - toolCallId?: string; - args?: unknown; - result?: unknown; - state?: string; - // ← 新增 - image?: { - url: string; - mimeType: string; - name?: string; - }; -} -``` - -### 2.3 `ModelOption` 统一扩展 - -在 `AgentChatArea.vue`、`AgentModelSelector.vue`、`AgentToolbar.vue`、`app/pages/chat/[agentSlug]/index.vue` 中各自定义的 `ModelOption` 接口统一新增 `type` 字段: - -```typescript -export interface ModelOption { - id: number; - name: string; - modelId: string; - providerName: string; - supportsTools: number; - type?: "text" | "vision" | "multimodal"; // ← 新增 -} -``` - -> **设计决策**:`type` 为可选字段,因为 guest models API 当前未返回该字段(需同步修复)。前端用 `type === 'vision' || type === 'multimodal'` 判断是否支持图片。 - ---- - -## 3. 前端改造 - -### 3.1 `AgentInput.vue` — 输入框组件 - -**当前状态**:仅 `inputText: ref` + textarea,emit `send: [content: string]`。 - -**改造点**: - -1. **新增图片状态**: - ```typescript - const images = ref<{ url: string; mimeType: string; name: string }[]>([]); - const isUploading = ref(false); - ``` - -2. **模型能力检测**:接收 `supportsVision: boolean` prop,为 false 时隐藏上传按钮、禁用粘贴/拖拽图片。 - -3. **上传交互**: - - 工具栏新增图片上传按钮(``) - - 支持粘贴图片(`paste` 事件检测 `clipboardData.items` 中的 image) - - 支持拖拽图片(`dragover` + `drop` 事件) - - 上传时调用 `POST /api/file/upload`(FormData),成功后 push 到 `images` - - `images.length >= 4` 时禁用上传入口 - -4. **图片预览区**:textarea 上方显示已添加图片的缩略图列表,每张带删除按钮。 - -5. **emit 签名变更**: - ```typescript - // 从 - emit: { send: [content: string] } - // 改为 - emit: { send: [parts: ContentPart[]] } - ``` - 发送时构建 `ContentPart[]`: - ```typescript - const parts: ContentPart[] = []; - if (inputText.value.trim()) { - parts.push({ type: "text", text: inputText.value.trim() }); - } - for (const img of images.value) { - parts.push({ type: "image", image: img.url, mimeType: img.mimeType, name: img.name }); - } - emit("send", parts); - // 清空 - inputText.value = ""; - images.value = []; - ``` - -6. **编辑模式**:接收 `editParts?: ContentPart[]` prop,填充 textarea 文本 + 图片列表。 - -### 3.2 `AgentChatArea.vue` — 聊天区域父组件 - -**改造点**: - -1. `ModelOption` 接口新增 `type` 字段。 -2. `send` emit 签名从 `[content: string]` 改为 `[parts: ContentPart[]]`。 -3. `edit` emit 签名从 `[messageId, content: string]` 改为 `[messageId, parts: ContentPart[]]`。 -4. 新增 `supportsVision` computed,传给 `AgentInput`: - ```typescript - const supportsVision = computed(() => { - const mid = props.modelId; - if (mid === null) return false; - const model = props.models.find((m) => m.id === mid); - return model?.type === "vision" || model?.type === "multimodal"; - }); - ``` -5. 将 `supportsVision` 传给 `AgentInput`。 - -### 3.3 `AgentMessageItem.vue` — 消息项组件 - -**当前状态**:用户消息渲染 `
{{ message.content }}
`。 - -**改造点**: - -1. 用户消息渲染改为遍历 `message.parts`(或 `message.contentParts`): - ```vue -
- -
- ``` - -2. 编辑按钮 emit 改为传 `ContentPart[]`: - ```typescript - // 从 - emit("edit", messageId, message.content) - // 改为 - emit("edit", messageId, userContentParts) - ``` - -3. 新增图片预览(点击放大,可用简单的 modal/overlay)。 - -### 3.4 `AgentMessageParts.vue` — assistant 消息 parts 渲染 - -**改造点**:新增 `image` 类型处理(虽然 assistant 消息一般不含图片,但保持一致性): -```vue - -``` - -### 3.5 `useAgentChat.ts` — 前端 composable - -**当前状态**:`send(content: string, opts?)` → fetch POST `{ content }`。 - -**改造点**: - -1. `send` 方法签名改为 `send(parts: ContentPart[], opts?)`。 - -2. `buildRequestBody` 改为传 `parts`: - ```typescript - body: { - parts, // ← 替代 content - modelId, - enableThinking, - enableTools, - ...opts - } - ``` - -3. **兼容映射** — `loadMessages` 时旧消息 `content: string` 自动映射: - ```typescript - // 加载历史消息时 - function normalizeMessage(m: RawMessage): ChatMessage { - const contentParts: ContentPart[] = m.parts - ? (JSON.parse(m.parts).filter((p: any) => p.type === "text" || p.type === "image")) - : [{ type: "text", text: m.content }]; - // ... - } - ``` - -4. **编辑消息**:`send(parts, { editMessageId })` 时,将 `parts` 发送给后端,后端更新消息内容。 - -5. **重生成**:`send(parts, { regenerate: true })` 时,`parts` 从最后一条用户消息的 `contentParts` 提取,沿用原图片。 - -### 3.6 `app/pages/chat/[agentSlug]/index.vue` — 页面入口 - -**改造点**: - -1. `ModelOption` 接口新增 `type` 字段。 -2. `loadModels()` 中 map 时保留 `type`: - ```typescript - models.value = list.map((m) => ({ - id: m.id, - name: m.name, - modelId: m.modelId, - providerName: m.providerName ?? "—", - supportsTools: m.supportsTools ?? 1, - type: m.type ?? "text", // ← 新增 - })); - ``` -3. `handleSend` 签名从 `(content: string)` 改为 `(parts: ContentPart[])`。 -4. `handleEdit` 签名从 `(messageId, content: string)` 改为 `(messageId, parts: ContentPart[])`。 -5. `handleRegenerate` 改为从最后一条用户消息提取 `contentParts`: - ```typescript - function handleRegenerate() { - const lastUserMsg = [...chat.messages.value].reverse().find((m) => m.role === "user"); - if (lastUserMsg && lastUserMsg.contentParts) { - chat.send(lastUserMsg.contentParts, { regenerate: true }); - } - } - ``` - -### 3.7 `ChatMessage` 类型扩展 - -在 `useAgentChat.ts` 或 `app/types/chat.ts` 中的 `ChatMessage` 接口新增: - -```typescript -export interface ChatMessage { - // ... 现有字段 - content: string; // 保留,纯文本提取(用于标题生成等) - contentParts?: ContentPart[]; // ← 新增,多模态内容 - parts?: MessagePart[]; // 现有,assistant 消息的 reasoning/tool 等 -} -``` - ---- - -## 4. 后端改造 - -### 4.1 `server/api/agents/[agentSlug]/chat/index.post.ts` — Agent chat POST API - -**改造点**: - -1. body 从 `content: string` 改为 `parts: ContentPart[]`。 -2. **兼容处理**:若 `body.content` 为 string(旧客户端),自动包装为 `[{ type: "text", text: content }]`。 - ```typescript - const parts: ContentPart[] = body.parts - ?? (body.content ? [{ type: "text", text: body.content }] : []); - ``` -3. 将 `parts` 传给 `ChatEngine`。 - -### 4.2 `server/service/agent/chat-engine.ts` — 核心聊天引擎 - -**改造点**: - -1. `ChatEngineParams.body` 从 `content: string` 改为 `parts: ContentPart[]`。 - -2. `executeChat` 中 `saveMessage` 调用改为: - ```typescript - // 从 parts 提取纯文本 content - const textContent = parts - .filter((p) => p.type === "text") - .map((p) => p.text) - .join("\n"); - // parts JSON 包含 image part - const partsJson = JSON.stringify(parts); - await saveMessage(sessionId, "user", textContent, partsJson); - ``` - -3. `buildModelMessages` 中 user 消息处理改为: - ```typescript - // 当前:{ type: "text", text: m.content } - // 改为:解析 m.parts JSON,映射为 AI SDK UserContent - const userParts = JSON.parse(m.parts ?? "[]") as ContentPart[]; - if (userParts.length === 0) { - // 兼容旧数据 - return { type: "text", text: m.content }; - } - return userParts.map((p) => { - if (p.type === "text") return { type: "text", text: p.text }; - if (p.type === "image") return { type: "image", image: p.image }; // AI SDK 接受 URL - }); - ``` - > AI SDK 的 `streamText` 的 `messages` 参数中,user 消息的 `content` 可以是 `UserContent` 数组,包含 `{ type: "image", image: url | URL }`。 - -### 4.3 `server/service/agent/session.ts` — 消息持久化 - -**当前状态**:`saveMessage(sessionId, role, content, parts?)` — `content: string`, `parts?: string | null`。 - -**改造点**:无需修改签名。调用方(`chat-engine.ts`)负责从 `ContentPart[]` 提取纯文本 `content` 和序列化 `parts` JSON。 - -### 4.4 `server/api/llm/chat/index.post.ts` — 独立 chat API - -**改造点**: - -1. body 新增 `parts?: ContentPart[]` 字段(与 `content: string` 二选一)。 -2. 若 `parts` 存在,使用 `parts` 构建消息;否则用 `content`。 -3. `messages` 参数中 user 消息支持 `ContentPart[]` 格式。 - -### 4.5 `server/api/agents/[agentSlug]/models/index.get.ts` — Guest models API - -**改造点**:map 时保留 `type` 字段: -```typescript -const list = systemModels.map((m) => ({ - id: m.id, - name: m.name, - modelId: m.modelId, - providerName: "—", - supportsTools: m.supportsTools, - type: m.type, // ← 新增 -})); -``` -同样在 `defaultModelId` 补充逻辑中也加上 `type: row.model.type`。 - ---- - -## 5. 数据流 - -### 5.1 发送消息(含图片) - -``` -用户输入文本 + 选择/粘贴/拖拽图片 - → AgentInput.vue: 上传图片到 /api/file/upload,获得 URL - → 构建 ContentPart[] = [{type:"text",text}, {type:"image",image:url,mimeType}, ...] - → emit("send", parts) - → AgentChatArea → index.vue handleSend(parts) - → useAgentChat.send(parts) - → POST /api/agents/[slug]/chat { parts, modelId, ... } - → chat-engine.executeChat: - - 提取纯文本 → saveMessage(sessionId, "user", textContent, partsJson) - - buildModelMessages: 解析 parts JSON → AI SDK UserContent[] - - streamText({ messages, model, ... }) - → SSE 流式返回 -``` - -### 5.2 加载历史消息 - -``` -loadMessages(sessionId) - → GET /api/agents/[slug]/sessions/[id]/messages - → 返回 { content: string, parts: string | null, ... } - → normalizeMessage: - - 若 parts JSON 含 image part → contentParts = parsed parts - - 若 parts 为空 → contentParts = [{type:"text", text: content}] - → ChatMessage { content, contentParts, parts (assistant reasoning/tools), ... } - → 渲染: AgentMessageItem 遍历 contentParts 显示文本 + 图片 -``` - -### 5.3 编辑消息 - -``` -用户点击编辑 → AgentMessageItem emit("edit", messageId, contentParts) - → AgentChatArea → index.vue handleEdit(messageId, parts) - → useAgentChat.send(parts, { editMessageId: messageId }) - → POST /api/agents/[slug]/chat { parts, editMessageId, ... } - → 后端更新消息: saveMessage(..., textContent, partsJson) - → 重新生成 assistant 回复 -``` - -### 5.4 重生成 - -``` -用户点击重生成 - → index.vue handleRegenerate() - → 找到最后一条 user 消息,提取 contentParts - → useAgentChat.send(contentParts, { regenerate: true }) - → 后端: 删除最后一条 assistant 消息,用原 user parts 重新生成 -``` - ---- - -## 6. 模型能力检测 - -### 前端拦截逻辑 - -```typescript -const supportsVision = computed(() => { - const model = models.value.find((m) => m.id === currentModelId.value); - return model?.type === "vision" || model?.type === "multimodal"; -}); -``` - -- `supportsVision = false` 时: - - `AgentInput` 隐藏图片上传按钮 - - 禁用粘贴图片(paste 事件中忽略 image items) - - 禁用拖拽图片(drop 事件中忽略 image files) - - 若用户切换到非 vision 模型且当前已有图片,toast 提示「当前模型不支持图片,请切换到支持视觉的模型」 - -### 后端无拦截 - -后端不拦截,直接将 `parts` 传给 AI SDK。若模型不支持图片,AI SDK / LLM provider 会返回错误,前端展示错误消息即可。 - ---- - -## 7. 兼容性处理 - -| 场景 | 旧格式 | 新格式 | 兼容策略 | -|------|--------|--------|----------| -| 前端发送 | `{ content: "text" }` | `{ parts: [{type:"text",text}] }` | 后端:`body.parts ?? (body.content ? [{type:"text",text:content}] : [])` | -| 数据库存储 | `content: "text", parts: null` | `content: "text", parts: "[{type:\"text\",...},{type:\"image\",...}]"` | `buildModelMessages` 解析 parts,为空时 fallback 到 content | -| 前端加载 | `content: "text"` | `contentParts: ContentPart[]` | `normalizeMessage`:parts 为空时映射为 `[{type:"text",text:content}]` | -| 编辑消息 | `emit("edit", id, "text")` | `emit("edit", id, parts)` | `handleEdit` 签名变更,旧消息 contentParts 仅含 text | -| 重生成 | `send(content, {regenerate})` | `send(contentParts, {regenerate})` | 从 `ChatMessage.contentParts` 提取 | - ---- - -## 8. 文件变更清单 - -### 前端 - -| 文件 | 变更类型 | 说明 | -|------|----------|------| -| `app/types/chat.ts` | 修改 | 新增 `ContentPart` 类型,扩展 `MessagePartType` + `MessagePart` | -| `app/components/agent/AgentInput.vue` | 修改 | 图片上传、预览、粘贴/拖拽,emit 改为 `ContentPart[]` | -| `app/components/agent/AgentChatArea.vue` | 修改 | `ModelOption` 加 `type`,`send`/`edit` emit 签名变更,传 `supportsVision` | -| `app/components/agent/AgentModelSelector.vue` | 修改 | `ModelOption` 加 `type` | -| `app/components/agent/AgentToolbar.vue` | 修改 | `ModelOption` 加 `type` | -| `app/components/agent/AgentMessageItem.vue` | 修改 | 用户消息渲染遍历 `contentParts`,编辑 emit 传 `ContentPart[]` | -| `app/components/agent/AgentMessageParts.vue` | 修改 | 新增 `image` part 渲染 | -| `app/composables/useAgentChat.ts` | 修改 | `send` 接收 `ContentPart[]`,`buildRequestBody` 传 `parts`,`normalizeMessage` 兼容映射 | -| `app/pages/chat/[agentSlug]/index.vue` | 修改 | `ModelOption` 加 `type`,`loadModels` 保留 `type`,`handleSend`/`handleEdit`/`handleRegenerate` 签名变更 | - -### 后端 - -| 文件 | 变更类型 | 说明 | -|------|----------|------| -| `server/api/agents/[agentSlug]/chat/index.post.ts` | 修改 | body 从 `content` 改为 `parts`,兼容 string | -| `server/service/agent/chat-engine.ts` | 修改 | `ChatEngineParams.body` 改为 `parts`,`executeChat` 提取文本 + 存 parts JSON,`buildModelMessages` 解析 parts 映射为 AI SDK UserContent | -| `server/api/llm/chat/index.post.ts` | 修改 | 新增 `parts` 字段支持 | -| `server/api/agents/[agentSlug]/models/index.get.ts` | 修改 | map 时保留 `type` 字段 | - -### 无需变更 - -| 文件 | 原因 | -|------|------| -| `server/api/file/upload.post.ts` | 已有上传 API,直接复用 | -| `server/constants/upload.ts` | 常量已定义 | -| `packages/drizzle-pkg/lib/schema/agent.ts` | `content` + `parts` 字段已足够,无需迁移 | -| `packages/drizzle-pkg/lib/schema/llm.ts` | `type` 字段已存在 | -| `server/service/agent/session.ts` | `saveMessage` 签名不变,调用方负责提取 | -| `server/api/agents/[agentSlug]/chat/stream.get.ts` | SSE 恢复机制不变 | -| `server/service/llm/model-resolver.ts` | 模型解析逻辑不变 | - ---- - -## 9. 边界情况 - -1. **空消息**:`parts` 仅含图片无文本 → 允许发送(LLM 可纯图片理解) -2. **仅文本**:`parts` 仅含一个 text part → 正常发送 -3. **图片上传失败**:toast 提示,不阻塞文本发送 -4. **切换模型**:从 vision 模型切到 text 模型,已有图片时 toast 提示,不自动清除图片(用户可选择切换回去或手动删除图片) -5. **编辑旧消息**:旧消息 `contentParts` 仅含 text,编辑时可添加图片 -6. **重生成**:沿用原 user 消息的 `contentParts`(含图片) -7. **超过 4 张图片**:禁用上传入口,toast 提示「最多 4 张图片」 -8. **非图片文件**:粘贴/拖拽时过滤非 image/* MIME 类型 - ---- - -## 10. 测试要点 - -- [ ] 文本 + 图片混合发送,LLM 正确接收并回复 -- [ ] 纯文本发送(向后兼容) -- [ ] 纯图片发送(无文本) -- [ ] 粘贴图片(clipboard) -- [ ] 拖拽图片 -- [ ] 文件选择器上传 -- [ ] 图片预览 + 删除 -- [ ] 最多 4 张限制 -- [ ] 非 vision 模型时上传入口禁用 -- [ ] 切换模型时已有图片的提示 -- [ ] 编辑消息:增删图片后重新发送 -- [ ] 重生成:沿用原图片 -- [ ] 加载历史消息:旧消息(无 parts)正确映射 -- [ ] 加载历史消息:新消息(含 image parts)正确渲染图片 -- [ ] 独立 chat API(`/api/llm/chat`)支持 parts -- [ ] Guest 用户 models API 返回 type 字段 diff --git a/docs/SPEC-multimodal-input.md b/docs/SPEC-multimodal-input.md new file mode 100644 index 0000000..46ac03d --- /dev/null +++ b/docs/SPEC-multimodal-input.md @@ -0,0 +1,517 @@ +# 多模态输入支持 — 设计规格 + +## 1. 概述 + +为 Agent 对话输入框增加图片上传能力,使消息支持「文本 + 图片」混合内容。后端将图片以 URL 引用方式传递给 LLM(Vercel AI SDK 的 `image` content part),实现多模态对话。 + +### 范围 + +- **支持模态**:仅图片(PNG / JPG / WebP) +- **传输方式**:先上传到服务器再引用 URL(复用已有 `server/api/file/upload.post.ts`) +- **模型能力检测**:前端根据 `llmModels.type` 字段(`text | vision | multimodal`)拦截,非 vision/multimodal 模型时禁用图片上传 +- **限制**:单条消息最多 4 张图片,单张 ≤ 5MB(已有上传限制) + +### 不在范围内 + +- 音频、视频、文件附件 +- Base64 内嵌传输 +- 新建上传 API(复用已有) +- 重连/刷新恢复机制改造(现有 `stream.get.ts` + stream buffer 已够用) +- 数据库 schema 迁移(`agentMessages.content` + `agentMessages.parts` JSON 字段已足够) + +--- + +## 2. 核心数据结构 + +### 2.1 `ContentPart` — 统一多模态消息内容 + +```typescript +// app/types/chat.ts (新增) + +export interface ContentTextPart { + type: "text"; + text: string; +} + +export interface ContentImagePart { + type: "image"; + image: string; // 服务器返回的 URL,如 /static/upload/xxx.png + mimeType: string; // image/png | image/jpeg | image/webp + name?: string; // 原始文件名(用于编辑时展示) +} + +export type ContentPart = ContentTextPart | ContentImagePart; +``` + +### 2.2 `MessagePart` 扩展 + +```typescript +// app/types/chat.ts (修改) + +export type MessagePartType = + | "text" + | "reasoning" + | "tool-call" + | "tool-result" + | "tool-approval" + | "image"; // ← 新增 + +export interface MessagePart { + id: string; + type: MessagePartType; + text?: string; + toolName?: string; + toolCallId?: string; + args?: unknown; + result?: unknown; + state?: string; + // ← 新增 + image?: { + url: string; + mimeType: string; + name?: string; + }; +} +``` + +### 2.3 `ModelOption` 统一扩展 + +在 `AgentChatArea.vue`、`AgentModelSelector.vue`、`AgentToolbar.vue`、`app/pages/chat/[agentSlug]/index.vue` 中各自定义的 `ModelOption` 接口统一新增 `type` 字段: + +```typescript +export interface ModelOption { + id: number; + name: string; + modelId: string; + providerName: string; + supportsTools: number; + type?: "text" | "vision" | "multimodal"; // ← 新增 +} +``` + +> **设计决策**:`type` 为可选字段,因为 guest models API 当前未返回该字段(需同步修复)。前端用 `type === 'vision' || type === 'multimodal'` 判断是否支持图片。 + +--- + +## 3. 前端改造 + +### 3.1 `AgentInput.vue` — 输入框组件 + +**当前状态**:仅 `inputText: ref` + textarea,emit `send: [content: string]`。 + +**改造点**: + +1. **新增图片状态**: + ```typescript + const images = ref<{ url: string; mimeType: string; name: string }[]>([]); + const isUploading = ref(false); + ``` + +2. **模型能力检测**:接收 `supportsVision: boolean` prop,为 false 时隐藏上传按钮、禁用粘贴/拖拽图片。 + +3. **上传交互**: + - 工具栏新增图片上传按钮(``) + - 支持粘贴图片(`paste` 事件检测 `clipboardData.items` 中的 image) + - 支持拖拽图片(`dragover` + `drop` 事件) + - 上传时调用 `POST /api/file/upload`(FormData),成功后 push 到 `images` + - `images.length >= 4` 时禁用上传入口 + +4. **图片预览区**:textarea 上方显示已添加图片的缩略图列表,每张带删除按钮。 + +5. **emit 签名变更**: + ```typescript + // 从 + emit: { send: [content: string] } + // 改为 + emit: { send: [parts: ContentPart[]] } + ``` + 发送时构建 `ContentPart[]`: + ```typescript + const parts: ContentPart[] = []; + if (inputText.value.trim()) { + parts.push({ type: "text", text: inputText.value.trim() }); + } + for (const img of images.value) { + parts.push({ type: "image", image: img.url, mimeType: img.mimeType, name: img.name }); + } + emit("send", parts); + // 清空 + inputText.value = ""; + images.value = []; + ``` + +6. **编辑模式**:接收 `editParts?: ContentPart[]` prop,填充 textarea 文本 + 图片列表。 + +### 3.2 `AgentChatArea.vue` — 聊天区域父组件 + +**改造点**: + +1. `ModelOption` 接口新增 `type` 字段。 +2. `send` emit 签名从 `[content: string]` 改为 `[parts: ContentPart[]]`。 +3. `edit` emit 签名从 `[messageId, content: string]` 改为 `[messageId, parts: ContentPart[]]`。 +4. 新增 `supportsVision` computed,传给 `AgentInput`: + ```typescript + const supportsVision = computed(() => { + const mid = props.modelId; + if (mid === null) return false; + const model = props.models.find((m) => m.id === mid); + return model?.type === "vision" || model?.type === "multimodal"; + }); + ``` +5. 将 `supportsVision` 传给 `AgentInput`。 + +### 3.3 `AgentMessageItem.vue` — 消息项组件 + +**当前状态**:用户消息渲染 `
{{ message.content }}
`。 + +**改造点**: + +1. 用户消息渲染改为遍历 `message.parts`(或 `message.contentParts`): + ```vue +
+ +
+ ``` + +2. 编辑按钮 emit 改为传 `ContentPart[]`: + ```typescript + // 从 + emit("edit", messageId, message.content) + // 改为 + emit("edit", messageId, userContentParts) + ``` + +3. 新增图片预览(点击放大,可用简单的 modal/overlay)。 + +### 3.4 `AgentMessageParts.vue` — assistant 消息 parts 渲染 + +**改造点**:新增 `image` 类型处理(虽然 assistant 消息一般不含图片,但保持一致性): +```vue + +``` + +### 3.5 `useAgentChat.ts` — 前端 composable + +**当前状态**:`send(content: string, opts?)` → fetch POST `{ content }`。 + +**改造点**: + +1. `send` 方法签名改为 `send(parts: ContentPart[], opts?)`。 + +2. `buildRequestBody` 改为传 `parts`: + ```typescript + body: { + parts, // ← 替代 content + modelId, + enableThinking, + enableTools, + ...opts + } + ``` + +3. **兼容映射** — `loadMessages` 时旧消息 `content: string` 自动映射: + ```typescript + // 加载历史消息时 + function normalizeMessage(m: RawMessage): ChatMessage { + const contentParts: ContentPart[] = m.parts + ? (JSON.parse(m.parts).filter((p: any) => p.type === "text" || p.type === "image")) + : [{ type: "text", text: m.content }]; + // ... + } + ``` + +4. **编辑消息**:`send(parts, { editMessageId })` 时,将 `parts` 发送给后端,后端更新消息内容。 + +5. **重生成**:`send(parts, { regenerate: true })` 时,`parts` 从最后一条用户消息的 `contentParts` 提取,沿用原图片。 + +### 3.6 `app/pages/chat/[agentSlug]/index.vue` — 页面入口 + +**改造点**: + +1. `ModelOption` 接口新增 `type` 字段。 +2. `loadModels()` 中 map 时保留 `type`: + ```typescript + models.value = list.map((m) => ({ + id: m.id, + name: m.name, + modelId: m.modelId, + providerName: m.providerName ?? "—", + supportsTools: m.supportsTools ?? 1, + type: m.type ?? "text", // ← 新增 + })); + ``` +3. `handleSend` 签名从 `(content: string)` 改为 `(parts: ContentPart[])`。 +4. `handleEdit` 签名从 `(messageId, content: string)` 改为 `(messageId, parts: ContentPart[])`。 +5. `handleRegenerate` 改为从最后一条用户消息提取 `contentParts`: + ```typescript + function handleRegenerate() { + const lastUserMsg = [...chat.messages.value].reverse().find((m) => m.role === "user"); + if (lastUserMsg && lastUserMsg.contentParts) { + chat.send(lastUserMsg.contentParts, { regenerate: true }); + } + } + ``` + +### 3.7 `ChatMessage` 类型扩展 + +在 `useAgentChat.ts` 或 `app/types/chat.ts` 中的 `ChatMessage` 接口新增: + +```typescript +export interface ChatMessage { + // ... 现有字段 + content: string; // 保留,纯文本提取(用于标题生成等) + contentParts?: ContentPart[]; // ← 新增,多模态内容 + parts?: MessagePart[]; // 现有,assistant 消息的 reasoning/tool 等 +} +``` + +--- + +## 4. 后端改造 + +### 4.1 `server/api/agents/[agentSlug]/chat/index.post.ts` — Agent chat POST API + +**改造点**: + +1. body 从 `content: string` 改为 `parts: ContentPart[]`。 +2. **兼容处理**:若 `body.content` 为 string(旧客户端),自动包装为 `[{ type: "text", text: content }]`。 + ```typescript + const parts: ContentPart[] = body.parts + ?? (body.content ? [{ type: "text", text: body.content }] : []); + ``` +3. 将 `parts` 传给 `ChatEngine`。 + +### 4.2 `server/service/agent/chat-engine.ts` — 核心聊天引擎 + +**改造点**: + +1. `ChatEngineParams.body` 从 `content: string` 改为 `parts: ContentPart[]`。 + +2. `executeChat` 中 `saveMessage` 调用改为: + ```typescript + // 从 parts 提取纯文本 content + const textContent = parts + .filter((p) => p.type === "text") + .map((p) => p.text) + .join("\n"); + // parts JSON 包含 image part + const partsJson = JSON.stringify(parts); + await saveMessage(sessionId, "user", textContent, partsJson); + ``` + +3. `buildModelMessages` 中 user 消息处理改为: + ```typescript + // 当前:{ type: "text", text: m.content } + // 改为:解析 m.parts JSON,映射为 AI SDK UserContent + const userParts = JSON.parse(m.parts ?? "[]") as ContentPart[]; + if (userParts.length === 0) { + // 兼容旧数据 + return { type: "text", text: m.content }; + } + return userParts.map((p) => { + if (p.type === "text") return { type: "text", text: p.text }; + if (p.type === "image") return { type: "image", image: p.image }; // AI SDK 接受 URL + }); + ``` + > AI SDK 的 `streamText` 的 `messages` 参数中,user 消息的 `content` 可以是 `UserContent` 数组,包含 `{ type: "image", image: url | URL }`。 + +### 4.3 `server/service/agent/session.ts` — 消息持久化 + +**当前状态**:`saveMessage(sessionId, role, content, parts?)` — `content: string`, `parts?: string | null`。 + +**改造点**:无需修改签名。调用方(`chat-engine.ts`)负责从 `ContentPart[]` 提取纯文本 `content` 和序列化 `parts` JSON。 + +### 4.4 `server/api/llm/chat/index.post.ts` — 独立 chat API + +**改造点**: + +1. body 新增 `parts?: ContentPart[]` 字段(与 `content: string` 二选一)。 +2. 若 `parts` 存在,使用 `parts` 构建消息;否则用 `content`。 +3. `messages` 参数中 user 消息支持 `ContentPart[]` 格式。 + +### 4.5 `server/api/agents/[agentSlug]/models/index.get.ts` — Guest models API + +**改造点**:map 时保留 `type` 字段: +```typescript +const list = systemModels.map((m) => ({ + id: m.id, + name: m.name, + modelId: m.modelId, + providerName: "—", + supportsTools: m.supportsTools, + type: m.type, // ← 新增 +})); +``` +同样在 `defaultModelId` 补充逻辑中也加上 `type: row.model.type`。 + +--- + +## 5. 数据流 + +### 5.1 发送消息(含图片) + +``` +用户输入文本 + 选择/粘贴/拖拽图片 + → AgentInput.vue: 上传图片到 /api/file/upload,获得 URL + → 构建 ContentPart[] = [{type:"text",text}, {type:"image",image:url,mimeType}, ...] + → emit("send", parts) + → AgentChatArea → index.vue handleSend(parts) + → useAgentChat.send(parts) + → POST /api/agents/[slug]/chat { parts, modelId, ... } + → chat-engine.executeChat: + - 提取纯文本 → saveMessage(sessionId, "user", textContent, partsJson) + - buildModelMessages: 解析 parts JSON → AI SDK UserContent[] + - streamText({ messages, model, ... }) + → SSE 流式返回 +``` + +### 5.2 加载历史消息 + +``` +loadMessages(sessionId) + → GET /api/agents/[slug]/sessions/[id]/messages + → 返回 { content: string, parts: string | null, ... } + → normalizeMessage: + - 若 parts JSON 含 image part → contentParts = parsed parts + - 若 parts 为空 → contentParts = [{type:"text", text: content}] + → ChatMessage { content, contentParts, parts (assistant reasoning/tools), ... } + → 渲染: AgentMessageItem 遍历 contentParts 显示文本 + 图片 +``` + +### 5.3 编辑消息 + +``` +用户点击编辑 → AgentMessageItem emit("edit", messageId, contentParts) + → AgentChatArea → index.vue handleEdit(messageId, parts) + → useAgentChat.send(parts, { editMessageId: messageId }) + → POST /api/agents/[slug]/chat { parts, editMessageId, ... } + → 后端更新消息: saveMessage(..., textContent, partsJson) + → 重新生成 assistant 回复 +``` + +### 5.4 重生成 + +``` +用户点击重生成 + → index.vue handleRegenerate() + → 找到最后一条 user 消息,提取 contentParts + → useAgentChat.send(contentParts, { regenerate: true }) + → 后端: 删除最后一条 assistant 消息,用原 user parts 重新生成 +``` + +--- + +## 6. 模型能力检测 + +### 前端拦截逻辑 + +```typescript +const supportsVision = computed(() => { + const model = models.value.find((m) => m.id === currentModelId.value); + return model?.type === "vision" || model?.type === "multimodal"; +}); +``` + +- `supportsVision = false` 时: + - `AgentInput` 隐藏图片上传按钮 + - 禁用粘贴图片(paste 事件中忽略 image items) + - 禁用拖拽图片(drop 事件中忽略 image files) + - 若用户切换到非 vision 模型且当前已有图片,toast 提示「当前模型不支持图片,请切换到支持视觉的模型」 + +### 后端无拦截 + +后端不拦截,直接将 `parts` 传给 AI SDK。若模型不支持图片,AI SDK / LLM provider 会返回错误,前端展示错误消息即可。 + +--- + +## 7. 兼容性处理 + +| 场景 | 旧格式 | 新格式 | 兼容策略 | +|------|--------|--------|----------| +| 前端发送 | `{ content: "text" }` | `{ parts: [{type:"text",text}] }` | 后端:`body.parts ?? (body.content ? [{type:"text",text:content}] : [])` | +| 数据库存储 | `content: "text", parts: null` | `content: "text", parts: "[{type:\"text\",...},{type:\"image\",...}]"` | `buildModelMessages` 解析 parts,为空时 fallback 到 content | +| 前端加载 | `content: "text"` | `contentParts: ContentPart[]` | `normalizeMessage`:parts 为空时映射为 `[{type:"text",text:content}]` | +| 编辑消息 | `emit("edit", id, "text")` | `emit("edit", id, parts)` | `handleEdit` 签名变更,旧消息 contentParts 仅含 text | +| 重生成 | `send(content, {regenerate})` | `send(contentParts, {regenerate})` | 从 `ChatMessage.contentParts` 提取 | + +--- + +## 8. 文件变更清单 + +### 前端 + +| 文件 | 变更类型 | 说明 | +|------|----------|------| +| `app/types/chat.ts` | 修改 | 新增 `ContentPart` 类型,扩展 `MessagePartType` + `MessagePart` | +| `app/components/agent/AgentInput.vue` | 修改 | 图片上传、预览、粘贴/拖拽,emit 改为 `ContentPart[]` | +| `app/components/agent/AgentChatArea.vue` | 修改 | `ModelOption` 加 `type`,`send`/`edit` emit 签名变更,传 `supportsVision` | +| `app/components/agent/AgentModelSelector.vue` | 修改 | `ModelOption` 加 `type` | +| `app/components/agent/AgentToolbar.vue` | 修改 | `ModelOption` 加 `type` | +| `app/components/agent/AgentMessageItem.vue` | 修改 | 用户消息渲染遍历 `contentParts`,编辑 emit 传 `ContentPart[]` | +| `app/components/agent/AgentMessageParts.vue` | 修改 | 新增 `image` part 渲染 | +| `app/composables/useAgentChat.ts` | 修改 | `send` 接收 `ContentPart[]`,`buildRequestBody` 传 `parts`,`normalizeMessage` 兼容映射 | +| `app/pages/chat/[agentSlug]/index.vue` | 修改 | `ModelOption` 加 `type`,`loadModels` 保留 `type`,`handleSend`/`handleEdit`/`handleRegenerate` 签名变更 | + +### 后端 + +| 文件 | 变更类型 | 说明 | +|------|----------|------| +| `server/api/agents/[agentSlug]/chat/index.post.ts` | 修改 | body 从 `content` 改为 `parts`,兼容 string | +| `server/service/agent/chat-engine.ts` | 修改 | `ChatEngineParams.body` 改为 `parts`,`executeChat` 提取文本 + 存 parts JSON,`buildModelMessages` 解析 parts 映射为 AI SDK UserContent | +| `server/api/llm/chat/index.post.ts` | 修改 | 新增 `parts` 字段支持 | +| `server/api/agents/[agentSlug]/models/index.get.ts` | 修改 | map 时保留 `type` 字段 | + +### 无需变更 + +| 文件 | 原因 | +|------|------| +| `server/api/file/upload.post.ts` | 已有上传 API,直接复用 | +| `server/constants/upload.ts` | 常量已定义 | +| `packages/drizzle-pkg/lib/schema/agent.ts` | `content` + `parts` 字段已足够,无需迁移 | +| `packages/drizzle-pkg/lib/schema/llm.ts` | `type` 字段已存在 | +| `server/service/agent/session.ts` | `saveMessage` 签名不变,调用方负责提取 | +| `server/api/agents/[agentSlug]/chat/stream.get.ts` | SSE 恢复机制不变 | +| `server/service/llm/model-resolver.ts` | 模型解析逻辑不变 | + +--- + +## 9. 边界情况 + +1. **空消息**:`parts` 仅含图片无文本 → 允许发送(LLM 可纯图片理解) +2. **仅文本**:`parts` 仅含一个 text part → 正常发送 +3. **图片上传失败**:toast 提示,不阻塞文本发送 +4. **切换模型**:从 vision 模型切到 text 模型,已有图片时 toast 提示,不自动清除图片(用户可选择切换回去或手动删除图片) +5. **编辑旧消息**:旧消息 `contentParts` 仅含 text,编辑时可添加图片 +6. **重生成**:沿用原 user 消息的 `contentParts`(含图片) +7. **超过 4 张图片**:禁用上传入口,toast 提示「最多 4 张图片」 +8. **非图片文件**:粘贴/拖拽时过滤非 image/* MIME 类型 + +--- + +## 10. 测试要点 + +- [ ] 文本 + 图片混合发送,LLM 正确接收并回复 +- [ ] 纯文本发送(向后兼容) +- [ ] 纯图片发送(无文本) +- [ ] 粘贴图片(clipboard) +- [ ] 拖拽图片 +- [ ] 文件选择器上传 +- [ ] 图片预览 + 删除 +- [ ] 最多 4 张限制 +- [ ] 非 vision 模型时上传入口禁用 +- [ ] 切换模型时已有图片的提示 +- [ ] 编辑消息:增删图片后重新发送 +- [ ] 重生成:沿用原图片 +- [ ] 加载历史消息:旧消息(无 parts)正确映射 +- [ ] 加载历史消息:新消息(含 image parts)正确渲染图片 +- [ ] 独立 chat API(`/api/llm/chat`)支持 parts +- [ ] Guest 用户 models API 返回 type 字段