feat: v0.5.4 多模态增强 — 多轮图片记忆 + 多模态总开关 + DeepSeek vision 模型支持
多轮图片记忆:
- 此前历史轮次的图片不回传 LLM(attachments 仅存压缩 preview,历史组装
时被丢弃)— 跨轮对话中模型对图片内容"失忆"
- 修复:SessionSummaryService.buildHistoryMessages 从持久化的 attachments
恢复 images(type=image 的 preview base64),历史图片随上下文回传
- Token 控制:最多注入最近 10 张(MAX_HISTORY_IMAGES,从最新消息向前
收集)— 每张 1024px 压缩图约数百至千余 token,无上限会吃满上下文
- 摘要区间(summarizedUntilRowid 之前)的图片不恢复,符合滚动摘要语义
多模态总开关 llm.multimodalEnabled(默认关闭):
- 新配置项:CONFIG_DEFAULTS 种子 + 设置弹框 LLM 配置 Switch +
首次引导向导 LLM 步骤 Switch(含说明文案)
- 上传入口双重判断:总开关 × 模型能力 — 未开启时即使模型支持多模态
也不能上传图片(ChatInput 的选择/拖拽/粘贴统一拦截,Toast 区分
"开关未开启"与"当前模型不支持"两种原因)
- 保存成功后同步 Agent Store 立即生效;App 启动时随 setProvider 加载
DeepSeek vision 模型支持:
- 新增 deepseek-v4-flash-vision-exp(OpenAI image_url content parts 格式,
128K 上下文 / 8K 输出)
- adapter 按 isVisionModel() 判断:vision 模型将带 images 的消息转换为
[{type:'text'},{type:'image_url'}] parts;非 vision 模型保持 images
静默丢弃(防 API 400)
测试(236 → 243 用例):
- 多轮图片记忆 ×3(session-summary.test.ts):历史 attachments 恢复
images / 上限 10 张从最新向前 / 摘要区间图片不恢复
- DeepSeek vision 请求格式 ×4(deepseek-vision.test.ts,契约级 mock
fetch 断言请求体):image_url parts 转换 / 非 vision 模型丢弃 /
max_tokens 钳制 8192 / 无图不转换
- 测试顺序修正:多轮图片用例置于 describe 末尾(插入新行消耗全局自增
rowid,插在中间会破坏既有用例对 rowid 数值的断言)
文档: README 同步(DeepSeek 模型表 + vision 多模态列、llm.multimodalEnabled
配置项、多轮图片记忆特性行、243 用例数)
验证: lint 0 / typecheck 双工程 0 / test:electron 243 全过 / build 成功
This commit is contained in:
@@ -10,6 +10,7 @@
|
||||
* @see apis/deepseek-api-docs-20260518.html
|
||||
*/
|
||||
|
||||
import log from 'electron-log';
|
||||
import { BaseAdapter } from './base-adapter';
|
||||
import type { MetonaRequest, MetonaResponse, MetonaStreamEvent } from '../types';
|
||||
import { MetonaFinishReason } from '../types';
|
||||
@@ -20,7 +21,11 @@ import { parseSSEStream, parseOpenAICompatibleResponse } from './shared/sse-stre
|
||||
export class DeepSeekAdapter extends BaseAdapter {
|
||||
// H-2 修复: provider → providerId(规范要求)
|
||||
override readonly providerId: string = 'deepseek';
|
||||
readonly supportedModels = ['deepseek-v4-pro', 'deepseek-v4-flash'];
|
||||
readonly supportedModels = [
|
||||
'deepseek-v4-pro',
|
||||
'deepseek-v4-flash',
|
||||
'deepseek-v4-flash-vision-exp',
|
||||
];
|
||||
readonly supportsToolCalling = true;
|
||||
readonly supportsThinking = true;
|
||||
|
||||
@@ -44,8 +49,29 @@ export class DeepSeekAdapter extends BaseAdapter {
|
||||
supportsThinking: true,
|
||||
description: 'DeepSeek 快速版,1M 上下文,低延迟推理',
|
||||
},
|
||||
// v0.5.4: DeepSeek 多模态实验模型(OpenAI image_url content parts 格式)
|
||||
'deepseek-v4-flash-vision-exp': {
|
||||
id: 'deepseek-v4-flash-vision-exp',
|
||||
name: 'DeepSeek V4 Flash Vision (Exp)',
|
||||
contextWindow: 128_000,
|
||||
maxOutputTokens: 8_192,
|
||||
supportsToolCalling: true,
|
||||
supportsThinking: false,
|
||||
description: 'DeepSeek 多模态实验模型,支持图片输入(image_url content parts)',
|
||||
},
|
||||
};
|
||||
|
||||
/**
|
||||
* v0.5.4: 当前模型是否支持多模态图片输入
|
||||
*
|
||||
* DeepSeek 仅 vision 系列模型支持图片(命名含 'vision');
|
||||
* 非 vision 模型收到 images 时静默丢弃(避免 API 400)。
|
||||
* 前端上传入口由 llm.multimodalEnabled 配置总开关控制,此处是 adapter 侧的模型级防线。
|
||||
*/
|
||||
private isVisionModel(): boolean {
|
||||
return this.config.defaultModel.includes('vision');
|
||||
}
|
||||
|
||||
// ===== POST /chat/completions (非流式) =====
|
||||
|
||||
// H-2 修复: chat → send(规范要求)
|
||||
@@ -252,6 +278,34 @@ export class DeepSeekAdapter extends BaseAdapter {
|
||||
stream,
|
||||
};
|
||||
|
||||
// v0.5.4: vision 模型的图片处理(OpenAI image_url content parts 格式)
|
||||
// 非 vision 模型保持 images 静默丢弃(共享层行为,避免 API 400)
|
||||
if (this.isVisionModel()) {
|
||||
const nonSystemMsgs = request.messages.filter((m) => m.role !== 'system');
|
||||
let imageCount = 0;
|
||||
// messages[0] 是 system,非 system 消息从 messages[1] 开始(与 nonSystemMsgs 对齐)
|
||||
for (let i = 1; i < messages.length; i++) {
|
||||
const origMsg = nonSystemMsgs[i - 1];
|
||||
if (!origMsg?.images?.length) continue;
|
||||
|
||||
imageCount += origMsg.images.length;
|
||||
const contentParts: Array<Record<string, unknown>> = [];
|
||||
if (origMsg.content) {
|
||||
contentParts.push({ type: 'text', text: origMsg.content });
|
||||
}
|
||||
for (const img of origMsg.images) {
|
||||
contentParts.push({
|
||||
type: 'image_url',
|
||||
image_url: { url: img.url },
|
||||
});
|
||||
}
|
||||
messages[i].content = contentParts;
|
||||
}
|
||||
if (imageCount > 0) {
|
||||
log.info(`[DeepSeek] Vision model processing ${imageCount} image(s)`);
|
||||
}
|
||||
}
|
||||
|
||||
if (stream) {
|
||||
body.stream_options = { include_usage: true };
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user