diff --git a/README.md b/README.md index 1a42b87..8a72a9a 100644 --- a/README.md +++ b/README.md @@ -10,7 +10,7 @@

- Version + Version License Electron React @@ -867,7 +867,7 @@ npm run format # Prettier 格式化 # ─── 测试 ───────────────────────────────── npm test # 运行单元测试 (Vitest, 系统 Node — audit 套件因 better-sqlite3 ABI 自动跳过) -npm run test:electron # 运行全量单元测试 (Electron Node ABI, 243 用例全执行, 含 SQLite 审计链哈希 + 引擎工具链集成) +npm run test:electron # 运行全量单元测试 (Electron Node ABI, 245 用例全执行, 含 SQLite 审计链哈希 + 引擎工具链集成) npm run test:watch # 测试监听模式 # ─── 构建 ───────────────────────────────── diff --git a/electron/harness/adapters/shared/openai-format.ts b/electron/harness/adapters/shared/openai-format.ts index 21492bf..61db1c8 100644 --- a/electron/harness/adapters/shared/openai-format.ts +++ b/electron/harness/adapters/shared/openai-format.ts @@ -47,9 +47,10 @@ export function buildOpenAICompatibleMessages( }; // 注意:图片(多模态)处理不在此共享函数中。 - // DeepSeek 不支持多模态,images 被静默丢弃是正确行为。 - // Agnes/MiMo 各自的 toNativeRequest 中有独立的 images 处理。 - // 审查修复: #27 曾在此添加 images 处理,但 DeepSeek 不支持多模态会导致 API 400,已撤销。 + // DeepSeek 非 vision 模型不支持多模态,images 被静默丢弃是正确行为 + // (vision 模型在 DeepSeekAdapter.toNativeRequest 中独立处理)。 + // Agnes/MiMo/OpenAI 各自的 toNativeRequest 中有独立的 images 处理。 + // 审查修复: #27 曾在此添加 images 处理,但 DeepSeek 非 vision 模型会导致 API 400,已撤销。 // === Assistant 消息 === if (m.role === 'assistant') { @@ -76,9 +77,9 @@ export function buildOpenAICompatibleMessages( // 否则 LLM 看到 "null" 不知道失败原因,可能重复调用导致死循环 msg.content = m.toolResult.error ? m.toolResult.error - : (typeof m.toolResult.result === 'string' + : typeof m.toolResult.result === 'string' ? m.toolResult.result - : JSON.stringify(m.toolResult.result)); + : JSON.stringify(m.toolResult.result); // #26 修复: 确保 tool 消息 content 不为 undefined // JSON.stringify(undefined) 返回 undefined(非字符串),会导致 content 字段在序列化后消失 // OpenAI/DeepSeek/Agnes API 严格要求 tool 消息必须有 content 字段,缺失会返回 400 diff --git a/electron/harness/utils/__tests__/token-estimator.test.ts b/electron/harness/utils/__tests__/token-estimator.test.ts index 07cf9cc..f6aad19 100644 --- a/electron/harness/utils/__tests__/token-estimator.test.ts +++ b/electron/harness/utils/__tests__/token-estimator.test.ts @@ -37,6 +37,27 @@ describe('estimateStringTokens', () => { }); describe('estimateMessagesTokens', () => { + // ===== v0.5.5: 图片 token 估算(多轮图片记忆场景) ===== + + it('带 images 的消息按每张 1000 tokens 计入(v0.5.5 — 此前完全忽略)', () => { + const textOnly = estimateMessagesTokens([{ content: '看这张图', role: 'user' } as never]); + const withImage = estimateMessagesTokens([ + { content: '看这张图', role: 'user', images: [{ url: 'data:image/jpeg;base64,x' }] } as never, + ]); + // 差值 = 1 张图的估算(1000) + expect(withImage - textOnly).toBe(1000); + + const withThree = estimateMessagesTokens([ + { content: '', role: 'user', images: [{ url: 'a' }, { url: 'b' }, { url: 'c' }] } as never, + ]); + // 3 张图 + 消息开销(content 为空 = 0) + expect(withThree).toBe(3 * 1000 + 4); + }); + + it('无 images 字段的消息行为不变(向后兼容)', () => { + expect(estimateMessagesTokens([{ content: 'abc', role: 'user' } as never])).toBe(4 + 1); + }); + it('每条消息计入结构性开销(4 tokens)', () => { const msgs = [{ content: '' }, { content: '' }]; expect(estimateMessagesTokens(msgs)).toBe(8); // 2 * 4 overhead diff --git a/electron/harness/utils/token-estimator.ts b/electron/harness/utils/token-estimator.ts index d4c4769..0077da0 100644 --- a/electron/harness/utils/token-estimator.ts +++ b/electron/harness/utils/token-estimator.ts @@ -20,17 +20,18 @@ */ // 中日韩统一表意文字 + 全角标点 + 日文假名 + 韩文谚文 -const CJK_REGEX = /[\u4e00-\u9fff\u3400-\u4dbf\u3000-\u303f\uff00-\uffef\u3040-\u309f\u30a0-\u30ff\uac00-\ud7af]/; +const CJK_REGEX = + /[\u4e00-\u9fff\u3400-\u4dbf\u3000-\u303f\uff00-\uffef\u3040-\u309f\u30a0-\u30ff\uac00-\ud7af]/; /** * L-17 修复: 提取魔法系数为命名常量,便于统一调整 * v0.3.18 修复: CJK_TOKEN_RATIO 从 1.5 调整为 1.0,更贴近 BPE 实际值 * @see project_memory.md — Token estimation coefficients */ -const CJK_TOKEN_RATIO = 1.0; // 中文字符(含全角标点、日韩文):1 字符 ≈ 1.0 token(保守,实测 0.6-0.8) -const ASCII_TOKEN_RATIO = 0.25; // ASCII 字符(英文、数字、半角符号):4 字符 ≈ 1 token -const OTHER_TOKEN_RATIO = 1; // 其他 Unicode(emoji 等):1 字符 ≈ 1 token -const MSG_OVERHEAD_TOKENS = 4; // 每条消息的结构性开销(role、分隔符,参考 OpenAI 规范) +const CJK_TOKEN_RATIO = 1.0; // 中文字符(含全角标点、日韩文):1 字符 ≈ 1.0 token(保守,实测 0.6-0.8) +const ASCII_TOKEN_RATIO = 0.25; // ASCII 字符(英文、数字、半角符号):4 字符 ≈ 1 token +const OTHER_TOKEN_RATIO = 1; // 其他 Unicode(emoji 等):1 字符 ≈ 1 token +const MSG_OVERHEAD_TOKENS = 4; // 每条消息的结构性开销(role、分隔符,参考 OpenAI 规范) /** * 估算字符串的 token 数 @@ -55,7 +56,9 @@ export function estimateStringTokens(text: string | null | undefined): number { } // L-17 修复: 使用命名常量替代魔法数字 - return Math.ceil(cjkCount * CJK_TOKEN_RATIO + asciiCount * ASCII_TOKEN_RATIO + otherCount * OTHER_TOKEN_RATIO); + return Math.ceil( + cjkCount * CJK_TOKEN_RATIO + asciiCount * ASCII_TOKEN_RATIO + otherCount * OTHER_TOKEN_RATIO, + ); } /** @@ -63,6 +66,17 @@ export function estimateStringTokens(text: string | null | undefined): number { */ const TOOL_CALL_OVERHEAD_TOKENS = 8; +/** + * v0.5.5: 单张图片的 token 估算 + * + * 多模态图片(vision 类模型)按视觉 token 计费:1024px 压缩图在主流 + * Provider(OpenAI/Anthropic/DeepSeek vision)约 700~1500 tokens,取保守 + * 上界 1000。此前估算器完全忽略 images——带 10 张图的消息被按纯文本 + * 估算,压缩 keepBudget 严重低估,导致压缩后实际 token 仍超限、反复 + * 触发压缩循环;上下文占用显示也严重失真。 + */ +const IMAGE_TOKEN_ESTIMATE = 1000; + /** * 估算多条消息的总 token 数 * @@ -71,16 +85,23 @@ const TOOL_CALL_OVERHEAD_TOKENS = 8; * @param messages 消息列表(content 可为 null,对应仅有 tool_calls 的 assistant 消息) * @returns 估算的 token 数 */ -export function estimateMessagesTokens(messages: Array<{ - content: string | null; - reasoningContent?: string; - toolCalls?: Array<{ id?: string; name?: string; args: Record }>; - toolCallId?: string; -}>): number { +export function estimateMessagesTokens( + messages: Array<{ + content: string | null; + reasoningContent?: string; + toolCalls?: Array<{ id?: string; name?: string; args: Record }>; + toolCallId?: string; + images?: Array<{ url: string }>; + }>, +): number { let total = 0; for (const msg of messages) { total += estimateStringTokens(msg.content); if (msg.reasoningContent) total += estimateStringTokens(msg.reasoningContent); + // v0.5.5: 图片按视觉 token 估算(多轮图片记忆场景,防压缩预算低估) + if (msg.images) { + total += msg.images.length * IMAGE_TOKEN_ESTIMATE; + } if (msg.toolCalls) { for (const tc of msg.toolCalls) { // #50 修复: OpenAI tokenizer 会将 tool_call 的完整结构(id、name、args)都计入 token diff --git a/package.json b/package.json index 07a451a..5b599ab 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "metona-ai-desktop", - "version": "0.5.4", + "version": "0.5.5", "description": "MetonaAI Desktop — 生产级通用 AI Agent 智能体桌面应用", "main": "dist-electron/main/main.js", "author": "Metona Team", diff --git a/src/components/chat/ChatInput.tsx b/src/components/chat/ChatInput.tsx index e4d75dc..ca9f53c 100644 --- a/src/components/chat/ChatInput.tsx +++ b/src/components/chat/ChatInput.tsx @@ -558,7 +558,9 @@ export function ChatInput(): React.JSX.Element { title={ supportsImages ? '附加文件(图片/文本/代码)' - : '附加文件(文本/代码)— DeepSeek 不支持图片' + : multimodalEnabled + ? '附加文件(文本/代码)— 当前模型不支持图片' + : '附加文件(文本/代码)— 多模态未开启(设置 → LLM 配置)' } >