/** * Token 估算工具 — 跨 Provider 通用 * * 策略:智能字符估算,区分中文字符与 ASCII 字符 * - 中文字符(含全角标点、日韩文):1 字符 ≈ 1.5 token * - ASCII 字符(英文、数字、半角符号):4 字符 ≈ 1 token * - 其他 Unicode(emoji 等):1 字符 ≈ 1 token * * 对比旧的 `length / 2` 方案: * - 中文场景:估算准确度从 ~50% 提升到 ~90% * - 英文场景:从偏低变为接近真实 * - 混合场景:更贴近实际 token 消耗 * * 仍为估算值(无 tiktoken 依赖),但留了 80% 触发阈值的缓冲。 */ // 中日韩统一表意文字 + 全角标点 + 日文假名 + 韩文谚文 const CJK_REGEX = /[\u4e00-\u9fff\u3400-\u4dbf\u3000-\u303f\uff00-\uffef\u3040-\u309f\u30a0-\u30ff\uac00-\ud7af]/; /** * L-17 修复: 提取魔法系数为命名常量,便于统一调整 * @see project_memory.md — Token estimation coefficients */ const CJK_TOKEN_RATIO = 1.5; // 中文字符(含全角标点、日韩文):1 字符 ≈ 1.5 token const ASCII_TOKEN_RATIO = 0.25; // ASCII 字符(英文、数字、半角符号):4 字符 ≈ 1 token const OTHER_TOKEN_RATIO = 1; // 其他 Unicode(emoji 等):1 字符 ≈ 1 token const MSG_OVERHEAD_TOKENS = 4; // 每条消息的结构性开销(role、分隔符,参考 OpenAI 规范) /** * 估算字符串的 token 数 * @param text 待估算的字符串(可为 null/undefined,视为 0 token) * @returns 估算的 token 数 */ export function estimateStringTokens(text: string | null | undefined): number { if (!text || text.length === 0) return 0; let cjkCount = 0; let asciiCount = 0; let otherCount = 0; for (const ch of text) { if (CJK_REGEX.test(ch)) { cjkCount++; } else if (ch.charCodeAt(0) < 128) { asciiCount++; } else { otherCount++; } } // L-17 修复: 使用命名常量替代魔法数字 return Math.ceil(cjkCount * CJK_TOKEN_RATIO + asciiCount * ASCII_TOKEN_RATIO + otherCount * OTHER_TOKEN_RATIO); } /** * 估算多条消息的总 token 数 * * 每条消息额外加 4 token 的结构性开销(role、分隔符等,参考 OpenAI 规范) * * @param messages 消息列表(content 可为 null,对应仅有 tool_calls 的 assistant 消息) * @returns 估算的 token 数 */ export function estimateMessagesTokens(messages: Array<{ content: string | null; reasoningContent?: string; toolCalls?: Array<{ args: Record }>; }>): number { let total = 0; for (const msg of messages) { total += estimateStringTokens(msg.content); if (msg.reasoningContent) total += estimateStringTokens(msg.reasoningContent); if (msg.toolCalls) { for (const tc of msg.toolCalls) { total += estimateStringTokens(JSON.stringify(tc.args)); } } // L-17 修复: 使用命名常量替代魔法数字 total += MSG_OVERHEAD_TOKENS; } return total; }