Files
metona-ai-desktop/electron/harness/utils/token-estimator.ts
T

98 lines
3.8 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
/**
* Token 估算工具 — 跨 Provider 通用
*
* 策略:智能字符估算,区分中文字符与 ASCII 字符
* - 中文字符(含全角标点、日韩文):1 字符 ≈ 1.5 token
* - ASCII 字符(英文、数字、半角符号):4 字符 ≈ 1 token
* - 其他 Unicodeemoji 等):1 字符 ≈ 1 token
*
* 对比旧的 `length / 2` 方案:
* - 中文场景:估算准确度从 ~50% 提升到 ~90%
* - 英文场景:从偏低变为接近真实
* - 混合场景:更贴近实际 token 消耗
*
* 仍为估算值(无 tiktoken 依赖),但留了 80% 触发阈值的缓冲。
*/
// 中日韩统一表意文字 + 全角标点 + 日文假名 + 韩文谚文
const CJK_REGEX = /[\u4e00-\u9fff\u3400-\u4dbf\u3000-\u303f\uff00-\uffef\u3040-\u309f\u30a0-\u30ff\uac00-\ud7af]/;
/**
* L-17 修复: 提取魔法系数为命名常量,便于统一调整
* @see project_memory.md — Token estimation coefficients
*/
const CJK_TOKEN_RATIO = 1.5; // 中文字符(含全角标点、日韩文):1 字符 ≈ 1.5 token
const ASCII_TOKEN_RATIO = 0.25; // ASCII 字符(英文、数字、半角符号):4 字符 ≈ 1 token
const OTHER_TOKEN_RATIO = 1; // 其他 Unicodeemoji 等):1 字符 ≈ 1 token
const MSG_OVERHEAD_TOKENS = 4; // 每条消息的结构性开销(role、分隔符,参考 OpenAI 规范)
/**
* 估算字符串的 token 数
* @param text 待估算的字符串(可为 null/undefined,视为 0 token
* @returns 估算的 token 数
*/
export function estimateStringTokens(text: string | null | undefined): number {
if (!text || text.length === 0) return 0;
let cjkCount = 0;
let asciiCount = 0;
let otherCount = 0;
for (const ch of text) {
if (CJK_REGEX.test(ch)) {
cjkCount++;
} else if (ch.charCodeAt(0) < 128) {
asciiCount++;
} else {
otherCount++;
}
}
// L-17 修复: 使用命名常量替代魔法数字
return Math.ceil(cjkCount * CJK_TOKEN_RATIO + asciiCount * ASCII_TOKEN_RATIO + otherCount * OTHER_TOKEN_RATIO);
}
/**
* #50 修复: tool_call 结构开销({"id":"","name":"","arguments":""} 等结构字符,参考 OpenAI 规范)
*/
const TOOL_CALL_OVERHEAD_TOKENS = 8;
/**
* 估算多条消息的总 token 数
*
* 每条消息额外加 4 token 的结构性开销(role、分隔符等,参考 OpenAI 规范)
*
* @param messages 消息列表(content 可为 null,对应仅有 tool_calls 的 assistant 消息)
* @returns 估算的 token 数
*/
export function estimateMessagesTokens(messages: Array<{
content: string | null;
reasoningContent?: string;
toolCalls?: Array<{ id?: string; name?: string; args: Record<string, unknown> }>;
toolCallId?: string;
}>): number {
let total = 0;
for (const msg of messages) {
total += estimateStringTokens(msg.content);
if (msg.reasoningContent) total += estimateStringTokens(msg.reasoningContent);
if (msg.toolCalls) {
for (const tc of msg.toolCalls) {
// #50 修复: OpenAI tokenizer 会将 tool_call 的完整结构(id、name、args)都计入 token
// 之前仅估算 args,忽略 id(通常 24 字符 call_xxx)和 name(通常 5-20 字符),导致每个 tool_call 少算 5-10 tokens
total += estimateStringTokens(tc.id ?? '');
total += estimateStringTokens(tc.name ?? '');
total += estimateStringTokens(JSON.stringify(tc.args ?? {}));
// 结构开销({"id":"","name":"","arguments":""} 等结构字符)
total += TOOL_CALL_OVERHEAD_TOKENS;
}
}
// #50 修复: tool 消息的 tool_call_id 字段也计入 token
if (msg.toolCallId) {
total += estimateStringTokens(msg.toolCallId);
}
// L-17 修复: 使用命名常量替代魔法数字
total += MSG_OVERHEAD_TOKENS;
}
return total;
}