Files
metona-ai-desktop/electron/harness/tools/built-in/network-utils.ts
T
thzxx 3940716dc2
CI / 类型检查 + Lint + 单元测试 (push) Failing after 5m45s
CI / 全量测试 (Electron ABI) (push) Failing after 5m22s
CI / 产物编译验证 (push) Successful in 10m3s
feat: v0.7.0 四阶段全量迭代 — 修复面收口 · 安全纵深 · 架构还债 · 能力演进
P1 修复面收口: v0.6.3 截断自愈推全量(Anthropic/Ollama/非流式/引擎兜底); SSE 上游错误帧检测进重试通道;
clearMessages 摘要游标根治; truncateResult 内联图片白名单统一; 前端四 bug(确认弹窗锁死/MemoryViewer/
Virtuoso Footer/abort 尾部过滤) + reasoning 缓冲跨迭代污染; 托盘通知过滤与新建会话死链接线

P2 安全纵深: MCP 审批闭环(ConfirmationHook×PolicyEngine 联动+重名拒注册); SSRF 收敛 ssrf-guard 共享模块
(web_fetch 双通道校验+重定向终态复检); Electron 加固(preload CJS 化→sandbox:true/CSP/权限白名单/will-navigate);
run_command cmd.exe 白名单通道元字符守门; diff_viewer 10MB 预检; Anthropic thinking 预算下限; Agnes 思考显式关闭

P3 架构还债: OpenAICompatibleAdapter 中间基类收敛四家样板; 错误分类单轨化(删 mapError/getFetchSignal,
超时显式 ETIMEDOUT); PRAGMA user_version 迁移版本化; 死代码清理专项(cn.ts/SHORTCUTS/ContextMenu 分支/
getWindowState/modifiedArgs/sandbox 空壳); i18next 引入; a11y 第一轮; SearXNG 页批量草稿模型统一

P4 能力演进: Ollama pull 可取消/capabilities 探测/num_ctx 实测缓存; UpdateService feed 比对式自动更新
(app:updateCheck IPC + StatusBar 入口); MiMo providerOptions(web_search 服务端工具/strict JSON);
web_fetch extract_mode=markdown(turndown); network.proxyUrl 全局代理(Chromium sessions+undici dispatcher)

测试: 264 → 507 用例(Electron ABI 全绿零跳过), 覆盖引擎压缩管线/重试竞速/MEMORY.md 闸门/file_editor 五操作/
filesystem 七工具实体夹具/git 真实仓库/SSE 错误帧/全线截断自愈/Provider 请求形态矩阵/SSRF 表测/钩子分级矩阵/
OutputValidator 全量/SLO 指标/MCP 安全纯函数/task_manager 链路/渲染层纯域/i18n 桥契约
2026-08-27 17:06:58 +08:00

326 lines
11 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
/**
* 网络工具共享模块
*
* 提供 UA 轮换池、反爬请求头、HTML 转文本、拦截检测、超时 fetch、LRU 缓存等通用能力。
* web_search 和 web_fetch 共享此模块。
*
* @see docs/Agent网络工具通用设计-v2.md — 第 3 章 web_fetch 抓取设计
*/
import { LRUCache } from 'lru-cache';
import log from 'electron-log';
// ===== LRU 缓存 =====
/** 搜索结果缓存(200 条,5 分钟 TTL) */
export const searchCache = new LRUCache<string, Record<string, unknown>>({
max: 200,
ttl: 300_000,
});
/** 浏览器回退缓存(100 条,10 分钟 TTL) */
export const fetchCache = new LRUCache<string, string>({
max: 100,
ttl: 600_000,
});
// ===== UA 轮换池 =====
export const UA_POOL = [
'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36',
'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Edg/131.0.0.0 Safari/537.36',
'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36',
'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/18.1 Safari/605.1.15',
'Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:133.0) Gecko/20100101 Firefox/133.0',
];
export const MOBILE_UA = 'Mozilla/5.0 (iPhone; CPU iPhone OS 17_5 like Mac OS X) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/17.5 Mobile/15E148 Safari/604.1';
export const ACCEPT_LANGUAGE_POOL = [
'zh-CN,zh;q=0.9,en;q=0.8',
'zh-CN,zh;q=0.9',
'en-US,en;q=0.9,zh-CN;q=0.8',
];
// ===== 反爬请求头构建 =====
export function buildAntiCrawlHeaders(
url: string,
attempt: number,
mobileUA = false,
): Record<string, string> {
const uaIdx = attempt % UA_POOL.length;
const langIdx = attempt % ACCEPT_LANGUAGE_POOL.length;
const userAgent = mobileUA ? MOBILE_UA : UA_POOL[uaIdx];
let origin = '';
try { origin = new URL(url).origin; } catch { /* ignore */ }
return {
'User-Agent': userAgent,
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8',
'Accept-Language': ACCEPT_LANGUAGE_POOL[langIdx],
'Accept-Encoding': 'gzip, deflate, br',
'Cache-Control': 'no-cache',
'DNT': '1',
'Referer': origin || '',
'Sec-Fetch-Dest': 'document',
'Sec-Fetch-Mode': 'navigate',
'Sec-Fetch-Site': 'none',
'Sec-Fetch-User': '?1',
'Pragma': 'no-cache',
};
}
// ===== 超时 fetch =====
export async function fetchWithTimeout(
url: string,
options: RequestInit = {},
timeoutMs = 20_000,
): Promise<Response> {
const controller = new AbortController();
const timer = setTimeout(() => controller.abort(), timeoutMs);
try {
return await fetch(url, { ...options, signal: controller.signal });
} finally {
clearTimeout(timer);
}
}
// ===== URL 标准化(去重用) =====
/**
* H-6 增强: URL 标准化用于搜索结果去重
*
* 规范推荐使用 normalize-url 库(@see docs/Agent网络工具通用设计-v2.md §6.1.3),
* 但当前实现已覆盖核心场景,且避免 ESM-only 依赖兼容性风险,
* 故在现有基础上增强以下能力(对标 normalize-url 默认行为):
* 1. 去除追踪参数(utm_*, gclid, fbclid
* 2. 强制小写 host
* 3. 去除尾部斜杠(根路径除外)
* 4. H-6 新增: 去除默认端口(http→:80, https→:443
* 5. H-6 新增: 排序查询参数(避免 ?a=1&b=2 vs ?b=2&a=1 被视为不同 URL
*/
export function normalizeUrl(url: string): string {
try {
const u = new URL(url);
// 去除追踪参数
const trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content', 'gclid', 'fbclid'];
for (const p of trackingParams) u.searchParams.delete(p);
// H-6 增强: 排序查询参数(确保参数顺序一致,便于去重)
// searchParams.sort() 原地排序,URLSearchParams 按码点顺序
u.searchParams.sort();
// 去除尾部斜杠(根路径除外)
let path = u.pathname;
if (path.length > 1 && path.endsWith('/')) path = path.slice(0, -1);
// H-6 增强: 去除默认端口
// http 默认 :80, https 默认 :443, ws 默认 :80, wss 默认 :443
const isDefaultPort =
(u.protocol === 'http:' && u.port === '80') ||
(u.protocol === 'https:' && u.port === '443') ||
(u.protocol === 'ws:' && u.port === '80') ||
(u.protocol === 'wss:' && u.port === '443');
const portSuffix = isDefaultPort ? '' : (u.port ? `:${u.port}` : '');
// 强制小写 host
return `${u.protocol}//${u.hostname.toLowerCase()}${portSuffix}${path}${u.search}${u.hash}`;
} catch {
return url;
}
}
// ===== 拦截页面检测 =====
const INTERCEPTION_PATTERNS = [
/just a moment/i,
/attention required/i,
/challenge-platform/i,
/\.cf-challenge-/i,
/access denied/i,
/403 forbidden/i,
/请启用\s*javascript/i,
/please enable javascript/i,
/checking your browser/i,
/ddos protection/i,
];
export function isInterceptedPage(html: string): boolean {
if (html.length < 80) return true;
const lower = html.toLowerCase();
return INTERCEPTION_PATTERNS.some((p) => p.test(lower));
}
// ===== HTML → 纯文本转换 =====
const HTML_ENTITY_MAP: Record<string, string> = {
'&nbsp;': ' ', '&lt;': '<', '&gt;': '>', '&amp;': '&', '&quot;': '"',
'&apos;': "'", '&hellip;': '…', '&mdash;': '—', '&ndash;': '',
'&laquo;': '«', '&raquo;': '»', '&times;': '×', '&divide;': '÷',
'&copy;': '©', '&reg;': '®', '&trade;': '™', '&euro;': '€',
'&pound;': '£', '&yen;': '¥', '&cent;': '¢', '&deg;': '°',
};
export function htmlToText(html: string): string {
return html
// 移除噪声标签及内容
.replace(/<script[^>]*>[\s\S]*?<\/script>/gi, '')
.replace(/<style[^>]*>[\s\S]*?<\/style>/gi, '')
.replace(/<noscript[^>]*>[\s\S]*?<\/noscript>/gi, '')
.replace(/<nav[^>]*>[\s\S]*?<\/nav>/gi, '')
.replace(/<header[^>]*>[\s\S]*?<\/header>/gi, '')
.replace(/<footer[^>]*>[\s\S]*?<\/footer>/gi, '')
.replace(/<aside[^>]*>[\s\S]*?<\/aside>/gi, '')
.replace(/<iframe[^>]*>[\s\S]*?<\/iframe>/gi, '')
.replace(/<svg[^>]*>[\s\S]*?<\/svg>/gi, '')
// 移除 HTML 注释
.replace(/<!--[\s\S]*?-->/g, '')
// 块级标签转换行
.replace(/<\/?(p|div|h[1-6]|li|tr|blockquote|section|article|pre|br|hr)[^>]*>/gi, '\n')
// 表格单元格转制表符
.replace(/<\/?(td|th)[^>]*>/gi, '\t')
// 移除剩余标签
.replace(/<[^>]+>/g, '')
// 解码 HTML 实体
.replace(/&#(\d+);/g, (_, n) => String.fromCharCode(Number(n)))
.replace(/&#x([0-9a-f]+);/gi, (_, h) => String.fromCharCode(parseInt(h, 16)))
.replace(/&[a-z]+;/gi, (m) => HTML_ENTITY_MAP[m.toLowerCase()] ?? m)
// 清理空白
.replace(/\n{3,}/g, '\n\n')
.replace(/[ \t]+/g, ' ')
.replace(/^[ \t]+/gm, '')
.trim();
}
// ===== 流式读取(大文件保护,10MB 上限) =====
export async function readBodyWithLimit(response: Response, maxBytes = 10 * 1024 * 1024): Promise<string> {
const contentLength = response.headers.get('content-length');
if (contentLength && parseInt(contentLength) > maxBytes) {
throw new Error(`Response too large: ${contentLength} bytes (limit: ${maxBytes})`);
}
const reader = response.body?.getReader();
if (!reader) return '';
const chunks: Uint8Array[] = [];
let totalBytes = 0;
try {
while (true) {
const { done, value } = await reader.read();
if (done) break;
if (value) {
totalBytes += value.byteLength;
if (totalBytes > maxBytes) {
throw new Error(`Response exceeded ${maxBytes} bytes limit`);
}
chunks.push(value);
}
}
} finally {
reader.releaseLock();
}
const decoder = new TextDecoder('utf-8', { fatal: false });
return decoder.decode(Buffer.concat(chunks));
}
// ===== SearXNG 认证头构建 =====
export function buildSearXNGAuthHeaders(authKey: string, authType: string): Record<string, string> {
if (!authKey) return {};
if (authType === 'bearer') {
return { Authorization: `Bearer ${authKey}` };
}
if (authType === 'basic') {
return { Authorization: `Basic ${Buffer.from(authKey).toString('base64')}` };
}
return {};
}
// ===== 日志辅助 =====
export function logTool(toolName: string, message: string): void {
log.info(`[Tool:${toolName}] ${message}`);
}
// ===== v0.6.4 P4-4: HTML → Markdown 转换(web_fetch extract_mode='markdown' =====
//
// v0.6.4 收尾:私有 npm 凭据解锁后,按开发规范第一铁律把第一轮的临时自写实现
// 替换为 turndown(成熟库)。对外函数签名与行为契约保持不变:
// h1-h6(atx) / 段落 / 链接 / 图片 / strong+em+code 行内 / pre 围栏代码块 /
// ul('-') 与 ol(数字) 列表(跨空行合并为紧凑形态) / blockquote / hr('---') /
// 表格等未知块降级为纯文本、<script/style/svg/noscript/iframe> 整体剔除。
import TurndownService from 'turndown';
const turndown = new TurndownService({
headingStyle: 'atx',
bulletListMarker: '-',
codeBlockStyle: 'fenced',
emDelimiter: '*',
});
// 噪声节点显式剔除(与 htmlToText 的剥离口径一致)
turndown.remove(['script', 'style', 'noscript', 'iframe', 'svg']);
// hr 输出 GitHub 风格 '---'turndown 默认 '* * *'
turndown.addRule('hr-rule', {
filter: ['hr'],
replacement: () => '\n\n---\n\n',
});
/** 列表项行判定:'- xxx' 或 '1. xxx'(允许前导空白) */
const LIST_LINE = /^\s*(?:- |\d+\. )/;
/**
* 紧凑化 + 规范化列表 —— turndown 对松散列表(li 之间带空白文本节点的常见书写)
* 输出条目间空行,且标记为 '- ' / '1. ' 多空格形态。这里做单趟扫描:
* 1. 归一化条目标记为紧凑形态('- ' / 'N. ');
* 2. 仅当"空行两侧都是同一列表的条目行"时移除该空行(绝不吞条目、不影响段落间距)。
*/
function collapseListGaps(markdown: string): string {
const lines = markdown.split('\n').map((line) =>
line
.replace(/^(\s*)- {2,}/, '$1- ')
.replace(/^(\s*\d+\.)\s{2,}/, '$1 '),
);
const isListItem = (l: string | undefined): boolean => (l ?? '').length > 0 && LIST_LINE.test(l!);
const out: string[] = [];
for (let i = 0; i < lines.length; i++) {
const line = lines[i];
if (line.trim() === '') {
const prev = out.length > 0 ? out[out.length - 1] : undefined;
const next = i + 1 < lines.length ? lines[i + 1] : undefined;
// 空行夹在两个列表项之间 → 移除;否则保留原始段落间隔
if (isListItem(prev) && isListItem(next)) continue;
out.push(line);
continue;
}
out.push(line);
}
return out.join('\n');
}
export function htmlToMarkdown(html: string): string {
if (!html || !html.trim()) return '';
let md: string;
try {
md = turndown.turndown(html);
} catch {
// 极端畸形输入时降级为空串(调用方已具备 Phase1 文本回退能力)
logTool?.('htmlToMarkdown', 'turndown conversion failed');
return '';
}
return collapseListGaps(md).replace(/\n{3,}/g, '\n\n').trim();
}