Files
thzxx e4d81d8247 feat: 升级至 v0.3.1 — 全量代码审计修复 + 安全增强
本次升级基于完整代码审查,修复 Critical/High/Medium/Low 四级共 96 项问题,
并通过返工审计修复 10 项遗留问题,tsc 双端类型检查零错误。

Critical (10/10 完成):
- C-4: command.ts 接入 shell-quote 进行 token-level 注入检测,替代原有正则匹配
  可防御 r"m" -rf /、$'rm'、$(echo rm) 等字符串拼接绕过

High (11/11 完成):
- 竞态保护、Promise.allSettled、AbortController 资源泄漏、IPC 参数校验等

Medium (55/55 完成):
- 事务保护、敏感数据脱敏、枚举校验、MUI v9 Stack prop 迁移、
  React 组件 cancelled 标志、类型收窄等

Low (20/20 完成):
- 辅助方法提取(flushToolCallBuffer/scoreAndPushMemory/tryAddColumn 等)
- nanoid 统一替代 Date.now()+Math.random()
- confirm() 替换为 MUI Dialog、useMemo 缓存、魔法数字命名化等

返工审计修复 (10/10 完成):
- L-11: LogsSettings 残留的原生 confirm()/alert() 全部替换为 MUI Dialog/Alert
- M-53: MemoryViewer handleSearch 独立 ref,修复 searching 状态卡死
- M-42: 脱敏短值(length <= 4)泄露修复
- M-47: tasks:update 补全 title/description 类型校验
- L-9: ollama.adapter 非流式路径 nanoid 统一
- M-45: audit:query limit 策略与 memory:listAll 一致化
- SettingsModal handleConfirmRemove 补全 try/catch + loadServers cleanup
- L-15: CommandPalette useMemo 补全 sessions 响应式依赖
- useAgentStream 事件类型补全 seq/timestamp 字段

新增依赖: shell-quote + @types/shell-quote
版本号: 0.3.0 -> 0.3.1
2026-07-13 22:36:58 +08:00

260 lines
9.9 KiB
TypeScript
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
/**
* web_fetch — 网页抓取工具
*
* 三阶段回退策略:
* Phase 1: HTTP 抓取(UA 轮换 + 反爬请求头 + 指数退避重试 + 拦截检测)
* Phase 2: 内容过短自动升级(< 200 字符 → 浏览器渲染)
* Phase 3: 浏览器回退(共享 BrowserWindowManager + JS 渲染 + 内容提取)
*
* 浏览器回退使用与 web_browser 相同的 BrowserWindowManager 单例,
* 避免创建多个独立浏览器窗口,支持窗口复用。
*
* @see docs/Agent网络工具通用设计-v2.md — 第 3 章 web_fetch 抓取设计
*/
import type { IMetonaTool, ToolExecutionContext } from '../../types/metona-tool';
import type { MetonaToolDef } from '../../../harness/types';
import { MetonaToolCategory, MetonaRiskLevel } from '../../../harness/types';
import {
fetchCache,
buildAntiCrawlHeaders,
fetchWithTimeout,
htmlToText,
isInterceptedPage,
readBodyWithLimit,
logTool,
} from './network-utils';
import { getBrowserManager } from './browser';
// ===== 跳过重试的状态码 =====
const SKIP_RETRY_STATUS = new Set([403, 429, 502, 503]);
// ===== WebFetchTool =====
export class WebFetchTool implements IMetonaTool {
readonly definition: MetonaToolDef = {
name: 'web_fetch',
description: 'Fetch a web page and convert to plain text. Uses a three-phase fallback strategy: HTTP fetch with anti-crawl headers → SPA auto-upgrade → browser rendering. Handles Cloudflare interception, JavaScript-rendered pages, and large files (10MB limit).',
parameters: {
type: 'object',
properties: {
url: { type: 'string', description: 'Target URL (http/https only)' },
// H-3/H-4 修复: 补齐规范要求的 max_chars 和 extract_mode 参数
// @see docs/Agent网络工具通用设计-v2.md — 第 3 章 web_fetch 抓取设计
max_chars: { type: 'number', description: 'Maximum characters to return (default 50000, truncated with notice)' },
extract_mode: { type: 'string', enum: ['text', 'html'], description: 'Content extraction mode: "text"=plain text (default), "html"=cleaned HTML with scripts/styles removed' },
mobile_ua: { type: 'boolean', description: 'Use mobile User-Agent (default false)' },
retry: { type: 'boolean', description: 'Enable retry with exponential backoff (default true)' },
},
required: ['url'],
},
category: MetonaToolCategory.NETWORK,
riskLevel: MetonaRiskLevel.LOW,
requiresPermission: false,
timeoutMs: 120_000,
};
async execute(args: Record<string, unknown>, _context: ToolExecutionContext): Promise<unknown> {
const url = args.url as string;
const mobileUA = (args.mobile_ua as boolean) ?? false;
const enableRetry = (args.retry as boolean) ?? true;
// H-3/H-4 修复: 读取 max_chars 和 extract_mode 参数
const maxChars = (args.max_chars as number) ?? 50_000;
const extractMode = ((args.extract_mode as string) ?? 'text') as 'text' | 'html';
if (!url || !/^https?:\/\//i.test(url)) {
return { url, content: '', success: false, error: 'URL must start with http:// or https://' };
}
// 先查缓存(HTTP 和浏览器阶段共享同一缓存)
const cached = fetchCache.get(url);
if (cached) {
logTool('web_fetch', `Cache hit: ${url}`);
return this.buildSuccess(url, cached, 'cache', maxChars);
}
logTool('web_fetch', `Fetching: ${url}`);
// ===== Phase 1: HTTP 抓取 =====
const phase1Result = await this.httpFetch(url, mobileUA, enableRetry);
if (phase1Result.success && !phase1Result.intercepted) {
// 根据 extract_mode 选择返回内容:'html' 模式返回清理后的 HTML'text' 模式返回纯文本
const phase1Content = extractMode === 'html' ? phase1Result.html : phase1Result.text;
// 内容过短检测 → Phase 2 升级(仅对 text 模式生效,html 模式不升级)
if (extractMode === 'text' && phase1Content.length < 200) {
logTool('web_fetch', `Phase 2: Content too short (${phase1Content.length} chars), upgrading to browser`);
const browserResult = await this.browserFetch(url);
if (browserResult) {
return this.buildSuccess(url, browserResult, 'browser', maxChars);
}
}
// 写入缓存(仅缓存 text 模式的内容,html 模式不缓存以避免模式混淆)
if (extractMode === 'text') {
fetchCache.set(url, phase1Content);
}
return this.buildSuccess(url, phase1Content, 'http', maxChars, extractMode);
}
// ===== Phase 3: 浏览器回退 =====
logTool('web_fetch', `Phase 3: Falling back to browser (${phase1Result.reason})`);
const browserResult = await this.browserFetch(url);
if (browserResult) {
return this.buildSuccess(url, browserResult, 'browser', maxChars);
}
// 全部失败
return {
url,
content: '',
success: false,
error: `All phases failed. HTTP: ${phase1Result.reason}. Browser fallback also failed.`,
};
}
// ===== Phase 1: HTTP 抓取 =====
private async httpFetch(
url: string,
mobileUA: boolean,
enableRetry: boolean,
): Promise<{ success: boolean; html: string; text: string; intercepted: boolean; reason: string }> {
const maxRetries = enableRetry ? 3 : 1;
const backoffBase = 2_000;
for (let attempt = 0; attempt < maxRetries; attempt++) {
try {
const headers = buildAntiCrawlHeaders(url, attempt, mobileUA);
const response = await fetchWithTimeout(url, { headers, redirect: 'follow' }, 20_000);
// 跳过重试的状态码 → 直接进入浏览器回退
if (SKIP_RETRY_STATUS.has(response.status)) {
return { success: false, html: '', text: '', intercepted: true, reason: `HTTP ${response.status}` };
}
if (!response.ok) {
// 5xx 可重试
if (response.status >= 500 && attempt < maxRetries - 1) {
await this.sleep(backoffBase * Math.pow(2, attempt) + Math.random() * backoffBase * 0.6);
continue;
}
return { success: false, html: '', text: '', intercepted: false, reason: `HTTP ${response.status} ${response.statusText}` };
}
// 读取正文(10MB 限制)
const html = await readBodyWithLimit(response, 10 * 1024 * 1024);
// 拦截检测
if (isInterceptedPage(html)) {
return { success: false, html: '', text: '', intercepted: true, reason: 'Intercepted page detected' };
}
// HTML → 纯文本
const text = htmlToText(html);
// H-3/H-4 修复: 同时保留原始 HTML,供 extract_mode='html' 使用
return { success: true, html, text, intercepted: false, reason: '' };
} catch (err) {
const errorMsg = (err as Error).message;
if (attempt < maxRetries - 1) {
logTool('web_fetch', `Attempt ${attempt + 1} failed: ${errorMsg}, retrying...`);
await this.sleep(backoffBase * Math.pow(2, attempt) + Math.random() * backoffBase * 0.6);
continue;
}
return { success: false, html: '', text: '', intercepted: false, reason: errorMsg };
}
}
return { success: false, html: '', text: '', intercepted: false, reason: 'All retries exhausted' };
}
// ===== Phase 2/3: 浏览器回退(使用共享 BrowserWindowManager 单例) =====
private async browserFetch(url: string): Promise<string | null> {
// 查缓存
const cached = fetchCache.get(url);
if (cached) {
logTool('web_fetch', 'Browser cache hit');
return cached;
}
try {
const manager = getBrowserManager();
// 通过 manager 打开 URL(复用已打开的同 URL 窗口,避免重复创建)
await manager.open({ url });
// 等待 JS 渲染
await this.sleep(2_500);
// 提取页面正文
const text = await manager.evaluate(`
(function() {
var clone = document.body.cloneNode(true);
var noise = clone.querySelectorAll('script, style, noscript, nav, header, footer, aside, iframe, svg');
noise.forEach(function(el) { el.remove(); });
return clone.innerText || '';
})();
`) as string;
if (text && text.trim().length >= 80) {
// 拦截检测(浏览器渲染后仍可能是验证码挑战页)
if (isInterceptedPage(text)) {
logTool('web_fetch', `Browser fetch detected intercepted page: ${url}`);
return null;
}
// 内容大小限制(与 HTTP 阶段一致,防止超大页面耗尽上下文)
const MAX_BROWSER_TEXT = 500_000; // 500K chars
const safeText = text.length > MAX_BROWSER_TEXT
? text.slice(0, MAX_BROWSER_TEXT) + '\n\n[... content truncated ...]'
: text;
// 写缓存
fetchCache.set(url, safeText);
logTool('web_fetch', `Browser fetch success: ${safeText.length} chars`);
return safeText;
}
return null;
} catch (err) {
logTool('web_fetch', `Browser fetch failed: ${(err as Error).message}`);
return null;
}
// 注意:不关闭窗口 — manager 是单例,窗口由 web_browser 或 cleanupBrowser 管理
}
// ===== 辅助方法 =====
private buildSuccess(
url: string,
text: string,
method: string,
maxChars?: number,
extractMode?: 'text' | 'html',
): unknown {
// H-3/H-4 修复: 应用 max_chars 截断,防止过长内容消耗过多 token
let content = text;
let truncated = false;
if (maxChars !== undefined && maxChars > 0 && text.length > maxChars) {
content = text.slice(0, maxChars) + `\n\n[... content truncated at ${maxChars} chars ...]`;
truncated = true;
}
return {
url,
content,
success: true,
method,
length: content.length,
original_length: text.length,
truncated,
extract_mode: extractMode ?? 'text',
};
}
private sleep(ms: number): Promise<void> {
return new Promise((resolve) => setTimeout(resolve, ms));
}
}