Files
metona-ai-desktop/electron/harness/tools/built-in/web-fetch.ts
T
thzxx 80cf5b482c
CI / 类型检查 + Lint + 单元测试 (push) Failing after 5m41s
CI / 全量测试 (Electron ABI) (push) Failing after 5m21s
CI / 产物编译验证 (push) Successful in 10m1s
fix: v0.6.1 修复回复/推理期间偶发崩溃 — 浏览器回退窗口竞态 + 崩溃可观测性
【根因(实证归因,非猜测)】
分析 userData/logs/main.log 全部 33 次启动会话,定位 3 处异常终止点
(07-25 ×2 / 08-22 ×1,启动标记前无 Database closed)。三处 100% 共享
同一模式:web_search 并行抓取 → 多个 web_fetch 同时进入浏览器回退 →
共享单例 BrowserWindowManager 中后到 open() 销毁前一个正在加载/执行
JS 的窗口。关键统计:56 次浏览器回退中 ERR_ABORTED(并发互毁的直接
证据)仅 3 次,而这 3 次恰好全部对应 3 个崩溃点;无并发销毁的 53 次
回退从未崩溃 —— 触发条件完全收敛。

缺陷链(三层叠加):
1. browserFetch 直接 open/evaluate 共享单例,无跨调用序列化 — 并发
   回退互相销毁窗口(ERR_ABORTED / "Object has been destroyed")
2. destroy() 对仍在使用中的 partition fire-and-forget
   clearStorageData/clearCache,与紧随其后的新窗口创建并发 —
   原生存储层竞态(崩溃引爆点)
3. ensureReady 检查与实际 executeJavaScript/loadURL 之间存在竞态窗口;
   loadURLWithTimeout 的 Race 落败方 rejection 无人处理

【修复(browser-window-manager.ts + web-fetch.ts + browser.ts)】
- 新增 fetchPageText:排队版页面抓取,串行化完整 open→等待→evaluate
  序列(与 open 共用单一操作链,destroy 只会在链上发生,跨链互毁彻底
  消除);web_fetch 浏览器回退改走此入口
- open() 拆分 openInternal(链内直调);open 与 fetchPageText 共用
  单一串行链,排队不分死锁
- destroy() 移除 session 存储清理(终态清理迁移至 close(),await 执行,
  不再与窗口创建并发)
- safeWebContents() 即时校验替代 racy 的 ensureReady;evaluate/extract/
  screenshot/click/type/scroll/waitForSelector 全部加固,消除对已销毁
  webContents 的调用
- loadURLWithTimeout 落败方 rejection 兜底(防 unhandledRejection)
- cleanupBrowser/cleanup/close 异步化适配(main.ts 退出链路 await)

【崩溃可观测性(此前崩溃无迹可查 — 日志无声截断)】
- process.on(uncaughtException/unhandledRejection) → [FATAL] 落盘
- app.on(render-process-gone/child-process-gone) → [FATAL] 落盘
- WindowManager: 每窗口 render-process-gone 日志 + 自动 reload 自愈
  (渲染进程 OOM/崩溃不再白屏卡死,可自动恢复)

【验证】
- lint 0/0;typecheck 双工程 0 错误;test:electron 252/252;build 通过
2026-08-22 19:16:41 +08:00

295 lines
10 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
/**
* web_fetch — 网页抓取工具
*
* 三阶段回退策略:
* Phase 1: HTTP 抓取(UA 轮换 + 反爬请求头 + 指数退避重试 + 拦截检测)
* Phase 2: 内容过短自动升级(< 200 字符 → 浏览器渲染)
* Phase 3: 浏览器回退(共享 BrowserWindowManager + JS 渲染 + 内容提取)
*
* 浏览器回退使用与 web_browser 相同的 BrowserWindowManager 单例,
* 避免创建多个独立浏览器窗口,支持窗口复用。
*
* @see docs/Agent网络工具通用设计-v2.md — 第 3 章 web_fetch 抓取设计
*/
import type { IMetonaTool, ToolExecutionContext } from '../../types/metona-tool';
import type { MetonaToolDef } from '../../../harness/types';
import { MetonaToolCategory, MetonaRiskLevel } from '../../../harness/types';
import {
fetchCache,
buildAntiCrawlHeaders,
fetchWithTimeout,
htmlToText,
isInterceptedPage,
readBodyWithLimit,
logTool,
} from './network-utils';
import { getBrowserManager } from './browser';
// ===== 跳过重试的状态码 =====
const SKIP_RETRY_STATUS = new Set([403, 429, 502, 503]);
// ===== WebFetchTool =====
export class WebFetchTool implements IMetonaTool {
readonly definition: MetonaToolDef = {
name: 'web_fetch',
description:
'Fetch a web page and convert to plain text. Uses a three-phase fallback strategy: HTTP fetch with anti-crawl headers → SPA auto-upgrade → browser rendering. Handles Cloudflare interception, JavaScript-rendered pages, and large files (10MB limit).',
parameters: {
type: 'object',
properties: {
url: { type: 'string', description: 'Target URL (http/https only)' },
// H-3/H-4 修复: 补齐规范要求的 max_chars 和 extract_mode 参数
// @see docs/Agent网络工具通用设计-v2.md — 第 3 章 web_fetch 抓取设计
max_chars: {
type: 'number',
description: 'Maximum characters to return (default 50000, truncated with notice)',
},
extract_mode: {
type: 'string',
enum: ['text', 'html'],
description:
'Content extraction mode: "text"=plain text (default), "html"=cleaned HTML with scripts/styles removed',
},
mobile_ua: { type: 'boolean', description: 'Use mobile User-Agent (default false)' },
retry: {
type: 'boolean',
description: 'Enable retry with exponential backoff (default true)',
},
},
required: ['url'],
},
category: MetonaToolCategory.NETWORK,
riskLevel: MetonaRiskLevel.LOW,
requiresPermission: false,
timeoutMs: 120_000,
};
async execute(args: Record<string, unknown>, _context: ToolExecutionContext): Promise<unknown> {
const url = args.url as string;
const mobileUA = (args.mobile_ua as boolean) ?? false;
const enableRetry = (args.retry as boolean) ?? true;
// H-3/H-4 修复: 读取 max_chars 和 extract_mode 参数
const maxChars = (args.max_chars as number) ?? 50_000;
const extractMode = ((args.extract_mode as string) ?? 'text') as 'text' | 'html';
if (!url || !/^https?:\/\//i.test(url)) {
return { url, content: '', success: false, error: 'URL must start with http:// or https://' };
}
// 先查缓存(HTTP 和浏览器阶段共享同一缓存)
const cached = fetchCache.get(url);
if (cached) {
logTool('web_fetch', `Cache hit: ${url}`);
return this.buildSuccess(url, cached, 'cache', maxChars);
}
logTool('web_fetch', `Fetching: ${url}`);
// ===== Phase 1: HTTP 抓取 =====
const phase1Result = await this.httpFetch(url, mobileUA, enableRetry);
if (phase1Result.success && !phase1Result.intercepted) {
// 根据 extract_mode 选择返回内容:'html' 模式返回清理后的 HTML'text' 模式返回纯文本
const phase1Content = extractMode === 'html' ? phase1Result.html : phase1Result.text;
// 内容过短检测 → Phase 2 升级(仅对 text 模式生效,html 模式不升级)
if (extractMode === 'text' && phase1Content.length < 200) {
logTool(
'web_fetch',
`Phase 2: Content too short (${phase1Content.length} chars), upgrading to browser`,
);
const browserResult = await this.browserFetch(url);
if (browserResult) {
return this.buildSuccess(url, browserResult, 'browser', maxChars);
}
}
// 写入缓存(仅缓存 text 模式的内容,html 模式不缓存以避免模式混淆)
if (extractMode === 'text') {
fetchCache.set(url, phase1Content);
}
return this.buildSuccess(url, phase1Content, 'http', maxChars, extractMode);
}
// ===== Phase 3: 浏览器回退 =====
logTool('web_fetch', `Phase 3: Falling back to browser (${phase1Result.reason})`);
const browserResult = await this.browserFetch(url);
if (browserResult) {
return this.buildSuccess(url, browserResult, 'browser', maxChars);
}
// 全部失败
return {
url,
content: '',
success: false,
error: `All phases failed. HTTP: ${phase1Result.reason}. Browser fallback also failed.`,
};
}
// ===== Phase 1: HTTP 抓取 =====
private async httpFetch(
url: string,
mobileUA: boolean,
enableRetry: boolean,
): Promise<{
success: boolean;
html: string;
text: string;
intercepted: boolean;
reason: string;
}> {
const maxRetries = enableRetry ? 3 : 1;
const backoffBase = 2_000;
for (let attempt = 0; attempt < maxRetries; attempt++) {
try {
const headers = buildAntiCrawlHeaders(url, attempt, mobileUA);
const response = await fetchWithTimeout(url, { headers, redirect: 'follow' }, 20_000);
// 跳过重试的状态码 → 直接进入浏览器回退
if (SKIP_RETRY_STATUS.has(response.status)) {
return {
success: false,
html: '',
text: '',
intercepted: true,
reason: `HTTP ${response.status}`,
};
}
if (!response.ok) {
// 5xx 可重试
if (response.status >= 500 && attempt < maxRetries - 1) {
await this.sleep(
backoffBase * Math.pow(2, attempt) + Math.random() * backoffBase * 0.6,
);
continue;
}
return {
success: false,
html: '',
text: '',
intercepted: false,
reason: `HTTP ${response.status} ${response.statusText}`,
};
}
// 读取正文(10MB 限制)
const html = await readBodyWithLimit(response, 10 * 1024 * 1024);
// 拦截检测
if (isInterceptedPage(html)) {
return {
success: false,
html: '',
text: '',
intercepted: true,
reason: 'Intercepted page detected',
};
}
// HTML → 纯文本
const text = htmlToText(html);
// H-3/H-4 修复: 同时保留原始 HTML,供 extract_mode='html' 使用
return { success: true, html, text, intercepted: false, reason: '' };
} catch (err) {
const errorMsg = (err as Error).message;
if (attempt < maxRetries - 1) {
logTool('web_fetch', `Attempt ${attempt + 1} failed: ${errorMsg}, retrying...`);
await this.sleep(backoffBase * Math.pow(2, attempt) + Math.random() * backoffBase * 0.6);
continue;
}
return { success: false, html: '', text: '', intercepted: false, reason: errorMsg };
}
}
return {
success: false,
html: '',
text: '',
intercepted: false,
reason: 'All retries exhausted',
};
}
// ===== Phase 2/3: 浏览器回退(使用共享 BrowserWindowManager 单例) =====
private async browserFetch(url: string): Promise<string | null> {
// 查缓存
const cached = fetchCache.get(url);
if (cached) {
logTool('web_fetch', 'Browser cache hit');
return cached;
}
try {
// 崩溃修复: 走 manager.fetchPageText(内部串行化完整的 open→等待→evaluate 序列)。
// 原实现直接 open/evaluate 共享单例 —— web_search 并行抓取触发多个回退同时进入时,
// 后到者销毁前者的窗口(ERR_ABORTED ×3 = 应用崩溃 ×3,见 manager 注释)。
const text = await getBrowserManager().fetchPageText(url);
if (text && text.trim().length >= 80) {
// 拦截检测(浏览器渲染后仍可能是验证码挑战页)
if (isInterceptedPage(text)) {
logTool('web_fetch', `Browser fetch detected intercepted page: ${url}`);
return null;
}
// 内容大小限制(与 HTTP 阶段一致,防止超大页面耗尽上下文)
const MAX_BROWSER_TEXT = 500_000; // 500K chars
const safeText =
text.length > MAX_BROWSER_TEXT
? text.slice(0, MAX_BROWSER_TEXT) + '\n\n[... content truncated ...]'
: text;
// 写缓存
fetchCache.set(url, safeText);
logTool('web_fetch', `Browser fetch success: ${safeText.length} chars`);
return safeText;
}
return null;
} catch (err) {
logTool('web_fetch', `Browser fetch failed: ${(err as Error).message}`);
return null;
}
// 注意:不关闭窗口 — manager 是单例,窗口由 web_browser 或 cleanupBrowser 管理
}
// ===== 辅助方法 =====
private buildSuccess(
url: string,
text: string,
method: string,
maxChars?: number,
extractMode?: 'text' | 'html',
): unknown {
// H-3/H-4 修复: 应用 max_chars 截断,防止过长内容消耗过多 token
let content = text;
let truncated = false;
if (maxChars !== undefined && maxChars > 0 && text.length > maxChars) {
content = text.slice(0, maxChars) + `\n\n[... content truncated at ${maxChars} chars ...]`;
truncated = true;
}
return {
url,
content,
success: true,
method,
length: content.length,
original_length: text.length,
truncated,
extract_mode: extractMode ?? 'text',
};
}
private sleep(ms: number): Promise<void> {
return new Promise((resolve) => setTimeout(resolve, ms));
}
}