后端修复: Ollama adapter 移除死代码/修复超时硬编码/iteration硬编码/pullModel无超时/流异常断开补发DONE; SSE解析器支持非字符串arguments; Sandbox fail-closed安全加固; IPC移除未使用变量 新增功能: TaskOrchestrator子任务编排器; DelegateTaskTool委派工具; Agent Loop并行工具执行; 上下文自动压缩; 记忆检索注入System Prompt 前端修复: useAgentStream text_delta thought累积bug; 工具调用状态正确流转; 修复重复stateChange事件; TraceStep.thought正确填充
231 lines
8.2 KiB
TypeScript
231 lines
8.2 KiB
TypeScript
/**
|
|
* web_fetch — 网页抓取工具
|
|
*
|
|
* 三阶段回退策略:
|
|
* Phase 1: HTTP 抓取(UA 轮换 + 反爬请求头 + 指数退避重试 + 拦截检测)
|
|
* Phase 2: 内容过短自动升级(< 200 字符 → 浏览器渲染)
|
|
* Phase 3: 浏览器回退(共享 BrowserWindowManager + JS 渲染 + 内容提取)
|
|
*
|
|
* 浏览器回退使用与 web_browser 相同的 BrowserWindowManager 单例,
|
|
* 避免创建多个独立浏览器窗口,支持窗口复用。
|
|
*
|
|
* @see docs/Agent网络工具通用设计-v2.md — 第 3 章 web_fetch 抓取设计
|
|
*/
|
|
|
|
import type { IMetonaTool, ToolExecutionContext } from '../../types/metona-tool';
|
|
import type { MetonaToolDef } from '../../../harness/types';
|
|
import { MetonaToolCategory, MetonaRiskLevel } from '../../../harness/types';
|
|
import {
|
|
fetchCache,
|
|
buildAntiCrawlHeaders,
|
|
fetchWithTimeout,
|
|
htmlToText,
|
|
isInterceptedPage,
|
|
readBodyWithLimit,
|
|
logTool,
|
|
} from './network-utils';
|
|
import { getBrowserManager } from './browser';
|
|
|
|
// ===== 跳过重试的状态码 =====
|
|
|
|
const SKIP_RETRY_STATUS = new Set([403, 429, 502, 503]);
|
|
|
|
// ===== WebFetchTool =====
|
|
|
|
export class WebFetchTool implements IMetonaTool {
|
|
readonly definition: MetonaToolDef = {
|
|
name: 'web_fetch',
|
|
description: 'Fetch a web page and convert to plain text. Uses a three-phase fallback strategy: HTTP fetch with anti-crawl headers → SPA auto-upgrade → browser rendering. Handles Cloudflare interception, JavaScript-rendered pages, and large files (10MB limit).',
|
|
parameters: {
|
|
type: 'object',
|
|
properties: {
|
|
url: { type: 'string', description: 'Target URL (http/https only)' },
|
|
mobile_ua: { type: 'boolean', description: 'Use mobile User-Agent (default false)' },
|
|
retry: { type: 'boolean', description: 'Enable retry with exponential backoff (default true)' },
|
|
},
|
|
required: ['url'],
|
|
},
|
|
category: MetonaToolCategory.NETWORK,
|
|
riskLevel: MetonaRiskLevel.LOW,
|
|
requiresPermission: false,
|
|
timeoutMs: 120_000,
|
|
};
|
|
|
|
async execute(args: Record<string, unknown>, _context: ToolExecutionContext): Promise<unknown> {
|
|
const url = args.url as string;
|
|
const mobileUA = (args.mobile_ua as boolean) ?? false;
|
|
const enableRetry = (args.retry as boolean) ?? true;
|
|
|
|
if (!url || !/^https?:\/\//i.test(url)) {
|
|
return { url, content: '', success: false, error: 'URL must start with http:// or https://' };
|
|
}
|
|
|
|
// 先查缓存(HTTP 和浏览器阶段共享同一缓存)
|
|
const cached = fetchCache.get(url);
|
|
if (cached) {
|
|
logTool('web_fetch', `Cache hit: ${url}`);
|
|
return { url, content: cached, success: true, method: 'cache', length: cached.length };
|
|
}
|
|
|
|
logTool('web_fetch', `Fetching: ${url}`);
|
|
|
|
// ===== Phase 1: HTTP 抓取 =====
|
|
const phase1Result = await this.httpFetch(url, mobileUA, enableRetry);
|
|
|
|
if (phase1Result.success && !phase1Result.intercepted) {
|
|
// 内容过短检测 → Phase 2 升级
|
|
if (phase1Result.text.length < 200) {
|
|
logTool('web_fetch', `Phase 2: Content too short (${phase1Result.text.length} chars), upgrading to browser`);
|
|
const browserResult = await this.browserFetch(url);
|
|
if (browserResult) {
|
|
return this.buildSuccess(url, browserResult, 'browser');
|
|
}
|
|
}
|
|
// 写入缓存
|
|
fetchCache.set(url, phase1Result.text);
|
|
return this.buildSuccess(url, phase1Result.text, 'http');
|
|
}
|
|
|
|
// ===== Phase 3: 浏览器回退 =====
|
|
logTool('web_fetch', `Phase 3: Falling back to browser (${phase1Result.reason})`);
|
|
const browserResult = await this.browserFetch(url);
|
|
if (browserResult) {
|
|
return this.buildSuccess(url, browserResult, 'browser');
|
|
}
|
|
|
|
// 全部失败
|
|
return {
|
|
url,
|
|
content: '',
|
|
success: false,
|
|
error: `All phases failed. HTTP: ${phase1Result.reason}. Browser fallback also failed.`,
|
|
};
|
|
}
|
|
|
|
// ===== Phase 1: HTTP 抓取 =====
|
|
|
|
private async httpFetch(
|
|
url: string,
|
|
mobileUA: boolean,
|
|
enableRetry: boolean,
|
|
): Promise<{ success: boolean; text: string; intercepted: boolean; reason: string }> {
|
|
const maxRetries = enableRetry ? 3 : 1;
|
|
const backoffBase = 2_000;
|
|
|
|
for (let attempt = 0; attempt < maxRetries; attempt++) {
|
|
try {
|
|
const headers = buildAntiCrawlHeaders(url, attempt, mobileUA);
|
|
const response = await fetchWithTimeout(url, { headers, redirect: 'follow' }, 20_000);
|
|
|
|
// 跳过重试的状态码 → 直接进入浏览器回退
|
|
if (SKIP_RETRY_STATUS.has(response.status)) {
|
|
return { success: false, text: '', intercepted: true, reason: `HTTP ${response.status}` };
|
|
}
|
|
|
|
if (!response.ok) {
|
|
// 5xx 可重试
|
|
if (response.status >= 500 && attempt < maxRetries - 1) {
|
|
await this.sleep(backoffBase * Math.pow(2, attempt) + Math.random() * backoffBase * 0.6);
|
|
continue;
|
|
}
|
|
return { success: false, text: '', intercepted: false, reason: `HTTP ${response.status} ${response.statusText}` };
|
|
}
|
|
|
|
// 读取正文(10MB 限制)
|
|
const html = await readBodyWithLimit(response, 10 * 1024 * 1024);
|
|
|
|
// 拦截检测
|
|
if (isInterceptedPage(html)) {
|
|
return { success: false, text: '', intercepted: true, reason: 'Intercepted page detected' };
|
|
}
|
|
|
|
// HTML → 纯文本
|
|
const text = htmlToText(html);
|
|
return { success: true, text, intercepted: false, reason: '' };
|
|
} catch (err) {
|
|
const errorMsg = (err as Error).message;
|
|
if (attempt < maxRetries - 1) {
|
|
logTool('web_fetch', `Attempt ${attempt + 1} failed: ${errorMsg}, retrying...`);
|
|
await this.sleep(backoffBase * Math.pow(2, attempt) + Math.random() * backoffBase * 0.6);
|
|
continue;
|
|
}
|
|
return { success: false, text: '', intercepted: false, reason: errorMsg };
|
|
}
|
|
}
|
|
|
|
return { success: false, text: '', intercepted: false, reason: 'All retries exhausted' };
|
|
}
|
|
|
|
// ===== Phase 2/3: 浏览器回退(使用共享 BrowserWindowManager 单例) =====
|
|
|
|
private async browserFetch(url: string): Promise<string | null> {
|
|
// 查缓存
|
|
const cached = fetchCache.get(url);
|
|
if (cached) {
|
|
logTool('web_fetch', 'Browser cache hit');
|
|
return cached;
|
|
}
|
|
|
|
try {
|
|
const manager = getBrowserManager();
|
|
|
|
// 通过 manager 打开 URL(复用已打开的同 URL 窗口,避免重复创建)
|
|
await manager.open({ url });
|
|
|
|
// 等待 JS 渲染
|
|
await this.sleep(2_500);
|
|
|
|
// 提取页面正文
|
|
const text = await manager.evaluate(`
|
|
(function() {
|
|
var clone = document.body.cloneNode(true);
|
|
var noise = clone.querySelectorAll('script, style, noscript, nav, header, footer, aside, iframe, svg');
|
|
noise.forEach(function(el) { el.remove(); });
|
|
return clone.innerText || '';
|
|
})();
|
|
`) as string;
|
|
|
|
if (text && text.trim().length >= 80) {
|
|
// 拦截检测(浏览器渲染后仍可能是验证码挑战页)
|
|
if (isInterceptedPage(text)) {
|
|
logTool('web_fetch', `Browser fetch detected intercepted page: ${url}`);
|
|
return null;
|
|
}
|
|
|
|
// 内容大小限制(与 HTTP 阶段一致,防止超大页面耗尽上下文)
|
|
const MAX_BROWSER_TEXT = 500_000; // 500K chars
|
|
const safeText = text.length > MAX_BROWSER_TEXT
|
|
? text.slice(0, MAX_BROWSER_TEXT) + '\n\n[... content truncated ...]'
|
|
: text;
|
|
|
|
// 写缓存
|
|
fetchCache.set(url, safeText);
|
|
logTool('web_fetch', `Browser fetch success: ${safeText.length} chars`);
|
|
return safeText;
|
|
}
|
|
|
|
return null;
|
|
} catch (err) {
|
|
logTool('web_fetch', `Browser fetch failed: ${(err as Error).message}`);
|
|
return null;
|
|
}
|
|
// 注意:不关闭窗口 — manager 是单例,窗口由 web_browser 或 cleanupBrowser 管理
|
|
}
|
|
|
|
// ===== 辅助方法 =====
|
|
|
|
private buildSuccess(url: string, text: string, method: string): unknown {
|
|
return {
|
|
url,
|
|
content: text,
|
|
success: true,
|
|
method,
|
|
length: text.length,
|
|
};
|
|
}
|
|
|
|
private sleep(ms: number): Promise<void> {
|
|
return new Promise((resolve) => setTimeout(resolve, ms));
|
|
}
|
|
}
|