feat: v0.4.1 质量加固版 — 工程化基线 + 安全加固 + 测试补齐 + 体验升级
工程化(从零到一): - 新增 Gitea Actions CI(debian-latest):类型检查 + Lint + 单元测试 + 产物编译验证 - 新增 husky + lint-staged 预提交钩子(lint-staged + typecheck 门禁) - 移除坏脚本 test:e2e(无 Playwright 配置必失败);prebuild 改用内置 fs.rmSync - 依赖清理:移除死依赖 sql.js(2MB)/@playwright/test,@types/shell-quote 移至 devDependencies 安全加固: - PolicyEngine 频率限制按会话隔离(多会话并发不再互抢配额) - ConfirmationHook 拒绝记忆加 10 分钟 TTL + 恢复询问入口(新增 2 个 IPC 通道) - Windows run_command 白名单工具(git/node/npm/npx/pnpm/yarn/tsc)改走 cmd.exe /c + 参数数组执行,收窄 shell 注入面 - web_search 四引擎 HTML 解析迁移 node-html-parser(结构化主层 + 正则降级) 缺陷修复(测试驱动发现): - mapError 大小写缺陷:网络错误码永远落入 UNKNOWN 无法触发重试 - 搜狗解析器自我过滤:相对链接补全后又被 sogou.com 过滤导致结果全丢 - 百度复合类名重复收录:class="result c-container" 被双重匹配 测试补齐(113 → 194 用例): - 新增 5 个测试文件:sse-stream / base-adapter / confirmation-hook / ipc-agent 编排链路 / web-search 解析器 - 覆盖 sendMessage 全分支、SSE 流解析、错误映射、确认钩子竞态/超时/批量审批 体验升级: - OutputValidator 验证结果可见化(VALIDATION 流事件 → 聊天流提示卡) - SettingsModal 巨型组件拆分(1503 行 → 10 个文件,可独立维护) - MessageList 接入 react-virtuoso 真虚拟滚动(千条消息恒定开销) - MCP 新增 streamable HTTP 传输支持(SDK 内置传输 + DB 迁移 6 + UI 双模式)
This commit is contained in:
@@ -15,7 +15,9 @@ describe('RunCommandTool.validateCommand', () => {
|
||||
const tool = new RunCommandTool();
|
||||
// 访问私有方法
|
||||
const validate = (cmd: string) =>
|
||||
(tool as unknown as { validateCommand: (c: string) => { allowed: boolean; reason?: string } }).validateCommand(cmd);
|
||||
(
|
||||
tool as unknown as { validateCommand: (c: string) => { allowed: boolean; reason?: string } }
|
||||
).validateCommand(cmd);
|
||||
|
||||
const blocked = (cmd: string) => {
|
||||
const result = validate(cmd);
|
||||
@@ -87,3 +89,28 @@ describe('RunCommandTool.validateCommand', () => {
|
||||
allowed('rm -rf node_modules');
|
||||
});
|
||||
});
|
||||
|
||||
describe('RunCommandTool — Windows execFile 白名单(v0.4.1)', () => {
|
||||
const tool = new RunCommandTool();
|
||||
const parseSimple = (cmd: string) =>
|
||||
(
|
||||
tool as unknown as {
|
||||
parseCommandSimple: (c: string) => { command: string; args: string[] } | null;
|
||||
}
|
||||
).parseCommandSimple(cmd);
|
||||
|
||||
it('白名单命令解析为简单命令(无 shell 运算符)', () => {
|
||||
const npm = parseSimple('npm install');
|
||||
expect(npm).toEqual({ command: 'npm', args: ['install'] });
|
||||
const git = parseSimple('git commit -m "fix: bug"');
|
||||
expect(git).toEqual({ command: 'git', args: ['commit', '-m', 'fix: bug'] });
|
||||
const node = parseSimple('node dist/main.js');
|
||||
expect(node).toEqual({ command: 'node', args: ['dist/main.js'] });
|
||||
});
|
||||
|
||||
it('含 shell 运算符的命令不解析为简单命令(继续走 exec 双层校验)', () => {
|
||||
expect(parseSimple('npm install && npm test')).toBeNull();
|
||||
expect(parseSimple('git log | head -5')).toBeNull();
|
||||
expect(parseSimple('echo hi > out.txt')).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
@@ -0,0 +1,174 @@
|
||||
/**
|
||||
* web_search 搜索引擎 HTML 解析器单元测试(v0.4.1 测试补齐)
|
||||
* 覆盖:node-html-parser 结构化解析(主层)、自域名链接过滤、
|
||||
* 相对链接补全、空/异常 HTML 容错
|
||||
*/
|
||||
|
||||
import { describe, it, expect } from 'vitest';
|
||||
import { parseBing, parseBaidu, parseSogou, parse360 } from '../web-search';
|
||||
|
||||
describe('parseBing — 结构化解析', () => {
|
||||
const BING_HTML = `
|
||||
<html><body>
|
||||
<ol id="b_results">
|
||||
<li class="b_algo">
|
||||
<h2><a href="https://example.com/article-1">第一篇 TypeScript 文章</a></h2>
|
||||
<p>这是第一条结果的摘要内容,讲述 TypeScript 高级用法。</p>
|
||||
</li>
|
||||
<li class="b_algo">
|
||||
<h2><a href="https://example.com/article-2">第二篇 Node.js 文章</a></h2>
|
||||
<div class="b_caption"><p>第二条结果的摘要。</p></div>
|
||||
</li>
|
||||
<li class="b_algo">
|
||||
<!-- 自身域名链接应被过滤 -->
|
||||
<h2><a href="https://www.bing.com/video?q=x">Bing 内部视频链接</a></h2>
|
||||
<p>不应出现在结果中。</p>
|
||||
</li>
|
||||
</ol>
|
||||
</body></html>
|
||||
`;
|
||||
|
||||
it('解析结果块并提取 title/url/snippet', () => {
|
||||
const results = parseBing(BING_HTML);
|
||||
expect(results).toHaveLength(2);
|
||||
expect(results[0]).toMatchObject({
|
||||
title: '第一篇 TypeScript 文章',
|
||||
url: 'https://example.com/article-1',
|
||||
snippet: '这是第一条结果的摘要内容,讲述 TypeScript 高级用法。',
|
||||
engine: 'bing',
|
||||
weight: 90,
|
||||
});
|
||||
expect(results[1].url).toBe('https://example.com/article-2');
|
||||
});
|
||||
|
||||
it('过滤指向 bing.com 自身域名的链接', () => {
|
||||
const results = parseBing(BING_HTML);
|
||||
expect(results.some((r) => r.url.includes('bing.com'))).toBe(false);
|
||||
});
|
||||
|
||||
it('空 HTML 返回空数组', () => {
|
||||
expect(parseBing('')).toHaveLength(0);
|
||||
expect(parseBing('<html><body></body></html>')).toHaveLength(0);
|
||||
});
|
||||
});
|
||||
|
||||
describe('parseBaidu — 结构化解析', () => {
|
||||
const BAIDU_HTML = `
|
||||
<html><body>
|
||||
<div id="content_left">
|
||||
<div class="result c-container new-pmd" srcid="1">
|
||||
<h3 class="t"><a data-url="https://example.com/real-url-1" href="https://www.baidu.com/link?url=xyz">百度结果一</a></h3>
|
||||
<span class="content-right_8Zs40">第一条摘要内容。</span>
|
||||
</div>
|
||||
<div class="result c-container" srcid="2">
|
||||
<h3><a href="https://www.baidu.com/link?url=abc">跳转链接结果</a></h3>
|
||||
<span class="content-right_8Zs40">无 data-url 的结果(跳转链接被过滤)。</span>
|
||||
</div>
|
||||
</div>
|
||||
</body></html>
|
||||
`;
|
||||
|
||||
it('优先使用 data-url 真实链接', () => {
|
||||
const results = parseBaidu(BAIDU_HTML);
|
||||
expect(results).toHaveLength(1);
|
||||
expect(results[0]).toMatchObject({
|
||||
title: '百度结果一',
|
||||
url: 'https://example.com/real-url-1',
|
||||
engine: '百度',
|
||||
weight: 80,
|
||||
});
|
||||
});
|
||||
|
||||
it('baidu.com/link 跳转链接被过滤', () => {
|
||||
const results = parseBaidu(BAIDU_HTML);
|
||||
expect(results.some((r) => r.url.includes('baidu.com/link'))).toBe(false);
|
||||
});
|
||||
|
||||
it('空 HTML 返回空数组', () => {
|
||||
expect(parseBaidu('')).toHaveLength(0);
|
||||
});
|
||||
});
|
||||
|
||||
describe('parseSogou — 结构化解析', () => {
|
||||
const SOGOU_HTML = `
|
||||
<html><body>
|
||||
<div class="results">
|
||||
<div class="vrwrap">
|
||||
<h3><a href="/link?url=sogou-internal-1">搜狗结果一</a></h3>
|
||||
<div class="str_info">搜狗结果一的摘要文本。</div>
|
||||
</div>
|
||||
<div class="rb">
|
||||
<h3><a href="https://example.com/direct">搜狗直链结果</a></h3>
|
||||
<p class="space-txt">直链结果的摘要。</p>
|
||||
</div>
|
||||
</div>
|
||||
</body></html>
|
||||
`;
|
||||
|
||||
it('相对链接补全 sogou.com 前缀', () => {
|
||||
const results = parseSogou(SOGOU_HTML);
|
||||
expect(results).toHaveLength(2);
|
||||
expect(results[0]).toMatchObject({
|
||||
title: '搜狗结果一',
|
||||
url: 'https://www.sogou.com/link?url=sogou-internal-1',
|
||||
engine: '搜狗',
|
||||
weight: 75,
|
||||
});
|
||||
});
|
||||
|
||||
it('http 开头的直链不补全前缀', () => {
|
||||
const results = parseSogou(SOGOU_HTML);
|
||||
expect(results[1].url).toBe('https://example.com/direct');
|
||||
expect(results[1].snippet).toBe('直链结果的摘要。');
|
||||
});
|
||||
|
||||
it('空 HTML 返回空数组', () => {
|
||||
expect(parseSogou('')).toHaveLength(0);
|
||||
});
|
||||
});
|
||||
|
||||
describe('parse360 — 结构化解析', () => {
|
||||
const SO360_HTML = `
|
||||
<html><body>
|
||||
<div class="res-list">
|
||||
<ul>
|
||||
<li class="res-list">
|
||||
<h3 class="res-title"><a href="https://example.com/360-result">360 结果一</a></h3>
|
||||
<p class="res-desc">360 搜索结果的摘要描述。</p>
|
||||
</li>
|
||||
<li class="res-list">
|
||||
<h3><a href="https://www.so.com/internal">360 内部链接</a></h3>
|
||||
<p class="res-desc">不应出现。</p>
|
||||
</li>
|
||||
</ul>
|
||||
</div>
|
||||
</body></html>
|
||||
`;
|
||||
|
||||
it('解析结果并过滤 so.com 自身链接', () => {
|
||||
const results = parse360(SO360_HTML);
|
||||
expect(results).toHaveLength(1);
|
||||
expect(results[0]).toMatchObject({
|
||||
title: '360 结果一',
|
||||
url: 'https://example.com/360-result',
|
||||
snippet: '360 搜索结果的摘要描述。',
|
||||
engine: '360搜索',
|
||||
weight: 75,
|
||||
});
|
||||
});
|
||||
|
||||
it('空 HTML 返回空数组', () => {
|
||||
expect(parse360('')).toHaveLength(0);
|
||||
});
|
||||
});
|
||||
|
||||
describe('解析器降级路径', () => {
|
||||
it('结构化解析无结果且正则也无结果时返回空数组(不抛错)', () => {
|
||||
// 非搜索结果页 HTML(如错误页/验证码页)
|
||||
const notSearchPage = '<html><body><div class="captcha">请输入验证码</div></body></html>';
|
||||
expect(parseBing(notSearchPage)).toHaveLength(0);
|
||||
expect(parseBaidu(notSearchPage)).toHaveLength(0);
|
||||
expect(parseSogou(notSearchPage)).toHaveLength(0);
|
||||
expect(parse360(notSearchPage)).toHaveLength(0);
|
||||
});
|
||||
});
|
||||
@@ -58,13 +58,22 @@ function decodeBuffer(buf: Buffer): string {
|
||||
function buildSafeCommandEnv(isWindows: boolean): Record<string, string> {
|
||||
// 敏感变量后缀黑名单
|
||||
const SENSITIVE_SUFFIXES = [
|
||||
'_API_KEY', '_TOKEN', '_SECRET', '_PASSWORD', '_PASSWD',
|
||||
'_CREDENTIAL', '_CREDENTIALS', '_PRIVATE_KEY',
|
||||
'_API_KEY',
|
||||
'_TOKEN',
|
||||
'_SECRET',
|
||||
'_PASSWORD',
|
||||
'_PASSWD',
|
||||
'_CREDENTIAL',
|
||||
'_CREDENTIALS',
|
||||
'_PRIVATE_KEY',
|
||||
];
|
||||
// 敏感变量名黑名单(精确匹配)
|
||||
const SENSITIVE_KEYS = new Set([
|
||||
'DEEPSEEK_API_KEY', 'AGNES_API_KEY', 'MIMO_API_KEY',
|
||||
'GITEA_PASSWORD', 'DATABASE_PASSWORD',
|
||||
'DEEPSEEK_API_KEY',
|
||||
'AGNES_API_KEY',
|
||||
'MIMO_API_KEY',
|
||||
'GITEA_PASSWORD',
|
||||
'DATABASE_PASSWORD',
|
||||
]);
|
||||
|
||||
const env: Record<string, string> = {};
|
||||
@@ -86,12 +95,31 @@ function buildSafeCommandEnv(isWindows: boolean): Record<string, string> {
|
||||
return env;
|
||||
}
|
||||
|
||||
/**
|
||||
* v0.4.1: Windows 白名单命令集合 — 这些工具的简单命令(无 shell 运算符)走
|
||||
* execFile('cmd.exe', ['/c', ...words]) 执行:参数以数组形式显式传递,不经过
|
||||
* shell 解析,从根本上去掉 exec() 的字符串拼接注入面(无法通过参数注入新命令)。
|
||||
*
|
||||
* 仅收录最常见的开发工具(小步灰度);其余命令仍走 exec + 双层校验的既有路径。
|
||||
* Node 18.20+/Electron 35 在 Windows 上直接 spawn .cmd 批处理会被拒绝(EINVAL),
|
||||
* 因此必须通过 cmd.exe /c 中转,但参数分离已足够收窄注入面。
|
||||
*/
|
||||
const WINDOWS_EXEC_FILE_WHITELIST = new Set(['git', 'node', 'npm', 'npx', 'pnpm', 'yarn', 'tsc']);
|
||||
|
||||
/** v0.4.1: 提取命令 basename(处理 C:\Program Files\nodejs\npm.cmd 等路径形式) */
|
||||
function commandBasename(cmd: string): string {
|
||||
const base = cmd.split(/[\\/]/).pop() ?? cmd;
|
||||
// 去掉 .exe/.cmd/.bat 扩展名(大小写不敏感)
|
||||
return base.replace(/\.(exe|cmd|bat)$/i, '');
|
||||
}
|
||||
|
||||
// ===== 9. run_command =====
|
||||
|
||||
export class RunCommandTool implements IMetonaTool {
|
||||
readonly definition: MetonaToolDef = {
|
||||
name: 'run_command',
|
||||
description: 'Execute a shell command in a sandboxed environment. Commands run in the workspace directory. High-risk commands require user confirmation. Passes through SandboxManager static code scan and path validation.',
|
||||
description:
|
||||
'Execute a shell command in a sandboxed environment. Commands run in the workspace directory. High-risk commands require user confirmation. Passes through SandboxManager static code scan and path validation.',
|
||||
parameters: {
|
||||
type: 'object',
|
||||
properties: {
|
||||
@@ -123,7 +151,11 @@ export class RunCommandTool implements IMetonaTool {
|
||||
// 安全校验:workdir 必须在工作空间内
|
||||
const resolvedWorkdir = resolve(context.workspacePath, workdir);
|
||||
if (!isPathWithinWorkspace(workdir, context.workspacePath)) {
|
||||
return { success: false, error: `Working directory must be within workspace: ${workdir}`, command };
|
||||
return {
|
||||
success: false,
|
||||
error: `Working directory must be within workspace: ${workdir}`,
|
||||
command,
|
||||
};
|
||||
}
|
||||
|
||||
// v0.2.0: SandboxManager 双重安全校验 — fail-closed 设计
|
||||
@@ -183,19 +215,32 @@ export class RunCommandTool implements IMetonaTool {
|
||||
let stderr: Buffer;
|
||||
|
||||
// #8 修复 + 审查修复: 简单命令使用 execFile(不经过 shell,防止命令注入)
|
||||
// 但 Windows 上 npm/npx/yarn/pnpm/tsc 等是 .cmd 批处理,execFile 无法执行(ENOENT)
|
||||
// 因此 Windows 上仍用 exec(已有 SandboxManager.scanCode + validateCommand 双层校验)
|
||||
// 非 Windows 上对简单命令用 execFile
|
||||
// 但 Windows 上 npm/npx/yarn/pnpm/tsc 等是 .cmd 批处理,execFile 无法直接执行(ENOENT/EINVAL)
|
||||
// v0.4.1: Windows 上白名单工具(git/node/npm/npx/pnpm/yarn/tsc)的简单命令改用
|
||||
// execFile('cmd.exe', ['/c', ...args]) — 参数显式分离传递,不经 shell 字符串解析,
|
||||
// 相比 exec() 的整串拼接显著收窄注入面
|
||||
// 非 Windows 上对简单命令直接 execFile
|
||||
if (simpleCmd && !isWindows) {
|
||||
const result = await execFileAsync(simpleCmd.command, simpleCmd.args, execOpts);
|
||||
stdout = result.stdout;
|
||||
stderr = result.stderr;
|
||||
} else if (
|
||||
simpleCmd &&
|
||||
isWindows &&
|
||||
WINDOWS_EXEC_FILE_WHITELIST.has(commandBasename(simpleCmd.command))
|
||||
) {
|
||||
// v0.4.1: 白名单工具通过 cmd.exe /c + 参数数组执行(参数不经 shell 解析)
|
||||
const result = await execFileAsync(
|
||||
'cmd.exe',
|
||||
['/c', simpleCmd.command, ...simpleCmd.args],
|
||||
execOpts,
|
||||
);
|
||||
stdout = result.stdout;
|
||||
stderr = result.stderr;
|
||||
} else {
|
||||
// 复杂命令(含管道/重定向/&& 等 shell 语法)或 Windows — 使用 exec
|
||||
// 已有 SandboxManager.scanCode + validateCommand 双层安全校验
|
||||
const finalCommand = isWindows
|
||||
? `chcp 65001 >nul 2>&1 && ${command}`
|
||||
: command;
|
||||
const finalCommand = isWindows ? `chcp 65001 >nul 2>&1 && ${command}` : command;
|
||||
const result = await execAsync(finalCommand, execOpts);
|
||||
stdout = result.stdout;
|
||||
stderr = result.stderr;
|
||||
@@ -236,7 +281,11 @@ export class RunCommandTool implements IMetonaTool {
|
||||
|
||||
// 受保护文件检查:禁止通过命令行读写工作空间根目录的 MEMORY.md
|
||||
if (commandTouchesProtectedFile(command)) {
|
||||
return { allowed: false, reason: 'Access denied: MEMORY.md is managed by the memory system and cannot be accessed via command execution' };
|
||||
return {
|
||||
allowed: false,
|
||||
reason:
|
||||
'Access denied: MEMORY.md is managed by the memory system and cannot be accessed via command execution',
|
||||
};
|
||||
}
|
||||
|
||||
// P0-5: 剥离 Windows chcp 前缀("chcp 65001 >nul 2>&1 &&" 会破坏 shell-quote
|
||||
@@ -256,33 +305,61 @@ export class RunCommandTool implements IMetonaTool {
|
||||
const hardBlocks = [
|
||||
// 文件系统破坏
|
||||
{ pattern: /\brm\b.*\//, reason: 'rm with absolute path is forbidden' },
|
||||
{ pattern: /\brm\s+-rf?\s+\/(?:[^|;&\s]*\s)*?(?:bin|boot|dev|etc|lib|proc|root|sbin|sys|usr|var)\b/i, reason: 'rm on system directories is forbidden' },
|
||||
{
|
||||
pattern:
|
||||
/\brm\s+-rf?\s+\/(?:[^|;&\s]*\s)*?(?:bin|boot|dev|etc|lib|proc|root|sbin|sys|usr|var)\b/i,
|
||||
reason: 'rm on system directories is forbidden',
|
||||
},
|
||||
{ pattern: /\b(sudo|su|doas)\b/, reason: 'Privilege escalation commands are forbidden' },
|
||||
// 系统控制
|
||||
{ pattern: /\b(shutdown|reboot|halt|poweroff)\b/, reason: 'System shutdown commands are forbidden' },
|
||||
{
|
||||
pattern: /\b(shutdown|reboot|halt|poweroff)\b/,
|
||||
reason: 'System shutdown commands are forbidden',
|
||||
},
|
||||
{ pattern: /\b(killall|pkill)\s+-9\b/, reason: 'Force kill all processes is forbidden' },
|
||||
// 远程代码执行
|
||||
{ pattern: /curl.*\|\s*(ba)?sh/, reason: 'Remote code execution via pipe is forbidden' },
|
||||
{ pattern: /wget.*\|\s*(ba)?sh/, reason: 'Remote code execution via pipe is forbidden' },
|
||||
{ pattern: /\bcurl\s+.*\s*-o\s+\/etc\//i, reason: 'Writing to system directories via curl is forbidden' },
|
||||
{
|
||||
pattern: /\bcurl\s+.*\s*-o\s+\/etc\//i,
|
||||
reason: 'Writing to system directories via curl is forbidden',
|
||||
},
|
||||
// 设备文件
|
||||
{ pattern: /\bdd\b.*of=\/dev\//, reason: 'Writing to device files is forbidden' },
|
||||
// 磁盘格式化
|
||||
{ pattern: /\b(mkfs|fdisk)\b/, reason: 'Disk formatting commands are forbidden' },
|
||||
// 权限滥用
|
||||
{ pattern: /\bchmod\s+777\b/, reason: 'chmod 777 is forbidden' },
|
||||
{ pattern: /\bchown\s+-R\s+\S+\s+\/(?:\s|$)/i, reason: 'Recursive chown on root is forbidden' },
|
||||
{
|
||||
pattern: /\bchown\s+-R\s+\S+\s+\/(?:\s|$)/i,
|
||||
reason: 'Recursive chown on root is forbidden',
|
||||
},
|
||||
// 环境变量窃取
|
||||
{ pattern: /\b(env|export|printenv)\s*\|.*\b(curl|wget|nc|ncat)\b/i, reason: 'Exfiltrating environment variables is forbidden' },
|
||||
{
|
||||
pattern: /\b(env|export|printenv)\s*\|.*\b(curl|wget|nc|ncat)\b/i,
|
||||
reason: 'Exfiltrating environment variables is forbidden',
|
||||
},
|
||||
// 反向 shell
|
||||
{ pattern: /\b(bash|sh|zsh)\s+-i\s+>\s*&\s*\/dev\/tcp\//i, reason: 'Reverse shell via /dev/tcp is forbidden' },
|
||||
{
|
||||
pattern: /\b(bash|sh|zsh)\s+-i\s+>\s*&\s*\/dev\/tcp\//i,
|
||||
reason: 'Reverse shell via /dev/tcp is forbidden',
|
||||
},
|
||||
{ pattern: /\bnc\s+.*\s+-e\s+(bash|sh)/i, reason: 'Reverse shell via netcat is forbidden' },
|
||||
// Windows 危险命令
|
||||
{ pattern: /\b(format|diskpart)\b/i, reason: 'Disk formatting commands are forbidden' },
|
||||
{ pattern: /\bshutdown\s*\//i, reason: 'System shutdown commands are forbidden' },
|
||||
{ pattern: /\breg\s+(add|delete|import|restore)/i, reason: 'Registry modification commands are forbidden' },
|
||||
{ pattern: /\b(taskkill|kill)\s*\//i, reason: 'Process termination with system flags is forbidden' },
|
||||
{ pattern: /\bpowershell\s+-enc\s+/i, reason: 'PowerShell encoded command execution is forbidden' },
|
||||
{
|
||||
pattern: /\breg\s+(add|delete|import|restore)/i,
|
||||
reason: 'Registry modification commands are forbidden',
|
||||
},
|
||||
{
|
||||
pattern: /\b(taskkill|kill)\s*\//i,
|
||||
reason: 'Process termination with system flags is forbidden',
|
||||
},
|
||||
{
|
||||
pattern: /\bpowershell\s+-enc\s+/i,
|
||||
reason: 'PowerShell encoded command execution is forbidden',
|
||||
},
|
||||
// 后台进程与管道炸弹
|
||||
{ pattern: /&\s*\(/, reason: 'Background subshell execution is forbidden' },
|
||||
{ pattern: /\|\s*&/, reason: 'Pipe to background process is forbidden' },
|
||||
@@ -342,20 +419,29 @@ export class RunCommandTool implements IMetonaTool {
|
||||
prevWasPipe = false;
|
||||
} else if (typeof obj.op === 'string') {
|
||||
// 跟踪管道运算符,用于下一轮检测 `| sh`
|
||||
prevWasPipe = (obj.op === '|');
|
||||
prevWasPipe = obj.op === '|';
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// 危险命令名 token(精确匹配,大小写不敏感)
|
||||
const dangerousCommands = new Set([
|
||||
'sudo', 'su', 'doas',
|
||||
'shutdown', 'reboot', 'halt', 'poweroff',
|
||||
'mkfs', 'fdisk', 'format', 'diskpart',
|
||||
'sudo',
|
||||
'su',
|
||||
'doas',
|
||||
'shutdown',
|
||||
'reboot',
|
||||
'halt',
|
||||
'poweroff',
|
||||
'mkfs',
|
||||
'fdisk',
|
||||
'format',
|
||||
'diskpart',
|
||||
]);
|
||||
// 危险参数 token
|
||||
const dangerousArgs = new Set([
|
||||
'-enc', '-encodedcommand', // PowerShell 编码执行
|
||||
'-enc',
|
||||
'-encodedcommand', // PowerShell 编码执行
|
||||
]);
|
||||
|
||||
for (const word of words) {
|
||||
|
||||
@@ -6,9 +6,15 @@
|
||||
* 智能排序:引擎权重(50%) + 可达性(30%) + 摘要质量(20%)
|
||||
* 自动抓取:对前 N 条结果调用 web_fetch 获取完整正文
|
||||
*
|
||||
* v0.4.1: HTML 解析迁移至 node-html-parser(结构化解析)
|
||||
* 主层使用 DOM 结构解析(引擎改版时选择器更精确、可维护性远优于正则),
|
||||
* 正则解析保留为降级路径(结构化解析无结果时兜底)。
|
||||
* 此前纯正则方案违反项目开发规范第一铁律(HTML 解析应使用成熟库)。
|
||||
*
|
||||
* @see docs/Agent网络工具通用设计-v2.md — 第 2 章 web_search 搜索设计
|
||||
*/
|
||||
|
||||
import { parse as parseHtmlDom, type HTMLElement } from 'node-html-parser';
|
||||
import type { IMetonaTool, ToolExecutionContext } from '../../types/metona-tool';
|
||||
import type { MetonaToolDef } from '../../../harness/types';
|
||||
import { MetonaToolCategory, MetonaRiskLevel } from '../../../harness/types';
|
||||
@@ -64,7 +70,9 @@ const ENGINES: EngineDef[] = [
|
||||
name: 'bing',
|
||||
weight: 90,
|
||||
searchUrl: (q, tr) => {
|
||||
const freshness = tr ? `&filters=ex1:"ez${tr === 'day' ? '1' : tr === 'week' ? '2' : tr === 'month' ? '3' : '4'}"` : '';
|
||||
const freshness = tr
|
||||
? `&filters=ex1:"ez${tr === 'day' ? '1' : tr === 'week' ? '2' : tr === 'month' ? '3' : '4'}"`
|
||||
: '';
|
||||
return `https://www.bing.com/search?q=${encodeURIComponent(q)}${freshness}&count=20`;
|
||||
},
|
||||
parse: parseBing,
|
||||
@@ -89,9 +97,150 @@ const ENGINES: EngineDef[] = [
|
||||
},
|
||||
];
|
||||
|
||||
// ===== HTML 解析器(正则实现,后续可迁移至 cheerio) =====
|
||||
// ===== HTML 解析器(v0.4.1: node-html-parser 结构化解析为主层,正则为降级层) =====
|
||||
|
||||
function parseBing(html: string): SearchResult[] {
|
||||
/**
|
||||
* v0.4.1: 从结果块中提取标题链接 — 跳过指向搜索引擎自身域名的链接(favicon/子导航等)
|
||||
*/
|
||||
function extractTitleLink(
|
||||
block: HTMLElement,
|
||||
selfDomain: string,
|
||||
): { url: string; title: string } | null {
|
||||
for (const a of block.querySelectorAll('a[href]')) {
|
||||
const url = a.getAttribute('href') ?? '';
|
||||
const title = a.text.trim();
|
||||
if (title && url && !url.includes(selfDomain) && url.startsWith('http')) {
|
||||
return { url, title };
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/** v0.4.1: 提取第一个非空文本的选择器(按优先级尝试多个候选选择器) */
|
||||
function extractText(block: HTMLElement, selectors: string[]): string {
|
||||
for (const sel of selectors) {
|
||||
const el = block.querySelector(sel);
|
||||
if (el) {
|
||||
const text = el.text.trim();
|
||||
if (text) return text;
|
||||
}
|
||||
}
|
||||
return '';
|
||||
}
|
||||
|
||||
/** v0.4.1: Bing 结构化解析 — li.b_algo 结果块 */
|
||||
function parseBingStructured(html: string): SearchResult[] {
|
||||
const results: SearchResult[] = [];
|
||||
const root = parseHtmlDom(html);
|
||||
for (const block of root.querySelectorAll('li.b_algo')) {
|
||||
const link = extractTitleLink(block, 'bing.com');
|
||||
if (!link) continue;
|
||||
const snippet = extractText(block, ['p', '.b_caption']);
|
||||
results.push({ title: link.title, url: link.url, snippet, engine: 'bing', weight: 90 });
|
||||
}
|
||||
return results;
|
||||
}
|
||||
|
||||
/** v0.4.1: 百度结构化解析 — div.result / div.c-container 结果块,优先 a[data-url] 真实链接 */
|
||||
function parseBaiduStructured(html: string): SearchResult[] {
|
||||
const results: SearchResult[] = [];
|
||||
const root = parseHtmlDom(html);
|
||||
// 复合选择器去重:class="result c-container" 的元素同时命中两个类名,
|
||||
// 分别查询再拼接会重复收录同一结果块
|
||||
const blocks = root.querySelectorAll('div.result, div.c-container');
|
||||
for (const block of blocks) {
|
||||
// 百度标题链接: 优先 data-url 属性(真实目标 URL),href 通常是 baidu.com/link 跳转
|
||||
const dataUrlLink = block.querySelector('a[data-url]');
|
||||
let url = dataUrlLink?.getAttribute('data-url') ?? '';
|
||||
let title = dataUrlLink?.text.trim() ?? '';
|
||||
if (!url || !title) {
|
||||
const fallback = block.querySelector('h3 a[href]') ?? block.querySelector('a[href]');
|
||||
if (fallback) {
|
||||
const href = fallback.getAttribute('href') ?? '';
|
||||
url = href.startsWith('http') ? href : href ? `https://${href}` : '';
|
||||
title = fallback.text.trim();
|
||||
}
|
||||
}
|
||||
const snippet = extractText(block, ['.c-abstract', '[class^="content-right"]']);
|
||||
if (title && url && !url.includes('baidu.com/link')) {
|
||||
results.push({ title, url, snippet, engine: '百度', weight: 80 });
|
||||
}
|
||||
}
|
||||
return results;
|
||||
}
|
||||
|
||||
/**
|
||||
* v0.4.1: 搜狗结构化解析 — div.vrwrap / div.rb 结果块(相对链接补全 sogou.com 前缀)
|
||||
*
|
||||
* v0.4.1 修复(原正则实现遗留缺陷): 搜狗结果链接是 sogou.com/link?url=... 跳转形式,
|
||||
* 原 `!url.includes('sogou.com')` 过滤条件把所有跳转结果一并丢弃(相对链接补全后必含 sogou.com),
|
||||
* 导致搜狗引擎基本无法返回结果。现仅过滤 sogou 自身页面链接,保留 /link 跳转结果
|
||||
* (可达性预检会跟随重定向验证)。
|
||||
*/
|
||||
function parseSogouStructured(html: string): SearchResult[] {
|
||||
const results: SearchResult[] = [];
|
||||
const root = parseHtmlDom(html);
|
||||
// 复合选择器避免同一元素命中两个类名时重复收录
|
||||
const blocks = root.querySelectorAll('div.vrwrap, div.rb');
|
||||
for (const block of blocks) {
|
||||
const a = block.querySelector('h3 a[href]') ?? block.querySelector('a[href]');
|
||||
if (!a) continue;
|
||||
const href = a.getAttribute('href') ?? '';
|
||||
const url = href.startsWith('http') ? href : `https://www.sogou.com${href}`;
|
||||
const title = a.text.trim();
|
||||
const snippet = extractText(block, ['.star-wiki', '.space-txt', '.str_info']);
|
||||
// 过滤搜狗自身页面(保留 /link 跳转结果)
|
||||
const isSelfPage = url.includes('sogou.com') && !url.includes('/link');
|
||||
if (title && url && !isSelfPage) {
|
||||
results.push({ title, url, snippet, engine: '搜狗', weight: 75 });
|
||||
}
|
||||
}
|
||||
return results;
|
||||
}
|
||||
|
||||
/** v0.4.1: 360 结构化解析 — li.res-list / div.result 结果块 */
|
||||
function parse360Structured(html: string): SearchResult[] {
|
||||
const results: SearchResult[] = [];
|
||||
const root = parseHtmlDom(html);
|
||||
// 复合选择器避免同一元素命中多个类名时重复收录
|
||||
const blocks = root.querySelectorAll('li.res-list, div.result');
|
||||
for (const block of blocks) {
|
||||
const link = extractTitleLink(block, 'so.com');
|
||||
if (!link) continue;
|
||||
const snippet = extractText(block, ['.res-desc', '.res-rich', '.res-summary', 'dd']);
|
||||
results.push({ title: link.title, url: link.url, snippet, engine: '360搜索', weight: 75 });
|
||||
}
|
||||
return results;
|
||||
}
|
||||
|
||||
/** v0.4.1: 结构化解析 + 正则降级的组合入口(供 ENGINES 引用,测试导出) */
|
||||
export function parseBing(html: string): SearchResult[] {
|
||||
const structured = parseBingStructured(html);
|
||||
if (structured.length > 0) return structured;
|
||||
return parseBingRegex(html);
|
||||
}
|
||||
|
||||
export function parseBaidu(html: string): SearchResult[] {
|
||||
const structured = parseBaiduStructured(html);
|
||||
if (structured.length > 0) return structured;
|
||||
return parseBaiduRegex(html);
|
||||
}
|
||||
|
||||
export function parseSogou(html: string): SearchResult[] {
|
||||
const structured = parseSogouStructured(html);
|
||||
if (structured.length > 0) return structured;
|
||||
return parseSogouRegex(html);
|
||||
}
|
||||
|
||||
export function parse360(html: string): SearchResult[] {
|
||||
const structured = parse360Structured(html);
|
||||
if (structured.length > 0) return structured;
|
||||
return parse360Regex(html);
|
||||
}
|
||||
|
||||
// ===== 正则降级解析器(v0.4.1 前的主实现,结构化解析无结果时兜底) =====
|
||||
|
||||
function parseBingRegex(html: string): SearchResult[] {
|
||||
const results: SearchResult[] = [];
|
||||
const blocks = html.split(/<li[^>]*class="b_algo"/i).slice(1);
|
||||
for (const block of blocks) {
|
||||
@@ -99,7 +248,9 @@ function parseBing(html: string): SearchResult[] {
|
||||
if (!titleMatch) continue;
|
||||
const url = titleMatch[1];
|
||||
const title = titleMatch[2].replace(/<[^>]+>/g, '').trim();
|
||||
const snippetMatch = block.match(/<p[^>]*>([\s\S]*?)<\/p>/i) || block.match(/class="b_caption"[^>]*>([\s\S]*?)<\/div>/i);
|
||||
const snippetMatch =
|
||||
block.match(/<p[^>]*>([\s\S]*?)<\/p>/i) ||
|
||||
block.match(/class="b_caption"[^>]*>([\s\S]*?)<\/div>/i);
|
||||
const snippet = snippetMatch ? snippetMatch[1].replace(/<[^>]+>/g, '').trim() : '';
|
||||
if (title && url && !url.includes('bing.com')) {
|
||||
results.push({ title, url, snippet, engine: 'bing', weight: 90 });
|
||||
@@ -108,17 +259,19 @@ function parseBing(html: string): SearchResult[] {
|
||||
return results;
|
||||
}
|
||||
|
||||
function parseBaidu(html: string): SearchResult[] {
|
||||
function parseBaiduRegex(html: string): SearchResult[] {
|
||||
const results: SearchResult[] = [];
|
||||
const blocks = html.split(/<div[^>]*class="result[^"]*"/i).slice(1);
|
||||
for (const block of blocks) {
|
||||
const titleMatch = block.match(/<a[^>]*data-url="([^"]+)"[^>]*>([\s\S]*?)<\/a>/i)
|
||||
|| block.match(/<a[^>]*href="([^"]+)"[^>]*>([\s\S]*?)<\/a>/i);
|
||||
const titleMatch =
|
||||
block.match(/<a[^>]*data-url="([^"]+)"[^>]*>([\s\S]*?)<\/a>/i) ||
|
||||
block.match(/<a[^>]*href="([^"]+)"[^>]*>([\s\S]*?)<\/a>/i);
|
||||
if (!titleMatch) continue;
|
||||
const url = titleMatch[1].startsWith('http') ? titleMatch[1] : `https://${titleMatch[1]}`;
|
||||
const title = titleMatch[2].replace(/<[^>]+>/g, '').trim();
|
||||
const snippetMatch = block.match(/class="c-abstract[^"]*"[^>]*>([\s\S]*?)<\/span>/i)
|
||||
|| block.match(/class="content-right[^"]*"[^>]*>([\s\S]*?)<\/div>/i);
|
||||
const snippetMatch =
|
||||
block.match(/class="c-abstract[^"]*"[^>]*>([\s\S]*?)<\/span>/i) ||
|
||||
block.match(/class="content-right[^"]*"[^>]*>([\s\S]*?)<\/div>/i);
|
||||
const snippet = snippetMatch ? snippetMatch[1].replace(/<[^>]+>/g, '').trim() : '';
|
||||
if (title && url && !url.includes('baidu.com/link')) {
|
||||
results.push({ title, url, snippet, engine: '百度', weight: 80 });
|
||||
@@ -127,18 +280,23 @@ function parseBaidu(html: string): SearchResult[] {
|
||||
return results;
|
||||
}
|
||||
|
||||
function parseSogou(html: string): SearchResult[] {
|
||||
function parseSogouRegex(html: string): SearchResult[] {
|
||||
const results: SearchResult[] = [];
|
||||
const blocks = html.split(/<div[^>]*class="vrwrap"/i).slice(1)
|
||||
const blocks = html
|
||||
.split(/<div[^>]*class="vrwrap"/i)
|
||||
.slice(1)
|
||||
.concat(html.split(/<div[^>]*class="rb"/i).slice(1));
|
||||
for (const block of blocks) {
|
||||
const titleMatch = block.match(/<a[^>]*href="([^"]+)"[^>]*>([\s\S]*?)<\/a>/i);
|
||||
if (!titleMatch) continue;
|
||||
const url = titleMatch[1].startsWith('http') ? titleMatch[1] : `https://www.sogou.com${titleMatch[1]}`;
|
||||
const url = titleMatch[1].startsWith('http')
|
||||
? titleMatch[1]
|
||||
: `https://www.sogou.com${titleMatch[1]}`;
|
||||
const title = titleMatch[2].replace(/<[^>]+>/g, '').trim();
|
||||
const snippetMatch = block.match(/class="star-wiki[^"]*"[^>]*>([\s\S]*?)<\/div>/i)
|
||||
|| block.match(/class="space-txt[^"]*"[^>]*>([\s\S]*?)<\/p>/i)
|
||||
|| block.match(/class="str_info[^"]*"[^>]*>([\s\S]*?)<\/p>/i);
|
||||
const snippetMatch =
|
||||
block.match(/class="star-wiki[^"]*"[^>]*>([\s\S]*?)<\/div>/i) ||
|
||||
block.match(/class="space-txt[^"]*"[^>]*>([\s\S]*?)<\/p>/i) ||
|
||||
block.match(/class="str_info[^"]*"[^>]*>([\s\S]*?)<\/p>/i);
|
||||
const snippet = snippetMatch ? snippetMatch[1].replace(/<[^>]+>/g, '').trim() : '';
|
||||
if (title && url && !url.includes('sogou.com')) {
|
||||
results.push({ title, url, snippet, engine: '搜狗', weight: 75 });
|
||||
@@ -147,19 +305,22 @@ function parseSogou(html: string): SearchResult[] {
|
||||
return results;
|
||||
}
|
||||
|
||||
function parse360(html: string): SearchResult[] {
|
||||
function parse360Regex(html: string): SearchResult[] {
|
||||
const results: SearchResult[] = [];
|
||||
const blocks = html.split(/<li[^>]*class="res-list"/i).slice(1)
|
||||
const blocks = html
|
||||
.split(/<li[^>]*class="res-list"/i)
|
||||
.slice(1)
|
||||
.concat(html.split(/<div[^>]*class="result"/i).slice(1));
|
||||
for (const block of blocks) {
|
||||
const titleMatch = block.match(/<a[^>]*href="([^"]+)"[^>]*>([\s\S]*?)<\/a>/i);
|
||||
if (!titleMatch) continue;
|
||||
const url = titleMatch[1];
|
||||
const title = titleMatch[2].replace(/<[^>]+>/g, '').trim();
|
||||
const snippetMatch = block.match(/class="res-desc[^"]*"[^>]*>([\s\S]*?)<\/p>/i)
|
||||
|| block.match(/class="res-rich[^"]*"[^>]*>([\s\S]*?)<\/div>/i)
|
||||
|| block.match(/class="res-summary[^"]*"[^>]*>([\s\S]*?)<\/p>/i)
|
||||
|| block.match(/<dd[^>]*>([\s\S]*?)<\/dd>/i);
|
||||
const snippetMatch =
|
||||
block.match(/class="res-desc[^"]*"[^>]*>([\s\S]*?)<\/p>/i) ||
|
||||
block.match(/class="res-rich[^"]*"[^>]*>([\s\S]*?)<\/div>/i) ||
|
||||
block.match(/class="res-summary[^"]*"[^>]*>([\s\S]*?)<\/p>/i) ||
|
||||
block.match(/<dd[^>]*>([\s\S]*?)<\/dd>/i);
|
||||
const snippet = snippetMatch ? snippetMatch[1].replace(/<[^>]+>/g, '').trim() : '';
|
||||
if (title && url && !url.includes('so.com')) {
|
||||
results.push({ title, url, snippet, engine: '360搜索', weight: 75 });
|
||||
@@ -192,7 +353,7 @@ async function checkReachability(urls: string[], concurrency = 5): Promise<Map<s
|
||||
function smartSort(results: SearchResult[]): SearchResult[] {
|
||||
for (const r of results) {
|
||||
const reachability = r.reachable ? 30 : -20;
|
||||
const snippetQuality = Math.min(r.snippet.length, 100) / 100 * 20;
|
||||
const snippetQuality = (Math.min(r.snippet.length, 100) / 100) * 20;
|
||||
const weightScore = (r.weight / 100) * 50;
|
||||
r._score = weightScore + reachability + snippetQuality;
|
||||
}
|
||||
@@ -223,13 +384,20 @@ function computeRelevance(query: string, result: SearchResult): number {
|
||||
export class WebSearchTool implements IMetonaTool {
|
||||
readonly definition: MetonaToolDef = {
|
||||
name: 'web_search',
|
||||
description: 'Search the web for information. Returns titles, snippets, and URLs. When SearXNG is enabled, uses the configured SearXNG instance; otherwise uses built-in engines (Bing, Baidu, Sogou, 360). Automatically fetches full content for top results.',
|
||||
description:
|
||||
'Search the web for information. Returns titles, snippets, and URLs. When SearXNG is enabled, uses the configured SearXNG instance; otherwise uses built-in engines (Bing, Baidu, Sogou, 360). Automatically fetches full content for top results.',
|
||||
parameters: {
|
||||
type: 'object',
|
||||
properties: {
|
||||
query: { type: 'string', description: 'Search query keywords' },
|
||||
time_range: { type: 'string', description: 'Time filter: day, week, month, year (optional)' },
|
||||
enhance_snippets: { type: 'boolean', description: 'Auto-enhance short snippets (default true)' },
|
||||
time_range: {
|
||||
type: 'string',
|
||||
description: 'Time filter: day, week, month, year (optional)',
|
||||
},
|
||||
enhance_snippets: {
|
||||
type: 'boolean',
|
||||
description: 'Auto-enhance short snippets (default true)',
|
||||
},
|
||||
},
|
||||
required: ['query'],
|
||||
},
|
||||
@@ -265,7 +433,10 @@ export class WebSearchTool implements IMetonaTool {
|
||||
? Math.min(8, Math.max(3, searxngConfig.fetch_count > 0 ? searxngConfig.fetch_count : 5))
|
||||
: 5;
|
||||
|
||||
logTool('web_search', `Mode=${useSearXNG ? 'searxng' : 'builtin'}, maxResults=${maxResults}, fetchTop=${fetchTop}`);
|
||||
logTool(
|
||||
'web_search',
|
||||
`Mode=${useSearXNG ? 'searxng' : 'builtin'}, maxResults=${maxResults}, fetchTop=${fetchTop}`,
|
||||
);
|
||||
|
||||
// 缓存检查(key 含模式 + maxResults + fetchTop,避免配置变更后返回旧缓存)
|
||||
const cacheKey = `${searxngConfig.enabled ? 'searxng' : 'builtin'}:${maxResults}:${fetchTop}:${normalizeUrl(query).toLowerCase()}`;
|
||||
@@ -338,7 +509,10 @@ export class WebSearchTool implements IMetonaTool {
|
||||
|
||||
// 写入缓存
|
||||
searchCache.set(cacheKey, output);
|
||||
logTool('web_search', `Completed: ${sorted.length} results, ${fetchedContent.length} fetched, mode=${mode}`);
|
||||
logTool(
|
||||
'web_search',
|
||||
`Completed: ${sorted.length} results, ${fetchedContent.length} fetched, mode=${mode}`,
|
||||
);
|
||||
|
||||
return output;
|
||||
}
|
||||
@@ -394,7 +568,7 @@ export class WebSearchTool implements IMetonaTool {
|
||||
}
|
||||
}
|
||||
} else {
|
||||
const data = await response.json() as { results?: Array<Record<string, unknown>> };
|
||||
const data = (await response.json()) as { results?: Array<Record<string, unknown>> };
|
||||
for (const item of data.results ?? []) {
|
||||
const url = item.url as string;
|
||||
const title = item.title as string;
|
||||
@@ -418,7 +592,10 @@ export class WebSearchTool implements IMetonaTool {
|
||||
}
|
||||
|
||||
results.push(...pageResults);
|
||||
logTool('web_search', `[SearXNG] Page ${page}: +${pageResults.length} (total ${results.length})`);
|
||||
logTool(
|
||||
'web_search',
|
||||
`[SearXNG] Page ${page}: +${pageResults.length} (total ${results.length})`,
|
||||
);
|
||||
page++;
|
||||
}
|
||||
|
||||
@@ -440,12 +617,17 @@ export class WebSearchTool implements IMetonaTool {
|
||||
const searchPromises = ENGINES.map(async (engine) => {
|
||||
try {
|
||||
const url = engine.searchUrl(query, timeRange);
|
||||
const response = await fetchWithTimeout(url, {
|
||||
headers: {
|
||||
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/131.0.0.0 Safari/537.36',
|
||||
'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
|
||||
const response = await fetchWithTimeout(
|
||||
url,
|
||||
{
|
||||
headers: {
|
||||
'User-Agent':
|
||||
'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/131.0.0.0 Safari/537.36',
|
||||
'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
|
||||
},
|
||||
},
|
||||
}, 8_000);
|
||||
8_000,
|
||||
);
|
||||
|
||||
if (!response.ok) {
|
||||
logTool('web_search', `[内置] ${engine.name} HTTP ${response.status}`);
|
||||
@@ -501,10 +683,10 @@ export class WebSearchTool implements IMetonaTool {
|
||||
if (enhanced >= maxEnhance) break;
|
||||
if (r.snippet.length < 30 && r.reachable) {
|
||||
try {
|
||||
const fetchResult = await this.webFetchTool.execute(
|
||||
const fetchResult = (await this.webFetchTool.execute(
|
||||
{ url: r.url },
|
||||
{ sessionId: '', workspacePath: '', iteration: 0, requestId: '' },
|
||||
) as { success: boolean; content?: string };
|
||||
)) as { success: boolean; content?: string };
|
||||
|
||||
if (fetchResult.success && fetchResult.content) {
|
||||
const text = fetchResult.content.slice(0, 200);
|
||||
@@ -537,7 +719,10 @@ export class WebSearchTool implements IMetonaTool {
|
||||
fetchTop: number,
|
||||
): Promise<Array<{ url: string; title: string; content: string }>> {
|
||||
// 相关性评分(不过滤,relevance=0 的结果也参与抓取候选)
|
||||
const withRelevance = results.map((r) => ({ result: r, relevance: computeRelevance(query, r) }));
|
||||
const withRelevance = results.map((r) => ({
|
||||
result: r,
|
||||
relevance: computeRelevance(query, r),
|
||||
}));
|
||||
const filtered = withRelevance.length > 0 ? withRelevance : [];
|
||||
|
||||
// 确定抓取数量:fetchTop 已在 execute() 中综合了配置面板和工具参数
|
||||
@@ -554,27 +739,30 @@ export class WebSearchTool implements IMetonaTool {
|
||||
}
|
||||
toFetch = shuffled.slice(0, topN);
|
||||
} else {
|
||||
toFetch = filtered
|
||||
.sort((a, b) => b.relevance - a.relevance)
|
||||
.slice(0, topN);
|
||||
toFetch = filtered.sort((a, b) => b.relevance - a.relevance).slice(0, topN);
|
||||
}
|
||||
|
||||
const fetched: Array<{ url: string; title: string; content: string }> = [];
|
||||
|
||||
const fetchOne = async (item: { result: SearchResult }): Promise<{ url: string; title: string; content: string } | null> => {
|
||||
const fetchOne = async (item: {
|
||||
result: SearchResult;
|
||||
}): Promise<{ url: string; title: string; content: string } | null> => {
|
||||
try {
|
||||
// 委托给 WebFetchTool — 享受三阶段回退策略(HTTP + 反爬 + 浏览器渲染)
|
||||
const fetchResult = await this.webFetchTool.execute(
|
||||
const fetchResult = (await this.webFetchTool.execute(
|
||||
{ url: item.result.url },
|
||||
{ sessionId: '', workspacePath: '', iteration: 0, requestId: '' },
|
||||
) as { success: boolean; content?: string };
|
||||
)) as { success: boolean; content?: string };
|
||||
|
||||
if (fetchResult.success && fetchResult.content) {
|
||||
return { url: item.result.url, title: item.result.title, content: fetchResult.content };
|
||||
}
|
||||
return null;
|
||||
} catch (err) {
|
||||
logTool('web_search', `Auto-fetch failed for ${item.result.url}: ${(err as Error).message}`);
|
||||
logTool(
|
||||
'web_search',
|
||||
`Auto-fetch failed for ${item.result.url}: ${(err as Error).message}`,
|
||||
);
|
||||
return null;
|
||||
}
|
||||
};
|
||||
@@ -601,7 +789,9 @@ export class WebSearchTool implements IMetonaTool {
|
||||
lines.push(`${i + 1}. ${r.title}`);
|
||||
lines.push(` URL: ${r.url}`);
|
||||
if (r.snippet) lines.push(` 摘要: ${r.snippet.slice(0, 150)}`);
|
||||
lines.push(` 来源: ${r.engine}${r.reachable === false ? ' (不可达)' : ''}${r._enhanced ? ' [已增强]' : ''}\n`);
|
||||
lines.push(
|
||||
` 来源: ${r.engine}${r.reachable === false ? ' (不可达)' : ''}${r._enhanced ? ' [已增强]' : ''}\n`,
|
||||
);
|
||||
});
|
||||
return lines.join('\n');
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user