feat: v0.4.1 质量加固版 — 工程化基线 + 安全加固 + 测试补齐 + 体验升级
CI / 类型检查 + Lint + 单元测试 (push) Failing after 5m25s
CI / 全量测试 (Electron ABI, experimental) (push) Failing after 5m19s
CI / 产物编译验证 (push) Successful in 10m3s

工程化(从零到一):
- 新增 Gitea Actions CI(debian-latest):类型检查 + Lint + 单元测试 + 产物编译验证
- 新增 husky + lint-staged 预提交钩子(lint-staged + typecheck 门禁)
- 移除坏脚本 test:e2e(无 Playwright 配置必失败);prebuild 改用内置 fs.rmSync
- 依赖清理:移除死依赖 sql.js(2MB)/@playwright/test,@types/shell-quote 移至 devDependencies

安全加固:
- PolicyEngine 频率限制按会话隔离(多会话并发不再互抢配额)
- ConfirmationHook 拒绝记忆加 10 分钟 TTL + 恢复询问入口(新增 2 个 IPC 通道)
- Windows run_command 白名单工具(git/node/npm/npx/pnpm/yarn/tsc)改走 cmd.exe /c + 参数数组执行,收窄 shell 注入面
- web_search 四引擎 HTML 解析迁移 node-html-parser(结构化主层 + 正则降级)

缺陷修复(测试驱动发现):
- mapError 大小写缺陷:网络错误码永远落入 UNKNOWN 无法触发重试
- 搜狗解析器自我过滤:相对链接补全后又被 sogou.com 过滤导致结果全丢
- 百度复合类名重复收录:class="result c-container" 被双重匹配

测试补齐(113 → 194 用例):
- 新增 5 个测试文件:sse-stream / base-adapter / confirmation-hook / ipc-agent 编排链路 / web-search 解析器
- 覆盖 sendMessage 全分支、SSE 流解析、错误映射、确认钩子竞态/超时/批量审批

体验升级:
- OutputValidator 验证结果可见化(VALIDATION 流事件 → 聊天流提示卡)
- SettingsModal 巨型组件拆分(1503 行 → 10 个文件,可独立维护)
- MessageList 接入 react-virtuoso 真虚拟滚动(千条消息恒定开销)
- MCP 新增 streamable HTTP 传输支持(SDK 内置传输 + DB 迁移 6 + UI 双模式)
This commit is contained in:
2026-08-21 13:58:48 +08:00
parent 2230bcec3f
commit 49c9b25538
41 changed files with 6254 additions and 2608 deletions
@@ -15,7 +15,9 @@ describe('RunCommandTool.validateCommand', () => {
const tool = new RunCommandTool();
// 访问私有方法
const validate = (cmd: string) =>
(tool as unknown as { validateCommand: (c: string) => { allowed: boolean; reason?: string } }).validateCommand(cmd);
(
tool as unknown as { validateCommand: (c: string) => { allowed: boolean; reason?: string } }
).validateCommand(cmd);
const blocked = (cmd: string) => {
const result = validate(cmd);
@@ -87,3 +89,28 @@ describe('RunCommandTool.validateCommand', () => {
allowed('rm -rf node_modules');
});
});
describe('RunCommandTool — Windows execFile 白名单(v0.4.1', () => {
const tool = new RunCommandTool();
const parseSimple = (cmd: string) =>
(
tool as unknown as {
parseCommandSimple: (c: string) => { command: string; args: string[] } | null;
}
).parseCommandSimple(cmd);
it('白名单命令解析为简单命令(无 shell 运算符)', () => {
const npm = parseSimple('npm install');
expect(npm).toEqual({ command: 'npm', args: ['install'] });
const git = parseSimple('git commit -m "fix: bug"');
expect(git).toEqual({ command: 'git', args: ['commit', '-m', 'fix: bug'] });
const node = parseSimple('node dist/main.js');
expect(node).toEqual({ command: 'node', args: ['dist/main.js'] });
});
it('含 shell 运算符的命令不解析为简单命令(继续走 exec 双层校验)', () => {
expect(parseSimple('npm install && npm test')).toBeNull();
expect(parseSimple('git log | head -5')).toBeNull();
expect(parseSimple('echo hi > out.txt')).toBeNull();
});
});
@@ -0,0 +1,174 @@
/**
* web_search 搜索引擎 HTML 解析器单元测试(v0.4.1 测试补齐)
* 覆盖:node-html-parser 结构化解析(主层)、自域名链接过滤、
* 相对链接补全、空/异常 HTML 容错
*/
import { describe, it, expect } from 'vitest';
import { parseBing, parseBaidu, parseSogou, parse360 } from '../web-search';
describe('parseBing — 结构化解析', () => {
const BING_HTML = `
<html><body>
<ol id="b_results">
<li class="b_algo">
<h2><a href="https://example.com/article-1">第一篇 TypeScript 文章</a></h2>
<p>这是第一条结果的摘要内容,讲述 TypeScript 高级用法。</p>
</li>
<li class="b_algo">
<h2><a href="https://example.com/article-2">第二篇 Node.js 文章</a></h2>
<div class="b_caption"><p>第二条结果的摘要。</p></div>
</li>
<li class="b_algo">
<!-- 自身域名链接应被过滤 -->
<h2><a href="https://www.bing.com/video?q=x">Bing 内部视频链接</a></h2>
<p>不应出现在结果中。</p>
</li>
</ol>
</body></html>
`;
it('解析结果块并提取 title/url/snippet', () => {
const results = parseBing(BING_HTML);
expect(results).toHaveLength(2);
expect(results[0]).toMatchObject({
title: '第一篇 TypeScript 文章',
url: 'https://example.com/article-1',
snippet: '这是第一条结果的摘要内容,讲述 TypeScript 高级用法。',
engine: 'bing',
weight: 90,
});
expect(results[1].url).toBe('https://example.com/article-2');
});
it('过滤指向 bing.com 自身域名的链接', () => {
const results = parseBing(BING_HTML);
expect(results.some((r) => r.url.includes('bing.com'))).toBe(false);
});
it('空 HTML 返回空数组', () => {
expect(parseBing('')).toHaveLength(0);
expect(parseBing('<html><body></body></html>')).toHaveLength(0);
});
});
describe('parseBaidu — 结构化解析', () => {
const BAIDU_HTML = `
<html><body>
<div id="content_left">
<div class="result c-container new-pmd" srcid="1">
<h3 class="t"><a data-url="https://example.com/real-url-1" href="https://www.baidu.com/link?url=xyz">百度结果一</a></h3>
<span class="content-right_8Zs40">第一条摘要内容。</span>
</div>
<div class="result c-container" srcid="2">
<h3><a href="https://www.baidu.com/link?url=abc">跳转链接结果</a></h3>
<span class="content-right_8Zs40">无 data-url 的结果(跳转链接被过滤)。</span>
</div>
</div>
</body></html>
`;
it('优先使用 data-url 真实链接', () => {
const results = parseBaidu(BAIDU_HTML);
expect(results).toHaveLength(1);
expect(results[0]).toMatchObject({
title: '百度结果一',
url: 'https://example.com/real-url-1',
engine: '百度',
weight: 80,
});
});
it('baidu.com/link 跳转链接被过滤', () => {
const results = parseBaidu(BAIDU_HTML);
expect(results.some((r) => r.url.includes('baidu.com/link'))).toBe(false);
});
it('空 HTML 返回空数组', () => {
expect(parseBaidu('')).toHaveLength(0);
});
});
describe('parseSogou — 结构化解析', () => {
const SOGOU_HTML = `
<html><body>
<div class="results">
<div class="vrwrap">
<h3><a href="/link?url=sogou-internal-1">搜狗结果一</a></h3>
<div class="str_info">搜狗结果一的摘要文本。</div>
</div>
<div class="rb">
<h3><a href="https://example.com/direct">搜狗直链结果</a></h3>
<p class="space-txt">直链结果的摘要。</p>
</div>
</div>
</body></html>
`;
it('相对链接补全 sogou.com 前缀', () => {
const results = parseSogou(SOGOU_HTML);
expect(results).toHaveLength(2);
expect(results[0]).toMatchObject({
title: '搜狗结果一',
url: 'https://www.sogou.com/link?url=sogou-internal-1',
engine: '搜狗',
weight: 75,
});
});
it('http 开头的直链不补全前缀', () => {
const results = parseSogou(SOGOU_HTML);
expect(results[1].url).toBe('https://example.com/direct');
expect(results[1].snippet).toBe('直链结果的摘要。');
});
it('空 HTML 返回空数组', () => {
expect(parseSogou('')).toHaveLength(0);
});
});
describe('parse360 — 结构化解析', () => {
const SO360_HTML = `
<html><body>
<div class="res-list">
<ul>
<li class="res-list">
<h3 class="res-title"><a href="https://example.com/360-result">360 结果一</a></h3>
<p class="res-desc">360 搜索结果的摘要描述。</p>
</li>
<li class="res-list">
<h3><a href="https://www.so.com/internal">360 内部链接</a></h3>
<p class="res-desc">不应出现。</p>
</li>
</ul>
</div>
</body></html>
`;
it('解析结果并过滤 so.com 自身链接', () => {
const results = parse360(SO360_HTML);
expect(results).toHaveLength(1);
expect(results[0]).toMatchObject({
title: '360 结果一',
url: 'https://example.com/360-result',
snippet: '360 搜索结果的摘要描述。',
engine: '360搜索',
weight: 75,
});
});
it('空 HTML 返回空数组', () => {
expect(parse360('')).toHaveLength(0);
});
});
describe('解析器降级路径', () => {
it('结构化解析无结果且正则也无结果时返回空数组(不抛错)', () => {
// 非搜索结果页 HTML(如错误页/验证码页)
const notSearchPage = '<html><body><div class="captcha">请输入验证码</div></body></html>';
expect(parseBing(notSearchPage)).toHaveLength(0);
expect(parseBaidu(notSearchPage)).toHaveLength(0);
expect(parseSogou(notSearchPage)).toHaveLength(0);
expect(parse360(notSearchPage)).toHaveLength(0);
});
});
+113 -27
View File
@@ -58,13 +58,22 @@ function decodeBuffer(buf: Buffer): string {
function buildSafeCommandEnv(isWindows: boolean): Record<string, string> {
// 敏感变量后缀黑名单
const SENSITIVE_SUFFIXES = [
'_API_KEY', '_TOKEN', '_SECRET', '_PASSWORD', '_PASSWD',
'_CREDENTIAL', '_CREDENTIALS', '_PRIVATE_KEY',
'_API_KEY',
'_TOKEN',
'_SECRET',
'_PASSWORD',
'_PASSWD',
'_CREDENTIAL',
'_CREDENTIALS',
'_PRIVATE_KEY',
];
// 敏感变量名黑名单(精确匹配)
const SENSITIVE_KEYS = new Set([
'DEEPSEEK_API_KEY', 'AGNES_API_KEY', 'MIMO_API_KEY',
'GITEA_PASSWORD', 'DATABASE_PASSWORD',
'DEEPSEEK_API_KEY',
'AGNES_API_KEY',
'MIMO_API_KEY',
'GITEA_PASSWORD',
'DATABASE_PASSWORD',
]);
const env: Record<string, string> = {};
@@ -86,12 +95,31 @@ function buildSafeCommandEnv(isWindows: boolean): Record<string, string> {
return env;
}
/**
* v0.4.1: Windows 白名单命令集合 — 这些工具的简单命令(无 shell 运算符)走
* execFile('cmd.exe', ['/c', ...words]) 执行:参数以数组形式显式传递,不经过
* shell 解析,从根本上去掉 exec() 的字符串拼接注入面(无法通过参数注入新命令)。
*
* 仅收录最常见的开发工具(小步灰度);其余命令仍走 exec + 双层校验的既有路径。
* Node 18.20+/Electron 35 在 Windows 上直接 spawn .cmd 批处理会被拒绝(EINVAL),
* 因此必须通过 cmd.exe /c 中转,但参数分离已足够收窄注入面。
*/
const WINDOWS_EXEC_FILE_WHITELIST = new Set(['git', 'node', 'npm', 'npx', 'pnpm', 'yarn', 'tsc']);
/** v0.4.1: 提取命令 basename(处理 C:\Program Files\nodejs\npm.cmd 等路径形式) */
function commandBasename(cmd: string): string {
const base = cmd.split(/[\\/]/).pop() ?? cmd;
// 去掉 .exe/.cmd/.bat 扩展名(大小写不敏感)
return base.replace(/\.(exe|cmd|bat)$/i, '');
}
// ===== 9. run_command =====
export class RunCommandTool implements IMetonaTool {
readonly definition: MetonaToolDef = {
name: 'run_command',
description: 'Execute a shell command in a sandboxed environment. Commands run in the workspace directory. High-risk commands require user confirmation. Passes through SandboxManager static code scan and path validation.',
description:
'Execute a shell command in a sandboxed environment. Commands run in the workspace directory. High-risk commands require user confirmation. Passes through SandboxManager static code scan and path validation.',
parameters: {
type: 'object',
properties: {
@@ -123,7 +151,11 @@ export class RunCommandTool implements IMetonaTool {
// 安全校验:workdir 必须在工作空间内
const resolvedWorkdir = resolve(context.workspacePath, workdir);
if (!isPathWithinWorkspace(workdir, context.workspacePath)) {
return { success: false, error: `Working directory must be within workspace: ${workdir}`, command };
return {
success: false,
error: `Working directory must be within workspace: ${workdir}`,
command,
};
}
// v0.2.0: SandboxManager 双重安全校验 — fail-closed 设计
@@ -183,19 +215,32 @@ export class RunCommandTool implements IMetonaTool {
let stderr: Buffer;
// #8 修复 + 审查修复: 简单命令使用 execFile(不经过 shell,防止命令注入)
// 但 Windows 上 npm/npx/yarn/pnpm/tsc 等是 .cmd 批处理,execFile 无法执行(ENOENT
// 因此 Windows 上仍用 exec(已有 SandboxManager.scanCode + validateCommand 双层校验)
// 非 Windows 上对简单命令用 execFile
// 但 Windows 上 npm/npx/yarn/pnpm/tsc 等是 .cmd 批处理,execFile 无法直接执行(ENOENT/EINVAL
// v0.4.1: Windows 上白名单工具(git/node/npm/npx/pnpm/yarn/tsc)的简单命令改用
// execFile('cmd.exe', ['/c', ...args]) — 参数显式分离传递,不经 shell 字符串解析,
// 相比 exec() 的整串拼接显著收窄注入面
// 非 Windows 上对简单命令直接 execFile
if (simpleCmd && !isWindows) {
const result = await execFileAsync(simpleCmd.command, simpleCmd.args, execOpts);
stdout = result.stdout;
stderr = result.stderr;
} else if (
simpleCmd &&
isWindows &&
WINDOWS_EXEC_FILE_WHITELIST.has(commandBasename(simpleCmd.command))
) {
// v0.4.1: 白名单工具通过 cmd.exe /c + 参数数组执行(参数不经 shell 解析)
const result = await execFileAsync(
'cmd.exe',
['/c', simpleCmd.command, ...simpleCmd.args],
execOpts,
);
stdout = result.stdout;
stderr = result.stderr;
} else {
// 复杂命令(含管道/重定向/&& 等 shell 语法)或 Windows — 使用 exec
// 已有 SandboxManager.scanCode + validateCommand 双层安全校验
const finalCommand = isWindows
? `chcp 65001 >nul 2>&1 && ${command}`
: command;
const finalCommand = isWindows ? `chcp 65001 >nul 2>&1 && ${command}` : command;
const result = await execAsync(finalCommand, execOpts);
stdout = result.stdout;
stderr = result.stderr;
@@ -236,7 +281,11 @@ export class RunCommandTool implements IMetonaTool {
// 受保护文件检查:禁止通过命令行读写工作空间根目录的 MEMORY.md
if (commandTouchesProtectedFile(command)) {
return { allowed: false, reason: 'Access denied: MEMORY.md is managed by the memory system and cannot be accessed via command execution' };
return {
allowed: false,
reason:
'Access denied: MEMORY.md is managed by the memory system and cannot be accessed via command execution',
};
}
// P0-5: 剥离 Windows chcp 前缀("chcp 65001 >nul 2>&1 &&" 会破坏 shell-quote
@@ -256,33 +305,61 @@ export class RunCommandTool implements IMetonaTool {
const hardBlocks = [
// 文件系统破坏
{ pattern: /\brm\b.*\//, reason: 'rm with absolute path is forbidden' },
{ pattern: /\brm\s+-rf?\s+\/(?:[^|;&\s]*\s)*?(?:bin|boot|dev|etc|lib|proc|root|sbin|sys|usr|var)\b/i, reason: 'rm on system directories is forbidden' },
{
pattern:
/\brm\s+-rf?\s+\/(?:[^|;&\s]*\s)*?(?:bin|boot|dev|etc|lib|proc|root|sbin|sys|usr|var)\b/i,
reason: 'rm on system directories is forbidden',
},
{ pattern: /\b(sudo|su|doas)\b/, reason: 'Privilege escalation commands are forbidden' },
// 系统控制
{ pattern: /\b(shutdown|reboot|halt|poweroff)\b/, reason: 'System shutdown commands are forbidden' },
{
pattern: /\b(shutdown|reboot|halt|poweroff)\b/,
reason: 'System shutdown commands are forbidden',
},
{ pattern: /\b(killall|pkill)\s+-9\b/, reason: 'Force kill all processes is forbidden' },
// 远程代码执行
{ pattern: /curl.*\|\s*(ba)?sh/, reason: 'Remote code execution via pipe is forbidden' },
{ pattern: /wget.*\|\s*(ba)?sh/, reason: 'Remote code execution via pipe is forbidden' },
{ pattern: /\bcurl\s+.*\s*-o\s+\/etc\//i, reason: 'Writing to system directories via curl is forbidden' },
{
pattern: /\bcurl\s+.*\s*-o\s+\/etc\//i,
reason: 'Writing to system directories via curl is forbidden',
},
// 设备文件
{ pattern: /\bdd\b.*of=\/dev\//, reason: 'Writing to device files is forbidden' },
// 磁盘格式化
{ pattern: /\b(mkfs|fdisk)\b/, reason: 'Disk formatting commands are forbidden' },
// 权限滥用
{ pattern: /\bchmod\s+777\b/, reason: 'chmod 777 is forbidden' },
{ pattern: /\bchown\s+-R\s+\S+\s+\/(?:\s|$)/i, reason: 'Recursive chown on root is forbidden' },
{
pattern: /\bchown\s+-R\s+\S+\s+\/(?:\s|$)/i,
reason: 'Recursive chown on root is forbidden',
},
// 环境变量窃取
{ pattern: /\b(env|export|printenv)\s*\|.*\b(curl|wget|nc|ncat)\b/i, reason: 'Exfiltrating environment variables is forbidden' },
{
pattern: /\b(env|export|printenv)\s*\|.*\b(curl|wget|nc|ncat)\b/i,
reason: 'Exfiltrating environment variables is forbidden',
},
// 反向 shell
{ pattern: /\b(bash|sh|zsh)\s+-i\s+>\s*&\s*\/dev\/tcp\//i, reason: 'Reverse shell via /dev/tcp is forbidden' },
{
pattern: /\b(bash|sh|zsh)\s+-i\s+>\s*&\s*\/dev\/tcp\//i,
reason: 'Reverse shell via /dev/tcp is forbidden',
},
{ pattern: /\bnc\s+.*\s+-e\s+(bash|sh)/i, reason: 'Reverse shell via netcat is forbidden' },
// Windows 危险命令
{ pattern: /\b(format|diskpart)\b/i, reason: 'Disk formatting commands are forbidden' },
{ pattern: /\bshutdown\s*\//i, reason: 'System shutdown commands are forbidden' },
{ pattern: /\breg\s+(add|delete|import|restore)/i, reason: 'Registry modification commands are forbidden' },
{ pattern: /\b(taskkill|kill)\s*\//i, reason: 'Process termination with system flags is forbidden' },
{ pattern: /\bpowershell\s+-enc\s+/i, reason: 'PowerShell encoded command execution is forbidden' },
{
pattern: /\breg\s+(add|delete|import|restore)/i,
reason: 'Registry modification commands are forbidden',
},
{
pattern: /\b(taskkill|kill)\s*\//i,
reason: 'Process termination with system flags is forbidden',
},
{
pattern: /\bpowershell\s+-enc\s+/i,
reason: 'PowerShell encoded command execution is forbidden',
},
// 后台进程与管道炸弹
{ pattern: /&\s*\(/, reason: 'Background subshell execution is forbidden' },
{ pattern: /\|\s*&/, reason: 'Pipe to background process is forbidden' },
@@ -342,20 +419,29 @@ export class RunCommandTool implements IMetonaTool {
prevWasPipe = false;
} else if (typeof obj.op === 'string') {
// 跟踪管道运算符,用于下一轮检测 `| sh`
prevWasPipe = (obj.op === '|');
prevWasPipe = obj.op === '|';
}
}
}
// 危险命令名 token(精确匹配,大小写不敏感)
const dangerousCommands = new Set([
'sudo', 'su', 'doas',
'shutdown', 'reboot', 'halt', 'poweroff',
'mkfs', 'fdisk', 'format', 'diskpart',
'sudo',
'su',
'doas',
'shutdown',
'reboot',
'halt',
'poweroff',
'mkfs',
'fdisk',
'format',
'diskpart',
]);
// 危险参数 token
const dangerousArgs = new Set([
'-enc', '-encodedcommand', // PowerShell 编码执行
'-enc',
'-encodedcommand', // PowerShell 编码执行
]);
for (const word of words) {
+235 -45
View File
@@ -6,9 +6,15 @@
* 智能排序:引擎权重(50%) + 可达性(30%) + 摘要质量(20%)
* 自动抓取:对前 N 条结果调用 web_fetch 获取完整正文
*
* v0.4.1: HTML 解析迁移至 node-html-parser(结构化解析)
* 主层使用 DOM 结构解析(引擎改版时选择器更精确、可维护性远优于正则),
* 正则解析保留为降级路径(结构化解析无结果时兜底)。
* 此前纯正则方案违反项目开发规范第一铁律(HTML 解析应使用成熟库)。
*
* @see docs/Agent网络工具通用设计-v2.md — 第 2 章 web_search 搜索设计
*/
import { parse as parseHtmlDom, type HTMLElement } from 'node-html-parser';
import type { IMetonaTool, ToolExecutionContext } from '../../types/metona-tool';
import type { MetonaToolDef } from '../../../harness/types';
import { MetonaToolCategory, MetonaRiskLevel } from '../../../harness/types';
@@ -64,7 +70,9 @@ const ENGINES: EngineDef[] = [
name: 'bing',
weight: 90,
searchUrl: (q, tr) => {
const freshness = tr ? `&filters=ex1:"ez${tr === 'day' ? '1' : tr === 'week' ? '2' : tr === 'month' ? '3' : '4'}"` : '';
const freshness = tr
? `&filters=ex1:"ez${tr === 'day' ? '1' : tr === 'week' ? '2' : tr === 'month' ? '3' : '4'}"`
: '';
return `https://www.bing.com/search?q=${encodeURIComponent(q)}${freshness}&count=20`;
},
parse: parseBing,
@@ -89,9 +97,150 @@ const ENGINES: EngineDef[] = [
},
];
// ===== HTML 解析器(正则实现,后续可迁移至 cheerio =====
// ===== HTML 解析器(v0.4.1: node-html-parser 结构化解析为主层,正则为降级层 =====
function parseBing(html: string): SearchResult[] {
/**
* v0.4.1: 从结果块中提取标题链接 — 跳过指向搜索引擎自身域名的链接(favicon/子导航等)
*/
function extractTitleLink(
block: HTMLElement,
selfDomain: string,
): { url: string; title: string } | null {
for (const a of block.querySelectorAll('a[href]')) {
const url = a.getAttribute('href') ?? '';
const title = a.text.trim();
if (title && url && !url.includes(selfDomain) && url.startsWith('http')) {
return { url, title };
}
}
return null;
}
/** v0.4.1: 提取第一个非空文本的选择器(按优先级尝试多个候选选择器) */
function extractText(block: HTMLElement, selectors: string[]): string {
for (const sel of selectors) {
const el = block.querySelector(sel);
if (el) {
const text = el.text.trim();
if (text) return text;
}
}
return '';
}
/** v0.4.1: Bing 结构化解析 — li.b_algo 结果块 */
function parseBingStructured(html: string): SearchResult[] {
const results: SearchResult[] = [];
const root = parseHtmlDom(html);
for (const block of root.querySelectorAll('li.b_algo')) {
const link = extractTitleLink(block, 'bing.com');
if (!link) continue;
const snippet = extractText(block, ['p', '.b_caption']);
results.push({ title: link.title, url: link.url, snippet, engine: 'bing', weight: 90 });
}
return results;
}
/** v0.4.1: 百度结构化解析 — div.result / div.c-container 结果块,优先 a[data-url] 真实链接 */
function parseBaiduStructured(html: string): SearchResult[] {
const results: SearchResult[] = [];
const root = parseHtmlDom(html);
// 复合选择器去重:class="result c-container" 的元素同时命中两个类名,
// 分别查询再拼接会重复收录同一结果块
const blocks = root.querySelectorAll('div.result, div.c-container');
for (const block of blocks) {
// 百度标题链接: 优先 data-url 属性(真实目标 URL),href 通常是 baidu.com/link 跳转
const dataUrlLink = block.querySelector('a[data-url]');
let url = dataUrlLink?.getAttribute('data-url') ?? '';
let title = dataUrlLink?.text.trim() ?? '';
if (!url || !title) {
const fallback = block.querySelector('h3 a[href]') ?? block.querySelector('a[href]');
if (fallback) {
const href = fallback.getAttribute('href') ?? '';
url = href.startsWith('http') ? href : href ? `https://${href}` : '';
title = fallback.text.trim();
}
}
const snippet = extractText(block, ['.c-abstract', '[class^="content-right"]']);
if (title && url && !url.includes('baidu.com/link')) {
results.push({ title, url, snippet, engine: '百度', weight: 80 });
}
}
return results;
}
/**
* v0.4.1: 搜狗结构化解析 — div.vrwrap / div.rb 结果块(相对链接补全 sogou.com 前缀)
*
* v0.4.1 修复(原正则实现遗留缺陷): 搜狗结果链接是 sogou.com/link?url=... 跳转形式,
* 原 `!url.includes('sogou.com')` 过滤条件把所有跳转结果一并丢弃(相对链接补全后必含 sogou.com),
* 导致搜狗引擎基本无法返回结果。现仅过滤 sogou 自身页面链接,保留 /link 跳转结果
* (可达性预检会跟随重定向验证)。
*/
function parseSogouStructured(html: string): SearchResult[] {
const results: SearchResult[] = [];
const root = parseHtmlDom(html);
// 复合选择器避免同一元素命中两个类名时重复收录
const blocks = root.querySelectorAll('div.vrwrap, div.rb');
for (const block of blocks) {
const a = block.querySelector('h3 a[href]') ?? block.querySelector('a[href]');
if (!a) continue;
const href = a.getAttribute('href') ?? '';
const url = href.startsWith('http') ? href : `https://www.sogou.com${href}`;
const title = a.text.trim();
const snippet = extractText(block, ['.star-wiki', '.space-txt', '.str_info']);
// 过滤搜狗自身页面(保留 /link 跳转结果)
const isSelfPage = url.includes('sogou.com') && !url.includes('/link');
if (title && url && !isSelfPage) {
results.push({ title, url, snippet, engine: '搜狗', weight: 75 });
}
}
return results;
}
/** v0.4.1: 360 结构化解析 — li.res-list / div.result 结果块 */
function parse360Structured(html: string): SearchResult[] {
const results: SearchResult[] = [];
const root = parseHtmlDom(html);
// 复合选择器避免同一元素命中多个类名时重复收录
const blocks = root.querySelectorAll('li.res-list, div.result');
for (const block of blocks) {
const link = extractTitleLink(block, 'so.com');
if (!link) continue;
const snippet = extractText(block, ['.res-desc', '.res-rich', '.res-summary', 'dd']);
results.push({ title: link.title, url: link.url, snippet, engine: '360搜索', weight: 75 });
}
return results;
}
/** v0.4.1: 结构化解析 + 正则降级的组合入口(供 ENGINES 引用,测试导出) */
export function parseBing(html: string): SearchResult[] {
const structured = parseBingStructured(html);
if (structured.length > 0) return structured;
return parseBingRegex(html);
}
export function parseBaidu(html: string): SearchResult[] {
const structured = parseBaiduStructured(html);
if (structured.length > 0) return structured;
return parseBaiduRegex(html);
}
export function parseSogou(html: string): SearchResult[] {
const structured = parseSogouStructured(html);
if (structured.length > 0) return structured;
return parseSogouRegex(html);
}
export function parse360(html: string): SearchResult[] {
const structured = parse360Structured(html);
if (structured.length > 0) return structured;
return parse360Regex(html);
}
// ===== 正则降级解析器(v0.4.1 前的主实现,结构化解析无结果时兜底) =====
function parseBingRegex(html: string): SearchResult[] {
const results: SearchResult[] = [];
const blocks = html.split(/<li[^>]*class="b_algo"/i).slice(1);
for (const block of blocks) {
@@ -99,7 +248,9 @@ function parseBing(html: string): SearchResult[] {
if (!titleMatch) continue;
const url = titleMatch[1];
const title = titleMatch[2].replace(/<[^>]+>/g, '').trim();
const snippetMatch = block.match(/<p[^>]*>([\s\S]*?)<\/p>/i) || block.match(/class="b_caption"[^>]*>([\s\S]*?)<\/div>/i);
const snippetMatch =
block.match(/<p[^>]*>([\s\S]*?)<\/p>/i) ||
block.match(/class="b_caption"[^>]*>([\s\S]*?)<\/div>/i);
const snippet = snippetMatch ? snippetMatch[1].replace(/<[^>]+>/g, '').trim() : '';
if (title && url && !url.includes('bing.com')) {
results.push({ title, url, snippet, engine: 'bing', weight: 90 });
@@ -108,17 +259,19 @@ function parseBing(html: string): SearchResult[] {
return results;
}
function parseBaidu(html: string): SearchResult[] {
function parseBaiduRegex(html: string): SearchResult[] {
const results: SearchResult[] = [];
const blocks = html.split(/<div[^>]*class="result[^"]*"/i).slice(1);
for (const block of blocks) {
const titleMatch = block.match(/<a[^>]*data-url="([^"]+)"[^>]*>([\s\S]*?)<\/a>/i)
|| block.match(/<a[^>]*href="([^"]+)"[^>]*>([\s\S]*?)<\/a>/i);
const titleMatch =
block.match(/<a[^>]*data-url="([^"]+)"[^>]*>([\s\S]*?)<\/a>/i) ||
block.match(/<a[^>]*href="([^"]+)"[^>]*>([\s\S]*?)<\/a>/i);
if (!titleMatch) continue;
const url = titleMatch[1].startsWith('http') ? titleMatch[1] : `https://${titleMatch[1]}`;
const title = titleMatch[2].replace(/<[^>]+>/g, '').trim();
const snippetMatch = block.match(/class="c-abstract[^"]*"[^>]*>([\s\S]*?)<\/span>/i)
|| block.match(/class="content-right[^"]*"[^>]*>([\s\S]*?)<\/div>/i);
const snippetMatch =
block.match(/class="c-abstract[^"]*"[^>]*>([\s\S]*?)<\/span>/i) ||
block.match(/class="content-right[^"]*"[^>]*>([\s\S]*?)<\/div>/i);
const snippet = snippetMatch ? snippetMatch[1].replace(/<[^>]+>/g, '').trim() : '';
if (title && url && !url.includes('baidu.com/link')) {
results.push({ title, url, snippet, engine: '百度', weight: 80 });
@@ -127,18 +280,23 @@ function parseBaidu(html: string): SearchResult[] {
return results;
}
function parseSogou(html: string): SearchResult[] {
function parseSogouRegex(html: string): SearchResult[] {
const results: SearchResult[] = [];
const blocks = html.split(/<div[^>]*class="vrwrap"/i).slice(1)
const blocks = html
.split(/<div[^>]*class="vrwrap"/i)
.slice(1)
.concat(html.split(/<div[^>]*class="rb"/i).slice(1));
for (const block of blocks) {
const titleMatch = block.match(/<a[^>]*href="([^"]+)"[^>]*>([\s\S]*?)<\/a>/i);
if (!titleMatch) continue;
const url = titleMatch[1].startsWith('http') ? titleMatch[1] : `https://www.sogou.com${titleMatch[1]}`;
const url = titleMatch[1].startsWith('http')
? titleMatch[1]
: `https://www.sogou.com${titleMatch[1]}`;
const title = titleMatch[2].replace(/<[^>]+>/g, '').trim();
const snippetMatch = block.match(/class="star-wiki[^"]*"[^>]*>([\s\S]*?)<\/div>/i)
|| block.match(/class="space-txt[^"]*"[^>]*>([\s\S]*?)<\/p>/i)
|| block.match(/class="str_info[^"]*"[^>]*>([\s\S]*?)<\/p>/i);
const snippetMatch =
block.match(/class="star-wiki[^"]*"[^>]*>([\s\S]*?)<\/div>/i) ||
block.match(/class="space-txt[^"]*"[^>]*>([\s\S]*?)<\/p>/i) ||
block.match(/class="str_info[^"]*"[^>]*>([\s\S]*?)<\/p>/i);
const snippet = snippetMatch ? snippetMatch[1].replace(/<[^>]+>/g, '').trim() : '';
if (title && url && !url.includes('sogou.com')) {
results.push({ title, url, snippet, engine: '搜狗', weight: 75 });
@@ -147,19 +305,22 @@ function parseSogou(html: string): SearchResult[] {
return results;
}
function parse360(html: string): SearchResult[] {
function parse360Regex(html: string): SearchResult[] {
const results: SearchResult[] = [];
const blocks = html.split(/<li[^>]*class="res-list"/i).slice(1)
const blocks = html
.split(/<li[^>]*class="res-list"/i)
.slice(1)
.concat(html.split(/<div[^>]*class="result"/i).slice(1));
for (const block of blocks) {
const titleMatch = block.match(/<a[^>]*href="([^"]+)"[^>]*>([\s\S]*?)<\/a>/i);
if (!titleMatch) continue;
const url = titleMatch[1];
const title = titleMatch[2].replace(/<[^>]+>/g, '').trim();
const snippetMatch = block.match(/class="res-desc[^"]*"[^>]*>([\s\S]*?)<\/p>/i)
|| block.match(/class="res-rich[^"]*"[^>]*>([\s\S]*?)<\/div>/i)
|| block.match(/class="res-summary[^"]*"[^>]*>([\s\S]*?)<\/p>/i)
|| block.match(/<dd[^>]*>([\s\S]*?)<\/dd>/i);
const snippetMatch =
block.match(/class="res-desc[^"]*"[^>]*>([\s\S]*?)<\/p>/i) ||
block.match(/class="res-rich[^"]*"[^>]*>([\s\S]*?)<\/div>/i) ||
block.match(/class="res-summary[^"]*"[^>]*>([\s\S]*?)<\/p>/i) ||
block.match(/<dd[^>]*>([\s\S]*?)<\/dd>/i);
const snippet = snippetMatch ? snippetMatch[1].replace(/<[^>]+>/g, '').trim() : '';
if (title && url && !url.includes('so.com')) {
results.push({ title, url, snippet, engine: '360搜索', weight: 75 });
@@ -192,7 +353,7 @@ async function checkReachability(urls: string[], concurrency = 5): Promise<Map<s
function smartSort(results: SearchResult[]): SearchResult[] {
for (const r of results) {
const reachability = r.reachable ? 30 : -20;
const snippetQuality = Math.min(r.snippet.length, 100) / 100 * 20;
const snippetQuality = (Math.min(r.snippet.length, 100) / 100) * 20;
const weightScore = (r.weight / 100) * 50;
r._score = weightScore + reachability + snippetQuality;
}
@@ -223,13 +384,20 @@ function computeRelevance(query: string, result: SearchResult): number {
export class WebSearchTool implements IMetonaTool {
readonly definition: MetonaToolDef = {
name: 'web_search',
description: 'Search the web for information. Returns titles, snippets, and URLs. When SearXNG is enabled, uses the configured SearXNG instance; otherwise uses built-in engines (Bing, Baidu, Sogou, 360). Automatically fetches full content for top results.',
description:
'Search the web for information. Returns titles, snippets, and URLs. When SearXNG is enabled, uses the configured SearXNG instance; otherwise uses built-in engines (Bing, Baidu, Sogou, 360). Automatically fetches full content for top results.',
parameters: {
type: 'object',
properties: {
query: { type: 'string', description: 'Search query keywords' },
time_range: { type: 'string', description: 'Time filter: day, week, month, year (optional)' },
enhance_snippets: { type: 'boolean', description: 'Auto-enhance short snippets (default true)' },
time_range: {
type: 'string',
description: 'Time filter: day, week, month, year (optional)',
},
enhance_snippets: {
type: 'boolean',
description: 'Auto-enhance short snippets (default true)',
},
},
required: ['query'],
},
@@ -265,7 +433,10 @@ export class WebSearchTool implements IMetonaTool {
? Math.min(8, Math.max(3, searxngConfig.fetch_count > 0 ? searxngConfig.fetch_count : 5))
: 5;
logTool('web_search', `Mode=${useSearXNG ? 'searxng' : 'builtin'}, maxResults=${maxResults}, fetchTop=${fetchTop}`);
logTool(
'web_search',
`Mode=${useSearXNG ? 'searxng' : 'builtin'}, maxResults=${maxResults}, fetchTop=${fetchTop}`,
);
// 缓存检查(key 含模式 + maxResults + fetchTop,避免配置变更后返回旧缓存)
const cacheKey = `${searxngConfig.enabled ? 'searxng' : 'builtin'}:${maxResults}:${fetchTop}:${normalizeUrl(query).toLowerCase()}`;
@@ -338,7 +509,10 @@ export class WebSearchTool implements IMetonaTool {
// 写入缓存
searchCache.set(cacheKey, output);
logTool('web_search', `Completed: ${sorted.length} results, ${fetchedContent.length} fetched, mode=${mode}`);
logTool(
'web_search',
`Completed: ${sorted.length} results, ${fetchedContent.length} fetched, mode=${mode}`,
);
return output;
}
@@ -394,7 +568,7 @@ export class WebSearchTool implements IMetonaTool {
}
}
} else {
const data = await response.json() as { results?: Array<Record<string, unknown>> };
const data = (await response.json()) as { results?: Array<Record<string, unknown>> };
for (const item of data.results ?? []) {
const url = item.url as string;
const title = item.title as string;
@@ -418,7 +592,10 @@ export class WebSearchTool implements IMetonaTool {
}
results.push(...pageResults);
logTool('web_search', `[SearXNG] Page ${page}: +${pageResults.length} (total ${results.length})`);
logTool(
'web_search',
`[SearXNG] Page ${page}: +${pageResults.length} (total ${results.length})`,
);
page++;
}
@@ -440,12 +617,17 @@ export class WebSearchTool implements IMetonaTool {
const searchPromises = ENGINES.map(async (engine) => {
try {
const url = engine.searchUrl(query, timeRange);
const response = await fetchWithTimeout(url, {
headers: {
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/131.0.0.0 Safari/537.36',
'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
const response = await fetchWithTimeout(
url,
{
headers: {
'User-Agent':
'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/131.0.0.0 Safari/537.36',
'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
},
},
}, 8_000);
8_000,
);
if (!response.ok) {
logTool('web_search', `[内置] ${engine.name} HTTP ${response.status}`);
@@ -501,10 +683,10 @@ export class WebSearchTool implements IMetonaTool {
if (enhanced >= maxEnhance) break;
if (r.snippet.length < 30 && r.reachable) {
try {
const fetchResult = await this.webFetchTool.execute(
const fetchResult = (await this.webFetchTool.execute(
{ url: r.url },
{ sessionId: '', workspacePath: '', iteration: 0, requestId: '' },
) as { success: boolean; content?: string };
)) as { success: boolean; content?: string };
if (fetchResult.success && fetchResult.content) {
const text = fetchResult.content.slice(0, 200);
@@ -537,7 +719,10 @@ export class WebSearchTool implements IMetonaTool {
fetchTop: number,
): Promise<Array<{ url: string; title: string; content: string }>> {
// 相关性评分(不过滤,relevance=0 的结果也参与抓取候选)
const withRelevance = results.map((r) => ({ result: r, relevance: computeRelevance(query, r) }));
const withRelevance = results.map((r) => ({
result: r,
relevance: computeRelevance(query, r),
}));
const filtered = withRelevance.length > 0 ? withRelevance : [];
// 确定抓取数量:fetchTop 已在 execute() 中综合了配置面板和工具参数
@@ -554,27 +739,30 @@ export class WebSearchTool implements IMetonaTool {
}
toFetch = shuffled.slice(0, topN);
} else {
toFetch = filtered
.sort((a, b) => b.relevance - a.relevance)
.slice(0, topN);
toFetch = filtered.sort((a, b) => b.relevance - a.relevance).slice(0, topN);
}
const fetched: Array<{ url: string; title: string; content: string }> = [];
const fetchOne = async (item: { result: SearchResult }): Promise<{ url: string; title: string; content: string } | null> => {
const fetchOne = async (item: {
result: SearchResult;
}): Promise<{ url: string; title: string; content: string } | null> => {
try {
// 委托给 WebFetchTool — 享受三阶段回退策略(HTTP + 反爬 + 浏览器渲染)
const fetchResult = await this.webFetchTool.execute(
const fetchResult = (await this.webFetchTool.execute(
{ url: item.result.url },
{ sessionId: '', workspacePath: '', iteration: 0, requestId: '' },
) as { success: boolean; content?: string };
)) as { success: boolean; content?: string };
if (fetchResult.success && fetchResult.content) {
return { url: item.result.url, title: item.result.title, content: fetchResult.content };
}
return null;
} catch (err) {
logTool('web_search', `Auto-fetch failed for ${item.result.url}: ${(err as Error).message}`);
logTool(
'web_search',
`Auto-fetch failed for ${item.result.url}: ${(err as Error).message}`,
);
return null;
}
};
@@ -601,7 +789,9 @@ export class WebSearchTool implements IMetonaTool {
lines.push(`${i + 1}. ${r.title}`);
lines.push(` URL: ${r.url}`);
if (r.snippet) lines.push(` 摘要: ${r.snippet.slice(0, 150)}`);
lines.push(` 来源: ${r.engine}${r.reachable === false ? ' (不可达)' : ''}${r._enhanced ? ' [已增强]' : ''}\n`);
lines.push(
` 来源: ${r.engine}${r.reachable === false ? ' (不可达)' : ''}${r._enhanced ? ' [已增强]' : ''}\n`,
);
});
return lines.join('\n');
}