feat(v2.2.1): 修复 Brotli 乱码 + 缓存治理 + 多页聚合 + 研究模式增强

核心修复(v2.2.1):
- 修复 Brotli 乱码 bug: build_browser_headers 智能声明 Accept-Encoding,
  仅在 brotli 可用时才声明 br; fetch.py 双路径 br 解压(requests + stdlib)
  此前 Chrome/Edge UA 抓取 example.com 等返回 br 的站点输出乱码

v2.2.0 新功能:
- main() 拆分为 _handle_verify/_handle_research/_handle_batch/_handle_single
- --cache-max-size MB: 缓存大小上限 + LRU 淘汰(默认 100MB)
- --pages N: 多页聚合 + 跨页去重
- --research 跨角度合并: 新增 merged_results 字段
- --stream / --progress: JSON Lines 流式输出 + request_id 贯穿
- --dry-run / --save-config / --log-format json
- --similarity-dedup / --throttle-* 参数化
- 15-UA 池 + PDF/docx 解析 + error_code 字段

文档与测试:
- SKILL.md: 版本号唯一(元数据),删除版本标记干扰
- README.md: 测试数量 539 -> 544
- 544 passed (新增 5 个 Content-Encoding 解压测试)
This commit is contained in:
2026-08-03 17:14:33 +08:00
parent 0c8fdc1e45
commit 157219d982
10 changed files with 2044 additions and 563 deletions
+345 -77
View File
@@ -13,6 +13,14 @@ Storage location (in priority order):
Uses WAL journal mode for better read concurrency. Entries expire lazily
on read; :func:`clear` removes all rows. Schema is created on first use.
大小上限与 LRU 淘汰(v2:
* :class:`SearchCache` 支持 ``max_size_bytes`` 参数(默认 100 MB0 表示
不限制,向后兼容)。put 时若总大小超限,按 ``last_accessed_at`` 升序
淘汰最旧条目,直到总大小 <= max_size_bytes。
* get 命中时更新 ``last_accessed_at``,实现 LRU 语义。
* :meth:`SearchCache.evict_expired` 可主动清理已过期条目。
* ``$SEARXNG_CACHE_MAX_SIZE_BYTES`` 环境变量可覆盖默认上限。
Design notes:
* Only the search result dict is cached — fetched page content is NOT,
because it is large and changes independently of the search result set.
@@ -31,6 +39,8 @@ import time
from pathlib import Path
DEFAULT_CACHE_DIR = Path.home() / ".cache" / "searxng-cli"
# 默认缓存大小上限:100 MB;0 表示不限制(向后兼容)
DEFAULT_MAX_SIZE_BYTES = 104857600
def _cache_path() -> Path:
@@ -41,6 +51,48 @@ def _cache_path() -> Path:
return DEFAULT_CACHE_DIR / "cache.db"
def _ensure_schema(conn: sqlite3.Connection) -> None:
"""创建表结构并执行向后兼容的 schema 迁移。
v1 schema: key, created_at, ttl_seconds, payload
v2 新增列: last_accessed_at (LRU 排序依据), size_bytes (payload 字节大小)
对已存在的旧表用 ALTER TABLE ADD COLUMN 添加新列,并回填数据,
保证升级后现有条目也能参与大小统计与 LRU 淘汰。
"""
conn.execute(
"""
CREATE TABLE IF NOT EXISTS search_cache (
key TEXT PRIMARY KEY,
created_at REAL NOT NULL,
ttl_seconds INTEGER NOT NULL,
payload TEXT NOT NULL
)
"""
)
# 检查现有列,决定是否需要迁移
cols = {row[1] for row in conn.execute("PRAGMA table_info(search_cache)").fetchall()}
if "last_accessed_at" not in cols:
# 新增 LRU 访问时间列,回填为 created_at(视为从未被访问过)
conn.execute("ALTER TABLE search_cache ADD COLUMN last_accessed_at REAL")
conn.execute(
"UPDATE search_cache SET last_accessed_at = created_at "
"WHERE last_accessed_at IS NULL"
)
if "size_bytes" not in cols:
# 新增 payload 字节大小列,回填为 payload 的 UTF-8 字节长度
conn.execute("ALTER TABLE search_cache ADD COLUMN size_bytes INTEGER")
rows = conn.execute(
"SELECT key, payload FROM search_cache WHERE size_bytes IS NULL"
).fetchall()
for key, payload in rows:
conn.execute(
"UPDATE search_cache SET size_bytes = ? WHERE key = ?",
(len(payload.encode("utf-8")), key),
)
conn.commit()
def _connect(path: Path):
"""Open a connection with WAL mode and ensure the schema exists.
@@ -56,17 +108,7 @@ def _connect(path: Path):
# when --fetch spawns parallel page fetches that might also touch the cache.
conn.execute("PRAGMA journal_mode=WAL")
conn.execute("PRAGMA synchronous=NORMAL")
conn.execute(
"""
CREATE TABLE IF NOT EXISTS search_cache (
key TEXT PRIMARY KEY,
created_at REAL NOT NULL,
ttl_seconds INTEGER NOT NULL,
payload TEXT NOT NULL
)
"""
)
conn.commit()
_ensure_schema(conn)
return contextlib.closing(conn)
@@ -89,82 +131,308 @@ def _make_key(params: dict) -> str:
return hashlib.sha256(raw.encode("utf-8")).hexdigest()
def get(params: dict, ttl_seconds: int):
"""Return cached result if within TTL, else None.
def _payload_size(payload: str) -> int:
"""计算 payload 序列化后的 UTF-8 字节大小。"""
return len(payload.encode("utf-8"))
``ttl_seconds`` is the caller's current TTL setting. If the stored
entry was written with a longer TTL, the caller's shorter TTL wins
(so reducing --cache-ttl takes effect immediately without a clear).
class SearchCache:
"""带大小上限和 LRU 淘汰的 SQLite 缓存。
Args:
path: 缓存数据库路径。None 表示使用 ``$SEARXNG_CACHE_DIR`` 或
默认路径(每次操作动态解析,便于测试 monkeypatch 环境变量)。
max_size_bytes: 缓存总大小上限(字节)。0 表示不限制(向后兼容)。
大小跟踪与 LRU 语义:
* put 时计算 payload 字节大小并维护 ``_total_bytes`` 计数器。
* 若加入新条目后总大小超过 ``max_size_bytes``,按
``last_accessed_at`` 升序淘汰最旧条目。
* get 命中时更新 ``last_accessed_at``,将条目移到"最近使用"位置。
"""
if ttl_seconds <= 0:
return None
key = _make_key(params)
try:
with _connect(_cache_path()) as conn:
def __init__(self, path: Path = None, max_size_bytes: int = DEFAULT_MAX_SIZE_BYTES):
self._path_override = Path(path) if path else None
self._max_size_bytes = max_size_bytes
self._total_bytes = 0
self._evicted_count = 0
self._total_bytes_loaded = False
# 记录上次加载 _total_bytes 时的路径,路径变化时重新加载
self._loaded_path = None
def _resolve_path(self) -> Path:
"""解析当前应使用的缓存路径(未显式指定时动态读取环境变量)。"""
return self._path_override or _cache_path()
def _load_total_bytes(self, conn: sqlite3.Connection) -> None:
"""惰性从 DB 加载 _total_bytes;路径变化时重新加载。"""
current_path = str(self._resolve_path())
if self._total_bytes_loaded and self._loaded_path == current_path:
return
row = conn.execute(
"SELECT COALESCE(SUM(size_bytes), 0) FROM search_cache"
).fetchone()
self._total_bytes = row[0] or 0
self._total_bytes_loaded = True
self._loaded_path = current_path
def get(self, params: dict, ttl_seconds: int):
"""Return cached result if within TTL, else None.
``ttl_seconds`` is the caller's current TTL setting. If the stored
entry was written with a longer TTL, the caller's shorter TTL wins
(so reducing --cache-ttl takes effect immediately without a clear).
命中时更新 ``last_accessed_at`` 以实现 LRU 语义。
"""
if ttl_seconds <= 0:
return None
key = _make_key(params)
try:
with _connect(self._resolve_path()) as conn:
self._load_total_bytes(conn)
row = conn.execute(
"SELECT payload, created_at, ttl_seconds FROM search_cache "
"WHERE key = ?",
(key,),
).fetchone()
if row is None:
return None
payload, created_at, stored_ttl = row
effective_ttl = min(ttl_seconds, stored_ttl)
if time.time() - created_at > effective_ttl:
return None
# LRU: 命中时把条目移到"最近使用"位置
conn.execute(
"UPDATE search_cache SET last_accessed_at = ? WHERE key = ?",
(time.time(), key),
)
conn.commit()
return json.loads(payload)
except sqlite3.Error:
return None
except (ValueError, json.JSONDecodeError):
# Corrupt payload — treat as miss
return None
def put(self, params: dict, result: dict, ttl_seconds: int) -> None:
"""Store a result with the given TTL. Silently no-ops on TTL<=0 or error.
若加入新条目后总大小超过 ``max_size_bytes``,按 LRU 淘汰最旧条目。
"""
if ttl_seconds <= 0:
return
key = _make_key(params)
payload = json.dumps(result, ensure_ascii=False)
new_size = _payload_size(payload)
try:
with _connect(self._resolve_path()) as conn:
self._load_total_bytes(conn)
# 若 key 已存在,先减去旧条目大小,避免重复计入
old = conn.execute(
"SELECT size_bytes FROM search_cache WHERE key = ?", (key,)
).fetchone()
if old is not None:
self._total_bytes -= (old[0] or 0)
now = time.time()
conn.execute(
"INSERT OR REPLACE INTO search_cache "
"(key, created_at, ttl_seconds, payload, "
" last_accessed_at, size_bytes) VALUES (?, ?, ?, ?, ?, ?)",
(key, now, ttl_seconds, payload, now, new_size),
)
self._total_bytes += new_size
self._enforce_size_limit(conn)
conn.commit()
except sqlite3.Error:
pass
def _enforce_size_limit(self, conn: sqlite3.Connection) -> None:
"""总大小超限时按 LRU 淘汰最旧条目,直到总大小 <= max_size_bytes。
``max_size_bytes <= 0`` 表示不限制,直接返回。当仅剩一个条目时
停止淘汰(避免 put 后立即被淘汰导致 get 不到刚写入的条目)。
"""
if self._max_size_bytes <= 0:
return
while self._total_bytes > self._max_size_bytes:
row = conn.execute(
"SELECT payload, created_at, ttl_seconds FROM search_cache "
"WHERE key = ?",
(key,),
"SELECT key, size_bytes FROM search_cache "
"ORDER BY last_accessed_at ASC, created_at ASC LIMIT 1"
).fetchone()
if row is None:
return None
payload, created_at, stored_ttl = row
effective_ttl = min(ttl_seconds, stored_ttl)
if time.time() - created_at > effective_ttl:
return None
return json.loads(payload)
except sqlite3.Error:
return None
except (ValueError, json.JSONDecodeError):
# Corrupt payload — treat as miss
return None
if row is None:
break
evict_key, evict_size = row
conn.execute("DELETE FROM search_cache WHERE key = ?", (evict_key,))
self._total_bytes -= (evict_size or 0)
self._evicted_count += 1
# 安全阀:只剩一个条目时停止淘汰(即新插入的条目本身超限也保留)
count_row = conn.execute("SELECT COUNT(*) FROM search_cache").fetchone()
if count_row[0] <= 1:
break
def clear(self) -> int:
"""Remove all cache entries. Returns count deleted, or 0 on error."""
try:
with _connect(self._resolve_path()) as conn:
self._load_total_bytes(conn)
cur = conn.execute("DELETE FROM search_cache")
conn.commit()
self._total_bytes = 0
return cur.rowcount
except sqlite3.Error:
return 0
def stats(self) -> dict:
"""Return cache statistics.
现有字段:entries, oldest_created_at, newest_created_at, path, size_bytes
新增字段:total_bytes, max_size_bytes, evicted_count, utilization_pct
"""
path = self._resolve_path()
try:
with _connect(path) as conn:
row = conn.execute(
"SELECT COUNT(*), MIN(created_at), MAX(created_at) "
"FROM search_cache"
).fetchone()
count, oldest, newest = row
# 从 DB 校准 total_bytes,防止外部进程修改导致计数器漂移
sum_row = conn.execute(
"SELECT COALESCE(SUM(size_bytes), 0) FROM search_cache"
).fetchone()
self._total_bytes = sum_row[0] or 0
self._total_bytes_loaded = True
self._loaded_path = str(path)
utilization = (
round(self._total_bytes * 100.0 / self._max_size_bytes, 2)
if self._max_size_bytes > 0
else 0
)
return {
"entries": count or 0,
"oldest_created_at": oldest,
"newest_created_at": newest,
"path": str(path),
"size_bytes": path.stat().st_size if path.exists() else 0,
# 新增字段:大小上限与 LRU 统计
"total_bytes": self._total_bytes,
"max_size_bytes": self._max_size_bytes,
"evicted_count": self._evicted_count,
"utilization_pct": utilization,
}
except sqlite3.Error as e:
return {
"entries": 0,
"error": str(e),
"path": str(path),
"size_bytes": 0,
"total_bytes": 0,
"max_size_bytes": self._max_size_bytes,
"evicted_count": self._evicted_count,
"utilization_pct": 0,
}
def evict_expired(self) -> int:
"""主动扫描并删除已过期条目,返回清理的条目数。
过期条件:``now - created_at > ttl_seconds``。清理时同步更新
``_total_bytes`` 计数器。
"""
now = time.time()
try:
with _connect(self._resolve_path()) as conn:
self._load_total_bytes(conn)
rows = conn.execute(
"SELECT key, size_bytes FROM search_cache "
"WHERE ? - created_at > ttl_seconds",
(now,),
).fetchall()
if not rows:
return 0
for key, size in rows:
conn.execute("DELETE FROM search_cache WHERE key = ?", (key,))
self._total_bytes -= (size or 0)
conn.commit()
return len(rows)
except sqlite3.Error:
return 0
def set_max_size_bytes(self, max_size_bytes: int) -> None:
"""更新大小上限并立即触发 LRU 淘汰(若当前已超限)。
供 CLI ``--cache-max-size`` 在运行时注入参数用。``max_size_bytes <= 0``
表示不限制。
"""
self._max_size_bytes = max_size_bytes
if max_size_bytes <= 0:
return
try:
with _connect(self._resolve_path()) as conn:
self._load_total_bytes(conn)
self._enforce_size_limit(conn)
conn.commit()
except sqlite3.Error:
pass
# ----- 模块级便捷 API(向后兼容)-----
# 现有调用方(search.py、测试)使用模块级函数;这里委托给一个惰性创建的
# 全局实例。全局实例的路径动态解析,因此 monkeypatch $SEARXNG_CACHE_DIR
# 能正常隔离每个测试。
_default_cache = None
def _get_default_cache() -> SearchCache:
"""惰性创建全局默认缓存实例。
``max_size_bytes`` 从 ``$SEARXNG_CACHE_MAX_SIZE_BYTES`` 读取(非法值回退默认)。
创建后可被 :func:`set_max_size_bytes` 覆盖(CLI ``--cache-max-size`` 优先级
高于环境变量)。
"""
global _default_cache
if _default_cache is None:
max_size = DEFAULT_MAX_SIZE_BYTES
env = os.environ.get("SEARXNG_CACHE_MAX_SIZE_BYTES")
if env:
try:
max_size = int(env)
except ValueError:
pass
_default_cache = SearchCache(max_size_bytes=max_size)
return _default_cache
def get(params: dict, ttl_seconds: int):
"""模块级便捷函数:委托给全局默认实例。"""
return _get_default_cache().get(params, ttl_seconds)
def put(params: dict, result: dict, ttl_seconds: int) -> None:
"""Store a result with the given TTL. Silently no-ops on TTL<=0 or error."""
if ttl_seconds <= 0:
return
key = _make_key(params)
payload = json.dumps(result, ensure_ascii=False)
try:
with _connect(_cache_path()) as conn:
conn.execute(
"INSERT OR REPLACE INTO search_cache "
"(key, created_at, ttl_seconds, payload) VALUES (?, ?, ?, ?)",
(key, time.time(), ttl_seconds, payload),
)
conn.commit()
except sqlite3.Error:
pass
"""模块级便捷函数:委托给全局默认实例。"""
_get_default_cache().put(params, result, ttl_seconds)
def clear() -> int:
"""Remove all cache entries. Returns count deleted, or 0 on error."""
try:
with _connect(_cache_path()) as conn:
cur = conn.execute("DELETE FROM search_cache")
conn.commit()
return cur.rowcount
except sqlite3.Error:
return 0
"""模块级便捷函数:委托给全局默认实例。"""
return _get_default_cache().clear()
def stats() -> dict:
"""Return cache statistics (entry count, age range, path)."""
path = _cache_path()
try:
with _connect(path) as conn:
row = conn.execute(
"SELECT COUNT(*), MIN(created_at), MAX(created_at) "
"FROM search_cache"
).fetchone()
count, oldest, newest = row
return {
"entries": count or 0,
"oldest_created_at": oldest,
"newest_created_at": newest,
"path": str(path),
"size_bytes": path.stat().st_size if path.exists() else 0,
}
except sqlite3.Error as e:
return {"entries": 0, "error": str(e), "path": str(path), "size_bytes": 0}
"""模块级便捷函数:委托给全局默认实例。"""
return _get_default_cache().stats()
def evict_expired() -> int:
"""模块级便捷函数:委托给全局默认实例。"""
return _get_default_cache().evict_expired()
def set_max_size_bytes(max_size_bytes: int) -> None:
"""模块级便捷函数:更新全局默认实例的大小上限。
供 search.py 的 ``--cache-max-size`` CLI 参数在 main() 早期注入用,
优先级高于 ``$SEARXNG_CACHE_MAX_SIZE_BYTES`` 环境变量。
"""
_get_default_cache().set_max_size_bytes(max_size_bytes)