Files
searxng-use-cli/scripts/cache.py
T
thzxx 157219d982 feat(v2.2.1): 修复 Brotli 乱码 + 缓存治理 + 多页聚合 + 研究模式增强
核心修复(v2.2.1):
- 修复 Brotli 乱码 bug: build_browser_headers 智能声明 Accept-Encoding,
  仅在 brotli 可用时才声明 br; fetch.py 双路径 br 解压(requests + stdlib)
  此前 Chrome/Edge UA 抓取 example.com 等返回 br 的站点输出乱码

v2.2.0 新功能:
- main() 拆分为 _handle_verify/_handle_research/_handle_batch/_handle_single
- --cache-max-size MB: 缓存大小上限 + LRU 淘汰(默认 100MB)
- --pages N: 多页聚合 + 跨页去重
- --research 跨角度合并: 新增 merged_results 字段
- --stream / --progress: JSON Lines 流式输出 + request_id 贯穿
- --dry-run / --save-config / --log-format json
- --similarity-dedup / --throttle-* 参数化
- 15-UA 池 + PDF/docx 解析 + error_code 字段

文档与测试:
- SKILL.md: 版本号唯一(元数据),删除版本标记干扰
- README.md: 测试数量 539 -> 544
- 544 passed (新增 5 个 Content-Encoding 解压测试)
2026-08-03 17:14:33 +08:00

439 lines
18 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""SQLite-backed result cache for searxng-cli.
Avoids re-hitting the SearXNG instance for identical queries within a
configurable TTL. Cache key is a SHA-256 of the normalized search params
(query + engines + categories + language + time_range + safesearch +
pageno + method), so different parameter combinations get separate entries.
Storage location (in priority order):
1. ``$SEARXNG_CACHE_DIR`` env var (directory; ``cache.db`` is created inside)
2. ``~/.cache/searxng-cli/cache.db`` (XDG-style; on Windows this resolves
to ``C:\\Users\\<user>\\.cache\\searxng-cli\\cache.db``)
Uses WAL journal mode for better read concurrency. Entries expire lazily
on read; :func:`clear` removes all rows. Schema is created on first use.
大小上限与 LRU 淘汰(v2:
* :class:`SearchCache` 支持 ``max_size_bytes`` 参数(默认 100 MB0 表示
不限制,向后兼容)。put 时若总大小超限,按 ``last_accessed_at`` 升序
淘汰最旧条目,直到总大小 <= max_size_bytes。
* get 命中时更新 ``last_accessed_at``,实现 LRU 语义。
* :meth:`SearchCache.evict_expired` 可主动清理已过期条目。
* ``$SEARXNG_CACHE_MAX_SIZE_BYTES`` 环境变量可覆盖默认上限。
Design notes:
* Only the search result dict is cached — fetched page content is NOT,
because it is large and changes independently of the search result set.
* The cache key excludes auth headers and timeouts (transient concerns)
so two callers with the same query + params share an entry.
* All operations swallow ``sqlite3.Error`` and degrade gracefully — a
cache failure must never break a search.
"""
import contextlib
import hashlib
import json
import os
import sqlite3
import time
from pathlib import Path
DEFAULT_CACHE_DIR = Path.home() / ".cache" / "searxng-cli"
# 默认缓存大小上限:100 MB;0 表示不限制(向后兼容)
DEFAULT_MAX_SIZE_BYTES = 104857600
def _cache_path() -> Path:
"""Resolve the cache database path from env var or default location."""
env = os.environ.get("SEARXNG_CACHE_DIR")
if env:
return Path(env) / "cache.db"
return DEFAULT_CACHE_DIR / "cache.db"
def _ensure_schema(conn: sqlite3.Connection) -> None:
"""创建表结构并执行向后兼容的 schema 迁移。
v1 schema: key, created_at, ttl_seconds, payload
v2 新增列: last_accessed_at (LRU 排序依据), size_bytes (payload 字节大小)
对已存在的旧表用 ALTER TABLE ADD COLUMN 添加新列,并回填数据,
保证升级后现有条目也能参与大小统计与 LRU 淘汰。
"""
conn.execute(
"""
CREATE TABLE IF NOT EXISTS search_cache (
key TEXT PRIMARY KEY,
created_at REAL NOT NULL,
ttl_seconds INTEGER NOT NULL,
payload TEXT NOT NULL
)
"""
)
# 检查现有列,决定是否需要迁移
cols = {row[1] for row in conn.execute("PRAGMA table_info(search_cache)").fetchall()}
if "last_accessed_at" not in cols:
# 新增 LRU 访问时间列,回填为 created_at(视为从未被访问过)
conn.execute("ALTER TABLE search_cache ADD COLUMN last_accessed_at REAL")
conn.execute(
"UPDATE search_cache SET last_accessed_at = created_at "
"WHERE last_accessed_at IS NULL"
)
if "size_bytes" not in cols:
# 新增 payload 字节大小列,回填为 payload 的 UTF-8 字节长度
conn.execute("ALTER TABLE search_cache ADD COLUMN size_bytes INTEGER")
rows = conn.execute(
"SELECT key, payload FROM search_cache WHERE size_bytes IS NULL"
).fetchall()
for key, payload in rows:
conn.execute(
"UPDATE search_cache SET size_bytes = ? WHERE key = ?",
(len(payload.encode("utf-8")), key),
)
conn.commit()
def _connect(path: Path):
"""Open a connection with WAL mode and ensure the schema exists.
Returns a :class:`contextlib.closing` wrapper so ``with _connect(...) as conn:``
closes the connection on exit. ``sqlite3.Connection.__exit__`` only commits /
rolls back the transaction — it does **not** call ``close()``, which leaks
file descriptors across many cache operations (especially under
``--queries-file`` + ``--cache-ttl`` batch runs).
"""
path.parent.mkdir(parents=True, exist_ok=True)
conn = sqlite3.connect(str(path), timeout=10)
# WAL allows concurrent readers alongside a single writer, which matters
# when --fetch spawns parallel page fetches that might also touch the cache.
conn.execute("PRAGMA journal_mode=WAL")
conn.execute("PRAGMA synchronous=NORMAL")
_ensure_schema(conn)
return contextlib.closing(conn)
def _make_key(params: dict) -> str:
"""Build a stable cache key from search params.
Only params that affect the result set are included; transient fields
(auth, timeout, format) are excluded so the same logical query hits the
same cache entry regardless of output formatting.
"""
# Whitelist the params that actually change what SearXNG returns.
# fmt: off
relevant = (
"q", "categories", "language", "pageno",
"time_range", "safesearch", "engines", "method",
)
# fmt: on
normalized = {k: params[k] for k in relevant if params.get(k)}
raw = json.dumps(normalized, sort_keys=True, ensure_ascii=False)
return hashlib.sha256(raw.encode("utf-8")).hexdigest()
def _payload_size(payload: str) -> int:
"""计算 payload 序列化后的 UTF-8 字节大小。"""
return len(payload.encode("utf-8"))
class SearchCache:
"""带大小上限和 LRU 淘汰的 SQLite 缓存。
Args:
path: 缓存数据库路径。None 表示使用 ``$SEARXNG_CACHE_DIR`` 或
默认路径(每次操作动态解析,便于测试 monkeypatch 环境变量)。
max_size_bytes: 缓存总大小上限(字节)。0 表示不限制(向后兼容)。
大小跟踪与 LRU 语义:
* put 时计算 payload 字节大小并维护 ``_total_bytes`` 计数器。
* 若加入新条目后总大小超过 ``max_size_bytes``,按
``last_accessed_at`` 升序淘汰最旧条目。
* get 命中时更新 ``last_accessed_at``,将条目移到"最近使用"位置。
"""
def __init__(self, path: Path = None, max_size_bytes: int = DEFAULT_MAX_SIZE_BYTES):
self._path_override = Path(path) if path else None
self._max_size_bytes = max_size_bytes
self._total_bytes = 0
self._evicted_count = 0
self._total_bytes_loaded = False
# 记录上次加载 _total_bytes 时的路径,路径变化时重新加载
self._loaded_path = None
def _resolve_path(self) -> Path:
"""解析当前应使用的缓存路径(未显式指定时动态读取环境变量)。"""
return self._path_override or _cache_path()
def _load_total_bytes(self, conn: sqlite3.Connection) -> None:
"""惰性从 DB 加载 _total_bytes;路径变化时重新加载。"""
current_path = str(self._resolve_path())
if self._total_bytes_loaded and self._loaded_path == current_path:
return
row = conn.execute(
"SELECT COALESCE(SUM(size_bytes), 0) FROM search_cache"
).fetchone()
self._total_bytes = row[0] or 0
self._total_bytes_loaded = True
self._loaded_path = current_path
def get(self, params: dict, ttl_seconds: int):
"""Return cached result if within TTL, else None.
``ttl_seconds`` is the caller's current TTL setting. If the stored
entry was written with a longer TTL, the caller's shorter TTL wins
(so reducing --cache-ttl takes effect immediately without a clear).
命中时更新 ``last_accessed_at`` 以实现 LRU 语义。
"""
if ttl_seconds <= 0:
return None
key = _make_key(params)
try:
with _connect(self._resolve_path()) as conn:
self._load_total_bytes(conn)
row = conn.execute(
"SELECT payload, created_at, ttl_seconds FROM search_cache "
"WHERE key = ?",
(key,),
).fetchone()
if row is None:
return None
payload, created_at, stored_ttl = row
effective_ttl = min(ttl_seconds, stored_ttl)
if time.time() - created_at > effective_ttl:
return None
# LRU: 命中时把条目移到"最近使用"位置
conn.execute(
"UPDATE search_cache SET last_accessed_at = ? WHERE key = ?",
(time.time(), key),
)
conn.commit()
return json.loads(payload)
except sqlite3.Error:
return None
except (ValueError, json.JSONDecodeError):
# Corrupt payload — treat as miss
return None
def put(self, params: dict, result: dict, ttl_seconds: int) -> None:
"""Store a result with the given TTL. Silently no-ops on TTL<=0 or error.
若加入新条目后总大小超过 ``max_size_bytes``,按 LRU 淘汰最旧条目。
"""
if ttl_seconds <= 0:
return
key = _make_key(params)
payload = json.dumps(result, ensure_ascii=False)
new_size = _payload_size(payload)
try:
with _connect(self._resolve_path()) as conn:
self._load_total_bytes(conn)
# 若 key 已存在,先减去旧条目大小,避免重复计入
old = conn.execute(
"SELECT size_bytes FROM search_cache WHERE key = ?", (key,)
).fetchone()
if old is not None:
self._total_bytes -= (old[0] or 0)
now = time.time()
conn.execute(
"INSERT OR REPLACE INTO search_cache "
"(key, created_at, ttl_seconds, payload, "
" last_accessed_at, size_bytes) VALUES (?, ?, ?, ?, ?, ?)",
(key, now, ttl_seconds, payload, now, new_size),
)
self._total_bytes += new_size
self._enforce_size_limit(conn)
conn.commit()
except sqlite3.Error:
pass
def _enforce_size_limit(self, conn: sqlite3.Connection) -> None:
"""总大小超限时按 LRU 淘汰最旧条目,直到总大小 <= max_size_bytes。
``max_size_bytes <= 0`` 表示不限制,直接返回。当仅剩一个条目时
停止淘汰(避免 put 后立即被淘汰导致 get 不到刚写入的条目)。
"""
if self._max_size_bytes <= 0:
return
while self._total_bytes > self._max_size_bytes:
row = conn.execute(
"SELECT key, size_bytes FROM search_cache "
"ORDER BY last_accessed_at ASC, created_at ASC LIMIT 1"
).fetchone()
if row is None:
break
evict_key, evict_size = row
conn.execute("DELETE FROM search_cache WHERE key = ?", (evict_key,))
self._total_bytes -= (evict_size or 0)
self._evicted_count += 1
# 安全阀:只剩一个条目时停止淘汰(即新插入的条目本身超限也保留)
count_row = conn.execute("SELECT COUNT(*) FROM search_cache").fetchone()
if count_row[0] <= 1:
break
def clear(self) -> int:
"""Remove all cache entries. Returns count deleted, or 0 on error."""
try:
with _connect(self._resolve_path()) as conn:
self._load_total_bytes(conn)
cur = conn.execute("DELETE FROM search_cache")
conn.commit()
self._total_bytes = 0
return cur.rowcount
except sqlite3.Error:
return 0
def stats(self) -> dict:
"""Return cache statistics.
现有字段:entries, oldest_created_at, newest_created_at, path, size_bytes
新增字段:total_bytes, max_size_bytes, evicted_count, utilization_pct
"""
path = self._resolve_path()
try:
with _connect(path) as conn:
row = conn.execute(
"SELECT COUNT(*), MIN(created_at), MAX(created_at) "
"FROM search_cache"
).fetchone()
count, oldest, newest = row
# 从 DB 校准 total_bytes,防止外部进程修改导致计数器漂移
sum_row = conn.execute(
"SELECT COALESCE(SUM(size_bytes), 0) FROM search_cache"
).fetchone()
self._total_bytes = sum_row[0] or 0
self._total_bytes_loaded = True
self._loaded_path = str(path)
utilization = (
round(self._total_bytes * 100.0 / self._max_size_bytes, 2)
if self._max_size_bytes > 0
else 0
)
return {
"entries": count or 0,
"oldest_created_at": oldest,
"newest_created_at": newest,
"path": str(path),
"size_bytes": path.stat().st_size if path.exists() else 0,
# 新增字段:大小上限与 LRU 统计
"total_bytes": self._total_bytes,
"max_size_bytes": self._max_size_bytes,
"evicted_count": self._evicted_count,
"utilization_pct": utilization,
}
except sqlite3.Error as e:
return {
"entries": 0,
"error": str(e),
"path": str(path),
"size_bytes": 0,
"total_bytes": 0,
"max_size_bytes": self._max_size_bytes,
"evicted_count": self._evicted_count,
"utilization_pct": 0,
}
def evict_expired(self) -> int:
"""主动扫描并删除已过期条目,返回清理的条目数。
过期条件:``now - created_at > ttl_seconds``。清理时同步更新
``_total_bytes`` 计数器。
"""
now = time.time()
try:
with _connect(self._resolve_path()) as conn:
self._load_total_bytes(conn)
rows = conn.execute(
"SELECT key, size_bytes FROM search_cache "
"WHERE ? - created_at > ttl_seconds",
(now,),
).fetchall()
if not rows:
return 0
for key, size in rows:
conn.execute("DELETE FROM search_cache WHERE key = ?", (key,))
self._total_bytes -= (size or 0)
conn.commit()
return len(rows)
except sqlite3.Error:
return 0
def set_max_size_bytes(self, max_size_bytes: int) -> None:
"""更新大小上限并立即触发 LRU 淘汰(若当前已超限)。
供 CLI ``--cache-max-size`` 在运行时注入参数用。``max_size_bytes <= 0``
表示不限制。
"""
self._max_size_bytes = max_size_bytes
if max_size_bytes <= 0:
return
try:
with _connect(self._resolve_path()) as conn:
self._load_total_bytes(conn)
self._enforce_size_limit(conn)
conn.commit()
except sqlite3.Error:
pass
# ----- 模块级便捷 API(向后兼容)-----
# 现有调用方(search.py、测试)使用模块级函数;这里委托给一个惰性创建的
# 全局实例。全局实例的路径动态解析,因此 monkeypatch $SEARXNG_CACHE_DIR
# 能正常隔离每个测试。
_default_cache = None
def _get_default_cache() -> SearchCache:
"""惰性创建全局默认缓存实例。
``max_size_bytes`` 从 ``$SEARXNG_CACHE_MAX_SIZE_BYTES`` 读取(非法值回退默认)。
创建后可被 :func:`set_max_size_bytes` 覆盖(CLI ``--cache-max-size`` 优先级
高于环境变量)。
"""
global _default_cache
if _default_cache is None:
max_size = DEFAULT_MAX_SIZE_BYTES
env = os.environ.get("SEARXNG_CACHE_MAX_SIZE_BYTES")
if env:
try:
max_size = int(env)
except ValueError:
pass
_default_cache = SearchCache(max_size_bytes=max_size)
return _default_cache
def get(params: dict, ttl_seconds: int):
"""模块级便捷函数:委托给全局默认实例。"""
return _get_default_cache().get(params, ttl_seconds)
def put(params: dict, result: dict, ttl_seconds: int) -> None:
"""模块级便捷函数:委托给全局默认实例。"""
_get_default_cache().put(params, result, ttl_seconds)
def clear() -> int:
"""模块级便捷函数:委托给全局默认实例。"""
return _get_default_cache().clear()
def stats() -> dict:
"""模块级便捷函数:委托给全局默认实例。"""
return _get_default_cache().stats()
def evict_expired() -> int:
"""模块级便捷函数:委托给全局默认实例。"""
return _get_default_cache().evict_expired()
def set_max_size_bytes(max_size_bytes: int) -> None:
"""模块级便捷函数:更新全局默认实例的大小上限。
供 search.py 的 ``--cache-max-size`` CLI 参数在 main() 早期注入用,
优先级高于 ``$SEARXNG_CACHE_MAX_SIZE_BYTES`` 环境变量。
"""
_get_default_cache().set_max_size_bytes(max_size_bytes)