Files
searxng-use-cli/tests/test_v250_features.py
T
thzxx 10c01a3abd feat(v2.5.0): Sec-Ch-Ua 头修复 + 结构化提取 + token 预算 + 正文去重 + 连接复用
正确性修复:
- 修复 Sec-Ch-Ua 构造 bug: 原实现产出 ""Not_A Brand";v="99"" 双重引号
  畸形头(Chrome/Edge 两路径), 严格校验的 WAF 会忽略; 改为品牌数组拼接
- search 403 快速失败: 实例级 403 不再退避重试(~10.5s 空等), 立即 failover
- AdaptiveThrottle: --throttle-failure-threshold 0 现为真正禁用语义
- number_of_results 缺失时用 len(results) 兜底(JSON/HTML 路径契约对齐)
- 版本对齐: pyproject.toml 与 _config.py 同步 2.5.0

AI 代理体验:
- fetch.py --extract json: 结构化骨架(title/meta/headings/links/images)
- fetch.py --max-chars N: 提取后语义级截断(区别于 --max-size 字节截断)
- search --fetch-total-chars N: --fetch 全局字符预算, 耗尽后 status=skipped
- --dedup-fetched-content: 抓取正文 SimHash 去重, status=duplicate
- --progress 新增 angle_start/ok/fail + fetch_skip/fetch_duplicate 事件
- fetch.py 补齐 --log-format json + --dump-schema
- CSV 媒体列自适应(images/videos 类别自动追加媒体字段列)
- --dry-run 批量模式打印实际查询列表
- search 连接复用: requests 可用时走模块级 Session(连接池)

工程治理:
- 新增 scripts/release_check.py 发布一致性检查(版本/错误码表漂移)
- 新增 tests/test_v250_features.py 46+4 个回归测试(全量 622 通过)
- tests/conftest.py: autouse fixture 强制 stdlib 路径(本机有 requests 时
  既有 urllib mock 测试不失效)
2026-08-07 15:19:17 +08:00

919 lines
36 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Tests for v2.5.0 changes.
Covers:
* Sec-Ch-Ua header construction fix — Chrome/Edge headers must be valid
(no double-quoted first brand, no rstrip-clipped version quotes)
* AdaptiveThrottle ``--throttle-failure-threshold 0`` = disabled semantics
* search 403 fast-fail — 403 no longer backoff-retried (immediate
failover); 429/5xx still retried
* fetch.py ``--extract json`` structured skeleton output (bs4 + stdlib
parity, caps, CLI integration, implies --format json)
* fetch.py ``--max-chars N`` semantic truncation (CLI, truncated flag)
* fetch.py ``--log-format json`` + ``--dump-schema`` (CLI)
* ``--fetch-total-chars N`` global budget (allocation, truncation,
skipped status, disabled-by-default)
* research-mode angle progress events (angle_start/ok/fail)
* CSV media columns for images/videos categories (plus backward compat)
* ``--dedup-fetched-content`` SimHash body dedup (mirror/footer-noise/
distinct/error-preserved/threshold)
* ``--dry-run`` batch mode prints the actual query list
* search requests-backed HTTP path (Session reuse, 404/403 handling)
* release_check.py version-drift + missing-hint detection
"""
import json
import logging
import os
import subprocess
import sys
import urllib.error
from pathlib import Path
from types import SimpleNamespace
from unittest.mock import MagicMock, patch
import pytest
import common as common_mod
import _config
import search as search_mod
import fetch as fetch_mod
from common import build_browser_headers
from fetch import FetchResult, _emit_fetch_result, extract_structure
from search import (
AdaptiveThrottle,
_retry_with_backoff,
deduplicate_fetched_content,
fetch_top_results,
)
PROJECT_ROOT = Path(__file__).resolve().parent.parent
sys.path.insert(0, str(PROJECT_ROOT / "scripts"))
def _make_args(**overrides):
"""args object matching the argparse.Namespace shape main() produces.
Extends the v2.3.0 baseline with v2.5.0 fields (fetch_total_chars,
dedup_fetched_content) and the fields the research/batch/dry-run paths
read (research, research_angles, stream, dry_run, parallel_queries...).
"""
base = dict(
query="test", format="json", method="GET", timeout=15, retry=0,
serial=False, no_dedup=False, sort_by="score", max_results=None,
include_domain=None, exclude_domain=None, fetch=0, fetch_timeout=10,
fetch_retries=3, fetch_total_chars=0, dedup_fetched_content=False,
max_size=None, cache_ttl=0, snippet_len=0,
categories=None, language=None, pageno=1, pages=1,
time_range="year", safesearch=0, engines="google,bing",
referer=None, no_fallback=False, fetch_report=False,
request_delay=0.3, throttle_failure_threshold=3,
throttle_pause_seconds=30, throttle_max_delay=10,
similarity_dedup=False, similarity_threshold=0.85,
encoding=None, output=None, parallel_queries=0,
research=None, research_angles=None, stream=False,
progress=False, dry_run=False, verify=False, fail_fast=False,
queries_file=None, save_config=None, dump_schema=False,
cache_max_size=0, clear_cache=False, cache_stats=False,
proxy=None, log_format="text", config=None, instance=None,
auth_bearer=None, auth_basic=None, auth_bearer_file=None,
auth_basic_file=None, verbose=False, quiet=False,
include_domains=None, snippet_len_brief=0,
)
base.update(overrides)
return SimpleNamespace(**base)
# ===== Sec-Ch-Ua header construction fix (v2.5.0) =====
def _chrome_ua():
return next(ua for ua in _config.UA_POOL
if "Chrome/" in ua and "Edg/" not in ua and "Firefox/" not in ua)
def _edge_ua():
return next(ua for ua in _config.UA_POOL if "Edg/" in ua)
def test_sec_ch_ua_chrome_first_brand_has_single_quotes():
"""Chrome Sec-Ch-Ua must not have double-quoted first brand.
Regression (v2.4.0): the first brand variable already contained quotes
and was wrapped in the f-string again, producing
``""Not_A Brand";v="99""`` — a malformed header strict WAFs ignore.
"""
h = build_browser_headers(_chrome_ua())
value = h["Sec-Ch-Ua"]
assert '"Not_A Brand";v="99"' in value
assert '""Not_A Brand' not in value # no doubled opening quote
def test_sec_ch_ua_chrome_brands_complete():
"""Chrome header carries Not_A Brand + Chromium + Google Chrome."""
value = build_browser_headers(_chrome_ua())["Sec-Ch-Ua"]
assert '"Not_A Brand";v="99"' in value
assert '"Chromium";v="' in value
assert '"Google Chrome";v="' in value
def test_sec_ch_ua_edge_includes_edge_brand_valid():
"""Edge Sec-Ch-Ua includes Microsoft Edge AND keeps valid syntax.
Regression (v2.4.0): rstrip('"') clipped the previous brand's version
quotes, producing ``"Google Chrome";v="138, "Microsoft Edge";v="138"`` —
detectable as a `"138, ` (version value directly followed by comma).
"""
value = build_browser_headers(_edge_ua())["Sec-Ch-Ua"]
assert '"Microsoft Edge";v="' in value
# Each brand is a self-contained quoted pair — no dangling unclosed quote
assert value.count('"') % 2 == 0
assert '"138, ' not in value # no clipped version value
def test_sec_ch_ua_version_matches_ua():
"""Brand version must equal the UA's Chrome major version."""
ua = _chrome_ua()
import re
ver = re.search(r"Chrome/(\d+)", ua).group(1)
value = build_browser_headers(ua)["Sec-Ch-Ua"]
assert f'"Chromium";v="{ver}"' in value
assert f'"Google Chrome";v="{ver}"' in value
# ===== AdaptiveThrottle threshold=0 = disabled semantics (v2.5.0) =====
def test_throttle_zero_threshold_never_escalates():
"""threshold=0 disables adaptive backoff entirely (delay/concurrency)."""
t = AdaptiveThrottle(0.3, 5, failure_threshold=0)
for _ in range(10):
t.report_failure()
assert t.delay == 0.3 # never doubled
assert t.concurrency == 5 # never halved
def test_throttle_zero_threshold_still_counts_failures():
"""threshold=0 still tracks the failure counter for stats/reports."""
t = AdaptiveThrottle(0.3, 5, failure_threshold=0)
t.report_failure()
t.report_failure()
assert t.stats()["consecutive_failures"] == 2
def test_throttle_zero_threshold_429_pause_still_works():
"""threshold=0 disables backoff but NOT the 429 global pause."""
t = AdaptiveThrottle(0.3, 5, failure_threshold=0, pause_seconds=30)
t.report_failure("HTTP 429 Too Many Requests")
assert t.stats()["global_paused"] is True
# ===== search 403 fast-fail (v2.5.0) =====
def _http_error(code, headers=None):
return urllib.error.HTTPError("https://inst.example.com", code,
"reason", headers or {}, None)
def test_retry_with_backoff_403_raises_immediately():
"""403 is NOT in the search retry set — raises without sleeping.
Regression (v2.4.0): 403 was retried ~3x with backoff (~10.5s wasted)
even though search never switches UAs, so instance-level 403s (auth/
ban) only failed over after the backoff budget was exhausted.
"""
err = _http_error(403)
def raiser():
raise err
with patch("search.time.sleep") as mock_sleep:
with pytest.raises(urllib.error.HTTPError) as exc_info:
_retry_with_backoff(raiser, max_retries=3)
assert exc_info.value.code == 403
mock_sleep.assert_not_called()
def test_retry_with_backoff_429_still_retries():
"""429 remains retryable (backoff honored) — behavior unchanged."""
calls = [0]
err = _http_error(429, {"Retry-After": "0"})
def flaky():
calls[0] += 1
if calls[0] == 1:
raise err
return "ok"
with patch("search.time.sleep"):
assert _retry_with_backoff(flaky, max_retries=3) == "ok"
assert calls[0] == 2
def test_retry_with_backoff_500_still_retries():
"""5xx remains retryable — behavior unchanged."""
calls = [0]
err = _http_error(503)
def flaky():
calls[0] += 1
if calls[0] == 1:
raise err
return "ok"
with patch("search.time.sleep"):
assert _retry_with_backoff(flaky, max_retries=3) == "ok"
assert calls[0] == 2
def test_search_multi_403_fails_over_fast():
"""Serial failover: instance1 403 → instance2 succeeds, no sleeps."""
err = _http_error(403)
ok = {"results": [{"url": "https://ok.com", "title": "OK"}]}
with patch.object(search_mod, "search_single",
side_effect=[err, ok]) as mock_ss:
with patch("search.time.sleep") as mock_sleep:
r = search_mod.search_multi(
["https://a.example.com", "https://b.example.com"],
{"q": "t"}, parallel=False)
assert r["results"][0]["url"] == "https://ok.com"
assert mock_ss.call_count == 2
mock_sleep.assert_not_called() # 403 fast-fail: no backoff wait
# ===== fetch.py --extract json structured skeleton (v2.5.0) =====
def _sample_html():
return (
"<html><head><title>Test Page Title</title>"
'<meta name="description" content="A meta description">'
"</head><body>"
"<h1>Main Heading</h1><p>Intro.</p>"
"<h2>Sub Section</h2><p>More.</p>"
'<a href="/link1">L1</a><a href="/link2">L2</a>'
'<img src="/img1.png"><img data-src="/img2.jpg">'
"</body></html>"
)
def test_extract_structure_bs4_full():
"""bs4 path extracts title/meta/headings/links/images."""
s = extract_structure(_sample_html())
assert s["title"] == "Test Page Title"
assert s["meta_description"] == "A meta description"
assert s["headings"] == ["h1: Main Heading", "h2: Sub Section"]
assert s["links"] == ["/link1", "/link2"]
assert "/img1.png" in s["images"] and "/img2.jpg" in s["images"]
def test_extract_structure_stdlib_fallback_parity():
"""stdlib HTMLParser path outputs the same shape as bs4."""
fetch_mod._HAS_BS4 = False
try:
s = extract_structure(_sample_html())
finally:
fetch_mod._HAS_BS4 = True
assert s["title"] == "Test Page Title"
assert s["headings"] == ["h1: Main Heading", "h2: Sub Section"]
assert s["links"] == ["/link1", "/link2"]
assert s["images"] == ["/img1.png", "/img2.jpg"]
def test_extract_structure_limits_caps():
"""headings/links/images are capped to avoid huge outputs."""
html = "<html><body>"
html += "".join(f"<h1>Heading {i}</h1>" for i in range(80))
html += "".join(f'<a href="/l{i}">L</a>' for i in range(200))
html += "".join(f'<img src="/i{i}">' for i in range(80))
html += "</body></html>"
s = extract_structure(html)
assert len(s["headings"]) == 50
assert len(s["links"]) == 100
assert len(s["images"]) == 50
def test_extract_structure_empty():
assert extract_structure("") == {
"title": "", "meta_description": "", "headings": [], "links": [],
"images": [],
}
def test_emit_fetch_result_json_includes_structure(capsys):
"""--extract json structure merges into the json success payload."""
structure = {"title": "T", "meta_description": "M", "headings": ["h1: H"],
"links": ["/a"], "images": []}
args = SimpleNamespace(url="https://x.example.com", extract="json",
format="json", output=None)
_emit_fetch_result(args, "body text", "https://x.example.com",
"https://x.example.com/", "text/html", False,
user_agent="UA", structure=structure)
out, _ = capsys.readouterr()
data = json.loads(out)
assert data["title"] == "T"
assert data["headings"] == ["h1: H"]
assert data["text_length"] == len("body text")
# ===== fetch.py CLI: --extract json / --max-chars / --dump-schema =====
def _run_fetch_main(args_list, capsys, fetch_result):
"""Run fetch.main() with patched fetch_url + argv.
Returns (stdout, stderr) via capsys — callers must use the returned
tuple, NOT call capsys.readouterr() again (it is consumed here).
"""
_fresh_logging() # rebind handlers to this test's capsys stderr
with patch.object(fetch_mod, "fetch_url", return_value=fetch_result):
with patch.object(sys, "argv", ["fetch.py"] + args_list):
fetch_mod.main()
return capsys.readouterr()
def _fresh_logging():
"""Clear the module-level 'searxng' logger handlers.
setup_logging() caches a StreamHandler bound to the stderr of whichever
test first called fetch/main().main() — a later test's capsys captures a
different stderr, so handler output silently goes to the stale stream.
Removing handlers forces setup_logging to re-create one for the current
capsys context.
"""
import logging as _logging
root = _logging.getLogger("searxng")
for h in list(root.handlers):
root.removeHandler(h)
def _fetch_result(content, content_type="text/html"):
return FetchResult(content=content, content_type=content_type,
final_url="https://x.example.com", truncated=False,
user_agent="TestUA")
def test_fetch_cli_extract_json(capsys):
"""fetch.py -u URL --extract json emits structured JSON with skeleton."""
html = _sample_html()
out, _ = _run_fetch_main(["-u", "https://x.example.com", "--extract", "json"],
capsys, _fetch_result(html))
data = json.loads(out)
assert data["status"] == "ok"
assert data["title"] == "Test Page Title"
assert data["headings"] == ["h1: Main Heading", "h2: Sub Section"]
assert data["extract"] == "json"
def test_fetch_cli_extract_json_implies_format_json(capsys):
"""--extract json with --format text still emits JSON (not raw text)."""
html = _sample_html()
out, _ = _run_fetch_main(["-u", "https://x.example.com", "--extract", "json",
"--format", "text"], capsys, _fetch_result(html))
data = json.loads(out) # would fail if raw text leaked to stdout
assert data["status"] == "ok"
def test_fetch_cli_max_chars_truncates(capsys):
"""--max-chars N truncates extracted text and sets truncated=true."""
long_html = "<html><body><article>" + ("lorem ipsum dolor " * 100) + \
"</article></body></html>"
out, _ = _run_fetch_main(["-u", "https://x.example.com", "--format", "json",
"--max-chars", "30"], capsys,
_fetch_result(long_html))
data = json.loads(out)
assert data["truncated"] is True
assert data["text_length"] == 30
def test_fetch_cli_max_chars_no_truncation_within_limit(capsys):
"""--max-chars larger than the text leaves it untouched."""
out, _ = _run_fetch_main(["-u", "https://x.example.com", "--format", "json",
"--max-chars", "5000"], capsys,
_fetch_result(_sample_html()))
data = json.loads(out)
assert data["truncated"] is False
def test_fetch_cli_dump_schema(capsys):
"""fetch.py --dump-schema exits 0 and prints a JSON schema (no --url)."""
_fresh_logging() # avoid stale-handler pollution across capsys tests
with pytest.raises(SystemExit) as exc_info:
with patch.object(sys, "argv", ["fetch.py", "--dump-schema"]):
fetch_mod.main()
assert exc_info.value.code == 0
out, _ = capsys.readouterr()
schema = json.loads(out)
assert "properties" in schema
assert schema["properties"]["status"]["enum"] == ["ok", "error"]
# v2.5.0: extract enum includes 'json'
assert "json" in schema["properties"]["extract"]["enum"]
def test_fetch_cli_log_format_json(capsys):
"""fetch.py --log-format json emits JSON log lines on stderr.
stdout stays pure data (extracted text by default); the JSON-formatted
logs go to stderr as one JSON object per line.
"""
html = _sample_html()
out, err = _run_fetch_main(["-u", "https://x.example.com",
"--log-format", "json"], capsys,
_fetch_result(html))
# stdout is the extracted text (data), not JSON
assert "Main Heading" in out
# stderr carries JSON log lines
assert any(line.lstrip().startswith("{") for line in err.splitlines())
# ===== --fetch-total-chars global budget (v2.5.0) =====
def _make_search_results(n=3):
return {"results": [{"url": f"https://example{i}.com", "title": f"T{i}"}
for i in range(n)]}
def _make_ok_page(url, size=500):
return {"url": url, "status": "ok", "text": "x" * size,
"text_length": size, "truncated": False,
"anti_bot_detected": False, "waf_type": None,
"fallback_used": None, "title": "T", "latency": 0.1,
"user_agent_used": "TestUA", "final_url": url}
def test_fetch_top_results_budget_allocation_and_skip():
"""Budget 800 with 3×500-char pages: 500 + 300(truncated) + skipped."""
calls = []
def _fake_fetch(url, **kwargs):
calls.append(url)
return _make_ok_page(url)
with patch.object(search_mod, "fetch_page", side_effect=_fake_fetch):
out = fetch_top_results(_make_search_results(3), 3, total_chars=800,
request_delay=0)
assert len(calls) == 2 # third page skipped without a request
statuses = {f["url"]: f["status"] for f in out}
assert statuses["https://example0.com"] == "ok"
assert statuses["https://example1.com"] == "ok"
assert statuses["https://example2.com"] == "skipped"
ok_items = [f for f in out if f["status"] == "ok"]
assert ok_items[0]["text_length"] == 500
assert ok_items[1]["text_length"] == 300 # remaining budget applied
assert ok_items[1]["truncated"] is True
skip = next(f for f in out if f["status"] == "skipped")
assert "budget" in skip["error"].lower()
def test_fetch_top_results_budget_disabled_by_default():
"""total_chars=0 (or None) → full fetch, no truncation, no skipping."""
for budget in (0, None):
calls = []
def _fake_fetch(url, **kwargs):
calls.append(url)
return _make_ok_page(url)
with patch.object(search_mod, "fetch_page", side_effect=_fake_fetch):
out = fetch_top_results(_make_search_results(3), 3,
total_chars=budget, request_delay=0)
assert len(calls) == 3
assert all(f["status"] == "ok" for f in out)
assert all(f["text_length"] == 500 for f in out)
def test_fetch_top_results_budget_small_caps_first_only():
"""Budget 100: first page truncated to 100, remaining two skipped."""
calls = []
def _fake_fetch(url, **kwargs):
calls.append(url)
return _make_ok_page(url)
with patch.object(search_mod, "fetch_page", side_effect=_fake_fetch):
out = fetch_top_results(_make_search_results(3), 3, total_chars=100,
request_delay=0)
assert len(calls) == 1
assert out[0]["status"] == "ok" and out[0]["text_length"] == 100
assert sum(1 for f in out if f["status"] == "skipped") == 2
def test_fetch_top_results_budget_rolls_over_unused():
"""Unused budget on a short page rolls over to the next."""
def _fake_fetch(url, **kwargs):
if url == "https://example0.com":
return _make_ok_page(url, size=200) # consumes only 200 of 500
return _make_ok_page(url, size=500)
with patch.object(search_mod, "fetch_page", side_effect=_fake_fetch):
out = fetch_top_results(_make_search_results(3), 3, total_chars=500,
request_delay=0)
assert out[0]["status"] == "ok" and out[0]["text_length"] == 200
assert out[1]["status"] == "ok" and out[1]["text_length"] == 300 # rollover
assert out[2]["status"] == "skipped"
# ===== research-mode angle progress events (v2.5.0) =====
def test_research_emits_angle_progress_events(monkeypatch, capsys):
"""angle_start/angle_ok events carry angle + index/total on stderr."""
search_mod.set_progress_enabled(True)
try:
def fake_multi(urls, params, **kw):
return {"results": [{"url": "https://r.com/1", "title": "R"}]}
args = _make_args(research="some topic", format="json")
with patch.object(search_mod, "search_multi", side_effect=fake_multi):
with pytest.raises(SystemExit) as exc_info:
search_mod._handle_research(
args, ["https://x.example.com"], {}, 0)
assert exc_info.value.code == 0
out, err = capsys.readouterr()
assert '"angle_start"' in err
assert '"angle_ok"' in err
assert '"index"' in err and '"total"' in err
assert json.loads(out)["research_topic"] == "some topic"
finally:
search_mod.set_progress_enabled(False)
def test_research_emits_angle_fail_event(monkeypatch, capsys):
"""A failing angle emits angle_fail with error info."""
search_mod.set_progress_enabled(True)
try:
def fake_multi(urls, params, **kw):
raise RuntimeError("network down")
args = _make_args(research="topic", format="json")
with patch.object(search_mod, "search_multi", side_effect=fake_multi):
with pytest.raises(SystemExit) as exc_info:
search_mod._handle_research(
args, ["https://x.example.com"], {}, 0)
assert exc_info.value.code == 1 # all angles errored
_, err = capsys.readouterr()
assert '"angle_fail"' in err
assert "network down" in err
finally:
search_mod.set_progress_enabled(False)
# ===== CSV media columns for images/videos (v2.5.0) =====
def _csv_args():
return SimpleNamespace(format="csv", fetch=0, snippet_len=0)
def test_csv_appends_media_columns_when_present():
"""images/videos results get img_src/thumbnail_src/resolution/iframe_src."""
from search import _format_results, _detect_media_columns
results = {"results": [
{"title": "img1", "url": "https://x/i1", "engine": "google",
"score": 1.0, "published_date": "", "content": "s",
"template": "images.html", "img_src": "https://x/1.jpg",
"thumbnail_src": "https://x/1t.jpg", "resolution": "800x600"},
{"title": "vid1", "url": "https://x/v1", "engine": "bing",
"score": 0.5, "published_date": "", "content": "s",
"template": "videos.html", "iframe_src": "https://x/v1.html"},
]}
csv_text = _format_results(results, _csv_args())
header = csv_text.splitlines()[0]
assert "img_src" in header and "thumbnail_src" in header
assert "resolution" in header and "iframe_src" in header
rows = csv_text.splitlines()[1:]
assert rows[0].startswith("img1,https://x/i1,google,1.0")
assert rows[0].endswith("800x600,") # img_src + thumb + res present
assert "https://x/v1.html" in rows[1]
def test_csv_no_media_columns_for_general_results():
"""Plain general results keep the original 6 columns (backward compat)."""
from search import _format_results
results = {"results": [
{"title": "t", "url": "https://x/t", "engine": "google",
"score": 1.0, "published_date": "", "content": "c"},
]}
csv_text = _format_results(results, _csv_args())
header = csv_text.splitlines()[0]
assert header == "title,url,engine,score,published_date,content"
assert "img_src" not in header
def test_detect_media_columns_empty():
from search import _detect_media_columns
assert _detect_media_columns([{"url": "https://x/1"},
{"url": "https://x/2"}]) == []
assert _detect_media_columns([{"img_src": ""}]) == []
# ===== --dedup-fetched-content SimHash body dedup (v2.5.0) =====
def _long_body():
return ("This is a long article about machine learning and neural "
"networks. It covers supervised and unsupervised learning "
"approaches in detail. ") * 20
def test_deduplicate_fetched_content_identical():
"""Exact duplicate body → status=duplicate, text cleared."""
body = _long_body()
fetched = [
{"url": "https://orig.example/1", "status": "ok", "text": body,
"text_length": len(body)},
{"url": "https://mirror.example/2", "status": "ok", "text": body,
"text_length": len(body)},
]
n = deduplicate_fetched_content(fetched)
assert n == 1
assert fetched[0]["status"] == "ok"
assert fetched[1]["status"] == "duplicate"
assert fetched[1]["text"] == "" and fetched[1]["text_length"] == 0
assert "similar" in fetched[1]["error"]
def test_deduplicate_fetched_content_footer_noise():
"""Original + site footer noise is still a duplicate (first-1000 window)."""
body = _long_body()
with_footer = body + ("\n\nCopyright 2026 Example Network. All rights "
"reserved. Privacy Policy. Terms of Service.")
fetched = [
{"url": "https://orig.example/1", "status": "ok", "text": body,
"text_length": len(body)},
{"url": "https://repub.example/2", "status": "ok",
"text": with_footer, "text_length": len(with_footer)},
]
assert deduplicate_fetched_content(fetched) == 1
assert fetched[1]["status"] == "duplicate"
def test_deduplicate_fetched_content_distinct_kept():
"""Different bodies are kept (no false positive)."""
a = ("Completely different content about cooking recipes. ") * 30
b = ("Sports news covering the latest football match results. ") * 30
fetched = [
{"url": "https://a.example", "status": "ok", "text": a,
"text_length": len(a)},
{"url": "https://b.example", "status": "ok", "text": b,
"text_length": len(b)},
]
assert deduplicate_fetched_content(fetched) == 0
assert fetched[0]["status"] == "ok" and fetched[1]["status"] == "ok"
def test_deduplicate_fetched_content_error_preserved():
"""error/skipped entries pass through untouched."""
body = _long_body()
fetched = [
{"url": "https://ok.example", "status": "ok", "text": body,
"text_length": len(body)},
{"url": "https://err.example", "status": "error", "text": "",
"text_length": 0},
{"url": "https://skip.example", "status": "skipped", "text": "",
"text_length": 0},
]
assert deduplicate_fetched_content(fetched) == 0
assert fetched[1]["status"] == "error" and fetched[2]["status"] == "skipped"
def test_deduplicate_fetched_content_tighter_threshold():
"""Higher threshold keeps near-identical-but-not-exact bodies.
Uses two variants whose SimHash fingerprints differ by exactly 2 bits
in the 1000-char window: 0.85 maps to hamming<=3 (duplicate), 0.95 maps
to hamming<=1 (kept).
"""
body = _long_body()
v_full = body.replace("machine learning", "deep learning") # dist 6
v_light = body.replace("machine learning", "deep learning", 1) # dist 4
a = {"url": "https://a.example", "status": "ok", "text": v_full,
"text_length": len(v_full)}
b = {"url": "https://b.example", "status": "ok", "text": v_light,
"text_length": len(v_light)}
# lenient (0.85) → duplicate; strict (0.95) → kept
assert deduplicate_fetched_content([dict(a), dict(b)]) == 1
assert deduplicate_fetched_content([dict(a), dict(b)], threshold=0.95) == 0
# ===== --dry-run batch lists queries (v2.5.0) =====
def test_dry_run_batch_includes_queries(tmp_path, capsys):
"""--dry-run --queries-file prints the actual query list (no HTTP)."""
qf = tmp_path / "queries.txt"
qf.write_text("alpha\n# comment\nbeta\n\n", encoding="utf-8")
# query must be None so the batch branch is selected (preview priority:
# query > research > queries_file)
args = _make_args(dry_run=True, query=None, queries_file=str(qf),
format="json")
with pytest.raises(SystemExit) as exc_info:
search_mod._dry_run_preview(args, ["https://x.example.com"], {})
assert exc_info.value.code == 0
out, _ = capsys.readouterr()
data = json.loads(out)
assert data["action"] == "batch"
assert data["queries"] == ["alpha", "beta"]
def test_dry_run_batch_missing_file_reports_error(tmp_path, capsys):
"""Unreadable queries file degrades gracefully in dry-run."""
args = _make_args(dry_run=True, query=None,
queries_file=str(tmp_path / "missing.txt"),
format="json")
with pytest.raises(SystemExit):
search_mod._dry_run_preview(args, ["https://x.example.com"], {})
out, _ = capsys.readouterr()
data = json.loads(out)
assert data["queries_error"]
# ===== search requests-backed HTTP path (v2.5.0) =====
def _enable_requests_path(monkeypatch):
"""Temporarily enable the requests backend (conftest forces stdlib)."""
monkeypatch.setattr(search_mod, "_HAS_REQUESTS", True)
monkeypatch.setattr(search_mod, "_session", None)
def test_search_json_requests_path_success(monkeypatch):
"""requests backend parses a JSON response via session.get."""
_enable_requests_path(monkeypatch)
resp = MagicMock()
resp.status_code = 200
resp.content = json.dumps({"results": [{"title": "R", "url": "https://u"}]}
).encode("utf-8")
resp.headers = {"Content-Type": "application/json"}
session = MagicMock()
session.get.return_value = resp
monkeypatch.setattr(search_mod, "_get_session", lambda: session)
r = search_mod.search_json("https://s.example.com",
{"q": "t", "format": "json"})
assert r["results"][0]["title"] == "R"
session.get.assert_called_once()
def test_search_json_requests_path_404_returns_none(monkeypatch):
"""requests backend: 404 → None (fall back to HTML scraping)."""
_enable_requests_path(monkeypatch)
resp = MagicMock()
resp.status_code = 404
resp.content = b"not found"
resp.headers = {}
session = MagicMock()
session.get.return_value = resp
monkeypatch.setattr(search_mod, "_get_session", lambda: session)
assert search_mod.search_json("https://s.example.com",
{"q": "t", "format": "json"}) is None
def test_search_json_requests_path_403_raises(monkeypatch):
"""requests backend: 403 raises (auth/ban — not silently masked)."""
import requests as requests_mod
_enable_requests_path(monkeypatch)
resp = MagicMock()
resp.status_code = 403
resp.content = b"forbidden"
resp.headers = {}
resp.raise_for_status.side_effect = requests_mod.exceptions.HTTPError(
response=resp)
session = MagicMock()
session.get.return_value = resp
monkeypatch.setattr(search_mod, "_get_session", lambda: session)
with pytest.raises(requests_mod.exceptions.HTTPError):
search_mod.search_json("https://s.example.com",
{"q": "t", "format": "json"})
def test_search_html_requests_path_decodes(monkeypatch):
"""requests backend HTML fallback decodes + parses results."""
_enable_requests_path(monkeypatch)
html = ('<html><body><article class="result"><h3>'
'<a href="https://x.com">Title</a></h3>'
'<p class="content">snip</p></article></body></html>').encode("utf-8")
resp = MagicMock()
resp.status_code = 200
resp.content = html
resp.headers = {"Content-Type": "text/html; charset=utf-8"}
session = MagicMock()
session.get.return_value = resp
monkeypatch.setattr(search_mod, "_get_session", lambda: session)
r = search_mod.search_html("https://s.example.com", {"q": "t"})
assert r["results"][0]["title"] == "Title"
def test_search_session_singleton_reused(monkeypatch):
"""_get_session returns the same Session instance across calls."""
_enable_requests_path(monkeypatch)
s1 = search_mod._get_session()
s2 = search_mod._get_session()
assert s1 is s2
assert isinstance(s1, MagicMock) is False # real Session on requests env
search_mod._reset_session()
s3 = search_mod._get_session()
assert s3 is not s1 # reset creates a fresh one
# ===== number_of_results fallback (v2.5.0 fix) =====
def test_search_json_fills_missing_number_of_results(monkeypatch):
"""Instance JSON missing number_of_results → filled with len(results).
Some instances/versions omit the field (verified against a real
instance: `number_of_results` absent, 20 results present). The HTML
fallback path always fills it; JSON must too, or the output contract
differs between paths.
"""
raw = json.dumps({"query": "t",
"results": [{"title": "A", "url": "https://u"}]}).encode()
resp = MagicMock()
resp.read.return_value = raw
resp.__enter__.return_value = resp
with patch("urllib.request.urlopen", return_value=resp):
r = search_mod.search_json("https://s.example.com",
{"q": "t", "format": "json"})
assert r["number_of_results"] == 1
def test_search_json_preserves_existing_number_of_results(monkeypatch):
"""Instance-provided number_of_results is never overwritten."""
raw = json.dumps({"query": "t", "results": [{"title": "A"}],
"number_of_results": 999}).encode()
resp = MagicMock()
resp.read.return_value = raw
resp.__enter__.return_value = resp
with patch("urllib.request.urlopen", return_value=resp):
r = search_mod.search_json("https://s.example.com",
{"q": "t", "format": "json"})
assert r["number_of_results"] == 999
def test_search_json_requests_path_fills_number_of_results(monkeypatch):
"""requests backend: same fallback applies."""
_enable_requests_path(monkeypatch)
resp = MagicMock()
resp.status_code = 200
resp.content = json.dumps({"query": "t", "results": [{"title": "A"},
{"title": "B"}]}
).encode("utf-8")
resp.headers = {"Content-Type": "application/json"}
session = MagicMock()
session.get.return_value = resp
monkeypatch.setattr(search_mod, "_get_session", lambda: session)
r = search_mod.search_json("https://s.example.com",
{"q": "t", "format": "json"})
assert r["number_of_results"] == 2
def test_search_json_non_dict_response_untouched(monkeypatch):
"""List responses (rare) are passed through without dict access."""
raw = json.dumps([{"x": 1}]).encode()
resp = MagicMock()
resp.read.return_value = raw
resp.__enter__.return_value = resp
with patch("urllib.request.urlopen", return_value=resp):
r = search_mod.search_json("https://s.example.com",
{"q": "t", "format": "json"})
assert isinstance(r, list) and r[0]["x"] == 1
# ===== release_check.py (v2.5.0) =====
def test_release_check_ok_on_consistent_state(tmp_path, monkeypatch):
"""Matching versions + complete hints → no errors."""
import release_check
scripts = tmp_path / "scripts"
scripts.mkdir()
(scripts / "_config.py").write_text('VERSION = "1.2.3"\n', encoding="utf-8")
(scripts / "common.py").write_text(
'E_TEST = "E_TEST"\n\nRECOVERY_HINTS = {\n "E_TEST": "hint",\n}\n',
encoding="utf-8")
(tmp_path / "pyproject.toml").write_text('version = "1.2.3"\n',
encoding="utf-8")
monkeypatch.setattr(release_check, "ROOT", tmp_path)
monkeypatch.setattr(release_check, "SCRIPTS_DIR", scripts)
assert release_check.check_version(strict=False) == []
assert release_check.check_error_codes(strict=False) == []
def test_release_check_detects_version_drift(tmp_path, monkeypatch):
"""pyproject.toml vs _config.VERSION mismatch is reported."""
import release_check
scripts = tmp_path / "scripts"
scripts.mkdir()
(scripts / "_config.py").write_text('VERSION = "2.0.0"\n', encoding="utf-8")
(tmp_path / "pyproject.toml").write_text('version = "1.0.0"\n',
encoding="utf-8")
monkeypatch.setattr(release_check, "ROOT", tmp_path)
monkeypatch.setattr(release_check, "SCRIPTS_DIR", scripts)
errors = release_check.check_version(strict=False)
assert any("pyproject.toml" in e and "1.0.0" in e for e in errors)
def test_release_check_detects_missing_hint(tmp_path, monkeypatch):
"""An E_* constant without a RECOVERY_HINTS entry is reported."""
import release_check
scripts = tmp_path / "scripts"
scripts.mkdir()
(scripts / "common.py").write_text('E_ORPHAN = "E_ORPHAN"\n',
encoding="utf-8")
monkeypatch.setattr(release_check, "ROOT", tmp_path)
monkeypatch.setattr(release_check, "SCRIPTS_DIR", scripts)
errors = release_check.check_error_codes(strict=False)
assert any("E_ORPHAN" in e and "RECOVERY_HINTS" in e for e in errors)