Files
thzxx d9a08716bc fix(v2.0.1): 修复反爬误判 + Wayback 兜底未触发两个 bug
Bug 1: Wayback 兜底未触发 (search.py fetch_page)

- 根因: Cloudflare JS 质询页常返回 HTTP 200 (非 403), 主抓取 result 非 None

- _should_try_fallback 在 result 非 None 时直接返回 False, 跳过兜底

- 反爬检测在兜底判断之后执行, 错过兜底入口

- 修复: 反爬检测提前到兜底判断之前, anti_bot_detected=True 也触发 Wayback

- Wayback 结果重新做反爬检测 (防御性)

Bug 2: WAF 指纹库误判正常内容 (search.py WAF_FINGERPRINTS)

- 根因: 裸公司名 (cloudflare/akamai) 和宽泛词 (captcha/challenge/dd-) 做全文匹配

- DataCamp 文章引用 cloudflare.com 文档链接 -> 误判为 cloudflare WAF

- 'coding challenges' 正常内容 -> 误判为 generic 反爬

- Wayback 归档正文被误判, 兜底返回的有效内容被丢弃

- 修复: 移除裸公司名和宽泛词, 改用技术标识符 (cf-ray/incap_ses/bm_sz 等)

- 通用文案用完整短语 (please complete the captcha) 替代单词

- 增加 <title> 标签精准检测 (反爬页 title 是特征文案, 误判率极低)

- 新增 Anubis 反爬系统检测 (anubis_challenge/miserere)

真实测试验证 (search.metona.cn 实例):

- v2.0.0: --fetch 3 全部失败 (3 ERR: cloudflare/generic, Wayback 未触发)

- v2.0.1: --fetch 3 全部成功 (3 OK: 38100/93442/1945 chars, UA 轮换绕过 Cloudflare)

测试: 458 个全部通过 (新增 7 个测试覆盖修复行为)
2026-08-01 21:44:58 +08:00

311 lines
11 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Tests for the --fetch auto-fetch feature in search.py.
Covers: _is_blocked_page (CAPTCHA/bot detection heuristics), fetch_page
(ok/error/blocked paths), fetch_top_results (URL dedup, ordering,
concurrency, empty input).
"""
from unittest.mock import patch, MagicMock
import search as search_mod
from search import _is_blocked_page, fetch_page, fetch_top_results
import fetch as fetch_mod
from fetch import FetchResult
# ----- _is_blocked_page -----
def test_blocked_page_detects_captcha():
assert _is_blocked_page("<html>Please complete the CAPTCHA</html>")
def test_blocked_page_detects_cloudflare():
assert _is_blocked_page("<html>Just a moment... cf-browser-verification</html>")
def test_blocked_page_detects_challenge():
assert _is_blocked_page("<html>Checking your browser before accessing</html>")
def test_blocked_page_detects_anubis():
assert _is_blocked_page("<html>anubis_challenge</html>")
def test_blocked_page_clean_html():
"""Normal HTML must not be flagged as blocked."""
assert not _is_blocked_page("<html><body><article>Real content</article></body></html>")
def test_blocked_page_empty():
assert not _is_blocked_page("")
def test_blocked_page_checks_first_2000_chars():
"""v2.0.0: detection now scans the FULL document, not just first 2000 chars.
Previously detection only scanned the first 2000 chars for performance.
v2.0.0 changed this to full-document scanning because large anti-bot
pages (e.g. Cloudflare challenges with big JS blobs) may place the
telltale keyword beyond the 2000-char boundary.
v2.0.1: 用完整短语 'please complete the captcha' 替代裸 'captcha'
避免正常内容误判。
"""
padding = "x" * 2500
html = f"<html>{padding}please complete the captcha</html>"
# v2.0.0: now detected (was: not detected)
assert _is_blocked_page(html)
# ----- fetch_page -----
def _mock_fetch_result(content: str, content_type="text/html",
final_url="https://example.com",
truncated=False, ua="searxng-cli/1.6.0"):
return FetchResult(
content=content,
content_type=content_type,
final_url=final_url,
truncated=truncated,
user_agent=ua,
)
def test_fetch_page_ok_html():
"""Successful HTML fetch extracts text and returns status=ok."""
html = "<html><body><article>Hello world</article></body></html>"
with patch.object(search_mod, "fetch_url",
return_value=_mock_fetch_result(html)):
r = fetch_page("https://example.com")
assert r["status"] == "ok"
assert "Hello world" in r["text"]
assert r["text_length"] > 0
assert r["truncated"] is False
def test_fetch_page_ok_non_html():
"""Non-HTML content is returned as-is without text extraction."""
with patch.object(search_mod, "fetch_url",
return_value=_mock_fetch_result(
"plain text", content_type="text/plain")):
r = fetch_page("https://example.com/file.txt")
assert r["status"] == "ok"
assert r["text"] == "plain text"
def test_fetch_page_blocked_detection():
"""CAPTCHA pages are marked as error, not ok."""
html = "<html>Please complete the CAPTCHA to continue</html>"
with patch.object(search_mod, "fetch_url",
return_value=_mock_fetch_result(html)):
r = fetch_page("https://example.com")
assert r["status"] == "error"
assert "Bot protection" in r["error"]
assert r["text"] == ""
def test_fetch_page_network_error():
"""Network exceptions return status=error with message."""
with patch.object(search_mod, "fetch_url",
side_effect=RuntimeError("HTTP 503 for url")):
r = fetch_page("https://example.com")
assert r["status"] == "error"
assert "503" in r["error"]
assert r["text"] == ""
def test_fetch_page_truncated_flag():
"""truncated=True is propagated from fetch_url."""
html = "<html>" + "x" * 100 + "</html>"
with patch.object(search_mod, "fetch_url",
return_value=_mock_fetch_result(html, truncated=True)):
r = fetch_page("https://example.com", max_size=50)
assert r["status"] == "ok"
assert r["truncated"] is True
assert r["truncated_at"] == 50
def test_fetch_page_preserves_final_url():
"""Redirect final_url is captured in the result."""
with patch.object(search_mod, "fetch_url",
return_value=_mock_fetch_result(
"<html>x</html>",
final_url="https://final.example.com/page")):
r = fetch_page("https://example.com")
assert r["final_url"] == "https://final.example.com/page"
def test_fetch_page_records_fallback_ua():
"""user_agent_used is propagated for logging/debugging."""
with patch.object(search_mod, "fetch_url",
return_value=_mock_fetch_result(
"<html>x</html>",
ua="Mozilla/5.0 fallback")):
r = fetch_page("https://example.com")
assert r["user_agent_used"] == "Mozilla/5.0 fallback"
# ----- v2.0.0: fetch_page new fields -----
def test_fetch_page_ok_has_v2_fields():
"""v2.0.0: ok result includes anti_bot_detected/waf_type/fallback_used."""
with patch.object(search_mod, "fetch_url",
return_value=_mock_fetch_result("<html>ok</html>")):
r = fetch_page("https://example.com", fallback_enabled=False)
assert r["status"] == "ok"
assert r["anti_bot_detected"] is False
assert r["waf_type"] is None
assert r["fallback_used"] is None
def test_fetch_page_error_has_v2_fields():
"""v2.0.0: error result includes anti_bot_detected/waf_type/fallback_used."""
with patch.object(search_mod, "fetch_url",
side_effect=RuntimeError("HTTP 500")):
r = fetch_page("https://example.com", fallback_enabled=False)
assert r["status"] == "error"
assert r["anti_bot_detected"] is False
assert r["waf_type"] is None
assert r["fallback_used"] is None
def test_fetch_page_cloudflare_detected():
"""v2.0.0: Cloudflare page detected with waf_type."""
html = "<html><body>Just a moment... cf-ray: 123</body></html>"
with patch.object(search_mod, "fetch_url",
return_value=_mock_fetch_result(html)):
r = fetch_page("https://example.com", fallback_enabled=False)
assert r["status"] == "error"
assert r["anti_bot_detected"] is True
assert r["waf_type"] == "cloudflare"
assert "cloudflare" in r["error"]
def test_fetch_page_404_triggers_wayback():
"""v2.0.0: 404 triggers Wayback fallback (default enabled)."""
wb_result = _mock_fetch_result("<html>archived</html>",
final_url="https://web.archive.org/web/2024/https://example.com")
with patch.object(search_mod, "fetch_url",
side_effect=[RuntimeError("HTTP 404"), wb_result]):
r = fetch_page("https://example.com", fallback_enabled=True)
assert r["status"] == "ok"
assert r["fallback_used"] == "wayback"
def test_fetch_page_no_fallback_when_disabled():
"""v2.0.0: --no-fallback disables Wayback."""
with patch.object(search_mod, "fetch_url",
side_effect=RuntimeError("HTTP 404")) as mock_fu:
r = fetch_page("https://example.com", fallback_enabled=False)
assert r["status"] == "error"
assert r["fallback_used"] is None
# 只调用一次(主抓取),不调用 Wayback
assert mock_fu.call_count == 1
def test_fetch_page_referer_passed():
"""v2.0.0: referer is passed to fetch_url."""
with patch.object(search_mod, "fetch_url",
return_value=_mock_fetch_result("<html>x</html>")) as mock_fu:
fetch_page("https://example.com", referer="https://ref.com/",
fallback_enabled=False)
kwargs = mock_fu.call_args[1]
assert kwargs.get("referer") == "https://ref.com/"
# ----- fetch_top_results -----
def test_fetch_top_results_empty():
"""No results → empty list, no fetch attempts."""
assert fetch_top_results({"results": []}, 3) == []
def test_fetch_top_results_no_results_key():
assert fetch_top_results({}, 3) == []
def test_fetch_top_results_dedup_urls():
"""Duplicate URLs are fetched only once."""
results = {"results": [
{"url": "https://a.com"},
{"url": "https://a.com"}, # dup
{"url": "https://b.com"},
]}
fetched_urls = []
def _fake_fetch(url, **kwargs):
fetched_urls.append(url)
return {"url": url, "status": "ok", "text": "x", "text_length": 1,
"truncated": False}
with patch.object(search_mod, "fetch_page", side_effect=_fake_fetch):
out = fetch_top_results(results, 3, request_delay=0)
assert len(fetched_urls) == 2 # dedup
assert "https://a.com" in fetched_urls
assert "https://b.com" in fetched_urls
def test_fetch_top_results_limits_count():
"""Only top N unique URLs are fetched."""
results = {"results": [{"url": f"https://x{i}.com"} for i in range(10)]}
def _fake_fetch(url, **kwargs):
return {"url": url, "status": "ok", "text": "x", "text_length": 1,
"truncated": False}
with patch.object(search_mod, "fetch_page", side_effect=_fake_fetch):
out = fetch_top_results(results, 3, request_delay=0)
assert len(out) == 3
def test_fetch_top_results_preserves_order():
"""Fetched results are reordered to match original result order."""
results = {"results": [
{"url": "https://a.com"},
{"url": "https://b.com"},
{"url": "https://c.com"},
]}
def _fake_fetch(url, **kwargs):
return {"url": url, "status": "ok", "text": url[-1], "text_length": 1,
"truncated": False}
with patch.object(search_mod, "fetch_page", side_effect=_fake_fetch):
out = fetch_top_results(results, 3, request_delay=0)
urls = [f["url"] for f in out]
assert urls == ["https://a.com", "https://b.com", "https://c.com"]
def test_fetch_top_results_handles_errors():
"""A failed fetch still appears in output with status=error."""
results = {"results": [{"url": "https://ok.com"}, {"url": "https://bad.com"}]}
def _fake_fetch(url, **kwargs):
if "bad" in url:
return {"url": url, "status": "error", "error": "503",
"text": "", "text_length": 0, "truncated": False}
return {"url": url, "status": "ok", "text": "content",
"text_length": 7, "truncated": False}
with patch.object(search_mod, "fetch_page", side_effect=_fake_fetch):
out = fetch_top_results(results, 2, request_delay=0)
statuses = {f["url"]: f["status"] for f in out}
assert statuses["https://ok.com"] == "ok"
assert statuses["https://bad.com"] == "error"
def test_fetch_top_results_skips_empty_urls():
"""Results without a URL are skipped."""
results = {"results": [
{"url": ""},
{"url": "https://real.com"},
]}
def _fake_fetch(url, **kwargs):
return {"url": url, "status": "ok", "text": "x", "text_length": 1,
"truncated": False}
with patch.object(search_mod, "fetch_page", side_effect=_fake_fetch):
out = fetch_top_results(results, 3, request_delay=0)
assert len(out) == 1
assert out[0]["url"] == "https://real.com"