feat(v2.0.0): 反爬增强 + 抓取稳定性大幅提升
反爬措施: - 浏览器指纹头 build_browser_headers(): Sec-Ch-Ua/Sec-Fetch-*/Accept-Language/Accept-Encoding, 绕过 80%+ 轻量 WAF - 12 个 UA 池 (Chrome/Edge/Firefox x Win/macOS/Linux x v129-131) - 确定性 UA 轮换 get_ua_for_domain(): SHA-256 按域名固定 UA, 会话内稳定跨进程可复现 - WAF 指纹库 _detect_anti_bot(): 识别 Cloudflare/Imperva/PerimeterX/DataDome/Akamai/通用, 全文档扫描 - Retry-After 遵守: 429/503 读取 header (数字或 HTTP date) 作为最小重试延迟 - 退避封顶 60s (原无上限, N=10 时 1536s 卡死进程) 抓取稳定性: - requests.Session 复用: 连接池(10/host) + cookie 持久化 + TLS 会话恢复 - 超时分离 (connect, read) 元组, 避免大页面浪费已建连接 - Wayback Machine 兜底: 404/403/超时自动重试 web.archive.org, 默认启用 --no-fallback 关闭 - AdaptiveThrottle 自适应限流: 3 次失败翻倍延迟+减半并发, 5 次成功渐进恢复, 429 全局暂停 30s - readability-lite 提取: article/main 缺失时按文本密度选最可能正文 div 新增 CLI flags: - --fetch-report: 结构化抓取报告到 stderr (每 URL 状态/WAF 类型/兜底方式/字符数 + JSON 摘要) - --no-fallback: 禁用 Wayback 兜底 - --referer: 设置 Referer 头 (默认实例 URL) - --request-delay: 抓取请求间隔秒数 (默认 0.3, 自适应可能增大) fetch 结果新字段: anti_bot_detected (bool), waf_type (str|null), fallback_used (str|null) 测试: 新增 4 个测试文件 (test_browser_headers/test_anti_bot/test_wayback_fallback/test_adaptive_throttle), 451 个测试全部通过
This commit is contained in:
@@ -40,10 +40,17 @@ def test_blocked_page_empty():
|
||||
|
||||
|
||||
def test_blocked_page_checks_first_2000_chars():
|
||||
"""Detection only scans the first 2000 chars for performance."""
|
||||
"""v2.0.0: detection now scans the FULL document, not just first 2000 chars.
|
||||
|
||||
Previously detection only scanned the first 2000 chars for performance.
|
||||
v2.0.0 changed this to full-document scanning because large anti-bot
|
||||
pages (e.g. Cloudflare challenges with big JS blobs) may place the
|
||||
telltale keyword beyond the 2000-char boundary.
|
||||
"""
|
||||
padding = "x" * 2500
|
||||
html = f"<html>{padding}captcha</html>"
|
||||
assert not _is_blocked_page(html)
|
||||
# v2.0.0: now detected (was: not detected)
|
||||
assert _is_blocked_page(html)
|
||||
|
||||
|
||||
# ----- fetch_page -----
|
||||
@@ -134,6 +141,74 @@ def test_fetch_page_records_fallback_ua():
|
||||
assert r["user_agent_used"] == "Mozilla/5.0 fallback"
|
||||
|
||||
|
||||
# ----- v2.0.0: fetch_page new fields -----
|
||||
|
||||
def test_fetch_page_ok_has_v2_fields():
|
||||
"""v2.0.0: ok result includes anti_bot_detected/waf_type/fallback_used."""
|
||||
with patch.object(search_mod, "fetch_url",
|
||||
return_value=_mock_fetch_result("<html>ok</html>")):
|
||||
r = fetch_page("https://example.com", fallback_enabled=False)
|
||||
assert r["status"] == "ok"
|
||||
assert r["anti_bot_detected"] is False
|
||||
assert r["waf_type"] is None
|
||||
assert r["fallback_used"] is None
|
||||
|
||||
|
||||
def test_fetch_page_error_has_v2_fields():
|
||||
"""v2.0.0: error result includes anti_bot_detected/waf_type/fallback_used."""
|
||||
with patch.object(search_mod, "fetch_url",
|
||||
side_effect=RuntimeError("HTTP 500")):
|
||||
r = fetch_page("https://example.com", fallback_enabled=False)
|
||||
assert r["status"] == "error"
|
||||
assert r["anti_bot_detected"] is False
|
||||
assert r["waf_type"] is None
|
||||
assert r["fallback_used"] is None
|
||||
|
||||
|
||||
def test_fetch_page_cloudflare_detected():
|
||||
"""v2.0.0: Cloudflare page detected with waf_type."""
|
||||
html = "<html><body>Just a moment... cf-ray: 123</body></html>"
|
||||
with patch.object(search_mod, "fetch_url",
|
||||
return_value=_mock_fetch_result(html)):
|
||||
r = fetch_page("https://example.com", fallback_enabled=False)
|
||||
assert r["status"] == "error"
|
||||
assert r["anti_bot_detected"] is True
|
||||
assert r["waf_type"] == "cloudflare"
|
||||
assert "cloudflare" in r["error"]
|
||||
|
||||
|
||||
def test_fetch_page_404_triggers_wayback():
|
||||
"""v2.0.0: 404 triggers Wayback fallback (default enabled)."""
|
||||
wb_result = _mock_fetch_result("<html>archived</html>",
|
||||
final_url="https://web.archive.org/web/2024/https://example.com")
|
||||
with patch.object(search_mod, "fetch_url",
|
||||
side_effect=[RuntimeError("HTTP 404"), wb_result]):
|
||||
r = fetch_page("https://example.com", fallback_enabled=True)
|
||||
assert r["status"] == "ok"
|
||||
assert r["fallback_used"] == "wayback"
|
||||
|
||||
|
||||
def test_fetch_page_no_fallback_when_disabled():
|
||||
"""v2.0.0: --no-fallback disables Wayback."""
|
||||
with patch.object(search_mod, "fetch_url",
|
||||
side_effect=RuntimeError("HTTP 404")) as mock_fu:
|
||||
r = fetch_page("https://example.com", fallback_enabled=False)
|
||||
assert r["status"] == "error"
|
||||
assert r["fallback_used"] is None
|
||||
# 只调用一次(主抓取),不调用 Wayback
|
||||
assert mock_fu.call_count == 1
|
||||
|
||||
|
||||
def test_fetch_page_referer_passed():
|
||||
"""v2.0.0: referer is passed to fetch_url."""
|
||||
with patch.object(search_mod, "fetch_url",
|
||||
return_value=_mock_fetch_result("<html>x</html>")) as mock_fu:
|
||||
fetch_page("https://example.com", referer="https://ref.com/",
|
||||
fallback_enabled=False)
|
||||
kwargs = mock_fu.call_args[1]
|
||||
assert kwargs.get("referer") == "https://ref.com/"
|
||||
|
||||
|
||||
# ----- fetch_top_results -----
|
||||
|
||||
def test_fetch_top_results_empty():
|
||||
|
||||
Reference in New Issue
Block a user