` weighting), avoiding nav/sidebar/footer noise - **`--fetch-report`** — structured per-URL report to stderr after `--fetch N`: status, WAF type, fallback used, char count, plus adaptive throttle stats and a JSON summary line diff --git a/scripts/_config.py b/scripts/_config.py index cc84971..b7696f9 100644 --- a/scripts/_config.py +++ b/scripts/_config.py @@ -7,6 +7,6 @@ Retry settings and shared HTTP utilities now live in ``common.py`` so that both ``search.py`` and ``fetch.py`` share one consistent implementation. """ -VERSION = "2.0.0" +VERSION = "2.0.1" SCHEMA_VERSION = "1.0" USER_AGENT = f"searxng-cli/{VERSION}" diff --git a/scripts/search.py b/scripts/search.py index 5e67dba..851905e 100644 --- a/scripts/search.py +++ b/scripts/search.py @@ -13,6 +13,7 @@ import json import logging import os import random +import re import sys import threading import time @@ -855,9 +856,35 @@ def fetch_page(url: str, timeout: int = 10, auth_headers: dict = None, except Exception as e: error_msg = str(e) if str(e) else e.__class__.__name__ - # Wayback 兜底:主抓取失败或被反爬拦截时尝试 + # 反爬检测(v2.0.0 增强:全文档扫描 + WAF 指纹库) + # 必须在 Wayback 兜底判断之前执行:Cloudflare 质询页常返回 HTTP 200, + # 此时 result 非 None 但内容是反爬页,必须识别出来才能触发兜底。 + anti_bot_detected = False + waf_type = None + if result is not None: + content = result.content + content_type = result.content_type or "" + is_html = ("html" in content_type.lower() or + content.strip().startswith(" 标签检测——反爬页 title 是特征文案,最精准 WAF_FINGERPRINTS = [ ("cloudflare", [ + # Cloudflare 技术标识符(cookie/header/JS 变量名,不会出现在正文) "cf-ray", "cf-chl-bypass", "cf-mitigated", - "cloudflare", "cf-browser-verification", - "attention required! | cloudflare", "just a moment", - "checking your browser before accessing", + "cf-browser-verification", "cf-error-details", "cf-error-code", + # 质询页特征文案(足够具体,正常内容不会完整出现) + "just a moment", "checking your browser before accessing", + "attention required! | cloudflare", + "enable javascript and cookies to continue", ]), ("imperva", [ + # Incapsula cookie/技术标识符 "incap_ses", "visid_incap", "incap_ses_", - "imperva", "incapsula", - "request unsuccessful. incapsula incident id", + "incapsula incident id", "request unsuccessful. incapsula", + "visit denied by incapsula", ]), ("perimeterx", [ - "_px", "px-captcha", "pxhd", "pxcts", "pxcookie", - "perimeterx", "press & hold to confirm you are a human", + # PerimeterX 专有标识符(_px 太短会匹配 CSS 类名,已移除) + "px-captcha", "pxhd", "pxcts", "pxcookie", + "_pxff", "_pxhd", + "press & hold to confirm you are a human", ]), ("datadome", [ - "datadome", "dd-", "data-dome", - "protected by datadome", + # DataDome 专有标识符(dd- 太宽泛会匹配 dd-class 等,已移除) + "datadome", "data-dome", + "protected by datadome", "datadome-bot-protect", ]), ("akamai", [ - "akamai", "bm_sz", "_abck", - "reference #", "akamaighost", + # Akamai Bot Manager cookie/标识符(akamai 裸名会匹配正文引用,已移除) + "bm_sz", "_abck", "akamaighost", "akamai-bot-manager", + "ak_bmsc", ]), - # 通用反爬指示词(无明确 WAF 归属) + # 通用反爬指示词(v2.0.1 收窄:完整短语而非单词,避免正文误判) ("generic", [ - "captcha", "challenge", "verify you are human", - "making sure you're not a bot", - "please enable javascript", "enable javascript to continue", - "ddos protection", "access denied", + "verify you are human", "verify that you are human", + "making sure you're not a bot", "are you a robot", + "robot or human", "human verification", + "please complete the captcha", "complete the security check", + "please enable javascript to continue", + "enable javascript to continue", + "ddos protection by", "access denied - sucuri", "you have been blocked", "unusual traffic from your computer", - "robot or human", "are you a robot", "pardon our interruption", "we'll be right back", + "bot protection", "anti-bot protection", + "anubis_challenge", "miserere", # Anubis 反爬系统(拦截 AI 爬虫) ]), ] +#
This article explains how CAPTCHA works.
" - # 关键词匹配会误报为 generic - assert _detect_anti_bot(html) == "generic" + # v2.0.1:收窄后不再误判 + assert _detect_anti_bot(html) is None + + +def test_detect_anti_bot_article_mentions_cloudflare_in_context(): + """v2.0.1 修复:文章引用 Cloudflare 文档链接,不再误判为 cloudflare WAF。 + + v2.0.0 用裸 "cloudflare" 做全文匹配,Wayback 归档的 DataCamp 文章 + 正文里有 cloudflare.com 链接,被误判为反爬页导致兜底失败。 + """ + html = ("Learn more at Cloudflare
" + "Join our daily coding challenges!
" + "Daily 5-minute coding challenges.
" + assert _detect_anti_bot(html) is None # ----- Priority: specialized WAF before generic ----- @@ -197,7 +227,8 @@ def test_waf_fingerprints_all_have_indicators(): def test_is_blocked_page_delegates_to_detect(): """_is_blocked_page 应委托给 _detect_anti_bot。""" - assert _is_blocked_page("captcha") is True + # v2.0.1:用完整反爬短语而非裸 captcha + assert _is_blocked_page("please complete the captcha") is True assert _is_blocked_page("normal content") is False diff --git a/tests/test_fetch_auto.py b/tests/test_fetch_auto.py index 393de8c..806fea1 100644 --- a/tests/test_fetch_auto.py +++ b/tests/test_fetch_auto.py @@ -46,9 +46,12 @@ def test_blocked_page_checks_first_2000_chars(): v2.0.0 changed this to full-document scanning because large anti-bot pages (e.g. Cloudflare challenges with big JS blobs) may place the telltale keyword beyond the 2000-char boundary. + + v2.0.1: 用完整短语 'please complete the captcha' 替代裸 'captcha', + 避免正常内容误判。 """ padding = "x" * 2500 - html = f"{padding}captcha" + html = f"{padding}please complete the captcha" # v2.0.0: now detected (was: not detected) assert _is_blocked_page(html) diff --git a/tests/test_wayback_fallback.py b/tests/test_wayback_fallback.py index f8e92ff..23a9c15 100644 --- a/tests/test_wayback_fallback.py +++ b/tests/test_wayback_fallback.py @@ -208,22 +208,67 @@ def test_fetch_page_normal_page_no_anti_bot(): def test_fetch_page_anti_bot_triggers_wayback(): - """被反爬拦截后也应尝试 Wayback(_should_try_fallback 防御性检查)。 + """被反爬拦截后应尝试 Wayback 兜底(v2.0.1 修复)。 - 当前实现:主抓取成功返回反爬页内容 → result 非 None → 不触发兜底。 - 此测试记录此行为:反爬页被当作"成功抓取"返回,在 fetch_page 内部检测。 + v2.0.0 bug:主抓取"成功"返回反爬页(HTTP 200 + Cloudflare 质询)时, + result 非 None 导致 _should_try_fallback 返回 False,Wayback 永不触发。 + v2.0.1 修复:反爬检测提前到兜底判断之前,反爬阳性也触发 Wayback。 + + 本测试 mock fetch_url 两次调用: + 1. 主抓取 → 返回 Cloudflare 质询页 + 2. Wayback 兜底 → 返回正常归档页 """ cf_html = "Just a moment... cf-ray: 123" + wb_html = "Archived article content.
" + cf_result = _make_result(content=cf_html, final_url="https://example.com") + wb_result = _make_result(content=wb_html, + final_url="https://web.archive.org/web/2024/https://example.com") with patch("search.fetch_url", - return_value=_make_result(content=cf_html)): + side_effect=[cf_result, wb_result]) as mock_fetch: result = fetch_page("https://example.com", fallback_enabled=True) - # 反爬页被检测到,标记为 error + anti_bot_detected + # Wayback 兜底成功,反爬标记清除,status=ok + assert result["status"] == "ok" + assert result["anti_bot_detected"] is False + assert result["waf_type"] is None + assert result["fallback_used"] == "wayback" + assert "Archived article content." in result["text"] + # 确认 fetch_url 被调用两次:主抓取 + Wayback + assert mock_fetch.call_count == 2 + # 第二次调用应该是 Wayback URL + assert "web.archive.org/web/2/" in mock_fetch.call_args_list[1][0][0] + + +def test_fetch_page_anti_bot_wayback_also_blocked(): + """主抓取反爬 + Wayback 也反爬/失败 → 最终返回 error。 + + Wayback 兜底返回 None(失败)时,保留主抓取的反爬检测结果。 + """ + cf_html = "Just a moment... cf-ray: 123" + cf_result = _make_result(content=cf_html, final_url="https://example.com") + with patch("search.fetch_url", + side_effect=[cf_result, RuntimeError("wayback timeout")]): + result = fetch_page("https://example.com", fallback_enabled=True) + # Wayback 失败,保留反爬 error assert result["status"] == "error" assert result["anti_bot_detected"] is True - # 但不会触发 Wayback(因为主抓取"成功"了,只是内容是反爬页) + assert result["waf_type"] == "cloudflare" assert result["fallback_used"] is None +def test_fetch_page_anti_bot_fallback_disabled(): + """--no-fallback 时反爬页直接返回 error,不尝试 Wayback。""" + cf_html = "Just a moment... cf-ray: 123" + with patch("search.fetch_url", + return_value=_make_result(content=cf_html)) as mock_fetch: + result = fetch_page("https://example.com", fallback_enabled=False) + assert result["status"] == "error" + assert result["anti_bot_detected"] is True + assert result["waf_type"] == "cloudflare" + assert result["fallback_used"] is None + # 只调用一次(主抓取),没有 Wayback + assert mock_fetch.call_count == 1 + + def test_fetch_page_passes_referer(): """referer 透传给 fetch_url。""" with patch("search.fetch_url", return_value=_make_result()) as mock_fetch: