"""Tests for v2.0.0 anti-bot / WAF detection enhancements. Covers: _detect_anti_bot (WAF fingerprint library: Cloudflare/Imperva/ PerimeterX/DataDome/Akamai/generic), full-document scanning (not just first 2000 chars), _is_blocked_page backward compatibility. """ import sys import os sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "scripts")) from search import _detect_anti_bot, _is_blocked_page, WAF_FINGERPRINTS # ----- Cloudflare detection ----- def test_detect_cloudflare_cf_ray(): html = "Just a moment..." \ "cf-ray: 8abc123" assert _detect_anti_bot(html) == "cloudflare" def test_detect_cloudflare_just_a_moment(): html = "Just a moment..." assert _detect_anti_bot(html) == "cloudflare" def test_detect_cloudflare_checking_browser(): html = "Checking your browser before accessing" assert _detect_anti_bot(html) == "cloudflare" def test_detect_cloudflare_attention_required(): html = "Attention Required! | Cloudflare" assert _detect_anti_bot(html) == "cloudflare" # ----- Imperva detection ----- def test_detect_imperva_incap_ses(): html = "incap_ses_123_cookie" assert _detect_anti_bot(html) == "imperva" def test_detect_imperva_incapsula(): html = "Request unsuccessful. Incapsula incident ID: 123" assert _detect_anti_bot(html) == "imperva" # ----- PerimeterX detection ----- def test_detect_perimeterx_px_captcha(): html = "px-captcha challenge" assert _detect_anti_bot(html) == "perimeterx" def test_detect_perimeterx_press_hold(): html = "Press & hold to confirm you are a human" assert _detect_anti_bot(html) == "perimeterx" # ----- DataDome detection ----- def test_detect_datadome_protected(): html = "Protected by DataDome" assert _detect_anti_bot(html) == "datadome" def test_detect_datadome_cookie(): html = "datadome cookie set" assert _detect_anti_bot(html) == "datadome" # ----- Akamai detection ----- def test_detect_akamai_bm_sz(): html = "bm_sz cookie" assert _detect_anti_bot(html) == "akamai" def test_detect_akamai_reference(): html = "Reference #123.akamaighost" assert _detect_anti_bot(html) == "akamai" # ----- Generic detection ----- def test_detect_generic_captcha(): """v2.0.1:收窄为完整短语 'please complete the captcha',避免正文误判。""" html = "Please complete the CAPTCHA" assert _detect_anti_bot(html) == "generic" def test_detect_generic_verify_human(): html = "Verify you are human" assert _detect_anti_bot(html) == "generic" def test_detect_generic_access_denied(): """v2.0.1:裸 'access denied' 从全文档扫描移除(避免权限文章误判), 改由 标签检测覆盖。反爬页 title 常为 'Access Denied'。""" html = "<html><head><title>Access DeniedAccess Denied" assert _detect_anti_bot(html) == "generic" def test_detect_generic_access_denied_sucuri(): """v2.0.1:Sucuri WAF 的完整文案在全文档扫描中保留。""" html = "Access Denied - Sucuri Website Firewall" assert _detect_anti_bot(html) == "generic" def test_detect_generic_blocked(): html = "You have been blocked" assert _detect_anti_bot(html) == "generic" def test_detect_generic_unusual_traffic(): html = "unusual traffic from your computer" assert _detect_anti_bot(html) == "generic" def test_detect_generic_robot(): html = "Are you a robot?" assert _detect_anti_bot(html) == "generic" # ----- Full-document scanning (v2.0.0 key improvement) ----- def test_detect_anti_bot_beyond_2000_chars(): """反爬指示词在前 2000 字符之外也能检测到。 v2.0.0 核心改进:旧版只扫前 2000 字符,大页面反爬页可能漏检。 """ # 构造 3000 字符的无意义填充 + 反爬关键词 padding = "x" * 2500 html = f"{padding}
Just a moment...
" assert _detect_anti_bot(html) == "cloudflare" def test_detect_anti_bot_large_page_end(): """反爬关键词在文档末尾也能检测到。""" padding = "y" * 5000 # v2.0.1:用完整短语而非裸 captcha,避免误判 html = f"{padding}please complete the captcha" assert _detect_anti_bot(html) == "generic" # ----- Negative cases ----- def test_detect_anti_bot_normal_page(): """正常页面不触发检测。""" html = "

Welcome

This is a normal article about Python programming.

" assert _detect_anti_bot(html) is None def test_detect_anti_bot_empty_content(): """空内容返回 None。""" assert _detect_anti_bot("") is None assert _detect_anti_bot(None) is None def test_detect_anti_bot_article_mentions_captcha_in_context(): """v2.0.1 修复:文章讨论 captcha 但不是反爬页,不再误判。 v2.0.0 用裸 "captcha" 做全文匹配,导致"讨论 captcha 工作原理的文章" 被误判为反爬页。v2.0.1 收窄为 "please complete the captcha" 等完整短语, 正常讨论 captcha 的文章不再触发。 """ html = "

This article explains how CAPTCHA works.

" # v2.0.1:收窄后不再误判 assert _detect_anti_bot(html) is None def test_detect_anti_bot_article_mentions_cloudflare_in_context(): """v2.0.1 修复:文章引用 Cloudflare 文档链接,不再误判为 cloudflare WAF。 v2.0.0 用裸 "cloudflare" 做全文匹配,Wayback 归档的 DataCamp 文章 正文里有 cloudflare.com 链接,被误判为反爬页导致兜底失败。 """ html = ("
" "

Learn more at Cloudflare

" "

Join our daily coding challenges!

" "
") assert _detect_anti_bot(html) is None def test_detect_anti_bot_article_mentions_challenge_in_context(): """v2.0.1 修复:文章含 'coding challenge' 等正常内容,不再误判。""" html = "

Daily 5-minute coding challenges.

" assert _detect_anti_bot(html) is None # ----- Priority: specialized WAF before generic ----- def test_detect_priority_cloudflare_over_generic(): """同时匹配 cloudflare 和 generic 时,返回 cloudflare(优先级)。""" # "just a moment" 是 cloudflare 专用,"captcha" 是 generic # cloudflare 在 WAF_FINGERPRINTS 中排在 generic 之前 html = "Just a moment... captcha" assert _detect_anti_bot(html) == "cloudflare" # ----- WAF_FINGERPRINTS structure ----- def test_waf_fingerprints_has_six_types(): """指纹库覆盖 6 种 WAF 类型。""" types = [waf_type for waf_type, _ in WAF_FINGERPRINTS] assert "cloudflare" in types assert "imperva" in types assert "perimeterx" in types assert "datadome" in types assert "akamai" in types assert "generic" in types def test_waf_fingerprints_generic_is_last(): """generic 排在最后(优先级最低)。""" assert WAF_FINGERPRINTS[-1][0] == "generic" def test_waf_fingerprints_all_have_indicators(): """每个 WAF 类型都有至少 2 个指示词。""" for waf_type, indicators in WAF_FINGERPRINTS: assert len(indicators) >= 2, f"{waf_type} has too few indicators" # ----- _is_blocked_page backward compat ----- def test_is_blocked_page_delegates_to_detect(): """_is_blocked_page 应委托给 _detect_anti_bot。""" # v2.0.1:用完整反爬短语而非裸 captcha assert _is_blocked_page("please complete the captcha") is True assert _is_blocked_page("normal content") is False def test_is_blocked_page_cloudflare(): """Cloudflare 页面被检测为 blocked。""" assert _is_blocked_page("cf-ray: 123") is True