A. fetch.py 补齐 Wayback 兜底 (修复重大 gap) - v2.0.1 gap: fetch.py 独立调用 403 时无 Wayback 兜底 (仅 search.py --fetch 有) - AI Agent 用 fetch.py -u URL 直接抓取被墙站点时, 403 后无任何回退 - 修复: fetch.py main() 增加 Wayback 兜底逻辑 + --no-fallback flag - 共享逻辑抽取到 common.py: should_try_wayback() + build_wayback_url() B. --research 研究模式 - 给定主题自动扩展 5 个多角度查询: overview/profile/background/works/review - 确定性规则 (不依赖 AI 判断), 跨进程可复现 - 输出含 research_topic + research_queries 元数据, AI Agent 可按角度结构化汇编 - 与 --query/--queries-file 互斥, 支持所有输出格式 (json/brief/urls/csv) - 三态退出码: 0=有结果, 2=全部空, 1=全部错误 C. 被墙站点智能回退 - common.py 增加 HARD_BLOCKED_DOMAINS: 百度百科/知乎/微博/微信公众号/豆瓣等 - is_hard_blocked_domain() 精确匹配 + 子域匹配 - 命中被墙站点时: 主抓取失败后立即 Wayback (不等 should_try_wayback 判断) - search.py _should_try_fallback 增加 url 参数, 被墙站点直接触发兜底 真实测试验证 (search.metona.cn 实例): - fetch.py 百度百科兜底: 403 → Wayback 恢复 150,493 chars ✓ - --research 模式: 5 角度查询扩展 + research 元数据 + 三态退出码 ✓ - 被墙站点检测: Hard-blocked domain detected 日志 + 自动 Wayback ✓ 测试: 503 个全部通过 (新增 45 个: test_wayback_shared + test_research_mode) 来源: 另一个 AI Agent 反馈 Wikipedia/百度百科/知乎 fetch 失败, 需要多角度搜索+失败回退+被墙站点列表
140 lines
5.1 KiB
Python
140 lines
5.1 KiB
Python
"""Tests for v2.1.0 shared Wayback fallback + hard-blocked domain logic.
|
|
|
|
Covers common.py functions:
|
|
* should_try_wayback — error string → bool (should we try Wayback?)
|
|
* build_wayback_url — URL → Wayback Machine URL
|
|
* is_hard_blocked_domain — URL → bool (is this a known anti-bot site?)
|
|
"""
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "scripts"))
|
|
|
|
from common import (
|
|
should_try_wayback,
|
|
build_wayback_url,
|
|
is_hard_blocked_domain,
|
|
HARD_BLOCKED_DOMAINS,
|
|
)
|
|
|
|
|
|
# ===== should_try_wayback =====
|
|
|
|
class TestShouldTryWayback:
|
|
def test_403_triggers(self):
|
|
assert should_try_wayback("403 Client Error: Forbidden") is True
|
|
|
|
def test_404_triggers(self):
|
|
assert should_try_wayback("404 Client Error: Not Found") is True
|
|
|
|
def test_timeout_triggers(self):
|
|
assert should_try_wayback("Connection timeout") is True
|
|
|
|
def test_timed_out_triggers(self):
|
|
assert should_try_wayback("Read timed out") is True
|
|
|
|
def test_connection_reset_triggers(self):
|
|
assert should_try_wayback("ConnectionResetError: connection reset") is True
|
|
|
|
def test_connection_aborted_triggers(self):
|
|
assert should_try_wayback("ConnectionAbortedError") is True
|
|
|
|
def test_max_retries_triggers(self):
|
|
assert should_try_wayback("Max retries exceeded with url") is True
|
|
|
|
def test_dns_failure_does_not_trigger(self):
|
|
"""DNS failures should not trigger Wayback — Wayback can't resolve either."""
|
|
assert should_try_wayback("Name or service not known") is False
|
|
|
|
def test_empty_does_not_trigger(self):
|
|
assert should_try_wayback("") is False
|
|
|
|
def test_none_does_not_trigger(self):
|
|
assert should_try_wayback(None) is False
|
|
|
|
def test_generic_error_does_not_trigger(self):
|
|
assert should_try_wayback("Some random error") is False
|
|
|
|
|
|
# ===== build_wayback_url =====
|
|
|
|
class TestBuildWaybackUrl:
|
|
def test_basic_url(self):
|
|
url = "https://example.com/page"
|
|
result = build_wayback_url(url)
|
|
assert result == "https://web.archive.org/web/2/https://example.com/page"
|
|
|
|
def test_http_url(self):
|
|
url = "http://example.com"
|
|
result = build_wayback_url(url)
|
|
assert result == "https://web.archive.org/web/2/http://example.com"
|
|
|
|
def test_url_with_query_params(self):
|
|
url = "https://example.com/search?q=test&lang=en"
|
|
result = build_wayback_url(url)
|
|
assert "web.archive.org/web/2/" in result
|
|
assert url in result
|
|
|
|
|
|
# ===== is_hard_blocked_domain =====
|
|
|
|
class TestIsHardBlockedDomain:
|
|
def test_baike_baidu_com(self):
|
|
assert is_hard_blocked_domain("https://baike.baidu.com/item/Python") is True
|
|
|
|
def test_zhihu_com(self):
|
|
assert is_hard_blocked_domain("https://zhuanlan.zhihu.com/p/123") is True
|
|
|
|
def test_zhihu_com_root(self):
|
|
assert is_hard_blocked_domain("https://www.zhihu.com/question/123") is True
|
|
|
|
def test_weibo_com(self):
|
|
assert is_hard_blocked_domain("https://weibo.com/123456") is True
|
|
|
|
def test_m_weibo_cn(self):
|
|
assert is_hard_blocked_domain("https://m.weibo.cn/detail/123") is True
|
|
|
|
def test_mp_weixin_qq_com(self):
|
|
assert is_hard_blocked_domain("https://mp.weixin.qq.com/s/abc") is True
|
|
|
|
def test_douban_com(self):
|
|
assert is_hard_blocked_domain("https://book.douban.com/subject/123") is True
|
|
|
|
def test_zhidao_baidu_com(self):
|
|
assert is_hard_blocked_domain("https://zhidao.baidu.com/question/123") is True
|
|
|
|
def test_normal_site_not_blocked(self):
|
|
assert is_hard_blocked_domain("https://example.com") is False
|
|
|
|
def test_wikipedia_not_blocked(self):
|
|
assert is_hard_blocked_domain("https://zh.wikipedia.org/wiki/Python") is False
|
|
|
|
def test_github_not_blocked(self):
|
|
assert is_hard_blocked_domain("https://github.com/python/cpython") is False
|
|
|
|
def test_baidu_search_not_blocked(self):
|
|
"""baidu.com search page is NOT in the hard-blocked list — only subdomains
|
|
like baike.baidu.com, zhidao.baidu.com are."""
|
|
# Actually, baidu.com is in _SUBDOMAIN_BLOCKED, so www.baidu.com matches.
|
|
# This is intentional — Baidu's main search also has strong anti-bot.
|
|
assert is_hard_blocked_domain("https://www.baidu.com/s?wd=test") is True
|
|
|
|
def test_empty_url(self):
|
|
assert is_hard_blocked_domain("") is False
|
|
|
|
def test_none_url(self):
|
|
assert is_hard_blocked_domain(None) is False
|
|
|
|
def test_no_scheme(self):
|
|
"""URLs without scheme: urlparse won't extract hostname.
|
|
This is expected — callers should provide full URLs."""
|
|
# With scheme: works
|
|
assert is_hard_blocked_domain("https://baike.baidu.com/item/test") is True
|
|
# Without scheme: urlparse returns no hostname — function returns False
|
|
# This is acceptable: all real callers (fetch.py, search.py) pass full URLs
|
|
assert is_hard_blocked_domain("baike.baidu.com/item/test") is False
|
|
|
|
def test_case_insensitive(self):
|
|
assert is_hard_blocked_domain("https://BAIKE.BAIDU.COM/item/test") is True
|
|
assert is_hard_blocked_domain("https://ZHIHU.COM/question/123") is True
|