"""Tests for the --fetch auto-fetch feature in search.py.
Covers: _is_blocked_page (CAPTCHA/bot detection heuristics), fetch_page
(ok/error/blocked paths), fetch_top_results (URL dedup, ordering,
concurrency, empty input).
"""
from unittest.mock import patch, MagicMock
import search as search_mod
from search import _is_blocked_page, fetch_page, fetch_top_results
import fetch as fetch_mod
from fetch import FetchResult
# ----- _is_blocked_page -----
def test_blocked_page_detects_captcha():
assert _is_blocked_page("Please complete the CAPTCHA")
def test_blocked_page_detects_cloudflare():
assert _is_blocked_page("Just a moment... cf-browser-verification")
def test_blocked_page_detects_challenge():
assert _is_blocked_page("Checking your browser before accessing")
def test_blocked_page_detects_anubis():
assert _is_blocked_page("anubis_challenge")
def test_blocked_page_clean_html():
"""Normal HTML must not be flagged as blocked."""
assert not _is_blocked_page("
Real content")
def test_blocked_page_empty():
assert not _is_blocked_page("")
def test_blocked_page_checks_first_2000_chars():
"""Detection only scans the first 2000 chars for performance."""
padding = "x" * 2500
html = f"{padding}captcha"
assert not _is_blocked_page(html)
# ----- fetch_page -----
def _mock_fetch_result(content: str, content_type="text/html",
final_url="https://example.com",
truncated=False, ua="searxng-cli/1.6.0"):
return FetchResult(
content=content,
content_type=content_type,
final_url=final_url,
truncated=truncated,
user_agent=ua,
)
def test_fetch_page_ok_html():
"""Successful HTML fetch extracts text and returns status=ok."""
html = "Hello world"
with patch.object(search_mod, "fetch_url",
return_value=_mock_fetch_result(html)):
r = fetch_page("https://example.com")
assert r["status"] == "ok"
assert "Hello world" in r["text"]
assert r["text_length"] > 0
assert r["truncated"] is False
def test_fetch_page_ok_non_html():
"""Non-HTML content is returned as-is without text extraction."""
with patch.object(search_mod, "fetch_url",
return_value=_mock_fetch_result(
"plain text", content_type="text/plain")):
r = fetch_page("https://example.com/file.txt")
assert r["status"] == "ok"
assert r["text"] == "plain text"
def test_fetch_page_blocked_detection():
"""CAPTCHA pages are marked as error, not ok."""
html = "Please complete the CAPTCHA to continue"
with patch.object(search_mod, "fetch_url",
return_value=_mock_fetch_result(html)):
r = fetch_page("https://example.com")
assert r["status"] == "error"
assert "Bot protection" in r["error"]
assert r["text"] == ""
def test_fetch_page_network_error():
"""Network exceptions return status=error with message."""
with patch.object(search_mod, "fetch_url",
side_effect=RuntimeError("HTTP 503 for url")):
r = fetch_page("https://example.com")
assert r["status"] == "error"
assert "503" in r["error"]
assert r["text"] == ""
def test_fetch_page_truncated_flag():
"""truncated=True is propagated from fetch_url."""
html = "" + "x" * 100 + ""
with patch.object(search_mod, "fetch_url",
return_value=_mock_fetch_result(html, truncated=True)):
r = fetch_page("https://example.com", max_size=50)
assert r["status"] == "ok"
assert r["truncated"] is True
assert r["truncated_at"] == 50
def test_fetch_page_preserves_final_url():
"""Redirect final_url is captured in the result."""
with patch.object(search_mod, "fetch_url",
return_value=_mock_fetch_result(
"x",
final_url="https://final.example.com/page")):
r = fetch_page("https://example.com")
assert r["final_url"] == "https://final.example.com/page"
def test_fetch_page_records_fallback_ua():
"""user_agent_used is propagated for logging/debugging."""
with patch.object(search_mod, "fetch_url",
return_value=_mock_fetch_result(
"x",
ua="Mozilla/5.0 fallback")):
r = fetch_page("https://example.com")
assert r["user_agent_used"] == "Mozilla/5.0 fallback"
# ----- fetch_top_results -----
def test_fetch_top_results_empty():
"""No results → empty list, no fetch attempts."""
assert fetch_top_results({"results": []}, 3) == []
def test_fetch_top_results_no_results_key():
assert fetch_top_results({}, 3) == []
def test_fetch_top_results_dedup_urls():
"""Duplicate URLs are fetched only once."""
results = {"results": [
{"url": "https://a.com"},
{"url": "https://a.com"}, # dup
{"url": "https://b.com"},
]}
fetched_urls = []
def _fake_fetch(url, **kwargs):
fetched_urls.append(url)
return {"url": url, "status": "ok", "text": "x", "text_length": 1,
"truncated": False}
with patch.object(search_mod, "fetch_page", side_effect=_fake_fetch):
out = fetch_top_results(results, 3, request_delay=0)
assert len(fetched_urls) == 2 # dedup
assert "https://a.com" in fetched_urls
assert "https://b.com" in fetched_urls
def test_fetch_top_results_limits_count():
"""Only top N unique URLs are fetched."""
results = {"results": [{"url": f"https://x{i}.com"} for i in range(10)]}
def _fake_fetch(url, **kwargs):
return {"url": url, "status": "ok", "text": "x", "text_length": 1,
"truncated": False}
with patch.object(search_mod, "fetch_page", side_effect=_fake_fetch):
out = fetch_top_results(results, 3, request_delay=0)
assert len(out) == 3
def test_fetch_top_results_preserves_order():
"""Fetched results are reordered to match original result order."""
results = {"results": [
{"url": "https://a.com"},
{"url": "https://b.com"},
{"url": "https://c.com"},
]}
def _fake_fetch(url, **kwargs):
return {"url": url, "status": "ok", "text": url[-1], "text_length": 1,
"truncated": False}
with patch.object(search_mod, "fetch_page", side_effect=_fake_fetch):
out = fetch_top_results(results, 3, request_delay=0)
urls = [f["url"] for f in out]
assert urls == ["https://a.com", "https://b.com", "https://c.com"]
def test_fetch_top_results_handles_errors():
"""A failed fetch still appears in output with status=error."""
results = {"results": [{"url": "https://ok.com"}, {"url": "https://bad.com"}]}
def _fake_fetch(url, **kwargs):
if "bad" in url:
return {"url": url, "status": "error", "error": "503",
"text": "", "text_length": 0, "truncated": False}
return {"url": url, "status": "ok", "text": "content",
"text_length": 7, "truncated": False}
with patch.object(search_mod, "fetch_page", side_effect=_fake_fetch):
out = fetch_top_results(results, 2, request_delay=0)
statuses = {f["url"]: f["status"] for f in out}
assert statuses["https://ok.com"] == "ok"
assert statuses["https://bad.com"] == "error"
def test_fetch_top_results_skips_empty_urls():
"""Results without a URL are skipped."""
results = {"results": [
{"url": ""},
{"url": "https://real.com"},
]}
def _fake_fetch(url, **kwargs):
return {"url": url, "status": "ok", "text": "x", "text_length": 1,
"truncated": False}
with patch.object(search_mod, "fetch_page", side_effect=_fake_fetch):
out = fetch_top_results(results, 3, request_delay=0)
assert len(out) == 1
assert out[0]["url"] == "https://real.com"