"""Tests for the --fetch auto-fetch feature in search.py. Covers: _is_blocked_page (CAPTCHA/bot detection heuristics), fetch_page (ok/error/blocked paths), fetch_top_results (URL dedup, ordering, concurrency, empty input). """ from unittest.mock import patch, MagicMock import search as search_mod from search import _is_blocked_page, fetch_page, fetch_top_results import fetch as fetch_mod from fetch import FetchResult # ----- _is_blocked_page ----- def test_blocked_page_detects_captcha(): assert _is_blocked_page("Please complete the CAPTCHA") def test_blocked_page_detects_cloudflare(): assert _is_blocked_page("Just a moment... cf-browser-verification") def test_blocked_page_detects_challenge(): assert _is_blocked_page("Checking your browser before accessing") def test_blocked_page_detects_anubis(): assert _is_blocked_page("anubis_challenge") def test_blocked_page_clean_html(): """Normal HTML must not be flagged as blocked.""" assert not _is_blocked_page("
Real content
") def test_blocked_page_empty(): assert not _is_blocked_page("") def test_blocked_page_checks_first_2000_chars(): """Detection only scans the first 2000 chars for performance.""" padding = "x" * 2500 html = f"{padding}captcha" assert not _is_blocked_page(html) # ----- fetch_page ----- def _mock_fetch_result(content: str, content_type="text/html", final_url="https://example.com", truncated=False, ua="searxng-cli/1.6.0"): return FetchResult( content=content, content_type=content_type, final_url=final_url, truncated=truncated, user_agent=ua, ) def test_fetch_page_ok_html(): """Successful HTML fetch extracts text and returns status=ok.""" html = "
Hello world
" with patch.object(search_mod, "fetch_url", return_value=_mock_fetch_result(html)): r = fetch_page("https://example.com") assert r["status"] == "ok" assert "Hello world" in r["text"] assert r["text_length"] > 0 assert r["truncated"] is False def test_fetch_page_ok_non_html(): """Non-HTML content is returned as-is without text extraction.""" with patch.object(search_mod, "fetch_url", return_value=_mock_fetch_result( "plain text", content_type="text/plain")): r = fetch_page("https://example.com/file.txt") assert r["status"] == "ok" assert r["text"] == "plain text" def test_fetch_page_blocked_detection(): """CAPTCHA pages are marked as error, not ok.""" html = "Please complete the CAPTCHA to continue" with patch.object(search_mod, "fetch_url", return_value=_mock_fetch_result(html)): r = fetch_page("https://example.com") assert r["status"] == "error" assert "Bot protection" in r["error"] assert r["text"] == "" def test_fetch_page_network_error(): """Network exceptions return status=error with message.""" with patch.object(search_mod, "fetch_url", side_effect=RuntimeError("HTTP 503 for url")): r = fetch_page("https://example.com") assert r["status"] == "error" assert "503" in r["error"] assert r["text"] == "" def test_fetch_page_truncated_flag(): """truncated=True is propagated from fetch_url.""" html = "" + "x" * 100 + "" with patch.object(search_mod, "fetch_url", return_value=_mock_fetch_result(html, truncated=True)): r = fetch_page("https://example.com", max_size=50) assert r["status"] == "ok" assert r["truncated"] is True assert r["truncated_at"] == 50 def test_fetch_page_preserves_final_url(): """Redirect final_url is captured in the result.""" with patch.object(search_mod, "fetch_url", return_value=_mock_fetch_result( "x", final_url="https://final.example.com/page")): r = fetch_page("https://example.com") assert r["final_url"] == "https://final.example.com/page" def test_fetch_page_records_fallback_ua(): """user_agent_used is propagated for logging/debugging.""" with patch.object(search_mod, "fetch_url", return_value=_mock_fetch_result( "x", ua="Mozilla/5.0 fallback")): r = fetch_page("https://example.com") assert r["user_agent_used"] == "Mozilla/5.0 fallback" # ----- fetch_top_results ----- def test_fetch_top_results_empty(): """No results → empty list, no fetch attempts.""" assert fetch_top_results({"results": []}, 3) == [] def test_fetch_top_results_no_results_key(): assert fetch_top_results({}, 3) == [] def test_fetch_top_results_dedup_urls(): """Duplicate URLs are fetched only once.""" results = {"results": [ {"url": "https://a.com"}, {"url": "https://a.com"}, # dup {"url": "https://b.com"}, ]} fetched_urls = [] def _fake_fetch(url, **kwargs): fetched_urls.append(url) return {"url": url, "status": "ok", "text": "x", "text_length": 1, "truncated": False} with patch.object(search_mod, "fetch_page", side_effect=_fake_fetch): out = fetch_top_results(results, 3, request_delay=0) assert len(fetched_urls) == 2 # dedup assert "https://a.com" in fetched_urls assert "https://b.com" in fetched_urls def test_fetch_top_results_limits_count(): """Only top N unique URLs are fetched.""" results = {"results": [{"url": f"https://x{i}.com"} for i in range(10)]} def _fake_fetch(url, **kwargs): return {"url": url, "status": "ok", "text": "x", "text_length": 1, "truncated": False} with patch.object(search_mod, "fetch_page", side_effect=_fake_fetch): out = fetch_top_results(results, 3, request_delay=0) assert len(out) == 3 def test_fetch_top_results_preserves_order(): """Fetched results are reordered to match original result order.""" results = {"results": [ {"url": "https://a.com"}, {"url": "https://b.com"}, {"url": "https://c.com"}, ]} def _fake_fetch(url, **kwargs): return {"url": url, "status": "ok", "text": url[-1], "text_length": 1, "truncated": False} with patch.object(search_mod, "fetch_page", side_effect=_fake_fetch): out = fetch_top_results(results, 3, request_delay=0) urls = [f["url"] for f in out] assert urls == ["https://a.com", "https://b.com", "https://c.com"] def test_fetch_top_results_handles_errors(): """A failed fetch still appears in output with status=error.""" results = {"results": [{"url": "https://ok.com"}, {"url": "https://bad.com"}]} def _fake_fetch(url, **kwargs): if "bad" in url: return {"url": url, "status": "error", "error": "503", "text": "", "text_length": 0, "truncated": False} return {"url": url, "status": "ok", "text": "content", "text_length": 7, "truncated": False} with patch.object(search_mod, "fetch_page", side_effect=_fake_fetch): out = fetch_top_results(results, 2, request_delay=0) statuses = {f["url"]: f["status"] for f in out} assert statuses["https://ok.com"] == "ok" assert statuses["https://bad.com"] == "error" def test_fetch_top_results_skips_empty_urls(): """Results without a URL are skipped.""" results = {"results": [ {"url": ""}, {"url": "https://real.com"}, ]} def _fake_fetch(url, **kwargs): return {"url": url, "status": "ok", "text": "x", "text_length": 1, "truncated": False} with patch.object(search_mod, "fetch_page", side_effect=_fake_fetch): out = fetch_top_results(results, 3, request_delay=0) assert len(out) == 1 assert out[0]["url"] == "https://real.com"