diff --git a/shelfmark/bypass/__init__.py b/shelfmark/bypass/__init__.py
index 2986e3a9..ed1ec9f3 100644
--- a/shelfmark/bypass/__init__.py
+++ b/shelfmark/bypass/__init__.py
@@ -3,3 +3,14 @@
class BypassCancelledError(Exception):
"""Raised when a bypass operation is cancelled."""
+
+
+class ChallengeNotSolvedError(Exception):
+ """Raised when a bypasser ran but the site still answered with a challenge.
+
+ Distinct from a bypasser that is broken or unreachable, which is what every
+ "the bypass failed" message used to say. A solver can do its job perfectly and
+ still be handed something it cannot clear - DDoS-Guard's manual CAPTCHA page is
+ the case from #1292 - and telling the user to go check that FlareSolverr is
+ reachable sends them to fix a service that is working.
+ """
diff --git a/shelfmark/bypass/external_bypasser.py b/shelfmark/bypass/external_bypasser.py
index 53dad163..f02f1ef5 100644
--- a/shelfmark/bypass/external_bypasser.py
+++ b/shelfmark/bypass/external_bypasser.py
@@ -6,7 +6,7 @@ from typing import TYPE_CHECKING, Any
import requests
-from shelfmark.bypass import BypassCancelledError
+from shelfmark.bypass import BypassCancelledError, ChallengeNotSolvedError
from shelfmark.bypass.challenge import challenge_marker
from shelfmark.bypass.cookie_store import store_extracted_cookies
from shelfmark.core.config import config
@@ -92,7 +92,13 @@ def _store_solution_clearance(target_url: str, solution: Mapping[str, Any]) -> N
def _fetch_via_bypasser(target_url: str) -> str | None:
- """Make a single request to the external bypasser service. Returns HTML or None."""
+ """Make a single request to the external bypasser service. Returns HTML or None.
+
+ Raises:
+ ChallengeNotSolvedError: the service answered with a page that is still a
+ challenge, whatever verdict it reported on itself.
+
+ """
raw_bypasser_url = _coerce_config_str(
config.get("EXT_BYPASSER_URL", "http://flaresolverr:8191"),
"http://flaresolverr:8191",
@@ -155,6 +161,12 @@ def _fetch_via_bypasser(target_url: str) -> str | None:
marker,
)
if marker:
+ # The solver's verdict is not evidence; the page is. Returning this one as a
+ # success is what made #1292 unrecoverable: the retry-and-rotate loop that
+ # could still have saved the search - the next mirror is a different
+ # DDoS-Guard host, in its own state - was never entered, and the challenge
+ # page's own __ddg cookies were filed as this host's clearance and replayed
+ # on every later request.
logger.warning(
"External bypasser reported success but returned a challenge page for "
"'%s' (%d bytes, marker=%r) - the solve did not clear the protection",
@@ -162,6 +174,7 @@ def _fetch_via_bypasser(target_url: str) -> str | None:
len(html),
marker,
)
+ raise ChallengeNotSolvedError(marker)
try:
_store_solution_clearance(target_url, solution)
@@ -212,16 +225,33 @@ def get_bypassed_page(
selector: network.AAMirrorSelector | None = None,
cancel_flag: Event | None = None,
) -> str | None:
- """Fetch HTML via external bypasser with retries and mirror rotation."""
+ """Fetch HTML via external bypasser with retries and mirror rotation.
+
+ Raises:
+ ChallengeNotSolvedError: every attempt came back still carrying a challenge.
+ Reported apart from returning None because the two ask the user for
+ opposite things: None means go and check the bypasser, this means the
+ bypasser is fine and the host is the one refusing.
+ BypassCancelledError: the caller's cancel flag was set.
+
+ """
from shelfmark.download import network as network_module
sel = selector or network_module.AAMirrorSelector()
+ unsolved_marker: str | None = None
for attempt in range(1, MAX_RETRY + 1):
_check_cancelled(cancel_flag, "by user")
attempt_url = sel.rewrite(url)
- result = _fetch_via_bypasser(attempt_url)
+ try:
+ result = _fetch_via_bypasser(attempt_url)
+ except ChallengeNotSolvedError as e:
+ # Worth the remaining attempts rather than an immediate give-up: the retry
+ # rotates onto the next mirror, and that is a different DDoS-Guard host with
+ # its own idea of whether this caller needs a CAPTCHA.
+ unsolved_marker = str(e) or unsolved_marker
+ result = None
if result:
return result
@@ -242,4 +272,11 @@ def get_bypassed_page(
if action in ("mirror", "dns") and new_base:
logger.info("Rotated %s for retry", action)
+ if unsolved_marker:
+ msg = (
+ "The bypasser ran, but the site kept answering with a protection challenge "
+ f"(marker={unsolved_marker!r}). That is usually a manual CAPTCHA, which no "
+ "bypasser can answer - the bypasser itself is working. Try again shortly."
+ )
+ raise ChallengeNotSolvedError(msg)
return None
diff --git a/shelfmark/download/http.py b/shelfmark/download/http.py
index 3fe5f18f..f58c0b78 100644
--- a/shelfmark/download/http.py
+++ b/shelfmark/download/http.py
@@ -5,12 +5,12 @@ import time
from http import HTTPStatus
from io import BytesIO
from typing import TYPE_CHECKING, NoReturn
-from urllib.parse import urljoin, urlparse
+from urllib.parse import parse_qsl, urlencode, urljoin, urlparse, urlunparse
import requests
from tqdm import tqdm
-from shelfmark.bypass import BypassCancelledError, cookie_store
+from shelfmark.bypass import BypassCancelledError, ChallengeNotSolvedError, cookie_store
from shelfmark.bypass.challenge import challenge_marker
from shelfmark.core import search_deadline
from shelfmark.core.config import config as app_config
@@ -29,6 +29,10 @@ logger = setup_logger(__name__)
_RNG = random.SystemRandom()
_MAX_REDIRECTS = 5
+# DDoS-Guard's re-check probe. Its 302 to `?check=1` is one hop of a handshake rather
+# than a page: the parameter asserts the caller already holds the cookies that hop
+# issued.
+_DDG_CHECK_PARAM = "check"
# Z-Library answers the first hit with a 503 whose only real payload is a Set-Cookie; echoing
# that cookie back returns the 302 to the real page. Two attempts cover the handshake without
# letting a server that keeps re-issuing cookies hold us in the loop.
@@ -48,6 +52,7 @@ _BYPASS_GRACE_SLACK_SECONDS = 30.0
_BYPASSER_ERRORS = (
AttributeError,
BypassCancelledError,
+ ChallengeNotSolvedError,
KeyError,
OSError,
RuntimeError,
@@ -252,6 +257,33 @@ def _response_challenge_marker(response: requests.Response) -> str | None:
return None
+def _solvable_url(url: str) -> str:
+ """The URL a solver should open, given one we may be mid-handshake on.
+
+ The manual AA redirect follower in `html_get_page` walks DDoS-Guard's handshake by
+ reassigning `current_url`, so by the time a 403, a 503 challenge or a redirect loop
+ hands that URL to a bypasser it is often the `?check=1` probe rather than the page
+ we actually wanted. A solver opens it in a fresh browser holding none of the cookies
+ the probe exists to collect, so DDoS-Guard cannot verify it automatically and answers
+ with the manual CAPTCHA page that nothing can solve - the failure in #1292, where
+ FlareSolverr reported "Challenge solved!" over a 4.7 KB DDOS-GUARD interstitial.
+
+ Handing over the pre-probe URL instead lets the solver's browser run the whole
+ handshake itself, which is what a real browser does and what the solver is for.
+
+ Scoped to the hosts whose redirects we follow manually: everywhere else `check` is
+ an ordinary query parameter and none of our business.
+ """
+ if not network.should_rotate_dns_for_url(url):
+ return url
+ parsed = urlparse(url)
+ params = parse_qsl(parsed.query, keep_blank_values=True)
+ kept = [(key, value) for key, value in params if key != _DDG_CHECK_PARAM]
+ if len(kept) == len(params):
+ return url
+ return urlunparse(parsed._replace(query=urlencode(kept)))
+
+
def _fatal_mirror_reason(e: Exception) -> str | None:
"""Return why ``e`` proves the mirror is unusable, or None if it may recover.
@@ -371,6 +403,9 @@ def html_get_page(
retry-loop branch above with `continue`, and with MAX_RETRY=1 there is no
later attempt for that branch to run on either.
"""
+ # Every handoff reaches the solver through here, so this is the one place the
+ # mid-handshake `?check=1` URL has to be unwound. See _solvable_url.
+ bypass_url = _solvable_url(bypass_url)
# Never start a minutes-long browser solve on a budget that has already run out:
# nothing downstream would get to report the real reason before the caller's
# deadline (or its reverse proxy) cut the request off.
@@ -405,6 +440,18 @@ def html_get_page(
except _STATUS_CALLBACK_ERRORS:
logger.debug("Rate-limit status callback failed", exc_info=True)
return _fail(str(e), bypass_url)
+ except ChallengeNotSolvedError as e:
+ # Not a bypasser malfunction: it ran, and the host answered with something it
+ # cannot clear - DDoS-Guard's manual CAPTCHA, typically. Must precede the
+ # generic handler below, whose "the protection bypasser failed" is what sent
+ # #1292 off to fix a FlareSolverr that was working perfectly.
+ logger.info("Bypass ran but did not clear the protection: %s", e)
+ if status_callback:
+ try:
+ status_callback("error", str(e))
+ except _STATUS_CALLBACK_ERRORS:
+ logger.debug("Unsolved-challenge status callback failed", exc_info=True)
+ return _fail(str(e), bypass_url)
except _BYPASSER_ERRORS as e:
logger.warning("Bypasser error: %s: %s", type(e).__name__, e)
# Surface the real reason. Without this the caller only sees an empty
diff --git a/shelfmark/release_sources/direct_download.py b/shelfmark/release_sources/direct_download.py
index 0a8687ae..abbb5d73 100644
--- a/shelfmark/release_sources/direct_download.py
+++ b/shelfmark/release_sources/direct_download.py
@@ -115,6 +115,16 @@ def _html_response_text(response: str | tuple[str, str]) -> str:
return response
+def _html_response_url(response: str | tuple[str, str]) -> str | None:
+ """The URL that actually answered, when the downloader was asked to report it.
+
+ None for the plain-string shape, so a caller can fall back to what it requested.
+ """
+ if isinstance(response, tuple):
+ return response[1] or None
+ return None
+
+
def _attr_to_str(value: object) -> str | None:
"""Convert a BeautifulSoup attribute value to a plain string."""
if isinstance(value, str):
@@ -696,10 +706,21 @@ def _fetch_search_table_uncached(
if search_deadline.expired():
raise SearchUnavailableError(search_deadline.deadline_message())
+ # include_response_url is what makes the diagnostics below name the mirror that
+ # actually answered. html_get_page rotates mirrors and follows redirects on its
+ # own, so `attempt_url` is only where this iteration started: #1298's bundle
+ # reported the untabled page against annas-archive.gl when the body had come
+ # from .pk, which is precisely the triage cost #1289 added the line to remove.
response = downloader.html_get_page(
- attempt_url, selector=selector, allow_bypasser_fallback=True
+ attempt_url,
+ selector=selector,
+ allow_bypasser_fallback=True,
+ include_response_url=True,
)
- if not response:
+ html = _html_response_text(response)
+ # Checked on the body, not on `response`: with include_response_url the give-up
+ # shape is the tuple ("", url), and a tuple is truthy.
+ if not html:
# Network/mirror exhaustion path bubbles up so API can notify clients.
# html_get_page records the concrete give-up reason on the selector; fall
# back to the generic line only if nothing was recorded.
@@ -708,7 +729,7 @@ def _fetch_search_table_uncached(
)
raise SearchUnavailableError(f"Unable to reach download source. {detail}")
- html = _html_response_text(response)
+ answered_url = _html_response_url(response) or attempt_url
soup = BeautifulSoup(html, "html.parser")
table = soup.find("table")
if isinstance(table, Tag):
@@ -724,7 +745,7 @@ def _fetch_search_table_uncached(
# alone, and the response body is not in the debug bundle. Fingerprint it here
# so the next report says which branch fired and why, rather than costing
# another round of guesswork - see #1289.
- _log_untabled_search_page(attempt_url, html)
+ _log_untabled_search_page(answered_url, html)
if _looks_like_aa_page(html):
# A real AA response in a shape the caller should report as drift. Checked
@@ -737,9 +758,18 @@ def _fetch_search_table_uncached(
# what came back. Rotating is pointless (every mirror shares the same
# protection) and reporting it as an empty result is worse: the user is
# told their query found nothing when the search never ran.
+ #
+ # The wording no longer blames the bypasser outright. In #1292 it was
+ # reachable and working, and the page it was handed was DDoS-Guard's manual
+ # CAPTCHA - so "check that the bypasser is working" was the one piece of
+ # advice guaranteed to waste the reporter's time. Name the marker instead
+ # and let the two causes be told apart.
msg = (
- "Anna's Archive answered with an unsolved protection challenge. "
- "Check that the bypasser is reachable and working."
+ "Anna's Archive answered with a protection challenge that was not "
+ f"cleared (marker={challenge_marker(html)!r}). If the bypasser reports "
+ "solving it, the host is serving a manual CAPTCHA that no bypasser can "
+ "answer - try again shortly. Otherwise check that the bypasser is "
+ "reachable and working."
)
raise SearchUnavailableError(msg)
diff --git a/tests/bypass/test_unsolved_challenge.py b/tests/bypass/test_unsolved_challenge.py
new file mode 100644
index 00000000..5451f93b
--- /dev/null
+++ b/tests/bypass/test_unsolved_challenge.py
@@ -0,0 +1,214 @@
+"""A challenge page is a failed solve, whatever verdict the solver reports on itself.
+
+Regression for #1292. FlareSolverr answers "Challenge solved!" for anything it does not
+recognise as a Cloudflare challenge, and DDoS-Guard's manual CAPTCHA page is one such
+thing. The external bypasser logged a warning that the solve had not cleared the
+protection and then returned the page as a success anyway, which had three consequences:
+the retry-and-rotate loop that could still have reached a working mirror was never
+entered, the CAPTCHA page's own __ddg cookies were filed as that host's clearance, and
+the user was told to go and check a bypasser that was working perfectly.
+"""
+
+import pytest
+
+from shelfmark.bypass import ChallengeNotSolvedError
+
+# Verbatim from the annas-archive.pk page in #1292, trimmed to the markers. This is the
+# *manual* CAPTCHA - "could not verify your browser automatically" - not the ~900 byte
+# JS interstitial that a browser clears on its own.
+DDOS_GUARD_CAPTCHA = (
+ '
DDOS-GUARD'
+ ''
+ ''
+ '
Checking your browser before '
+ 'accessing annas-archive.pk
Sorry, we could not verify your '
+ "browser automatically. Complete the manual check to continue
"
+ '
'
+)
+
+
+class _FakeResponse:
+ def __init__(self, payload: dict) -> None:
+ self._payload = payload
+
+ def raise_for_status(self) -> None:
+ return None
+
+ def json(self) -> dict:
+ return self._payload
+
+
+def _stub_solution(monkeypatch, external_bypasser, solution: dict) -> None:
+ """Answer every bypass with `solution`, with config and SSL stubbed out."""
+
+ def fake_get(key, default=""):
+ values = {
+ "EXT_BYPASSER_URL": "https://bypass.example",
+ "EXT_BYPASSER_PATH": "/v1",
+ "EXT_BYPASSER_TIMEOUT": 60000,
+ }
+ return values.get(key, default)
+
+ monkeypatch.setattr(external_bypasser.config, "get", fake_get)
+ monkeypatch.setattr(
+ external_bypasser.requests,
+ "post",
+ # "Challenge solved!" is the solver's verdict; the page is the evidence.
+ lambda *_a, **_k: _FakeResponse(
+ {"status": "ok", "message": "Challenge solved!", "solution": solution}
+ ),
+ )
+ monkeypatch.setattr(external_bypasser, "get_ssl_verify", lambda _url: False)
+
+
+def test_a_captcha_page_is_reported_as_unsolved_not_returned(monkeypatch):
+ import shelfmark.bypass.external_bypasser as external_bypasser
+
+ _stub_solution(monkeypatch, external_bypasser, {"response": DDOS_GUARD_CAPTCHA})
+
+ with pytest.raises(ChallengeNotSolvedError) as excinfo:
+ external_bypasser._fetch_via_bypasser("https://annas-archive.pk/search?q=dune")
+
+ # The marker travels with the failure so the user-facing message can name it.
+ assert str(excinfo.value) == "/.well-known/ddos-guard/"
+
+
+def test_cookies_from_a_captcha_page_are_never_filed_as_clearance(monkeypatch):
+ """They belong to an unsolved check, so replaying them only re-arms the gate."""
+ import shelfmark.bypass.cookie_store as cookie_store
+ import shelfmark.bypass.external_bypasser as external_bypasser
+
+ monkeypatch.setattr(cookie_store, "_cf_cookies", {})
+ monkeypatch.setattr(cookie_store, "_cf_user_agents", {})
+ _stub_solution(
+ monkeypatch,
+ external_bypasser,
+ {
+ "response": DDOS_GUARD_CAPTCHA,
+ "userAgent": "Mozilla/5.0 (solver)",
+ "cookies": [{"name": "__ddg1_", "value": "from-a-captcha"}],
+ },
+ )
+
+ with pytest.raises(ChallengeNotSolvedError):
+ external_bypasser._fetch_via_bypasser("https://annas-archive.pk/search?q=dune")
+
+ assert cookie_store.get_cf_cookies_for_domain("annas-archive.pk") == {}
+ assert cookie_store.get_cf_user_agent_for_domain("annas-archive.pk") is None
+
+
+class _FakeSelector:
+ """Two mirrors, rotated on demand - each is its own DDoS-Guard host."""
+
+ def __init__(self) -> None:
+ self.current_base = "https://mirror-one.example"
+ self.rotate_calls = 0
+
+ def rewrite(self, url: str) -> str:
+ return url.replace("https://orig.example", self.current_base, 1)
+
+ def next_mirror_or_rotate_dns(self) -> tuple[str | None, str]:
+ self.rotate_calls += 1
+ self.current_base = "https://mirror-two.example"
+ return self.current_base, "mirror"
+
+
+def _no_sleeping(monkeypatch, external_bypasser) -> None:
+ monkeypatch.setattr(external_bypasser, "_sleep_with_cancellation", lambda _seconds, _flag: None)
+
+
+def test_an_unsolved_challenge_rotates_to_the_next_mirror(monkeypatch):
+ """The recovery the old code skipped by calling the CAPTCHA page a success."""
+ import shelfmark.bypass.external_bypasser as external_bypasser
+
+ _no_sleeping(monkeypatch, external_bypasser)
+ fetched: list[str] = []
+
+ def fake_fetch(url: str) -> str | None:
+ fetched.append(url)
+ if "mirror-one" in url:
+ raise ChallengeNotSolvedError("/.well-known/ddos-guard/")
+ return "real page"
+
+ monkeypatch.setattr(external_bypasser, "_fetch_via_bypasser", fake_fetch)
+
+ selector = _FakeSelector()
+ result = external_bypasser.get_bypassed_page("https://orig.example/search", selector=selector)
+
+ assert result == "real page"
+ assert fetched == [
+ "https://mirror-one.example/search",
+ "https://mirror-two.example/search",
+ ]
+ assert selector.rotate_calls == 1
+
+
+def test_every_attempt_challenged_blames_the_host_not_the_bypasser(monkeypatch):
+ import shelfmark.bypass.external_bypasser as external_bypasser
+
+ _no_sleeping(monkeypatch, external_bypasser)
+
+ def always_challenged(_url: str) -> str | None:
+ raise ChallengeNotSolvedError("/.well-known/ddos-guard/")
+
+ monkeypatch.setattr(external_bypasser, "_fetch_via_bypasser", always_challenged)
+
+ with pytest.raises(ChallengeNotSolvedError) as excinfo:
+ external_bypasser.get_bypassed_page("https://orig.example/search", selector=_FakeSelector())
+
+ message = str(excinfo.value)
+ assert "manual CAPTCHA" in message
+ assert "the bypasser itself is working" in message
+
+
+def test_an_unreachable_bypasser_still_reports_as_such(monkeypatch):
+ """The other cause must stay distinguishable: None, not an unsolved challenge."""
+ import shelfmark.bypass.external_bypasser as external_bypasser
+
+ _no_sleeping(monkeypatch, external_bypasser)
+ monkeypatch.setattr(external_bypasser, "_fetch_via_bypasser", lambda _url: None)
+
+ assert (
+ external_bypasser.get_bypassed_page("https://orig.example/search", selector=_FakeSelector())
+ is None
+ )
+
+
+def test_html_get_page_surfaces_the_host_as_the_cause(monkeypatch):
+ """The message the user actually reads must not send them to fix FlareSolverr.
+
+ `_run_bypasser`'s generic handler says "the protection bypasser failed", and the
+ search layer's give-up used to add "check that the bypasser is reachable and
+ working" - which is what #1292 spent its investigation doing.
+ """
+ import shelfmark.download.http as http
+ import shelfmark.download.network as network
+
+ monkeypatch.setattr(http, "_is_cf_bypass_enabled", lambda: True)
+ monkeypatch.setattr(http.network, "should_rotate_dns_for_url", lambda _url: True)
+
+ def challenged(*_args, **_kwargs):
+ msg = "the site kept answering with a protection challenge - manual CAPTCHA"
+ raise ChallengeNotSolvedError(msg)
+
+ monkeypatch.setattr(http, "get_bypassed_page", challenged)
+
+ statuses: list[tuple[str, str | None]] = []
+ selector = network.AAMirrorSelector()
+
+ html = http.html_get_page(
+ "https://annas-archive.pk/search?q=dune",
+ retry=1,
+ selector=selector,
+ status_callback=lambda stage, detail: statuses.append((stage, detail)),
+ use_bypasser=True,
+ success_delay=0,
+ )
+
+ assert html == ""
+ assert selector.last_failure is not None
+ assert "manual CAPTCHA" in selector.last_failure
+ assert "reachable" not in selector.last_failure
+ assert ("error", "the site kept answering with a protection challenge - manual CAPTCHA") in (
+ statuses
+ )
diff --git a/tests/direct_download/test_search_queries.py b/tests/direct_download/test_search_queries.py
index fe106e5f..5e1e3944 100644
--- a/tests/direct_download/test_search_queries.py
+++ b/tests/direct_download/test_search_queries.py
@@ -306,8 +306,8 @@ def test_search_books_filters_locally_when_path_language_enabled(monkeypatch):
captured_url: dict[str, str] = {}
- def _fake_html_get_page(url: str, selector, allow_bypasser_fallback=False):
- del selector, allow_bypasser_fallback
+ def _fake_html_get_page(url: str, selector, **_kwargs):
+ del selector
captured_url["url"] = url
return r"""
diff --git a/tests/direct_download/test_untabled_page_names_the_answering_mirror.py b/tests/direct_download/test_untabled_page_names_the_answering_mirror.py
new file mode 100644
index 00000000..34f2a76b
--- /dev/null
+++ b/tests/direct_download/test_untabled_page_names_the_answering_mirror.py
@@ -0,0 +1,114 @@
+"""The untabled-page diagnostic must name the mirror that actually answered.
+
+Regression for #1298. `html_get_page` rotates mirrors and follows redirects internally,
+so the URL the search layer passed in is only where the attempt started. Logging that
+one made the debug bundle report the untabled page against annas-archive.gl when the
+body had come from .pk - the exact triage cost the #1289 diagnostics were added to
+remove, reintroduced by reading the wrong variable.
+"""
+
+import logging
+
+import pytest
+
+# A protection challenge, so the fingerprint line fires without looking like AA.
+CHALLENGE_PAGE = (
+ "DDOS-GUARD"
+ ''
+ "Complete the manual check to continue"
+)
+
+
+@pytest.fixture
+def search_logs():
+ """Collect this module's log messages.
+
+ setup_logger builds loggers outside the standard hierarchy, so their records never
+ reach the root handler caplog installs - see tests/bypass/test_ddg_cookie_reuse.py.
+ """
+ import shelfmark.release_sources.direct_download as dd
+
+ messages: list[str] = []
+
+ class _Capture(logging.Handler):
+ def emit(self, record: logging.LogRecord) -> None:
+ messages.append(record.getMessage())
+
+ handler = _Capture()
+ dd.logger.addHandler(handler)
+ previous = dd.logger.level
+ dd.logger.setLevel(logging.DEBUG)
+ dd.logger._cache.clear()
+ try:
+ yield messages
+ finally:
+ dd.logger.removeHandler(handler)
+ dd.logger.setLevel(previous)
+
+
+class _Selector:
+ current_base = "https://annas-archive.gl"
+ last_failure = None
+
+ def rewrite(self, url: str) -> str:
+ return url
+
+ def next_mirror_or_rotate_dns(self, *, fatal: bool = False, reason: str = ""):
+ return None, "exhausted"
+
+
+REQUESTED = "https://annas-archive.gl/search?q=Ken+follett"
+ANSWERED = "https://annas-archive.pk/search?q=Ken+follett"
+
+
+def test_the_fingerprint_names_the_mirror_that_answered(monkeypatch, search_logs):
+ import shelfmark.release_sources.direct_download as dd
+
+ def fake_get(url, **kwargs):
+ # The caller must ask for it, or there is nothing to report.
+ assert kwargs["include_response_url"] is True
+ assert url == REQUESTED
+ # What an internal rotation looks like from the outside: a different host.
+ return CHALLENGE_PAGE, ANSWERED
+
+ monkeypatch.setattr(dd.downloader, "html_get_page", fake_get)
+ monkeypatch.setattr(dd.network, "get_available_aa_urls", lambda: ["a"])
+
+ with pytest.raises(dd.SearchUnavailableError):
+ dd._fetch_search_table_uncached(REQUESTED, _Selector())
+
+ fingerprint = [m for m in search_logs if m.startswith("Search page has no results table")]
+ assert len(fingerprint) == 1
+ assert ANSWERED in fingerprint[0]
+ assert "annas-archive.gl" not in fingerprint[0]
+
+
+def test_a_downloader_that_reports_no_url_falls_back_to_the_request(monkeypatch, search_logs):
+ """The plain-string shape stays supported; the line is still worth having."""
+ import shelfmark.release_sources.direct_download as dd
+
+ monkeypatch.setattr(dd.downloader, "html_get_page", lambda _url, **_k: CHALLENGE_PAGE)
+ monkeypatch.setattr(dd.network, "get_available_aa_urls", lambda: ["a"])
+
+ with pytest.raises(dd.SearchUnavailableError):
+ dd._fetch_search_table_uncached(REQUESTED, _Selector())
+
+ fingerprint = [m for m in search_logs if m.startswith("Search page has no results table")]
+ assert len(fingerprint) == 1
+ assert REQUESTED in fingerprint[0]
+
+
+def test_the_empty_body_give_up_survives_the_tuple_shape(monkeypatch):
+ """`("", url)` is truthy, so the exhaustion check has to read the body."""
+ import shelfmark.release_sources.direct_download as dd
+
+ monkeypatch.setattr(dd.downloader, "html_get_page", lambda _url, **_k: ("", REQUESTED))
+ monkeypatch.setattr(dd.network, "get_available_aa_urls", lambda: ["a"])
+
+ selector = _Selector()
+ selector.last_failure = "Every mirror refused the connection."
+
+ with pytest.raises(dd.SearchUnavailableError) as excinfo:
+ dd._fetch_search_table_uncached(REQUESTED, selector)
+
+ assert "Every mirror refused the connection." in str(excinfo.value)
diff --git a/tests/download/test_ddg_check_probe_handoff.py b/tests/download/test_ddg_check_probe_handoff.py
new file mode 100644
index 00000000..3bddf502
--- /dev/null
+++ b/tests/download/test_ddg_check_probe_handoff.py
@@ -0,0 +1,139 @@
+"""A solver must never be handed DDoS-Guard's ?check=1 probe URL.
+
+Regression for #1292. `html_get_page` follows Anna's Archive redirects by hand, and
+DDoS-Guard's gate answers /search with a 302 to the same path plus `check=1`. Because
+the follower walks that handshake by reassigning `current_url`, every downstream handoff
+- the 403 branch, the 503-challenge branch, the redirect-loop rescues - passed the
+*probe* URL to the bypasser rather than the page we wanted.
+
+A solver opens that in a fresh browser holding none of the cookies the probe exists to
+collect, so DDoS-Guard cannot verify it automatically and serves the manual CAPTCHA page
+that nothing can solve. The reporter's log is exactly that: a 403 handed off on a
+`&check=1` URL, FlareSolverr answering "Challenge solved!", and a 4.7 KB DDOS-GUARD
+CAPTCHA page coming back.
+"""
+
+import requests
+
+
+class _FakeResponse:
+ """Minimal stand-in for requests.Response covering what html_get_page touches."""
+
+ def __init__(
+ self,
+ status_code: int,
+ *,
+ url: str,
+ text: str = "",
+ headers: dict[str, str] | None = None,
+ cookies: dict[str, str] | None = None,
+ ) -> None:
+ self.status_code = status_code
+ self.url = url
+ self.text = text
+ self.cookies = cookies or {}
+ self.headers = {"Content-Type": "text/html;charset=utf-8", **(headers or {})}
+ self.is_redirect = 300 <= status_code < 400
+
+ def raise_for_status(self) -> None:
+ if self.status_code >= 400:
+ error = requests.exceptions.HTTPError(f"{self.status_code} Error")
+ error.response = self
+ raise error
+
+
+def _aa_http(monkeypatch):
+ """Import http with the network stubbed out and AA treated as an AA host."""
+ import shelfmark.download.http as http
+
+ monkeypatch.setattr(http, "_apply_cf_bypass", lambda _url, _headers: {})
+ monkeypatch.setattr(http, "get_proxies", lambda _url: {})
+ monkeypatch.setattr(http, "get_ssl_verify", lambda _url: True)
+ monkeypatch.setattr(http.time, "sleep", lambda _seconds: None)
+ monkeypatch.setattr(http, "_is_cf_bypass_enabled", lambda: True)
+ monkeypatch.setattr(http.network, "should_rotate_dns_for_url", lambda _url: True)
+ monkeypatch.setattr(http.network, "is_aa_auto_mode", lambda: True)
+ return http
+
+
+SEARCH_URL = "https://annas-archive.pk/search?index=&display=table&q=Ken+follett"
+PROBE_URL = f"{SEARCH_URL}&check=1"
+
+
+def test_403_on_the_check_probe_hands_over_the_pre_probe_url(monkeypatch):
+ """The reporter's exact sequence: 302 to ?check=1, then 403 on the probe."""
+ http = _aa_http(monkeypatch)
+
+ bypassed: list[str] = []
+
+ def fake_get(url: str, **_kwargs):
+ if "check=1" not in url:
+ return _FakeResponse(302, url=url, headers={"Location": PROBE_URL})
+ return _FakeResponse(403, url=url)
+
+ monkeypatch.setattr(http.requests, "get", fake_get)
+ monkeypatch.setattr(
+ http, "get_bypassed_page", lambda url, *_a, **_k: bypassed.append(url) or "ok"
+ )
+
+ html = http.html_get_page(SEARCH_URL, retry=1, success_delay=0)
+
+ assert html == "ok"
+ # The page we wanted, not the handshake hop we happened to be standing on.
+ assert bypassed == [SEARCH_URL]
+
+
+def test_redirect_loop_hands_over_the_pre_probe_url(monkeypatch):
+ """Stale clearance turns the gate into an endless ?check=1 bounce."""
+ http = _aa_http(monkeypatch)
+
+ bypassed: list[str] = []
+
+ monkeypatch.setattr(
+ http.requests,
+ "get",
+ lambda url, **_kwargs: _FakeResponse(302, url=url, headers={"Location": PROBE_URL}),
+ )
+ monkeypatch.setattr(
+ http, "get_bypassed_page", lambda url, *_a, **_k: bypassed.append(url) or "ok"
+ )
+
+ html = http.html_get_page(SEARCH_URL, retry=1, success_delay=0)
+
+ assert html == "ok"
+ assert bypassed == [SEARCH_URL]
+
+
+def test_only_the_check_parameter_is_dropped(monkeypatch):
+ """Everything else about the URL survives - it is still the search we asked for."""
+ http = _aa_http(monkeypatch)
+
+ url = (
+ "https://annas-archive.pk/search?index=&page=1&display=table&acc=aa_download"
+ "&acc=external_download&ext=epub&q=Ken+follett&check=1"
+ )
+
+ assert http._solvable_url(url) == (
+ "https://annas-archive.pk/search?index=&page=1&display=table&acc=aa_download"
+ "&acc=external_download&ext=epub&q=Ken+follett"
+ )
+
+
+def test_a_url_without_the_probe_is_returned_untouched(monkeypatch):
+ """No rewriting, no re-encoding: an unrelated URL must come back identical."""
+ http = _aa_http(monkeypatch)
+
+ url = "https://annas-archive.pk/md5/abc?q=a%20b&empty="
+
+ assert http._solvable_url(url) is url
+
+
+def test_non_aa_hosts_keep_their_check_parameter(monkeypatch):
+ """Elsewhere `check` is an ordinary query parameter and none of our business."""
+ import shelfmark.download.http as http
+
+ monkeypatch.setattr(http.network, "should_rotate_dns_for_url", lambda _url: False)
+
+ url = "https://example.com/api?check=1"
+
+ assert http._solvable_url(url) == url