diff --git a/shelfmark/bypass/__init__.py b/shelfmark/bypass/__init__.py index 2986e3a9..ed1ec9f3 100644 --- a/shelfmark/bypass/__init__.py +++ b/shelfmark/bypass/__init__.py @@ -3,3 +3,14 @@ class BypassCancelledError(Exception): """Raised when a bypass operation is cancelled.""" + + +class ChallengeNotSolvedError(Exception): + """Raised when a bypasser ran but the site still answered with a challenge. + + Distinct from a bypasser that is broken or unreachable, which is what every + "the bypass failed" message used to say. A solver can do its job perfectly and + still be handed something it cannot clear - DDoS-Guard's manual CAPTCHA page is + the case from #1292 - and telling the user to go check that FlareSolverr is + reachable sends them to fix a service that is working. + """ diff --git a/shelfmark/bypass/external_bypasser.py b/shelfmark/bypass/external_bypasser.py index 53dad163..f02f1ef5 100644 --- a/shelfmark/bypass/external_bypasser.py +++ b/shelfmark/bypass/external_bypasser.py @@ -6,7 +6,7 @@ from typing import TYPE_CHECKING, Any import requests -from shelfmark.bypass import BypassCancelledError +from shelfmark.bypass import BypassCancelledError, ChallengeNotSolvedError from shelfmark.bypass.challenge import challenge_marker from shelfmark.bypass.cookie_store import store_extracted_cookies from shelfmark.core.config import config @@ -92,7 +92,13 @@ def _store_solution_clearance(target_url: str, solution: Mapping[str, Any]) -> N def _fetch_via_bypasser(target_url: str) -> str | None: - """Make a single request to the external bypasser service. Returns HTML or None.""" + """Make a single request to the external bypasser service. Returns HTML or None. + + Raises: + ChallengeNotSolvedError: the service answered with a page that is still a + challenge, whatever verdict it reported on itself. + + """ raw_bypasser_url = _coerce_config_str( config.get("EXT_BYPASSER_URL", "http://flaresolverr:8191"), "http://flaresolverr:8191", @@ -155,6 +161,12 @@ def _fetch_via_bypasser(target_url: str) -> str | None: marker, ) if marker: + # The solver's verdict is not evidence; the page is. Returning this one as a + # success is what made #1292 unrecoverable: the retry-and-rotate loop that + # could still have saved the search - the next mirror is a different + # DDoS-Guard host, in its own state - was never entered, and the challenge + # page's own __ddg cookies were filed as this host's clearance and replayed + # on every later request. logger.warning( "External bypasser reported success but returned a challenge page for " "'%s' (%d bytes, marker=%r) - the solve did not clear the protection", @@ -162,6 +174,7 @@ def _fetch_via_bypasser(target_url: str) -> str | None: len(html), marker, ) + raise ChallengeNotSolvedError(marker) try: _store_solution_clearance(target_url, solution) @@ -212,16 +225,33 @@ def get_bypassed_page( selector: network.AAMirrorSelector | None = None, cancel_flag: Event | None = None, ) -> str | None: - """Fetch HTML via external bypasser with retries and mirror rotation.""" + """Fetch HTML via external bypasser with retries and mirror rotation. + + Raises: + ChallengeNotSolvedError: every attempt came back still carrying a challenge. + Reported apart from returning None because the two ask the user for + opposite things: None means go and check the bypasser, this means the + bypasser is fine and the host is the one refusing. + BypassCancelledError: the caller's cancel flag was set. + + """ from shelfmark.download import network as network_module sel = selector or network_module.AAMirrorSelector() + unsolved_marker: str | None = None for attempt in range(1, MAX_RETRY + 1): _check_cancelled(cancel_flag, "by user") attempt_url = sel.rewrite(url) - result = _fetch_via_bypasser(attempt_url) + try: + result = _fetch_via_bypasser(attempt_url) + except ChallengeNotSolvedError as e: + # Worth the remaining attempts rather than an immediate give-up: the retry + # rotates onto the next mirror, and that is a different DDoS-Guard host with + # its own idea of whether this caller needs a CAPTCHA. + unsolved_marker = str(e) or unsolved_marker + result = None if result: return result @@ -242,4 +272,11 @@ def get_bypassed_page( if action in ("mirror", "dns") and new_base: logger.info("Rotated %s for retry", action) + if unsolved_marker: + msg = ( + "The bypasser ran, but the site kept answering with a protection challenge " + f"(marker={unsolved_marker!r}). That is usually a manual CAPTCHA, which no " + "bypasser can answer - the bypasser itself is working. Try again shortly." + ) + raise ChallengeNotSolvedError(msg) return None diff --git a/shelfmark/download/http.py b/shelfmark/download/http.py index 3fe5f18f..f58c0b78 100644 --- a/shelfmark/download/http.py +++ b/shelfmark/download/http.py @@ -5,12 +5,12 @@ import time from http import HTTPStatus from io import BytesIO from typing import TYPE_CHECKING, NoReturn -from urllib.parse import urljoin, urlparse +from urllib.parse import parse_qsl, urlencode, urljoin, urlparse, urlunparse import requests from tqdm import tqdm -from shelfmark.bypass import BypassCancelledError, cookie_store +from shelfmark.bypass import BypassCancelledError, ChallengeNotSolvedError, cookie_store from shelfmark.bypass.challenge import challenge_marker from shelfmark.core import search_deadline from shelfmark.core.config import config as app_config @@ -29,6 +29,10 @@ logger = setup_logger(__name__) _RNG = random.SystemRandom() _MAX_REDIRECTS = 5 +# DDoS-Guard's re-check probe. Its 302 to `?check=1` is one hop of a handshake rather +# than a page: the parameter asserts the caller already holds the cookies that hop +# issued. +_DDG_CHECK_PARAM = "check" # Z-Library answers the first hit with a 503 whose only real payload is a Set-Cookie; echoing # that cookie back returns the 302 to the real page. Two attempts cover the handshake without # letting a server that keeps re-issuing cookies hold us in the loop. @@ -48,6 +52,7 @@ _BYPASS_GRACE_SLACK_SECONDS = 30.0 _BYPASSER_ERRORS = ( AttributeError, BypassCancelledError, + ChallengeNotSolvedError, KeyError, OSError, RuntimeError, @@ -252,6 +257,33 @@ def _response_challenge_marker(response: requests.Response) -> str | None: return None +def _solvable_url(url: str) -> str: + """The URL a solver should open, given one we may be mid-handshake on. + + The manual AA redirect follower in `html_get_page` walks DDoS-Guard's handshake by + reassigning `current_url`, so by the time a 403, a 503 challenge or a redirect loop + hands that URL to a bypasser it is often the `?check=1` probe rather than the page + we actually wanted. A solver opens it in a fresh browser holding none of the cookies + the probe exists to collect, so DDoS-Guard cannot verify it automatically and answers + with the manual CAPTCHA page that nothing can solve - the failure in #1292, where + FlareSolverr reported "Challenge solved!" over a 4.7 KB DDOS-GUARD interstitial. + + Handing over the pre-probe URL instead lets the solver's browser run the whole + handshake itself, which is what a real browser does and what the solver is for. + + Scoped to the hosts whose redirects we follow manually: everywhere else `check` is + an ordinary query parameter and none of our business. + """ + if not network.should_rotate_dns_for_url(url): + return url + parsed = urlparse(url) + params = parse_qsl(parsed.query, keep_blank_values=True) + kept = [(key, value) for key, value in params if key != _DDG_CHECK_PARAM] + if len(kept) == len(params): + return url + return urlunparse(parsed._replace(query=urlencode(kept))) + + def _fatal_mirror_reason(e: Exception) -> str | None: """Return why ``e`` proves the mirror is unusable, or None if it may recover. @@ -371,6 +403,9 @@ def html_get_page( retry-loop branch above with `continue`, and with MAX_RETRY=1 there is no later attempt for that branch to run on either. """ + # Every handoff reaches the solver through here, so this is the one place the + # mid-handshake `?check=1` URL has to be unwound. See _solvable_url. + bypass_url = _solvable_url(bypass_url) # Never start a minutes-long browser solve on a budget that has already run out: # nothing downstream would get to report the real reason before the caller's # deadline (or its reverse proxy) cut the request off. @@ -405,6 +440,18 @@ def html_get_page( except _STATUS_CALLBACK_ERRORS: logger.debug("Rate-limit status callback failed", exc_info=True) return _fail(str(e), bypass_url) + except ChallengeNotSolvedError as e: + # Not a bypasser malfunction: it ran, and the host answered with something it + # cannot clear - DDoS-Guard's manual CAPTCHA, typically. Must precede the + # generic handler below, whose "the protection bypasser failed" is what sent + # #1292 off to fix a FlareSolverr that was working perfectly. + logger.info("Bypass ran but did not clear the protection: %s", e) + if status_callback: + try: + status_callback("error", str(e)) + except _STATUS_CALLBACK_ERRORS: + logger.debug("Unsolved-challenge status callback failed", exc_info=True) + return _fail(str(e), bypass_url) except _BYPASSER_ERRORS as e: logger.warning("Bypasser error: %s: %s", type(e).__name__, e) # Surface the real reason. Without this the caller only sees an empty diff --git a/shelfmark/release_sources/direct_download.py b/shelfmark/release_sources/direct_download.py index 0a8687ae..abbb5d73 100644 --- a/shelfmark/release_sources/direct_download.py +++ b/shelfmark/release_sources/direct_download.py @@ -115,6 +115,16 @@ def _html_response_text(response: str | tuple[str, str]) -> str: return response +def _html_response_url(response: str | tuple[str, str]) -> str | None: + """The URL that actually answered, when the downloader was asked to report it. + + None for the plain-string shape, so a caller can fall back to what it requested. + """ + if isinstance(response, tuple): + return response[1] or None + return None + + def _attr_to_str(value: object) -> str | None: """Convert a BeautifulSoup attribute value to a plain string.""" if isinstance(value, str): @@ -696,10 +706,21 @@ def _fetch_search_table_uncached( if search_deadline.expired(): raise SearchUnavailableError(search_deadline.deadline_message()) + # include_response_url is what makes the diagnostics below name the mirror that + # actually answered. html_get_page rotates mirrors and follows redirects on its + # own, so `attempt_url` is only where this iteration started: #1298's bundle + # reported the untabled page against annas-archive.gl when the body had come + # from .pk, which is precisely the triage cost #1289 added the line to remove. response = downloader.html_get_page( - attempt_url, selector=selector, allow_bypasser_fallback=True + attempt_url, + selector=selector, + allow_bypasser_fallback=True, + include_response_url=True, ) - if not response: + html = _html_response_text(response) + # Checked on the body, not on `response`: with include_response_url the give-up + # shape is the tuple ("", url), and a tuple is truthy. + if not html: # Network/mirror exhaustion path bubbles up so API can notify clients. # html_get_page records the concrete give-up reason on the selector; fall # back to the generic line only if nothing was recorded. @@ -708,7 +729,7 @@ def _fetch_search_table_uncached( ) raise SearchUnavailableError(f"Unable to reach download source. {detail}") - html = _html_response_text(response) + answered_url = _html_response_url(response) or attempt_url soup = BeautifulSoup(html, "html.parser") table = soup.find("table") if isinstance(table, Tag): @@ -724,7 +745,7 @@ def _fetch_search_table_uncached( # alone, and the response body is not in the debug bundle. Fingerprint it here # so the next report says which branch fired and why, rather than costing # another round of guesswork - see #1289. - _log_untabled_search_page(attempt_url, html) + _log_untabled_search_page(answered_url, html) if _looks_like_aa_page(html): # A real AA response in a shape the caller should report as drift. Checked @@ -737,9 +758,18 @@ def _fetch_search_table_uncached( # what came back. Rotating is pointless (every mirror shares the same # protection) and reporting it as an empty result is worse: the user is # told their query found nothing when the search never ran. + # + # The wording no longer blames the bypasser outright. In #1292 it was + # reachable and working, and the page it was handed was DDoS-Guard's manual + # CAPTCHA - so "check that the bypasser is working" was the one piece of + # advice guaranteed to waste the reporter's time. Name the marker instead + # and let the two causes be told apart. msg = ( - "Anna's Archive answered with an unsolved protection challenge. " - "Check that the bypasser is reachable and working." + "Anna's Archive answered with a protection challenge that was not " + f"cleared (marker={challenge_marker(html)!r}). If the bypasser reports " + "solving it, the host is serving a manual CAPTCHA that no bypasser can " + "answer - try again shortly. Otherwise check that the bypasser is " + "reachable and working." ) raise SearchUnavailableError(msg) diff --git a/tests/bypass/test_unsolved_challenge.py b/tests/bypass/test_unsolved_challenge.py new file mode 100644 index 00000000..5451f93b --- /dev/null +++ b/tests/bypass/test_unsolved_challenge.py @@ -0,0 +1,214 @@ +"""A challenge page is a failed solve, whatever verdict the solver reports on itself. + +Regression for #1292. FlareSolverr answers "Challenge solved!" for anything it does not +recognise as a Cloudflare challenge, and DDoS-Guard's manual CAPTCHA page is one such +thing. The external bypasser logged a warning that the solve had not cleared the +protection and then returned the page as a success anyway, which had three consequences: +the retry-and-rotate loop that could still have reached a working mirror was never +entered, the CAPTCHA page's own __ddg cookies were filed as that host's clearance, and +the user was told to go and check a bypasser that was working perfectly. +""" + +import pytest + +from shelfmark.bypass import ChallengeNotSolvedError + +# Verbatim from the annas-archive.pk page in #1292, trimmed to the markers. This is the +# *manual* CAPTCHA - "could not verify your browser automatically" - not the ~900 byte +# JS interstitial that a browser clears on its own. +DDOS_GUARD_CAPTCHA = ( + 'DDOS-GUARD' + '' + '' + '

Checking your browser before ' + 'accessing annas-archive.pk

Sorry, we could not verify your ' + "browser automatically. Complete the manual check to continue

" + '
' +) + + +class _FakeResponse: + def __init__(self, payload: dict) -> None: + self._payload = payload + + def raise_for_status(self) -> None: + return None + + def json(self) -> dict: + return self._payload + + +def _stub_solution(monkeypatch, external_bypasser, solution: dict) -> None: + """Answer every bypass with `solution`, with config and SSL stubbed out.""" + + def fake_get(key, default=""): + values = { + "EXT_BYPASSER_URL": "https://bypass.example", + "EXT_BYPASSER_PATH": "/v1", + "EXT_BYPASSER_TIMEOUT": 60000, + } + return values.get(key, default) + + monkeypatch.setattr(external_bypasser.config, "get", fake_get) + monkeypatch.setattr( + external_bypasser.requests, + "post", + # "Challenge solved!" is the solver's verdict; the page is the evidence. + lambda *_a, **_k: _FakeResponse( + {"status": "ok", "message": "Challenge solved!", "solution": solution} + ), + ) + monkeypatch.setattr(external_bypasser, "get_ssl_verify", lambda _url: False) + + +def test_a_captcha_page_is_reported_as_unsolved_not_returned(monkeypatch): + import shelfmark.bypass.external_bypasser as external_bypasser + + _stub_solution(monkeypatch, external_bypasser, {"response": DDOS_GUARD_CAPTCHA}) + + with pytest.raises(ChallengeNotSolvedError) as excinfo: + external_bypasser._fetch_via_bypasser("https://annas-archive.pk/search?q=dune") + + # The marker travels with the failure so the user-facing message can name it. + assert str(excinfo.value) == "/.well-known/ddos-guard/" + + +def test_cookies_from_a_captcha_page_are_never_filed_as_clearance(monkeypatch): + """They belong to an unsolved check, so replaying them only re-arms the gate.""" + import shelfmark.bypass.cookie_store as cookie_store + import shelfmark.bypass.external_bypasser as external_bypasser + + monkeypatch.setattr(cookie_store, "_cf_cookies", {}) + monkeypatch.setattr(cookie_store, "_cf_user_agents", {}) + _stub_solution( + monkeypatch, + external_bypasser, + { + "response": DDOS_GUARD_CAPTCHA, + "userAgent": "Mozilla/5.0 (solver)", + "cookies": [{"name": "__ddg1_", "value": "from-a-captcha"}], + }, + ) + + with pytest.raises(ChallengeNotSolvedError): + external_bypasser._fetch_via_bypasser("https://annas-archive.pk/search?q=dune") + + assert cookie_store.get_cf_cookies_for_domain("annas-archive.pk") == {} + assert cookie_store.get_cf_user_agent_for_domain("annas-archive.pk") is None + + +class _FakeSelector: + """Two mirrors, rotated on demand - each is its own DDoS-Guard host.""" + + def __init__(self) -> None: + self.current_base = "https://mirror-one.example" + self.rotate_calls = 0 + + def rewrite(self, url: str) -> str: + return url.replace("https://orig.example", self.current_base, 1) + + def next_mirror_or_rotate_dns(self) -> tuple[str | None, str]: + self.rotate_calls += 1 + self.current_base = "https://mirror-two.example" + return self.current_base, "mirror" + + +def _no_sleeping(monkeypatch, external_bypasser) -> None: + monkeypatch.setattr(external_bypasser, "_sleep_with_cancellation", lambda _seconds, _flag: None) + + +def test_an_unsolved_challenge_rotates_to_the_next_mirror(monkeypatch): + """The recovery the old code skipped by calling the CAPTCHA page a success.""" + import shelfmark.bypass.external_bypasser as external_bypasser + + _no_sleeping(monkeypatch, external_bypasser) + fetched: list[str] = [] + + def fake_fetch(url: str) -> str | None: + fetched.append(url) + if "mirror-one" in url: + raise ChallengeNotSolvedError("/.well-known/ddos-guard/") + return "real page" + + monkeypatch.setattr(external_bypasser, "_fetch_via_bypasser", fake_fetch) + + selector = _FakeSelector() + result = external_bypasser.get_bypassed_page("https://orig.example/search", selector=selector) + + assert result == "real page" + assert fetched == [ + "https://mirror-one.example/search", + "https://mirror-two.example/search", + ] + assert selector.rotate_calls == 1 + + +def test_every_attempt_challenged_blames_the_host_not_the_bypasser(monkeypatch): + import shelfmark.bypass.external_bypasser as external_bypasser + + _no_sleeping(monkeypatch, external_bypasser) + + def always_challenged(_url: str) -> str | None: + raise ChallengeNotSolvedError("/.well-known/ddos-guard/") + + monkeypatch.setattr(external_bypasser, "_fetch_via_bypasser", always_challenged) + + with pytest.raises(ChallengeNotSolvedError) as excinfo: + external_bypasser.get_bypassed_page("https://orig.example/search", selector=_FakeSelector()) + + message = str(excinfo.value) + assert "manual CAPTCHA" in message + assert "the bypasser itself is working" in message + + +def test_an_unreachable_bypasser_still_reports_as_such(monkeypatch): + """The other cause must stay distinguishable: None, not an unsolved challenge.""" + import shelfmark.bypass.external_bypasser as external_bypasser + + _no_sleeping(monkeypatch, external_bypasser) + monkeypatch.setattr(external_bypasser, "_fetch_via_bypasser", lambda _url: None) + + assert ( + external_bypasser.get_bypassed_page("https://orig.example/search", selector=_FakeSelector()) + is None + ) + + +def test_html_get_page_surfaces_the_host_as_the_cause(monkeypatch): + """The message the user actually reads must not send them to fix FlareSolverr. + + `_run_bypasser`'s generic handler says "the protection bypasser failed", and the + search layer's give-up used to add "check that the bypasser is reachable and + working" - which is what #1292 spent its investigation doing. + """ + import shelfmark.download.http as http + import shelfmark.download.network as network + + monkeypatch.setattr(http, "_is_cf_bypass_enabled", lambda: True) + monkeypatch.setattr(http.network, "should_rotate_dns_for_url", lambda _url: True) + + def challenged(*_args, **_kwargs): + msg = "the site kept answering with a protection challenge - manual CAPTCHA" + raise ChallengeNotSolvedError(msg) + + monkeypatch.setattr(http, "get_bypassed_page", challenged) + + statuses: list[tuple[str, str | None]] = [] + selector = network.AAMirrorSelector() + + html = http.html_get_page( + "https://annas-archive.pk/search?q=dune", + retry=1, + selector=selector, + status_callback=lambda stage, detail: statuses.append((stage, detail)), + use_bypasser=True, + success_delay=0, + ) + + assert html == "" + assert selector.last_failure is not None + assert "manual CAPTCHA" in selector.last_failure + assert "reachable" not in selector.last_failure + assert ("error", "the site kept answering with a protection challenge - manual CAPTCHA") in ( + statuses + ) diff --git a/tests/direct_download/test_search_queries.py b/tests/direct_download/test_search_queries.py index fe106e5f..5e1e3944 100644 --- a/tests/direct_download/test_search_queries.py +++ b/tests/direct_download/test_search_queries.py @@ -306,8 +306,8 @@ def test_search_books_filters_locally_when_path_language_enabled(monkeypatch): captured_url: dict[str, str] = {} - def _fake_html_get_page(url: str, selector, allow_bypasser_fallback=False): - del selector, allow_bypasser_fallback + def _fake_html_get_page(url: str, selector, **_kwargs): + del selector captured_url["url"] = url return r""" diff --git a/tests/direct_download/test_untabled_page_names_the_answering_mirror.py b/tests/direct_download/test_untabled_page_names_the_answering_mirror.py new file mode 100644 index 00000000..34f2a76b --- /dev/null +++ b/tests/direct_download/test_untabled_page_names_the_answering_mirror.py @@ -0,0 +1,114 @@ +"""The untabled-page diagnostic must name the mirror that actually answered. + +Regression for #1298. `html_get_page` rotates mirrors and follows redirects internally, +so the URL the search layer passed in is only where the attempt started. Logging that +one made the debug bundle report the untabled page against annas-archive.gl when the +body had come from .pk - the exact triage cost the #1289 diagnostics were added to +remove, reintroduced by reading the wrong variable. +""" + +import logging + +import pytest + +# A protection challenge, so the fingerprint line fires without looking like AA. +CHALLENGE_PAGE = ( + "DDOS-GUARD" + '' + "Complete the manual check to continue" +) + + +@pytest.fixture +def search_logs(): + """Collect this module's log messages. + + setup_logger builds loggers outside the standard hierarchy, so their records never + reach the root handler caplog installs - see tests/bypass/test_ddg_cookie_reuse.py. + """ + import shelfmark.release_sources.direct_download as dd + + messages: list[str] = [] + + class _Capture(logging.Handler): + def emit(self, record: logging.LogRecord) -> None: + messages.append(record.getMessage()) + + handler = _Capture() + dd.logger.addHandler(handler) + previous = dd.logger.level + dd.logger.setLevel(logging.DEBUG) + dd.logger._cache.clear() + try: + yield messages + finally: + dd.logger.removeHandler(handler) + dd.logger.setLevel(previous) + + +class _Selector: + current_base = "https://annas-archive.gl" + last_failure = None + + def rewrite(self, url: str) -> str: + return url + + def next_mirror_or_rotate_dns(self, *, fatal: bool = False, reason: str = ""): + return None, "exhausted" + + +REQUESTED = "https://annas-archive.gl/search?q=Ken+follett" +ANSWERED = "https://annas-archive.pk/search?q=Ken+follett" + + +def test_the_fingerprint_names_the_mirror_that_answered(monkeypatch, search_logs): + import shelfmark.release_sources.direct_download as dd + + def fake_get(url, **kwargs): + # The caller must ask for it, or there is nothing to report. + assert kwargs["include_response_url"] is True + assert url == REQUESTED + # What an internal rotation looks like from the outside: a different host. + return CHALLENGE_PAGE, ANSWERED + + monkeypatch.setattr(dd.downloader, "html_get_page", fake_get) + monkeypatch.setattr(dd.network, "get_available_aa_urls", lambda: ["a"]) + + with pytest.raises(dd.SearchUnavailableError): + dd._fetch_search_table_uncached(REQUESTED, _Selector()) + + fingerprint = [m for m in search_logs if m.startswith("Search page has no results table")] + assert len(fingerprint) == 1 + assert ANSWERED in fingerprint[0] + assert "annas-archive.gl" not in fingerprint[0] + + +def test_a_downloader_that_reports_no_url_falls_back_to_the_request(monkeypatch, search_logs): + """The plain-string shape stays supported; the line is still worth having.""" + import shelfmark.release_sources.direct_download as dd + + monkeypatch.setattr(dd.downloader, "html_get_page", lambda _url, **_k: CHALLENGE_PAGE) + monkeypatch.setattr(dd.network, "get_available_aa_urls", lambda: ["a"]) + + with pytest.raises(dd.SearchUnavailableError): + dd._fetch_search_table_uncached(REQUESTED, _Selector()) + + fingerprint = [m for m in search_logs if m.startswith("Search page has no results table")] + assert len(fingerprint) == 1 + assert REQUESTED in fingerprint[0] + + +def test_the_empty_body_give_up_survives_the_tuple_shape(monkeypatch): + """`("", url)` is truthy, so the exhaustion check has to read the body.""" + import shelfmark.release_sources.direct_download as dd + + monkeypatch.setattr(dd.downloader, "html_get_page", lambda _url, **_k: ("", REQUESTED)) + monkeypatch.setattr(dd.network, "get_available_aa_urls", lambda: ["a"]) + + selector = _Selector() + selector.last_failure = "Every mirror refused the connection." + + with pytest.raises(dd.SearchUnavailableError) as excinfo: + dd._fetch_search_table_uncached(REQUESTED, selector) + + assert "Every mirror refused the connection." in str(excinfo.value) diff --git a/tests/download/test_ddg_check_probe_handoff.py b/tests/download/test_ddg_check_probe_handoff.py new file mode 100644 index 00000000..3bddf502 --- /dev/null +++ b/tests/download/test_ddg_check_probe_handoff.py @@ -0,0 +1,139 @@ +"""A solver must never be handed DDoS-Guard's ?check=1 probe URL. + +Regression for #1292. `html_get_page` follows Anna's Archive redirects by hand, and +DDoS-Guard's gate answers /search with a 302 to the same path plus `check=1`. Because +the follower walks that handshake by reassigning `current_url`, every downstream handoff +- the 403 branch, the 503-challenge branch, the redirect-loop rescues - passed the +*probe* URL to the bypasser rather than the page we wanted. + +A solver opens that in a fresh browser holding none of the cookies the probe exists to +collect, so DDoS-Guard cannot verify it automatically and serves the manual CAPTCHA page +that nothing can solve. The reporter's log is exactly that: a 403 handed off on a +`&check=1` URL, FlareSolverr answering "Challenge solved!", and a 4.7 KB DDOS-GUARD +CAPTCHA page coming back. +""" + +import requests + + +class _FakeResponse: + """Minimal stand-in for requests.Response covering what html_get_page touches.""" + + def __init__( + self, + status_code: int, + *, + url: str, + text: str = "", + headers: dict[str, str] | None = None, + cookies: dict[str, str] | None = None, + ) -> None: + self.status_code = status_code + self.url = url + self.text = text + self.cookies = cookies or {} + self.headers = {"Content-Type": "text/html;charset=utf-8", **(headers or {})} + self.is_redirect = 300 <= status_code < 400 + + def raise_for_status(self) -> None: + if self.status_code >= 400: + error = requests.exceptions.HTTPError(f"{self.status_code} Error") + error.response = self + raise error + + +def _aa_http(monkeypatch): + """Import http with the network stubbed out and AA treated as an AA host.""" + import shelfmark.download.http as http + + monkeypatch.setattr(http, "_apply_cf_bypass", lambda _url, _headers: {}) + monkeypatch.setattr(http, "get_proxies", lambda _url: {}) + monkeypatch.setattr(http, "get_ssl_verify", lambda _url: True) + monkeypatch.setattr(http.time, "sleep", lambda _seconds: None) + monkeypatch.setattr(http, "_is_cf_bypass_enabled", lambda: True) + monkeypatch.setattr(http.network, "should_rotate_dns_for_url", lambda _url: True) + monkeypatch.setattr(http.network, "is_aa_auto_mode", lambda: True) + return http + + +SEARCH_URL = "https://annas-archive.pk/search?index=&display=table&q=Ken+follett" +PROBE_URL = f"{SEARCH_URL}&check=1" + + +def test_403_on_the_check_probe_hands_over_the_pre_probe_url(monkeypatch): + """The reporter's exact sequence: 302 to ?check=1, then 403 on the probe.""" + http = _aa_http(monkeypatch) + + bypassed: list[str] = [] + + def fake_get(url: str, **_kwargs): + if "check=1" not in url: + return _FakeResponse(302, url=url, headers={"Location": PROBE_URL}) + return _FakeResponse(403, url=url) + + monkeypatch.setattr(http.requests, "get", fake_get) + monkeypatch.setattr( + http, "get_bypassed_page", lambda url, *_a, **_k: bypassed.append(url) or "ok" + ) + + html = http.html_get_page(SEARCH_URL, retry=1, success_delay=0) + + assert html == "ok" + # The page we wanted, not the handshake hop we happened to be standing on. + assert bypassed == [SEARCH_URL] + + +def test_redirect_loop_hands_over_the_pre_probe_url(monkeypatch): + """Stale clearance turns the gate into an endless ?check=1 bounce.""" + http = _aa_http(monkeypatch) + + bypassed: list[str] = [] + + monkeypatch.setattr( + http.requests, + "get", + lambda url, **_kwargs: _FakeResponse(302, url=url, headers={"Location": PROBE_URL}), + ) + monkeypatch.setattr( + http, "get_bypassed_page", lambda url, *_a, **_k: bypassed.append(url) or "ok" + ) + + html = http.html_get_page(SEARCH_URL, retry=1, success_delay=0) + + assert html == "ok" + assert bypassed == [SEARCH_URL] + + +def test_only_the_check_parameter_is_dropped(monkeypatch): + """Everything else about the URL survives - it is still the search we asked for.""" + http = _aa_http(monkeypatch) + + url = ( + "https://annas-archive.pk/search?index=&page=1&display=table&acc=aa_download" + "&acc=external_download&ext=epub&q=Ken+follett&check=1" + ) + + assert http._solvable_url(url) == ( + "https://annas-archive.pk/search?index=&page=1&display=table&acc=aa_download" + "&acc=external_download&ext=epub&q=Ken+follett" + ) + + +def test_a_url_without_the_probe_is_returned_untouched(monkeypatch): + """No rewriting, no re-encoding: an unrelated URL must come back identical.""" + http = _aa_http(monkeypatch) + + url = "https://annas-archive.pk/md5/abc?q=a%20b&empty=" + + assert http._solvable_url(url) is url + + +def test_non_aa_hosts_keep_their_check_parameter(monkeypatch): + """Elsewhere `check` is an ordinary query parameter and none of our business.""" + import shelfmark.download.http as http + + monkeypatch.setattr(http.network, "should_rotate_dns_for_url", lambda _url: False) + + url = "https://example.com/api?check=1" + + assert http._solvable_url(url) == url