diff --git a/shelfmark/download/http.py b/shelfmark/download/http.py index 4b669511..241200bd 100644 --- a/shelfmark/download/http.py +++ b/shelfmark/download/http.py @@ -341,8 +341,11 @@ def html_get_page( is internal-bypasser only; with an external one get_cf_cookies_for_domain() already returns {}. """ - if not _is_using_external_bypasser(): - _get_internal_bypasser().clear_cf_cookies(urlparse(bypass_url).hostname or "") + hostname = urlparse(bypass_url).hostname or "" + # An empty domain means "clear every host" to the bypasser, so skip the purge + # rather than wipe clearance for sites that are working fine. + if hostname and not _is_using_external_bypasser(): + _get_internal_bypasser().clear_cf_cookies(hostname) return _run_bypasser(bypass_url) configured_retry = normalize_positive_int(app_config.MAX_RETRY) @@ -456,6 +459,12 @@ def html_get_page( return _result("", current_url) # Same-host redirect (relative or absolute) - follow manually. + # DDoS-Guard gates AA /search behind a cookie probe: the 302 to + # ?check=1 carries Set-Cookie (__ddg*) which must be echoed back on + # the next hop, or the server just re-issues the redirect forever. + issued = _new_cookies(response, handshake_cookies) + if issued: + handshake_cookies.update(issued) redirects_followed += 1 if redirects_followed > _MAX_REDIRECTS: # A same-host redirect loop on AA is not a network fault — it is @@ -490,12 +499,18 @@ def html_get_page( except Exception as e: status = _get_status_code(e) - # The same DDoS-Guard rescue for loops the manual AA follower above does not - # see: non-AA hosts keep allow_redirects=True, so `requests` follows the loop - # itself and raises, and an AA redirect missing its Location header raises - # too. TooManyRedirects carries no status, so the 403 rescue below never fires - # and every retry would re-send the dead cookies. - if isinstance(e, requests.exceptions.TooManyRedirects) and _bypass_handoff_allowed(): + # The same DDoS-Guard rescue, for the loops the manual AA follower above hands + # back rather than resolving inline — an AA redirect missing its Location + # header. TooManyRedirects carries no status, so the 403 rescue below never + # fires and every retry would re-send the dead cookies. Scoped to the hosts + # whose redirects we follow manually: elsewhere `requests` follows them itself, + # and a loop there is an ordinary misconfiguration that a cookie purge and a + # minutes-long browser solve would be the wrong answer to. + if ( + isinstance(e, requests.exceptions.TooManyRedirects) + and network.should_rotate_dns_for_url(current_url) + and _bypass_handoff_allowed() + ): logger.info("Redirect loop detected; switching to bypasser: %s", current_url) return _redirect_loop_handoff(current_url) diff --git a/tests/download/test_http_aa_redirects.py b/tests/download/test_http_aa_redirects.py index 0de05b5c..87ccaa2d 100644 --- a/tests/download/test_http_aa_redirects.py +++ b/tests/download/test_http_aa_redirects.py @@ -149,3 +149,50 @@ def test_html_get_page_locked_aa_does_not_fail_over_on_cross_host_redirect(monke assert html == "" assert calls == ["https://annas-archive.li/search?q=test"] + + +def test_html_get_page_echoes_cookies_across_same_host_redirects(monkeypatch): + """DDoS-Guard's ?check=1 probe is cleared by echoing the Set-Cookie it issues. + + Without this the __ddg* cookie is dropped on every hop, the server re-issues the same + redirect, and the request dies with TooManyRedirects. + """ + import shelfmark.download.http as http + + monkeypatch.setattr(http, "_is_cf_bypass_enabled", lambda: False) + monkeypatch.setattr(http, "get_proxies", lambda _url: {}) + monkeypatch.setattr(http.time, "sleep", lambda _s: None) + monkeypatch.setattr(http.network, "get_aa_base_url", lambda: "https://annas-archive.li") + monkeypatch.setattr(http.network, "is_aa_auto_mode", lambda: True) + + sent_cookies: list[dict[str, str]] = [] + + def fake_get(url: str, **kwargs): + sent_cookies.append(dict(kwargs["cookies"])) + if url == "https://annas-archive.li/search?q=test": + response = _FakeResponse(302, headers={"Location": "/search?q=test&check=1"}, url=url) + response.cookies = {"__ddg2_": "probe"} + return response + if url == "https://annas-archive.li/search?q=test&check=1": + # The probe only clears if the cookie comes back on this hop. + if kwargs["cookies"].get("__ddg2_") != "probe": + response = _FakeResponse( + 302, headers={"Location": "/search?q=test&check=1"}, url=url + ) + response.cookies = {"__ddg2_": "probe"} + return response + return _FakeResponse(200, text="RESULTS", url=url) + raise AssertionError(f"Unexpected URL: {url}") + + monkeypatch.setattr(http.requests, "get", fake_get) + + selector = _DummySelector(["https://annas-archive.li"]) + html = http.html_get_page( + "https://annas-archive.li/search?q=test", + selector=selector, + retry=1, + allow_bypasser_fallback=False, + ) + + assert html == "RESULTS" + assert sent_cookies == [{}, {"__ddg2_": "probe"}] diff --git a/tests/download/test_http_bypasser_fallbacks.py b/tests/download/test_http_bypasser_fallbacks.py index 47033889..a7d16409 100644 --- a/tests/download/test_http_bypasser_fallbacks.py +++ b/tests/download/test_http_bypasser_fallbacks.py @@ -330,49 +330,97 @@ def test_redirect_loop_gives_up_immediately_when_bypasser_not_allowed(monkeypatc assert len(requested) == http._MAX_REDIRECTS + 1 -def test_requests_raised_redirect_loop_still_reaches_bypasser(monkeypatch): - """Non-AA hosts keep allow_redirects=True, so `requests` raises the loop itself. +def test_html_get_page_redirect_loop_purges_cookies_and_bypasses(monkeypatch): + """A redirect loop is the challenge served against stale cookies, not a retryable error. - The AA follower above never sees that one, so the rescue in the exception path has - to stay — and it must purge the stale cookie exactly like the inline AA handoff. + TooManyRedirects carries no status code, so without an explicit branch it falls through + to the generic retry path and repeats the identical failure for the whole retry budget. """ import shelfmark.download.http as http - cleared: list[str] = [] - - class _FakeInternalBypasser: - @staticmethod - def clear_cf_cookies(domain: str) -> None: - cleared.append(domain) - monkeypatch.setattr(http, "_is_cf_bypass_enabled", lambda: True) monkeypatch.setattr(http, "_is_using_external_bypasser", lambda: False) - monkeypatch.setattr(http, "_get_internal_bypasser", lambda: _FakeInternalBypasser) - monkeypatch.setattr(http, "_bypass_grace_seconds", lambda: 100.0) - monkeypatch.setattr(http, "_apply_cf_bypass", lambda _url, _headers: {}) + monkeypatch.setattr(http, "_bypass_grace_seconds", lambda: 330.0) monkeypatch.setattr(http, "get_proxies", lambda _url: {}) - monkeypatch.setattr(http, "get_ssl_verify", lambda _url: True) - monkeypatch.setattr(http.network, "should_rotate_dns_for_url", lambda _url: False) - monkeypatch.setattr(http.time, "sleep", lambda _seconds: None) + monkeypatch.setattr(http.time, "sleep", lambda _s: None) + monkeypatch.setattr(http.network, "should_rotate_dns_for_url", lambda _url: True) + monkeypatch.setattr(http.network, "get_aa_base_url", lambda: "https://annas-archive.li") + monkeypatch.setattr(http.network, "is_aa_auto_mode", lambda: False) - def looping(_url: str, **_kwargs): - raise requests.exceptions.TooManyRedirects("too many redirects") + cleared: list[str] = [] - bypassed: list[str] = [] - monkeypatch.setattr(http.requests, "get", looping) - monkeypatch.setattr( - http, - "get_bypassed_page", - lambda url, *_a, **_k: bypassed.append(url) or "ok", - ) + class FakeInternalBypasser: + def clear_cf_cookies(self, domain: str) -> None: + cleared.append(domain) + + def get_cf_cookies_for_domain(self, _domain: str) -> dict[str, str]: + return {"__ddg2_": "stale"} + + def get_cf_user_agent_for_domain(self, _domain: str) -> str | None: + return None + + monkeypatch.setattr(http, "_get_internal_bypasser", lambda: FakeInternalBypasser()) + monkeypatch.setattr(http, "get_bypassed_page", lambda *_args, **_kwargs: "SOLVED") + + class _FakeRedirect: + """A 302 that always points at the same ?check=1 URL, cookies unchanged.""" + + is_redirect = True + status_code = 302 + cookies = {"__ddg2_": "stale"} + + def __init__(self, url: str) -> None: + self.url = url + self.headers = {"Location": "https://annas-archive.li/search?q=test&check=1"} + + hits: list[str] = [] + + def fake_get(url: str, **kwargs): + hits.append(url) + # Stale cookies: the server keeps re-issuing the same ?check=1 redirect. + return _FakeRedirect(url) + + monkeypatch.setattr(http.requests, "get", fake_get) html = http.html_get_page( - "https://z-lib.fm/s/dune", - retry=1, - allow_bypasser_fallback=True, + "https://annas-archive.li/search?q=test", + retry=2, success_delay=0, ) - assert html == "ok" - assert cleared == ["z-lib.fm"] - assert bypassed == ["https://z-lib.fm/s/dune"] + assert html == "SOLVED" + assert cleared == ["annas-archive.li"] + # The loop is cut short: no second attempt spent repeating the same redirects. + assert len(hits) == http._MAX_REDIRECTS + 1 + + +def test_html_get_page_redirect_loop_on_non_aa_host_is_left_alone(monkeypatch): + """Only hosts whose redirects we follow manually get the challenge treatment. + + Elsewhere requests follows redirects itself, so a loop is an ordinary misconfiguration - + purging that host's cookies and forcing a bypass would be the wrong response. + """ + import shelfmark.download.http as http + + monkeypatch.setattr(http, "_is_cf_bypass_enabled", lambda: True) + monkeypatch.setattr(http, "_is_using_external_bypasser", lambda: False) + monkeypatch.setattr(http, "_apply_cf_bypass", lambda _url, _headers: {}) + monkeypatch.setattr(http, "get_proxies", lambda _url: {}) + monkeypatch.setattr(http, "get_ssl_verify", lambda _url: True) + monkeypatch.setattr(http.time, "sleep", lambda _s: None) + monkeypatch.setattr(http.network, "should_rotate_dns_for_url", lambda _url: False) + + bypassed: list[str] = [] + monkeypatch.setattr( + http, "get_bypassed_page", lambda url, *_args, **_kwargs: bypassed.append(url) or "SOLVED" + ) + + def fake_get(_url: str, **_kwargs): + raise requests.exceptions.TooManyRedirects("Exceeded 30 redirects.") + + monkeypatch.setattr(http.requests, "get", fake_get) + + html = http.html_get_page("https://example.com/loop", retry=2, success_delay=0) + + assert html == "" + assert bypassed == []