From 41a4df01c4f48d0e667631e2eb61c366516e9783 Mon Sep 17 00:00:00 2001 From: vansh Date: Sat, 3 Oct 2026 01:56:04 +0530 Subject: [PATCH] fix: hand DiamWall 513 challenges to the bypasser (#1400) ## Summary Recognize DiamWall interstitials in both the HTTP retry path and the internal browser helper. HTTP 513 challenge responses go directly to the configured bypasser, while ordinary 513 responses keep the existing error behavior. Fixes #1386 ## Testing - `uv run --frozen --extra browser pytest -q -n 0 tests/download/test_http_challenge_513.py tests/bypass/test_internal_bypasser.py`: 27 passed - Ruff lint and format checks passed --- shelfmark/bypass/challenge.py | 11 ++- shelfmark/bypass/internal_bypasser.py | 18 ++++- shelfmark/download/http.py | 23 +++++++ tests/bypass/test_internal_bypasser.py | 21 ++++++ tests/download/test_http_challenge_513.py | 83 +++++++++++++++++++++++ 5 files changed, 152 insertions(+), 4 deletions(-) create mode 100644 tests/download/test_http_challenge_513.py diff --git a/shelfmark/bypass/challenge.py b/shelfmark/bypass/challenge.py index b24a0225..ebca3e49 100644 --- a/shelfmark/bypass/challenge.py +++ b/shelfmark/bypass/challenge.py @@ -21,6 +21,10 @@ DDOS_GUARD_INDICATORS = [ "could not verify your browser automatically", ] +OTHER_CHALLENGE_INDICATORS = [ + "diamwall", +] + # Markers that exist only in raw markup: the bypassers scan rendered innerText, where # a script src or a never appears. The title match is scoped to the tag on # purpose - hosts word the rest of that sentence differently, and matching "checking @@ -46,7 +50,12 @@ def challenge_marker(html: str) -> str | None: if not html or len(html) > MAX_CHALLENGE_HTML_CHARS: return None lowered = html.lower() - for marker in (*_RAW_HTML_MARKERS, *DDOS_GUARD_INDICATORS, *CLOUDFLARE_INDICATORS): + for marker in ( + *_RAW_HTML_MARKERS, + *DDOS_GUARD_INDICATORS, + *CLOUDFLARE_INDICATORS, + *OTHER_CHALLENGE_INDICATORS, + ): if marker in lowered: return marker return None diff --git a/shelfmark/bypass/internal_bypasser.py b/shelfmark/bypass/internal_bypasser.py index f1eeb4df..27154ce5 100644 --- a/shelfmark/bypass/internal_bypasser.py +++ b/shelfmark/bypass/internal_bypasser.py @@ -27,7 +27,11 @@ from seleniumbase import cdp_driver from seleniumbase.undetected.cdp_driver.connection import ProtocolException from shelfmark.bypass import BypassCancelledError -from shelfmark.bypass.challenge import CLOUDFLARE_INDICATORS, DDOS_GUARD_INDICATORS +from shelfmark.bypass.challenge import ( + CLOUDFLARE_INDICATORS, + DDOS_GUARD_INDICATORS, + OTHER_CHALLENGE_INDICATORS, +) from shelfmark.bypass.cookie_store import ( clear_cf_cookies, export_store, @@ -443,7 +447,7 @@ def _has_cloudflare_patterns(body: str, url: str) -> bool: async def _detect_challenge_type(page: Any) -> str: - """Detect challenge type: 'cloudflare', 'ddos_guard', or 'none'.""" + """Detect challenge type: 'cloudflare', 'ddos_guard', 'other', or 'none'.""" title, body, current_url = await _get_page_info(page) # DDOS-Guard indicators @@ -456,6 +460,10 @@ async def _detect_challenge_type(page: Any) -> str: logger.debug("Cloudflare indicator found: '%s'", found) return "cloudflare" + if found := _check_indicators(title, body, OTHER_CHALLENGE_INDICATORS): + logger.debug("Other challenge indicator found: '%s'", found) + return "other" + # Check URL patterns if _has_cloudflare_patterns(body, current_url): return "cloudflare" @@ -490,7 +498,11 @@ def _is_bypassed_content( return True # Check for protection indicators (means NOT bypassed) - if _check_indicators(title, body, CLOUDFLARE_INDICATORS + DDOS_GUARD_INDICATORS): + if _check_indicators( + title, + body, + CLOUDFLARE_INDICATORS + DDOS_GUARD_INDICATORS + OTHER_CHALLENGE_INDICATORS, + ): return False # Cloudflare URL patterns diff --git a/shelfmark/download/http.py b/shelfmark/download/http.py index 079f16fa..3df87e23 100644 --- a/shelfmark/download/http.py +++ b/shelfmark/download/http.py @@ -42,6 +42,8 @@ _HTTP_STATUS_FORBIDDEN = HTTPStatus.FORBIDDEN _HTTP_STATUS_NOT_FOUND = HTTPStatus.NOT_FOUND _HTTP_STATUS_RATE_LIMITED = HTTPStatus.TOO_MANY_REQUESTS _HTTP_STATUS_SERVICE_UNAVAILABLE = HTTPStatus.SERVICE_UNAVAILABLE +# DiamWall uses this non-standard status for its browser-verification page. +_HTTP_STATUS_CHALLENGE = 513 _HTTP_STATUS_OK = HTTPStatus.OK _HTTP_STATUS_RANGE_NOT_SATISFIABLE = HTTPStatus.REQUESTED_RANGE_NOT_SATISFIABLE _HTTP_STATUS_PARTIAL_CONTENT = HTTPStatus.PARTIAL_CONTENT @@ -619,6 +621,27 @@ def html_get_page( current_url, ) + if response.status_code == _HTTP_STATUS_CHALLENGE: + marker = _response_challenge_marker(response) + if marker and _bypass_handoff_allowed(): + if cookies: + logger.debug( + "513 challenge with cookies presented; purging: %s", current_url + ) + _purge_clearance(current_url) + logger.info( + "513 challenge detected (%s); switching to bypasser: %s", + marker, + current_url, + ) + return _run_bypasser(current_url) + if marker: + logger.debug( + "513 challenge (%s) but no bypasser handoff available: %s", + marker, + current_url, + ) + if is_aa_url and response.is_redirect: location = response.headers.get("Location", "") if not location: diff --git a/tests/bypass/test_internal_bypasser.py b/tests/bypass/test_internal_bypasser.py index 443479d6..d3c3cb2c 100644 --- a/tests/bypass/test_internal_bypasser.py +++ b/tests/bypass/test_internal_bypasser.py @@ -6,6 +6,27 @@ from pathlib import Path import pytest +def test_diamwall_is_detected_until_the_challenge_clears(): + import shelfmark.bypass.internal_bypasser as internal_bypasser + + class DiamWallPage: + async def get_title(self): + return "DiamWall" + + async def evaluate(self, _expression): + return "Please wait while DiamWall checks your browser" + + async def get_current_url(self): + return "https://example.com" + + assert asyncio.run(internal_bypasser._detect_challenge_type(DiamWallPage())) == "other" + assert not internal_bypasser._is_bypassed_content( + "DiamWall", + "Please wait while DiamWall checks your browser", + "https://example.com", + ) + + def test_bypass_tries_all_methods_before_abort(monkeypatch): """Regression test for issue #524: don't abort before cycling through bypass methods.""" import shelfmark.bypass.internal_bypasser as internal_bypasser diff --git a/tests/download/test_http_challenge_513.py b/tests/download/test_http_challenge_513.py new file mode 100644 index 00000000..f2c448b0 --- /dev/null +++ b/tests/download/test_http_challenge_513.py @@ -0,0 +1,83 @@ +"""Tests for handing a DiamWall 513 challenge to the browser bypasser.""" + +import requests + + +class _FakeResponse: + def __init__(self, text: str) -> None: + self.status_code = 513 + self.url = "https://z-lib.gd/book/abc" + self.text = text + self.cookies: dict[str, str] = {} + self.headers = {"Content-Type": "text/html; charset=utf-8"} + self.is_redirect = False + + def raise_for_status(self) -> None: + error = requests.exceptions.HTTPError("513 Error") + error.response = self + raise error + + +def _neutralize_network(monkeypatch, http) -> None: + monkeypatch.setattr(http, "_apply_cf_bypass", lambda _url, _headers: {}) + monkeypatch.setattr(http, "get_proxies", lambda _url: {}) + monkeypatch.setattr(http, "get_ssl_verify", lambda _url: True) + monkeypatch.setattr(http.network, "should_rotate_dns_for_url", lambda _url: False) + monkeypatch.setattr(http.time, "sleep", lambda _seconds: None) + monkeypatch.setattr(http, "_bypass_grace_seconds", lambda: 1.0) + + +def test_diamwall_513_is_handed_to_the_bypasser(monkeypatch) -> None: + import shelfmark.download.http as http + + _neutralize_network(monkeypatch, http) + monkeypatch.setattr(http, "_is_cf_bypass_enabled", lambda: True) + + attempts: list[str] = [] + bypassed: list[str] = [] + monkeypatch.setattr( + http.requests, + "get", + lambda url, **_kwargs: ( + attempts.append(url) + or _FakeResponse( + "<html><title>DiamWallBrowser verification" + ) + ), + ) + monkeypatch.setattr( + http, + "get_bypassed_page", + lambda url, *_args, **_kwargs: bypassed.append(url) or "book page", + ) + + url = "https://z-lib.gd/book/abc" + html = http.html_get_page(url, retry=10, success_delay=0) + + assert html == "book page" + assert attempts == [url] + assert bypassed == [url] + + +def test_plain_513_does_not_start_a_browser(monkeypatch) -> None: + import shelfmark.download.http as http + + _neutralize_network(monkeypatch, http) + monkeypatch.setattr(http, "_is_cf_bypass_enabled", lambda: True) + + bypassed: list[str] = [] + monkeypatch.setattr( + http.requests, + "get", + lambda _url, **_kwargs: _FakeResponse("Unknown server error"), + ) + monkeypatch.setattr( + http, + "get_bypassed_page", + lambda url, *_args, **_kwargs: bypassed.append(url) or "", + ) + + html = http.html_get_page("https://z-lib.gd/book/abc", retry=1, success_delay=0) + + assert html == "" + assert bypassed == []