diff --git a/shelfmark/bypass/challenge.py b/shelfmark/bypass/challenge.py
index b24a0225..ebca3e49 100644
--- a/shelfmark/bypass/challenge.py
+++ b/shelfmark/bypass/challenge.py
@@ -21,6 +21,10 @@ DDOS_GUARD_INDICATORS = [
"could not verify your browser automatically",
]
+OTHER_CHALLENGE_INDICATORS = [
+ "diamwall",
+]
+
# Markers that exist only in raw markup: the bypassers scan rendered innerText, where
# a script src or a
never appears. The title match is scoped to the tag on
# purpose - hosts word the rest of that sentence differently, and matching "checking
@@ -46,7 +50,12 @@ def challenge_marker(html: str) -> str | None:
if not html or len(html) > MAX_CHALLENGE_HTML_CHARS:
return None
lowered = html.lower()
- for marker in (*_RAW_HTML_MARKERS, *DDOS_GUARD_INDICATORS, *CLOUDFLARE_INDICATORS):
+ for marker in (
+ *_RAW_HTML_MARKERS,
+ *DDOS_GUARD_INDICATORS,
+ *CLOUDFLARE_INDICATORS,
+ *OTHER_CHALLENGE_INDICATORS,
+ ):
if marker in lowered:
return marker
return None
diff --git a/shelfmark/bypass/internal_bypasser.py b/shelfmark/bypass/internal_bypasser.py
index f1eeb4df..27154ce5 100644
--- a/shelfmark/bypass/internal_bypasser.py
+++ b/shelfmark/bypass/internal_bypasser.py
@@ -27,7 +27,11 @@ from seleniumbase import cdp_driver
from seleniumbase.undetected.cdp_driver.connection import ProtocolException
from shelfmark.bypass import BypassCancelledError
-from shelfmark.bypass.challenge import CLOUDFLARE_INDICATORS, DDOS_GUARD_INDICATORS
+from shelfmark.bypass.challenge import (
+ CLOUDFLARE_INDICATORS,
+ DDOS_GUARD_INDICATORS,
+ OTHER_CHALLENGE_INDICATORS,
+)
from shelfmark.bypass.cookie_store import (
clear_cf_cookies,
export_store,
@@ -443,7 +447,7 @@ def _has_cloudflare_patterns(body: str, url: str) -> bool:
async def _detect_challenge_type(page: Any) -> str:
- """Detect challenge type: 'cloudflare', 'ddos_guard', or 'none'."""
+ """Detect challenge type: 'cloudflare', 'ddos_guard', 'other', or 'none'."""
title, body, current_url = await _get_page_info(page)
# DDOS-Guard indicators
@@ -456,6 +460,10 @@ async def _detect_challenge_type(page: Any) -> str:
logger.debug("Cloudflare indicator found: '%s'", found)
return "cloudflare"
+ if found := _check_indicators(title, body, OTHER_CHALLENGE_INDICATORS):
+ logger.debug("Other challenge indicator found: '%s'", found)
+ return "other"
+
# Check URL patterns
if _has_cloudflare_patterns(body, current_url):
return "cloudflare"
@@ -490,7 +498,11 @@ def _is_bypassed_content(
return True
# Check for protection indicators (means NOT bypassed)
- if _check_indicators(title, body, CLOUDFLARE_INDICATORS + DDOS_GUARD_INDICATORS):
+ if _check_indicators(
+ title,
+ body,
+ CLOUDFLARE_INDICATORS + DDOS_GUARD_INDICATORS + OTHER_CHALLENGE_INDICATORS,
+ ):
return False
# Cloudflare URL patterns
diff --git a/shelfmark/download/http.py b/shelfmark/download/http.py
index 079f16fa..3df87e23 100644
--- a/shelfmark/download/http.py
+++ b/shelfmark/download/http.py
@@ -42,6 +42,8 @@ _HTTP_STATUS_FORBIDDEN = HTTPStatus.FORBIDDEN
_HTTP_STATUS_NOT_FOUND = HTTPStatus.NOT_FOUND
_HTTP_STATUS_RATE_LIMITED = HTTPStatus.TOO_MANY_REQUESTS
_HTTP_STATUS_SERVICE_UNAVAILABLE = HTTPStatus.SERVICE_UNAVAILABLE
+# DiamWall uses this non-standard status for its browser-verification page.
+_HTTP_STATUS_CHALLENGE = 513
_HTTP_STATUS_OK = HTTPStatus.OK
_HTTP_STATUS_RANGE_NOT_SATISFIABLE = HTTPStatus.REQUESTED_RANGE_NOT_SATISFIABLE
_HTTP_STATUS_PARTIAL_CONTENT = HTTPStatus.PARTIAL_CONTENT
@@ -619,6 +621,27 @@ def html_get_page(
current_url,
)
+ if response.status_code == _HTTP_STATUS_CHALLENGE:
+ marker = _response_challenge_marker(response)
+ if marker and _bypass_handoff_allowed():
+ if cookies:
+ logger.debug(
+ "513 challenge with cookies presented; purging: %s", current_url
+ )
+ _purge_clearance(current_url)
+ logger.info(
+ "513 challenge detected (%s); switching to bypasser: %s",
+ marker,
+ current_url,
+ )
+ return _run_bypasser(current_url)
+ if marker:
+ logger.debug(
+ "513 challenge (%s) but no bypasser handoff available: %s",
+ marker,
+ current_url,
+ )
+
if is_aa_url and response.is_redirect:
location = response.headers.get("Location", "")
if not location:
diff --git a/tests/bypass/test_internal_bypasser.py b/tests/bypass/test_internal_bypasser.py
index 443479d6..d3c3cb2c 100644
--- a/tests/bypass/test_internal_bypasser.py
+++ b/tests/bypass/test_internal_bypasser.py
@@ -6,6 +6,27 @@ from pathlib import Path
import pytest
+def test_diamwall_is_detected_until_the_challenge_clears():
+ import shelfmark.bypass.internal_bypasser as internal_bypasser
+
+ class DiamWallPage:
+ async def get_title(self):
+ return "DiamWall"
+
+ async def evaluate(self, _expression):
+ return "Please wait while DiamWall checks your browser"
+
+ async def get_current_url(self):
+ return "https://example.com"
+
+ assert asyncio.run(internal_bypasser._detect_challenge_type(DiamWallPage())) == "other"
+ assert not internal_bypasser._is_bypassed_content(
+ "DiamWall",
+ "Please wait while DiamWall checks your browser",
+ "https://example.com",
+ )
+
+
def test_bypass_tries_all_methods_before_abort(monkeypatch):
"""Regression test for issue #524: don't abort before cycling through bypass methods."""
import shelfmark.bypass.internal_bypasser as internal_bypasser
diff --git a/tests/download/test_http_challenge_513.py b/tests/download/test_http_challenge_513.py
new file mode 100644
index 00000000..f2c448b0
--- /dev/null
+++ b/tests/download/test_http_challenge_513.py
@@ -0,0 +1,83 @@
+"""Tests for handing a DiamWall 513 challenge to the browser bypasser."""
+
+import requests
+
+
+class _FakeResponse:
+ def __init__(self, text: str) -> None:
+ self.status_code = 513
+ self.url = "https://z-lib.gd/book/abc"
+ self.text = text
+ self.cookies: dict[str, str] = {}
+ self.headers = {"Content-Type": "text/html; charset=utf-8"}
+ self.is_redirect = False
+
+ def raise_for_status(self) -> None:
+ error = requests.exceptions.HTTPError("513 Error")
+ error.response = self
+ raise error
+
+
+def _neutralize_network(monkeypatch, http) -> None:
+ monkeypatch.setattr(http, "_apply_cf_bypass", lambda _url, _headers: {})
+ monkeypatch.setattr(http, "get_proxies", lambda _url: {})
+ monkeypatch.setattr(http, "get_ssl_verify", lambda _url: True)
+ monkeypatch.setattr(http.network, "should_rotate_dns_for_url", lambda _url: False)
+ monkeypatch.setattr(http.time, "sleep", lambda _seconds: None)
+ monkeypatch.setattr(http, "_bypass_grace_seconds", lambda: 1.0)
+
+
+def test_diamwall_513_is_handed_to_the_bypasser(monkeypatch) -> None:
+ import shelfmark.download.http as http
+
+ _neutralize_network(monkeypatch, http)
+ monkeypatch.setattr(http, "_is_cf_bypass_enabled", lambda: True)
+
+ attempts: list[str] = []
+ bypassed: list[str] = []
+ monkeypatch.setattr(
+ http.requests,
+ "get",
+ lambda url, **_kwargs: (
+ attempts.append(url)
+ or _FakeResponse(
+ "DiamWallBrowser verification"
+ )
+ ),
+ )
+ monkeypatch.setattr(
+ http,
+ "get_bypassed_page",
+ lambda url, *_args, **_kwargs: bypassed.append(url) or "book page",
+ )
+
+ url = "https://z-lib.gd/book/abc"
+ html = http.html_get_page(url, retry=10, success_delay=0)
+
+ assert html == "book page"
+ assert attempts == [url]
+ assert bypassed == [url]
+
+
+def test_plain_513_does_not_start_a_browser(monkeypatch) -> None:
+ import shelfmark.download.http as http
+
+ _neutralize_network(monkeypatch, http)
+ monkeypatch.setattr(http, "_is_cf_bypass_enabled", lambda: True)
+
+ bypassed: list[str] = []
+ monkeypatch.setattr(
+ http.requests,
+ "get",
+ lambda _url, **_kwargs: _FakeResponse("Unknown server error"),
+ )
+ monkeypatch.setattr(
+ http,
+ "get_bypassed_page",
+ lambda url, *_args, **_kwargs: bypassed.append(url) or "",
+ )
+
+ html = http.html_get_page("https://z-lib.gd/book/abc", retry=1, success_delay=0)
+
+ assert html == ""
+ assert bypassed == []