fix: prevent Anna's Archive download countdown resets by preserving browser sessions (#1325)

## Observed bug

Anna's Archive slow-download pages can return a JavaScript countdown
before a download link is available. The internal browser returns that
waiting-room HTML and closes its incognito session. The downloader then
sleeps and fetches the URL again, which can create a new queue session
and **restart the countdown instead of reaching the download link**.

## Fix

- **Preserve the queue session:** keep the original browser tab open
while the site's own countdown and automatic navigation finish. HTTP 200
and cached-cookie waiting-room responses enter the same flow.
- **Return a consistent page:** capture HTML and readiness together in
one browser evaluation so navigation cannot pair a new page's status
with stale protection-page HTML. Share cache validation and
page-readiness rules across their callers.
- **Keep waiting cancellable and bounded:** poll cancellation while a
slow browser read remains pending, rather than repeatedly cancelling and
reissuing it. Apply a **300-second waiting-room limit** within the
existing browser watchdog.
- **Report queue timeouts accurately:** preserve the timeout across the
helper-process boundary and stop the solve without restarting the
browser or rotating mirrors.

Waiting-room detection is limited to Anna's Archive `/slow_download/`
pages containing an actual `.js-partner-countdown` element. The site
controls the countdown and refresh. External-bypasser behavior and
file-transfer timeouts are unchanged; the PR adds no deployment
configuration or dependencies.

## Validation

Validated at `c81e0a2`:

| Check | Result |
| --- | --- |
| Full Linux unit suite | **2,978 passed** on Python 3.14 in a non-root
environment with entrypoint test stubs enabled |
| Focused regression coverage | **30 passed**, covering countdown
completion, zero timers, navigation, both cookie-cache paths,
cancellation, stuck queues, slow reads, and timeout propagation |
| Navigation-race regression | Fails against the previous PR
implementation and passes with the fix |
| Python static checks | Ruff lint/format, BasedPyright for backend and
tests, and Vulture passed |
| Real Chromium fixture | Queue cookie persisted through 1.5-second DOM
reads and one automatic refresh; the CDP connection survived multiple
polling intervals |
| Live source check | Observed **19 → 14 → 9 → 4 → download link** while
retaining the browser session; the patched browser path also completed
the waiting room |

Full unit-suite command:

```sh
pytest tests/ -n 2 --tb=short -m "not integration and not e2e"
```

The live check validates waiting-room completion and link resolution.
Remote file-host availability remains a separate concern. The unit suite
emitted two existing Authlib deprecation warnings.
This commit is contained in:
Austin Brogle
2026-09-11 00:51:48 -04:00
committed by GitHub
parent c45d342931
commit 99e0cfde3d
4 changed files with 423 additions and 3 deletions
+101 -3
View File
@@ -37,6 +37,7 @@ from shelfmark.bypass.cookie_store import (
store_extracted_cookies,
)
from shelfmark.bypass.fingerprint import get_screen_size
from shelfmark.bypass.waiting_room import WaitingRoomTimeoutError, is_aa_waiting_room
from shelfmark.config import env
from shelfmark.config.env import LOG_DIR
from shelfmark.config.settings import RECORDING_DIR
@@ -85,11 +86,22 @@ _HELPER_IDLE_TIMEOUT_DEFAULT = 180.0
# How long to wait for a solved page to produce its document before the attempt is
# abandoned. SeleniumBase's own get_page_source() allows one second; see _read_page_source.
_PAGE_SOURCE_TIMEOUT_DEFAULT = 20.0
# Leave time inside the existing browser watchdog for challenge solving and cleanup.
_AA_WAITING_ROOM_TIMEOUT_SECONDS = 300.0
_AA_WAITING_ROOM_POLL_SECONDS = 1.0
_PARENT_WATCHDOG_INTERVAL_SECONDS = 5.0
# How much of ffmpeg's stderr to quote when reporting that it died.
_FFMPEG_ERROR_TAIL_CHARS = 500
class _WaitingRoomSnapshot(TypedDict):
html: str
title: str
body: str
url: str
waiting: bool
class _DisplayState(TypedDict):
ffmpeg: subprocess.Popen[bytes] | None
ffmpeg_output: Path | None
@@ -454,6 +466,14 @@ async def _detect_challenge_type(page: Any) -> str:
async def _is_bypassed(page: Any, *, escape_emojis: bool = True) -> bool:
"""Check if the protection has been bypassed."""
title, body, current_url = await _get_page_info(page)
return _is_bypassed_content(title, body, current_url, escape_emojis=escape_emojis)
def _is_bypassed_content(
title: str, body: str, current_url: str, *, escape_emojis: bool = True
) -> bool:
"""Apply the same protection checks to one consistent page snapshot."""
title, body = title.lower(), body.lower()
body_len = len(body.strip())
# Long page content = probably bypassed
@@ -831,6 +851,75 @@ async def _read_page_source(page: Any) -> str:
return await element.get_html_async()
async def _read_waiting_room_snapshot(
page: Any, cancel_flag: Event | None
) -> _WaitingRoomSnapshot | None:
"""Read one DOM snapshot while still checking cancellation during a stalled read."""
task = asyncio.create_task(
page.evaluate("""({
html: document.documentElement?.outerHTML || '',
title: document.title,
body: document.body?.innerText || '',
url: location.href,
waiting: !!document.querySelector('.js-partner-countdown')
})""")
)
try:
while not task.done():
_check_cancellation(cancel_flag, "Bypass cancelled in Anna's waiting room")
# Keep a slow CDP request alive. Cancelling and reissuing it on every
# poll can break the listener when a late response targets a cancelled
# SeleniumBase transaction. Cancel only when this browser is unwinding.
await asyncio.wait({task}, timeout=_AA_WAITING_ROOM_POLL_SECONDS)
_check_cancellation(cancel_flag, "Bypass cancelled in Anna's waiting room")
return task.result()
finally:
task.cancel()
await asyncio.gather(task, return_exceptions=True)
async def _wait_for_aa_download_page(
page: Any, url: str, html: str, cancel_flag: Event | None = None
) -> str:
"""Let the waiting room's own JavaScript countdown and navigation finish.
Returning the timer HTML closes this incognito browser. Sleeping in the HTTP
downloader and fetching again then starts a different session, losing queue state.
Keep the original tab alive, including through zero and automatic page reloads.
"""
if not is_aa_waiting_room(url, html):
return html
logger.info("Waiting for Anna's Archive queue in the same browser session...")
started = time.monotonic()
try:
async with asyncio.timeout(_AA_WAITING_ROOM_TIMEOUT_SECONDS):
while True:
try:
# Read readiness and HTML atomically: navigation between separate
# CDP reads could validate a new page but return an old interstitial.
snapshot = await _read_waiting_room_snapshot(page, cancel_flag)
except _CDP_OPERATION_ERRORS:
# A frame/context can disappear during automatic navigation.
snapshot = None
if (
snapshot
and not snapshot["waiting"]
and _is_bypassed_content(snapshot["title"], snapshot["body"], snapshot["url"])
):
logger.info(
"Anna's Archive waiting room finished after %.0fs",
time.monotonic() - started,
)
return snapshot["html"]
# A zero timer, empty document, or protection page is not completion.
await asyncio.sleep(_AA_WAITING_ROOM_POLL_SECONDS)
except TimeoutError as exc:
raise WaitingRoomTimeoutError(
f"Anna's Archive waiting room did not finish within {_AA_WAITING_ROOM_TIMEOUT_SECONDS:g}s"
) from exc
async def _get(url: str, driver: Any, cancel_flag: Event | None = None) -> str:
"""Fetch URL with Cloudflare bypass using a CDP browser."""
_check_cancellation(cancel_flag, "Bypass cancelled before starting")
@@ -853,8 +942,10 @@ async def _get(url: str, driver: Any, cancel_flag: Event | None = None) -> str:
logger.debug("Starting bypass process...")
if await _bypass(page, cancel_flag=cancel_flag):
html = await _read_page_source(page)
html = await _wait_for_aa_download_page(page, url, html, cancel_flag)
await _extract_cookies_from_cdp(driver, page, url)
return await _read_page_source(page)
return html
logger.warning("Bypass completed but page still shows protection")
try:
@@ -904,7 +995,8 @@ def _run_bypass_in_current_process(url: str, retry: int, cancel_flag: Event | No
result = await _get(url, driver, cancel_flag)
if result:
return result
except BypassCancelledError:
except BypassCancelledError, WaitingRoomTimeoutError:
# Retrying would restart the same queue in another browser.
raise
except _CDP_OPERATION_ERRORS as e:
error_details = f"{type(e).__name__}: {e}"
@@ -1215,6 +1307,8 @@ def _get_via_subprocess(url: str, retry: int, cancel_flag: Event | None = None)
trace = result.get("traceback")
if trace:
logger.debug("Internal bypasser helper traceback: %s", trace)
if error_type == WaitingRoomTimeoutError.__name__:
raise WaitingRoomTimeoutError(error)
msg = f"{error_type}: {error}"
raise RuntimeError(msg)
@@ -1534,6 +1628,10 @@ def _try_with_cached_cookies(url: str, hostname: str) -> str | None:
verify=get_ssl_verify(url),
)
if response.status_code == HTTPStatus.OK:
if is_aa_waiting_room(url, response.text):
# Clearance is valid, but HTTP cannot run the queue's JavaScript.
# Enforce this here for both cache checks, including the locked one.
return None
logger.debug("Cached cookies worked, skipped Chrome bypass")
return response.text
if response.status_code == HTTPStatus.TOO_MANY_REQUESTS:
@@ -1606,7 +1704,7 @@ def get_bypassed_page(
try:
response_html = get(attempt_url, cancel_flag=cancel_flag)
except BypassCancelledError:
except BypassCancelledError, WaitingRoomTimeoutError:
raise
except _CDP_OPERATION_ERRORS + _REQUEST_OPERATION_ERRORS:
_check_cancellation(cancel_flag, "Bypass cancelled")
+16
View File
@@ -0,0 +1,16 @@
"""Recognize Anna's Archive pages that require a live JavaScript timer."""
from urllib.parse import urlparse
from bs4 import BeautifulSoup
class WaitingRoomTimeoutError(TimeoutError):
"""The source waiting room did not finish within the browser session budget."""
def is_aa_waiting_room(url: str, html: str) -> bool:
"""Match the download route and actual timer element, not a script reference."""
return urlparse(url).path.startswith("/slow_download/") and bool(
BeautifulSoup(html, "html.parser").select_one(".js-partner-countdown")
)
+12
View File
@@ -12,6 +12,7 @@ from tqdm import tqdm
from shelfmark.bypass import BypassCancelledError, ChallengeNotSolvedError, cookie_store
from shelfmark.bypass.challenge import challenge_marker
from shelfmark.bypass.waiting_room import WaitingRoomTimeoutError, is_aa_waiting_room
from shelfmark.core import search_deadline
from shelfmark.core.config import config as app_config
from shelfmark.core.logger import setup_logger
@@ -440,6 +441,9 @@ def html_get_page(
except _STATUS_CALLBACK_ERRORS:
logger.debug("Rate-limit status callback failed", exc_info=True)
return _fail(str(e), bypass_url)
except WaitingRoomTimeoutError as e:
logger.info("Waiting room timed out: %s", e)
return _fail(str(e), bypass_url)
except ChallengeNotSolvedError as e:
# Not a bypasser malfunction: it ran, and the host answered with something it
# cannot clear - DDoS-Guard's manual CAPTCHA, typically. Must precede the
@@ -705,6 +709,14 @@ def html_get_page(
continue
response.raise_for_status()
if (
_bypass_handoff_allowed()
and not _is_using_external_bypasser()
and is_aa_waiting_room(current_url, response.text)
):
# A successful HTTP response can still need a live browser: the
# queue's JavaScript must finish in the session that entered it.
return _run_bypasser(current_url)
if success_delay > 0:
time.sleep(success_delay)
return _result(response.text, response.url)
+294
View File
@@ -0,0 +1,294 @@
"""Keep Anna's JavaScript waiting room alive until its own navigation completes."""
import asyncio
import threading
from types import SimpleNamespace
from unittest.mock import AsyncMock, Mock
import pytest
import requests
import shelfmark.bypass.internal_bypasser as b
from shelfmark.bypass import BypassCancelledError
from shelfmark.bypass.waiting_room import WaitingRoomTimeoutError, is_aa_waiting_room
URL = "https://annas-archive.gl/slow_download/abc/0/0"
WAIT = '<html><span class="js-partner-countdown">25</span></html>'
ZERO = WAIT.replace(">25<", ">0<")
READY = '<html><a href="https://files.example/book.epub">Download now</a></html>'
def _snapshot(html=READY, *, waiting=False, title="Anna's Archive", body=None):
return {
"html": html,
"waiting": waiting,
"title": title,
"url": URL,
"body": body
if body is not None
else "Your download is ready. Follow the download link to obtain the requested file.",
}
@pytest.fixture
def cached_http(monkeypatch):
response = SimpleNamespace(status_code=200, text=WAIT)
cleared = Mock()
monkeypatch.setattr(b, "get_cf_cookies_for_domain", lambda _: {"cf_clearance": "valid"})
monkeypatch.setattr(b, "clear_cf_cookies", cleared)
monkeypatch.setattr(b.requests, "get", Mock(return_value=response))
monkeypatch.setattr(b.network, "host_cooldown_remaining", lambda _: 0)
return response, cleared
@pytest.mark.parametrize(
("url", "html", "expected"),
[
(URL, WAIT, True),
(URL, ZERO, True),
(URL, READY, False),
(URL.replace("/slow_download/abc/0/0", "/search"), WAIT, False),
(URL, '<script>document.querySelector(".js-partner-countdown")</script>', False),
],
)
def test_waiting_room_detection(url, html, expected):
assert is_aa_waiting_room(url, html) is expected
def test_get_keeps_same_tab_through_zero_reload_and_protection(monkeypatch):
page = SimpleNamespace(
wait=AsyncMock(),
get_current_url=AsyncMock(return_value=URL),
get_title=AsyncMock(return_value="Anna's Archive"),
)
driver = SimpleNamespace(get=AsyncMock(return_value=page))
monkeypatch.setattr(b, "_bypass", AsyncMock(return_value=True))
# A transient protocol error and a protection page can occur during navigation.
read = AsyncMock(return_value=WAIT)
page.evaluate = AsyncMock(
side_effect=[
_snapshot(WAIT, waiting=True),
_snapshot(ZERO, waiting=True),
b.ProtocolException("navigating"),
_snapshot("<html>DDOS-GUARD</html>", title="DDOS-GUARD"),
_snapshot("", body=""),
_snapshot(),
]
)
monkeypatch.setattr(b, "_read_page_source", read)
cookies = AsyncMock()
monkeypatch.setattr(b, "_extract_cookies_from_cdp", cookies)
monkeypatch.setattr(b.asyncio, "sleep", AsyncMock())
assert asyncio.run(b._get(URL, driver)) == READY
driver.get.assert_awaited_once_with(URL)
read.assert_awaited_once_with(page)
cookies.assert_awaited_once_with(driver, page, URL)
def test_wait_can_be_cancelled(monkeypatch):
cancel = threading.Event()
async def cancel_during_sleep(_delay):
cancel.set()
monkeypatch.setattr(b.asyncio, "sleep", cancel_during_sleep)
page = SimpleNamespace(evaluate=AsyncMock(return_value=_snapshot(WAIT, waiting=True)))
with pytest.raises(BypassCancelledError):
asyncio.run(b._wait_for_aa_download_page(page, URL, WAIT, cancel))
def test_stuck_queue_has_a_deadline(monkeypatch):
monkeypatch.setattr(b, "_AA_WAITING_ROOM_TIMEOUT_SECONDS", 0.01)
with pytest.raises(WaitingRoomTimeoutError, match="waiting room"):
page = SimpleNamespace(evaluate=AsyncMock(return_value=_snapshot(WAIT, waiting=True)))
asyncio.run(b._wait_for_aa_download_page(page, URL, WAIT))
def test_queue_timeout_does_not_open_another_browser(monkeypatch):
driver = object()
create = AsyncMock(return_value=driver)
close = AsyncMock()
get = AsyncMock(side_effect=WaitingRoomTimeoutError("Queue timed out"))
monkeypatch.setattr(b, "_create_cdp_browser", create)
monkeypatch.setattr(b, "_close_cdp_driver", close)
monkeypatch.setattr(b, "_get", get)
monkeypatch.setattr(b, "_CDP_WORKER", SimpleNamespace(run=lambda task, **kw: asyncio.run(task)))
with pytest.raises(WaitingRoomTimeoutError, match="Queue timed out"):
b._run_bypass_in_current_process(URL, retry=10)
create.assert_awaited_once()
get.assert_awaited_once()
close.assert_awaited_once_with(driver)
@pytest.mark.parametrize("html", [READY, "<html>ordinary page</html>"])
def test_non_waiting_page_returns_without_polling(html):
assert asyncio.run(b._wait_for_aa_download_page(object(), URL, html)) == html
@pytest.mark.parametrize(
("enabled", "external", "fallback", "handoff"),
[
(True, False, True, True),
(False, False, True, False),
(True, True, True, False),
(True, False, False, False),
],
)
def test_http_200_timer_handoff_honors_browser_settings(
monkeypatch, enabled, external, fallback, handoff
):
import shelfmark.download.http as http
response = requests.Response()
response.status_code = 200
response.url = URL
response._content = WAIT.encode()
monkeypatch.setattr(http.requests, "get", lambda *a, **kw: response)
monkeypatch.setattr(http, "_apply_cf_bypass", lambda *a: {})
monkeypatch.setattr(http, "_is_cf_bypass_enabled", lambda: enabled)
monkeypatch.setattr(http, "_is_using_external_bypasser", lambda: external)
monkeypatch.setattr(http, "_bypass_grace_seconds", lambda: 450)
calls = []
monkeypatch.setattr(http, "get_bypassed_page", lambda url, *a: calls.append(url) or READY)
selector = SimpleNamespace(rewrite=lambda url: url)
result = http.html_get_page(
URL, retry=1, selector=selector, success_delay=0, allow_bypasser_fallback=fallback
)
assert result == (READY if handoff else WAIT)
assert calls == ([URL] if handoff else [])
def test_deadline_also_bounds_a_stalled_page_read(monkeypatch):
async def stalled_read(_expression):
await asyncio.Event().wait()
monkeypatch.setattr(b, "_AA_WAITING_ROOM_TIMEOUT_SECONDS", 0.01)
page = SimpleNamespace(evaluate=stalled_read)
with pytest.raises(WaitingRoomTimeoutError):
asyncio.run(b._wait_for_aa_download_page(page, URL, WAIT))
def test_helper_preserves_waiting_room_timeout_type(monkeypatch):
monkeypatch.setattr(
b,
"_BYPASS_HELPER",
SimpleNamespace(
run=lambda *args: {
"ok": False,
"error_type": "WaitingRoomTimeoutError",
"error": "Queue timed out",
}
),
)
with pytest.raises(WaitingRoomTimeoutError, match="Queue timed out"):
b._get_via_subprocess(URL, retry=10)
def test_queue_timeout_does_not_rotate_mirror_and_restart_wait(monkeypatch):
monkeypatch.setattr(b.network, "host_cooldown_remaining", lambda _: 0)
monkeypatch.setattr(b, "_try_with_cached_cookies", lambda *_: None)
def timeout(*args, **kwargs):
raise WaitingRoomTimeoutError("Queue timed out")
monkeypatch.setattr(b, "get", timeout)
# No rotation method: trying to rotate after this failure is an error.
selector = SimpleNamespace(rewrite=lambda url: url)
with pytest.raises(WaitingRoomTimeoutError, match="Queue timed out"):
b.get_bypassed_page(URL, selector)
def test_http_reports_queue_timeout_without_blaming_cloudflare(monkeypatch):
import shelfmark.download.http as http
def timeout(*args, **kwargs):
raise WaitingRoomTimeoutError("Anna's Archive waiting room did not finish within 300s")
monkeypatch.setattr(http, "_is_cf_bypass_enabled", lambda: True)
monkeypatch.setattr(http, "_bypass_grace_seconds", lambda: 450)
monkeypatch.setattr(http, "get_bypassed_page", timeout)
selector = SimpleNamespace(rewrite=lambda url: url, last_failure=None)
result = http.html_get_page(URL, retry=10, selector=selector, use_bypasser=True)
assert result == ""
assert selector.last_failure == "Anna's Archive waiting room did not finish within 300s"
@pytest.mark.parametrize("docker_mode", [False, True])
@pytest.mark.parametrize("cached", [WAIT, READY])
def test_cached_pages_through_both_entry_points(monkeypatch, cached_http, docker_mode, cached):
response, cleared = cached_http
response.text = cached
monkeypatch.setattr(b.env, "DOCKERMODE", docker_mode)
monkeypatch.delenv(b._BYPASS_CHILD_ENV, raising=False)
calls = []
target = "_get_via_subprocess" if docker_mode else "_run_bypass_in_current_process"
monkeypatch.setattr(b, target, lambda url, *a, **kw: calls.append(url) or READY)
selector = SimpleNamespace(rewrite=lambda url: url)
assert b.get_bypassed_page(URL, selector) == READY
assert calls == ([URL] if cached == WAIT else [])
# A waiting room does not invalidate the clearance cookies it was served with.
cleared.assert_not_called()
def test_page_readiness_and_returned_html_belong_to_the_same_snapshot(monkeypatch):
old_page = "<html>DDOS-GUARD</html>"
page = SimpleNamespace(
evaluate=AsyncMock(
side_effect=[
_snapshot(old_page, title="DDOS-GUARD"),
_snapshot(),
]
)
)
# Reproduce navigation after the HTML read: a separate live-page check already
# sees a ready page. The old loop incorrectly returned the captured interstitial.
monkeypatch.setattr(b, "_read_page_source", AsyncMock(return_value=old_page))
monkeypatch.setattr(b, "_is_bypassed", AsyncMock(return_value=True))
monkeypatch.setattr(b.asyncio, "sleep", AsyncMock())
assert asyncio.run(b._wait_for_aa_download_page(page, URL, WAIT)) == READY
def test_cancellation_interrupts_a_stalled_browser_read(monkeypatch):
cancel = threading.Event()
async def stalled_read(_expression):
cancel.set()
await asyncio.Event().wait()
monkeypatch.setattr(b, "_AA_WAITING_ROOM_POLL_SECONDS", 0.01)
page = SimpleNamespace(evaluate=stalled_read)
with pytest.raises(BypassCancelledError):
asyncio.run(b._wait_for_aa_download_page(page, URL, WAIT, cancel))
def test_cancellation_after_read_wins_over_ready_result(monkeypatch):
cancel = threading.Event()
async def finish_and_cancel(_expression):
cancel.set()
return _snapshot()
page = SimpleNamespace(evaluate=finish_and_cancel)
with pytest.raises(BypassCancelledError):
asyncio.run(b._wait_for_aa_download_page(page, URL, WAIT, cancel))
@pytest.mark.parametrize("transient", [TimeoutError(), TypeError("missing CDP result"), None])
def test_transient_read_failure_keeps_the_same_tab(monkeypatch, transient):
page = SimpleNamespace(evaluate=AsyncMock(side_effect=[transient, _snapshot()]))
monkeypatch.setattr(b.asyncio, "sleep", AsyncMock())
assert asyncio.run(b._wait_for_aa_download_page(page, URL, WAIT)) == READY
assert page.evaluate.await_count == 2
def test_slow_snapshot_is_not_cancelled_and_reissued(monkeypatch):
async def slow_read(_expression):
await asyncio.sleep(0.03)
return _snapshot()
monkeypatch.setattr(b, "_AA_WAITING_ROOM_POLL_SECONDS", 0.005)
page = SimpleNamespace(evaluate=AsyncMock(side_effect=slow_read))
assert asyncio.run(b._wait_for_aa_download_page(page, URL, WAIT)) == READY
page.evaluate.assert_awaited_once()