fix/group archive extracted audiobooks (#1261)

- fix: group multi-file audiobooks that arrive as an archive
- Surface the concrete reason when a direct-download fetch fails
This commit is contained in:
CaliBrain
2026-08-24 01:25:00 -04:00
committed by GitHub
parent 7d56624ab6
commit 95e34670f7
4 changed files with 104 additions and 16 deletions
+69 -11
View File
@@ -333,11 +333,26 @@ def html_get_page(
"""
# Normalise before the closures below capture it: they touch selector.last_failure,
# so it must be a concrete selector, not the Optional parameter.
selector = selector or network.AAMirrorSelector()
def _result(html: str, response_url: str) -> str | tuple[str, str]:
if include_response_url:
return html, response_url
return html
def _fail(reason: str, response_url: str) -> str | tuple[str, str]:
"""Record why the fetch is giving up, then return the empty result.
Every give-up path returns an empty page, which is all the caller used to
see. Stashing the concrete reason on the shared selector lets the caller
surface it (see release_sources.direct_download) rather than reporting the
same generic "network restricted or mirrors blocked" for every cause.
"""
selector.last_failure = reason
return _result("", response_url)
def _run_bypasser(bypass_url: str) -> str | tuple[str, str]:
"""Run the active bypasser for one URL and return its result.
@@ -355,7 +370,13 @@ def html_get_page(
# bypasser that fails to load is still reported as a bypasser error.
request_activity_grace(status_callback, _bypass_grace_seconds())
result = get_bypassed_page(bypass_url, selector, cancel_flag)
return _result(result or "", bypass_url)
if result:
return _result(result, bypass_url)
return _fail(
"The protection bypasser returned an empty page — the challenge was "
"not solved. Check that FlareSolverr/the CF bypasser is reachable.",
bypass_url,
)
except _BYPASSER_ERRORS as e:
logger.warning("Bypasser error: %s: %s", type(e).__name__, e)
# Surface the real reason. Without this the caller only sees an empty
@@ -366,7 +387,9 @@ def html_get_page(
status_callback("error", f"Bypass failed: {type(e).__name__}: {e}")
except _STATUS_CALLBACK_ERRORS:
logger.debug("Bypass error status callback failed", exc_info=True)
return _result("", bypass_url)
if isinstance(e, BypassCancelledError):
return _fail("The protection bypass was cancelled.", bypass_url)
return _fail(f"The protection bypasser failed: {type(e).__name__}: {e}", bypass_url)
finally:
release_activity_grace(status_callback)
@@ -408,19 +431,21 @@ def html_get_page(
retry_limit = (
retry if retry is not None else (configured_retry if configured_retry is not None else 1)
)
selector = selector or network.AAMirrorSelector()
original_url = url
current_url = selector.rewrite(original_url)
use_bypasser_now = use_bypasser
# Survives across attempts so a cookie won once is still presented on later retries.
handshake_cookies: dict[str, str] = {}
handshake_retries = 0
# Last transport error seen, so the exhausted-retries path can name the real
# cause (timeout, connection refused, DNS, ...) instead of a generic message.
last_error: Exception | None = None
for attempt in range(1, retry_limit + 1):
# Check for cancellation before each attempt
if cancel_flag and cancel_flag.is_set():
logger.info("html_get_page cancelled before attempt %s", attempt)
return _result("", current_url)
return _fail("The request was cancelled.", current_url)
cookies: dict[str, str] = {}
try:
@@ -522,7 +547,12 @@ def html_get_page(
redirect_host,
current_url,
)
return _result("", current_url)
return _fail(
f"The configured mirror {current_host} redirected to "
f"{redirect_host}; it may be down or seized. Point MIRROR at "
"a working host or switch to auto mode.",
current_url,
)
new_url = _try_rotation(original_url, current_url, selector)
if new_url:
@@ -541,7 +571,11 @@ def html_get_page(
redirect_host,
current_url,
)
return _result("", current_url)
return _fail(
"Every Anna's Archive mirror redirected away to a dead host — "
"all configured mirrors are unreachable.",
current_url,
)
# Same-host redirect (relative or absolute) - follow manually.
# DDoS-Guard gates AA /search behind a cookie probe: the 302 to
@@ -572,7 +606,12 @@ def html_get_page(
logger.warning(
"Redirect loop and no bypasser available, giving up: %s", current_url
)
return _result("", current_url)
return _fail(
"Anna's Archive is behind a protection challenge (endless "
"redirect loop) and no bypasser is enabled to solve it. Enable "
"FlareSolverr/the CF bypasser.",
current_url,
)
current_url = redirect_url
continue
@@ -582,6 +621,7 @@ def html_get_page(
return _result(response.text, response.url)
except Exception as e:
last_error = e
status = _get_status_code(e)
# The same DDoS-Guard rescue, for the loops the manual AA follower above hands
@@ -608,7 +648,10 @@ def html_get_page(
current_url = new_url
continue
logger.warning("403 error, mirrors exhausted: %s", current_url)
return _result("", current_url)
return _fail(
"Anna's Archive returned 403 (blocked) and all mirrors are exhausted.",
current_url,
)
if _is_cf_bypass_enabled() and not use_bypasser_now:
# Before switching to bypasser, check if cookies have become available
@@ -642,12 +685,18 @@ def html_get_page(
# Same reasoning as the redirect-loop handoffs.
return _run_bypasser(current_url)
logger.warning("403 error, giving up: %s", current_url)
return _result("", current_url)
return _fail(
"Anna's Archive returned 403 (blocked) and no bypasser is enabled "
"to solve the protection challenge.",
current_url,
)
# 404 = Not found
if status == _HTTP_STATUS_NOT_FOUND:
logger.warning("404 error: %s", current_url)
return _result("", current_url)
return _fail(
f"Anna's Archive returned 404 Not Found for {current_url}.", current_url
)
# Try mirror/DNS rotation on retryable errors. A failure that proves the
# mirror is unusable also drops it from this process's rotation, so the
@@ -676,7 +725,16 @@ def html_get_page(
else:
logger.exception("Giving up after %s attempts: %s", retry_limit, current_url)
return _result("", current_url)
if last_error is not None:
return _fail(
f"Could not reach Anna's Archive after {retry_limit} attempt(s): "
f"{type(last_error).__name__}: {last_error}",
current_url,
)
return _fail(
"Could not reach Anna's Archive — all mirrors were exhausted without a usable response.",
current_url,
)
def download_url(
+5
View File
@@ -1486,6 +1486,11 @@ class AAMirrorSelector:
def __init__(self) -> None:
"""Initialize mirror state from the current AA configuration."""
# Set by html_get_page at each give-up path so a caller that only sees the
# returned empty page can still report *why* the fetch produced nothing
# (403, 404, redirect loop, bypasser error, mirrors exhausted, ...) instead
# of a blanket "network restricted" guess. None means "no failure recorded".
self.last_failure: str | None = None
self._ensure_fresh_state(reset_attempts=True)
def _ensure_fresh_state(self, *, reset_attempts: bool = False) -> None:
+11 -5
View File
@@ -590,9 +590,13 @@ def _fetch_search_table(url: str, selector: network.AAMirrorSelector) -> tuple[s
attempt_url, selector=selector, allow_bypasser_fallback=True
)
if not response:
# Network/mirror exhaustion path bubbles up so API can notify clients
msg = "Unable to reach download source. Network restricted or mirrors are blocked."
raise SearchUnavailableError(msg)
# Network/mirror exhaustion path bubbles up so API can notify clients.
# html_get_page records the concrete give-up reason on the selector; fall
# back to the generic line only if nothing was recorded.
detail = getattr(selector, "last_failure", None) or (
"Network restricted or mirrors are blocked."
)
raise SearchUnavailableError(f"Unable to reach download source. {detail}")
html = _html_response_text(response)
soup = BeautifulSoup(html, "html.parser")
@@ -747,8 +751,10 @@ def get_book_info(book_id: str, *, fetch_download_count: bool = True) -> BrowseR
html = downloader.html_get_page(url, selector=selector, allow_bypasser_fallback=True)
if not html:
msg = "Unable to reach download source. Network restricted or mirrors are blocked."
raise SearchUnavailableError(msg)
detail = getattr(selector, "last_failure", None) or (
"Network restricted or mirrors are blocked."
)
raise SearchUnavailableError(f"Unable to reach download source. {detail}")
soup = BeautifulSoup(_html_response_text(html), "html.parser")
@@ -110,3 +110,22 @@ def test_unreachable_mirror_raises_search_unavailable(monkeypatch):
except dd.SearchUnavailableError:
return
raise AssertionError("expected SearchUnavailableError")
def test_recorded_failure_reason_is_surfaced_to_the_caller(monkeypatch):
"""The concrete give-up reason html_get_page stashed replaces the generic line."""
import shelfmark.release_sources.direct_download as dd
reason = "Anna's Archive returned 403 (blocked) and no bypasser is enabled."
def fake_get(url, *, selector=None, **_kwargs):
selector.last_failure = reason
return ""
monkeypatch.setattr(dd.downloader, "html_get_page", fake_get)
monkeypatch.setattr(dd.network, "get_available_aa_urls", lambda: ["a"])
selector = _Selector(["https://real.test"])
selector.last_failure = None
with pytest.raises(dd.SearchUnavailableError, match="403 .blocked."):
dd._fetch_search_table("https://real.test/search?q=dune", selector)