From bc03ad062ec7746319ef3f7f495dd78cda0d1af3 Mon Sep 17 00:00:00 2001 From: splitsec2 <35583321+splitsec2@users.noreply.github.com> Date: Sun, 20 Sep 2026 21:04:22 -0600 Subject: [PATCH] fix(http): keep the host of a protocol-relative download link (#1368) `get_absolute_url()` replaced both `netloc` and `scheme` whenever either one was missing. A protocol-relative href such as `//cdn.example.org/f.epub`, scraped from a page on `https://annas-archive.org/...`, parses with a netloc and an empty scheme, so it came back pointing at the page's own host. The download then 404s and the source is skipped. Each field now falls back to the base URL only when the parsed URL does not supply it. Plain relative paths resolve exactly as before, which the control test covers. This affects the Z-Library, welib and generic download link handling in `release_sources/direct_download/annas_archive.py`. ## Verification - New `tests/download/test_http_absolute_url.py`: a protocol-relative link keeps its own host, and a plain `/path` still resolves against the base. The first fails on current main and passes here. - Full suite (3165), ruff, ruff format, basedpyright, vulture green. --- shelfmark/download/http.py | 4 +++- tests/download/test_http_absolute_url.py | 17 +++++++++++++++++ 2 files changed, 20 insertions(+), 1 deletion(-) create mode 100644 tests/download/test_http_absolute_url.py diff --git a/shelfmark/download/http.py b/shelfmark/download/http.py index 106c60c2..079f16fa 100644 --- a/shelfmark/download/http.py +++ b/shelfmark/download/http.py @@ -1104,5 +1104,7 @@ def get_absolute_url(base_url: str, url: str) -> str: parsed = urlparse(url) base = urlparse(base_url) if not parsed.netloc or not parsed.scheme: - parsed = parsed._replace(netloc=base.netloc, scheme=base.scheme) + parsed = parsed._replace( + netloc=parsed.netloc or base.netloc, scheme=parsed.scheme or base.scheme + ) return parsed.geturl() diff --git a/tests/download/test_http_absolute_url.py b/tests/download/test_http_absolute_url.py new file mode 100644 index 00000000..32f65029 --- /dev/null +++ b/tests/download/test_http_absolute_url.py @@ -0,0 +1,17 @@ +"""Tests for resolving a scraped link against the page it came from.""" + + +def test_get_absolute_url_keeps_a_protocol_relative_host(): + import shelfmark.download.http as http + + result = http.get_absolute_url("https://annas-archive.org/md5/abc", "//cdn.example.org/f.epub") + + assert result == "https://cdn.example.org/f.epub" + + +def test_get_absolute_url_resolves_a_relative_path_against_the_base(): + import shelfmark.download.http as http + + result = http.get_absolute_url("https://annas-archive.org/md5/abc", "/slow_download/abc/0/1") + + assert result == "https://annas-archive.org/slow_download/abc/0/1"