fix: resolve remaining "Could not determine torrent hash from URL" failures (#1012) (#1108)

This commit is contained in:
CaliBrain
2026-07-08 14:20:31 -04:00
committed by GitHub
parent d7a21ea248
commit 9b1d4322b7
6 changed files with 443 additions and 44 deletions
+54 -4
View File
@@ -258,7 +258,7 @@ class QBittorrentClient(DownloadClient):
self._tags = _normalize_tags(config.get("QBITTORRENT_TAG", []))
def _get_torrents_info(
self, torrent_hash: str | None = None
self, torrent_hash: str | None = None, category: str | None = None
) -> tuple[list[SimpleNamespace], str | None]:
"""Get torrent info using GET.
@@ -267,6 +267,7 @@ class QBittorrentClient(DownloadClient):
- Keep "API/auth/connect" errors distinct from "torrent missing".
- If a hash-specific query returns empty, fall back to listing by category
and matching locally.
- Without a hash, `category` narrows the listing to that category.
Returns:
(torrents, error_message)
@@ -301,6 +302,8 @@ class QBittorrentClient(DownloadClient):
primary_params: dict[str, str] = {}
if torrent_hash:
primary_params["hashes"] = torrent_hash
elif category:
primary_params["category"] = category
response = do_request(primary_params)
torrents, error = parse_response(response, request_params=primary_params)
@@ -358,6 +361,44 @@ class QBittorrentClient(DownloadClient):
else:
return torrents, None
def _list_category_hashes(self, category: str | None) -> set[str] | None:
"""Snapshot the hashes qBittorrent currently reports for a category."""
torrents, error = self._get_torrents_info(category=category)
if error:
logger.debug("Could not snapshot qBittorrent torrents: %s", error)
return None
return {str(torrent.hash).lower() for torrent in torrents if getattr(torrent, "hash", None)}
def _discover_added_torrent_hash(
self,
name: str,
category: str | None,
known_hashes: set[str] | None,
) -> str | None:
"""Recover the hash of a torrent that was added without a known info_hash.
A `known_hashes` of None means the pre-add snapshot failed, so only a
torrent matching the requested rename can identify the new arrival.
"""
for _ in range(20):
torrents, error = self._get_torrents_info(category=category)
if error:
logger.debug("qBittorrent hash discovery: %s", error)
else:
new_torrents = [
torrent
for torrent in torrents
if getattr(torrent, "hash", None)
and (known_hashes is None or str(torrent.hash).lower() not in known_hashes)
]
for torrent in new_torrents:
if getattr(torrent, "name", None) == name:
return str(torrent.hash).lower()
if known_hashes is not None and len(new_torrents) == 1:
return str(new_torrents[0].hash).lower()
time.sleep(0.5)
return None
@staticmethod
def is_configured() -> bool:
"""Check if qBittorrent is configured and selected as the torrent client."""
@@ -425,6 +466,10 @@ class QBittorrentClient(DownloadClient):
expected_hash = torrent_info.info_hash
torrent_data = torrent_info.torrent_data
known_hashes: set[str] | None = None
if not expected_hash:
known_hashes = self._list_category_hashes(category)
# Per-torrent seeding limits from indexer
seeding_time_limit_value = kwargs.get("seeding_time_limit")
seeding_time_limit = coerce_optional_int(seeding_time_limit_value)
@@ -459,12 +504,17 @@ class QBittorrentClient(DownloadClient):
result_text = _normalize_add_result(result)
logger.debug("qBittorrent add result: %s", result_text)
if not expected_hash:
_raise_runtime_error("Could not determine torrent hash from URL")
if _is_explicit_add_failure(result):
_raise_runtime_error(f"Failed to add torrent: {result_text}")
if not expected_hash:
# qBittorrent fetches .torrent URLs itself, so the add can succeed
# even when no hash could be extracted up front. Recover it by
# watching for the new torrent to appear.
expected_hash = self._discover_added_torrent_hash(name, category, known_hashes)
if not expected_hash:
_raise_runtime_error("Could not determine torrent hash from URL")
# Some qBittorrent-compatible clients return HTTP 200 with an empty body
# instead of qBittorrent's literal "Ok." response. Prefer verifying that
# the torrent becomes visible over trusting the response body alone.
+60
View File
@@ -4,6 +4,7 @@ Uses xmlrpc to communicate with rTorrent's RPC interface.
"""
import ssl
import time
import xmlrpc.client as stdlib_xmlrpc_client
from typing import Any, NoReturn, Protocol, cast
from urllib.parse import urlparse
@@ -160,6 +161,10 @@ class RTorrentClient(DownloadClient):
try:
torrent_info = extract_torrent_info(url, expected_hash=expected_hash)
known_hashes: set[str] | None = None
if not (torrent_info.info_hash or expected_hash):
known_hashes = self._list_torrent_hashes()
commands = []
is_audiobook = kwargs.get("content_type") == "audiobook"
@@ -195,6 +200,11 @@ class RTorrentClient(DownloadClient):
self._rpc.load.start("", add_url, ";".join(commands))
torrent_hash = torrent_info.info_hash or expected_hash
if not torrent_hash:
# rTorrent fetches .torrent URLs itself, so the add can succeed
# even when no hash could be extracted up front. Recover it by
# watching for the new download to appear.
torrent_hash = self._discover_added_torrent_hash(name, label, known_hashes)
if not torrent_hash:
_raise_runtime_error("Could not determine torrent hash from URL")
@@ -387,6 +397,56 @@ class RTorrentClient(DownloadClient):
except _RTORRENT_CLIENT_ERRORS:
return "/downloads"
def _list_torrent_hashes(self) -> set[str] | None:
"""Snapshot the hashes rTorrent currently reports."""
try:
all_torrents = self._rpc.d.multicall2("", "", "d.hash=")
except _RTORRENT_CLIENT_ERRORS as e:
logger.debug("Could not snapshot rTorrent downloads: %s", e)
return None
return {str(row[0]).lower() for row in all_torrents if row and row[0]}
def _discover_added_torrent_hash(
self,
name: str,
label: str,
known_hashes: set[str] | None,
) -> str | None:
"""Recover the hash of a torrent that was added without a known info_hash.
rTorrent fetches .torrent URLs itself, so the add can succeed even when
no hash could be extracted up front. A `known_hashes` of None means the
pre-add snapshot failed, so only an exact name match can identify the
new arrival.
"""
for _ in range(20):
try:
all_torrents = self._rpc.d.multicall2("", "", "d.hash=", "d.name=", "d.custom1=")
except _RTORRENT_CLIENT_ERRORS as e:
logger.debug("rTorrent hash discovery: %s", e)
else:
new_torrents = [
row
for row in all_torrents
if row
and row[0]
and (known_hashes is None or str(row[0]).lower() not in known_hashes)
]
# The label set at add time distinguishes concurrent arrivals,
# but rTorrent may not have applied it yet, so it only ever
# narrows a non-empty candidate list.
if label:
labeled = [row for row in new_torrents if len(row) > 2 and row[2] == label]
if labeled:
new_torrents = labeled
for row in new_torrents:
if len(row) > 1 and row[1] == name:
return str(row[0]).lower()
if known_hashes is not None and len(new_torrents) == 1:
return str(new_torrents[0][0]).lower()
time.sleep(0.5)
return None
def _get_torrent_path(self, download_id: str) -> str | None:
"""Get the file path of a torrent by hash.
+37 -36
View File
@@ -19,6 +19,7 @@ from shelfmark.download.network import get_ssl_verify
logger = setup_logger(__name__)
_MAGNET_RESPONSE_MAX_BYTES = 2000
_TORRENT_FETCH_MAX_REDIRECTS = 5
_BASE32_BTMH_TAG_BYTES = 34
_BTIH_INFO_BYTE_HEX = 0x20
_BTIH_PREFIX_BYTE = 0x12
@@ -98,19 +99,18 @@ def extract_torrent_info(
# A release source can legitimately hand us a download URL on a different
# origin than the configured Prowlarr/Newznab endpoint (e.g. a direct
# tracker link, or Prowlarr reached through a separate proxy). We still need
# to fetch the .torrent to recover the info_hash when the source did not
# provide one, so the prefetch runs regardless of origin. The Prowlarr API
# key, however, is only ever sent to a trusted origin so it can never leak
# to an arbitrary indexer/tracker host.
trusted_origin = _is_trusted_torrent_fetch_url(url)
# tracker link, or Prowlarr reached through a separate proxy), and a trusted
# Prowlarr download URL commonly redirects to the indexer's own download
# link. We still need to fetch the .torrent to recover the info_hash when
# the source did not provide one, so the prefetch runs regardless of origin
# and follows cross-origin redirects. The Prowlarr API key, however, is
# re-evaluated per hop and only ever sent to a trusted origin so it can
# never leak to an arbitrary indexer/tracker host.
headers: dict[str, str] = {"Accept": "application/x-bittorrent"}
if trusted_origin:
# TODO(shelfmark): Move this source-specific Prowlarr auth handling into a source hook.
api_key = str(config.get("PROWLARR_API_KEY", "") or "").strip()
if api_key:
headers["X-Api-Key"] = api_key
# TODO(shelfmark): Move this source-specific Prowlarr auth handling into a source hook.
api_key = str(config.get("PROWLARR_API_KEY", "") or "").strip()
if api_key:
headers["X-Api-Key"] = api_key
def resolve_url(current: str, location: str) -> str:
if not location:
@@ -121,19 +121,28 @@ def extract_torrent_info(
try:
logger.debug("Fetching torrent file from: %s...", url[:80])
# Use allow_redirects=False to handle magnet link redirects manually
# Some indexers redirect download URLs to magnet links
resp = requests.get(
url,
timeout=30,
allow_redirects=False,
headers=headers,
verify=get_ssl_verify(url),
)
# Redirects are followed manually: some indexers redirect download URLs
# to magnet links, and each hop must decide anew whether it may see the
# API key.
current_url = url
redirects_remaining = _TORRENT_FETCH_MAX_REDIRECTS
while True:
request_headers = dict(headers)
if not _is_trusted_torrent_fetch_url(current_url):
request_headers.pop("X-Api-Key", None)
# Check if this is a redirect to a magnet link
if resp.status_code in (301, 302, 303, 307, 308):
redirect_url = resolve_url(url, resp.headers.get("Location", ""))
resp = requests.get(
current_url,
timeout=30,
allow_redirects=False,
headers=request_headers,
verify=get_ssl_verify(current_url),
)
if resp.status_code not in (301, 302, 303, 307, 308):
break
redirect_url = resolve_url(current_url, resp.headers.get("Location", ""))
if redirect_url.startswith("magnet:"):
logger.debug("Download URL redirected to magnet link")
info_hash = extract_hash_from_magnet(redirect_url)
@@ -145,20 +154,12 @@ def extract_torrent_info(
is_magnet=True,
magnet_url=redirect_url,
)
if not _is_trusted_torrent_fetch_url(redirect_url):
logger.debug(
"Skipping torrent prefetch redirect to untrusted URL: %s...",
redirect_url[:80],
)
if redirects_remaining <= 0:
logger.debug("Too many redirects fetching torrent file: %s...", url[:80])
return TorrentInfo(info_hash=expected_hash, torrent_data=None, is_magnet=False)
# Not a magnet redirect, follow it manually
redirects_remaining -= 1
logger.debug("Following redirect to: %s...", redirect_url[:80])
resp = requests.get(
redirect_url,
timeout=30,
headers=headers,
verify=get_ssl_verify(redirect_url),
)
current_url = redirect_url
resp.raise_for_status()
torrent_data = resp.content
+113
View File
@@ -790,6 +790,119 @@ class TestQBittorrentClientAddDownload:
expected_hash=expected_hash,
)
def test_add_download_discovers_hash_when_extraction_fails(self, monkeypatch):
"""Regression for #1012: a URL add without a hash adopts the new torrent's hash.
qBittorrent fetches .torrent URLs itself, so the add succeeds even when
the prefetch could not determine the info_hash; the client must recover
the hash from the torrent that appears instead of raising.
"""
config_values = {
"QBITTORRENT_URL": "http://localhost:8080",
"QBITTORRENT_USERNAME": "admin",
"QBITTORRENT_PASSWORD": "password",
"QBITTORRENT_CATEGORY": "books",
}
monkeypatch.setattr(
"shelfmark.download.clients.qbittorrent.config.get",
lambda key, default="": config_values.get(key, default),
)
discovered_hash = "3b245504cf5f11bbdbe1201cea6a6bf45aee1bc0"
existing = MockTorrent(
hash_val="a1b2c3d4e5f6a1b2c3d4e5f6a1b2c3d4e5f6a1b2", name="Existing Torrent"
)
added = MockTorrent(hash_val=discovered_hash, name="Test Download")
mock_client_instance = MagicMock()
mock_client_instance.torrents_add.return_value = "Ok."
torrents_before_add = create_mock_session_response([existing])
torrents_after_add = create_mock_session_response([existing, added])
properties_ok = create_mock_session_response({}, status_code=200)
def session_get(request_url, params=None, timeout=None):
if request_url.endswith("/torrents/properties"):
return properties_ok
if mock_client_instance.torrents_add.called:
return torrents_after_add
return torrents_before_add
mock_client_instance._session.get.side_effect = session_get
mock_client_class = MagicMock(return_value=mock_client_instance)
with patch.dict("sys.modules", {"qbittorrentapi": MagicMock(Client=mock_client_class)}):
import importlib
import shelfmark.download.clients.qbittorrent as qb_module
importlib.reload(qb_module)
with patch(
"shelfmark.download.clients.qbittorrent.extract_torrent_info",
autospec=True,
) as mock_extract:
mock_extract.return_value = TorrentInfo(
info_hash=None,
torrent_data=None,
is_magnet=False,
magnet_url=None,
)
client = qb_module.QBittorrentClient()
result = client.add_download(
"http://tracker.example/download/book.torrent", "Test Download"
)
assert result == discovered_hash
add_kwargs = mock_client_instance.torrents_add.call_args.kwargs
assert add_kwargs["urls"] == "http://tracker.example/download/book.torrent"
def test_add_download_raises_when_hash_never_discovered(self, monkeypatch):
"""Keep failing loudly when no hash is known and no new torrent appears."""
config_values = {
"QBITTORRENT_URL": "http://localhost:8080",
"QBITTORRENT_USERNAME": "admin",
"QBITTORRENT_PASSWORD": "password",
"QBITTORRENT_CATEGORY": "books",
}
monkeypatch.setattr(
"shelfmark.download.clients.qbittorrent.config.get",
lambda key, default="": config_values.get(key, default),
)
monkeypatch.setattr(
"shelfmark.download.clients.qbittorrent.time.sleep", lambda _seconds: None
)
mock_client_instance = MagicMock()
mock_client_instance.torrents_add.return_value = "Ok."
mock_client_instance._session.get.return_value = create_mock_session_response([])
mock_client_class = MagicMock(return_value=mock_client_instance)
with patch.dict("sys.modules", {"qbittorrentapi": MagicMock(Client=mock_client_class)}):
import importlib
import shelfmark.download.clients.qbittorrent as qb_module
importlib.reload(qb_module)
with patch(
"shelfmark.download.clients.qbittorrent.extract_torrent_info",
autospec=True,
) as mock_extract:
mock_extract.return_value = TorrentInfo(
info_hash=None,
torrent_data=None,
is_magnet=False,
magnet_url=None,
)
client = qb_module.QBittorrentClient()
with pytest.raises(RuntimeError, match="Could not determine torrent hash"):
client.add_download(
"http://tracker.example/download/book.torrent", "Test Download"
)
def test_add_download_creates_category(self, monkeypatch):
"""Test that add_download creates category if needed."""
config_values = {
+105
View File
@@ -348,6 +348,111 @@ class TestRTorrentClientAddDownload:
assert "RPC Error" in str(excinfo.value)
def test_add_download_discovers_hash_when_extraction_fails(self, monkeypatch):
"""Regression for #1012: a URL add without a hash adopts the new download's hash.
rTorrent fetches .torrent URLs itself, so the add succeeds even when the
prefetch could not determine the info_hash; the client must recover the
hash from the download that appears instead of raising.
"""
config_values = {
"RTORRENT_URL": "http://localhost:8080/RPC2",
"RTORRENT_USERNAME": "",
"RTORRENT_PASSWORD": "",
"RTORRENT_DOWNLOAD_DIR": "/downloads",
"RTORRENT_LABEL": "cwabd",
}
monkeypatch.setattr(
"shelfmark.download.clients.rtorrent.config.get",
make_config_getter(config_values),
)
existing_hash = "A1B2C3D4E5F6A1B2C3D4E5F6A1B2C3D4E5F6A1B2"
discovered_hash = "3B245504CF5F11BBDBE1201CEA6A6BF45AEE1BC0"
mock_rpc = MagicMock()
# First multicall2 is the pre-add snapshot; the second is the discovery
# poll after load.start, where the new download has appeared.
mock_rpc.d.multicall2.side_effect = [
[[existing_hash]],
[
[existing_hash, "Existing Torrent", "cwabd"],
[discovered_hash, "Test Torrent", "cwabd"],
],
]
mock_xmlrpc = create_mock_xmlrpc_module()
mock_xmlrpc.ServerProxy.return_value = mock_rpc
mock_torrent_info = MagicMock()
mock_torrent_info.torrent_data = None
mock_torrent_info.magnet_url = None
mock_torrent_info.info_hash = None
mock_torrent_info.is_magnet = False
with patch.dict("sys.modules", {"xmlrpc.client": mock_xmlrpc}):
with patch(
"shelfmark.download.clients.torrent_utils.extract_torrent_info",
return_value=mock_torrent_info,
):
if "shelfmark.download.clients.rtorrent" in sys.modules:
del sys.modules["shelfmark.download.clients.rtorrent"]
from shelfmark.download.clients.rtorrent import (
RTorrentClient,
)
client = RTorrentClient()
result_hash = client.add_download(
"http://tracker.example/download/book.torrent", "Test Torrent"
)
assert result_hash == discovered_hash.lower()
mock_rpc.load.start.assert_called_once()
def test_add_download_raises_when_hash_never_discovered(self, monkeypatch):
"""Keep failing loudly when no hash is known and no new download appears."""
config_values = {
"RTORRENT_URL": "http://localhost:8080/RPC2",
"RTORRENT_USERNAME": "",
"RTORRENT_PASSWORD": "",
"RTORRENT_DOWNLOAD_DIR": "/downloads",
"RTORRENT_LABEL": "cwabd",
}
monkeypatch.setattr(
"shelfmark.download.clients.rtorrent.config.get",
make_config_getter(config_values),
)
existing_hash = "A1B2C3D4E5F6A1B2C3D4E5F6A1B2C3D4E5F6A1B2"
mock_rpc = MagicMock()
mock_rpc.d.multicall2.return_value = [[existing_hash, "Existing Torrent", "cwabd"]]
mock_xmlrpc = create_mock_xmlrpc_module()
mock_xmlrpc.ServerProxy.return_value = mock_rpc
mock_torrent_info = MagicMock()
mock_torrent_info.torrent_data = None
mock_torrent_info.magnet_url = None
mock_torrent_info.info_hash = None
mock_torrent_info.is_magnet = False
with patch.dict("sys.modules", {"xmlrpc.client": mock_xmlrpc}):
with patch(
"shelfmark.download.clients.torrent_utils.extract_torrent_info",
return_value=mock_torrent_info,
):
if "shelfmark.download.clients.rtorrent" in sys.modules:
del sys.modules["shelfmark.download.clients.rtorrent"]
from shelfmark.download.clients import rtorrent as rtorrent_module
monkeypatch.setattr(rtorrent_module.time, "sleep", lambda _seconds: None)
client = rtorrent_module.RTorrentClient()
with pytest.raises(RuntimeError, match="Could not determine torrent hash"):
client.add_download(
"http://tracker.example/download/book.torrent", "Test Torrent"
)
class TestRTorrentClientAudiobookLabel:
"""Regression tests for issue #1025 — rTorrent audiobook label selection."""
+74 -4
View File
@@ -499,15 +499,84 @@ class TestExtractTorrentInfo:
assert result.torrent_data == torrent_data
mock_get.assert_called_once()
def test_does_not_follow_trusted_torrent_url_redirect_to_untrusted_host(self, monkeypatch):
"""Trusted HTTP prefetch does not continue through arbitrary redirects."""
def test_follows_trusted_redirect_to_untrusted_host_without_api_key(self, monkeypatch):
"""Regression for #1012 on v1.3.2.
Prowlarr's download endpoint commonly redirects to the indexer's own
download link on another origin. The prefetch must follow that redirect
(it is often the only way to learn the info_hash) but must not forward
the Prowlarr API key to the untrusted origin.
"""
info_dict = {
b"name": b"book.txt",
b"length": 100,
b"piece length": 16384,
b"pieces": b"\x00" * 20,
}
torrent_data = bencode_encode({b"info": info_dict})
expected_hash = hashlib.sha1(bencode_encode(info_dict)).hexdigest().lower()
config_values = {
"PROWLARR_URL": "https://prowlarr.example",
"PROWLARR_API_KEY": "secret",
}
monkeypatch.setattr(
"shelfmark.download.clients.torrent_utils.config.get",
lambda key, default="": config_values.get(key, default),
)
redirect = MagicMock(status_code=302)
redirect.headers = {"Location": "https://tracker.example/download/book.torrent"}
final = MagicMock(status_code=200, content=torrent_data)
final.raise_for_status = MagicMock()
mock_get = MagicMock(side_effect=[redirect, final])
monkeypatch.setattr("shelfmark.download.clients.torrent_utils.requests.get", mock_get)
# No expected_hash supplied: the hash can only come from the prefetch.
result = extract_torrent_info(
"https://prowlarr.example/1/download?apikey=secret&indexer=7",
fetch_torrent=True,
)
assert result.info_hash == expected_hash
assert result.torrent_data == torrent_data
assert result.is_magnet is False
assert mock_get.call_count == 2
trusted_headers = mock_get.call_args_list[0].kwargs["headers"]
untrusted_headers = mock_get.call_args_list[1].kwargs["headers"]
assert trusted_headers.get("X-Api-Key") == "secret"
assert "X-Api-Key" not in untrusted_headers
def test_redirect_to_magnet_link_returns_magnet_info(self, monkeypatch):
"""A download URL redirecting to a magnet link yields the magnet's hash."""
magnet = "magnet:?xt=urn:btih:3b245504cf5f11bbdbe1201cea6a6bf45aee1bc0&dn=test"
monkeypatch.setattr(
"shelfmark.download.clients.torrent_utils.config.get",
lambda key, default="": "https://prowlarr.example" if key == "PROWLARR_URL" else "",
)
response = MagicMock(status_code=302)
response.headers = {"Location": magnet}
mock_get = MagicMock(return_value=response)
monkeypatch.setattr("shelfmark.download.clients.torrent_utils.requests.get", mock_get)
result = extract_torrent_info(
"https://prowlarr.example/1/download?apikey=secret&indexer=7",
fetch_torrent=True,
)
assert result.is_magnet is True
assert result.magnet_url == magnet
assert result.info_hash == "3b245504cf5f11bbdbe1201cea6a6bf45aee1bc0"
mock_get.assert_called_once()
def test_gives_up_after_too_many_redirects(self, monkeypatch):
"""A redirect loop falls back to the expected hash instead of spinning."""
expected_hash = "3b245504cf5f11bbdbe1201cea6a6bf45aee1bc0"
monkeypatch.setattr(
"shelfmark.download.clients.torrent_utils.config.get",
lambda key, default="": "https://prowlarr.example" if key == "PROWLARR_URL" else "",
)
response = MagicMock(status_code=302)
response.headers = {"Location": "https://attacker.example/book.torrent"}
response.headers = {"Location": "https://tracker.example/loop"}
mock_get = MagicMock(return_value=response)
monkeypatch.setattr("shelfmark.download.clients.torrent_utils.requests.get", mock_get)
@@ -520,7 +589,8 @@ class TestExtractTorrentInfo:
assert result.info_hash == expected_hash
assert result.torrent_data is None
assert result.is_magnet is False
mock_get.assert_called_once()
# Initial request plus the maximum of five followed redirects.
assert mock_get.call_count == 6
class TestExtractHashFromMagnet: