fix(mam): keep the session ID on MAM and rerun Prowlarr's exact search (#1399)

Follow-up to #1390.

- The origin came from each result's infoUrl, matched by a regex that
did
  not check the host, and the mam_id cookie had no domain. Any Prowlarr
  indexer returning a URL like https://evil.example?myanonamouse.net/t/1
  sent the session ID to evil.example. Requests now always go to
https://www.myanonamouse.net (Prowlarr's only MAM URL), with the cookie
  as a header and redirects off. Only results from Prowlarr's
  MyAnonamouse indexer are looked up, and their URLs must be on
  myanonamouse.net.
- The lookup searched every category and read one page, so for common
  titles most of Prowlarr's results were missed (a "Dune" audiobook
  search: 56 audiobooks on the first 100 of 328 matches). It now reruns
  Prowlarr's exact search: the same query clean-up, the MAM main
  categories behind the Torznab categories searched (13/15/16 for
  audiobooks, 14 for e-books, all once expanded), and the MAM indexer's
  own search type, search-in options and languages. Further pages are
  read while IDs are missing, page 1 of every title first, at most 4
  requests per search.
- Failed requests back off for 1, 2, 4 ... up to 30 minutes. The 10th
  consecutive failure stops enrichment until Test MAM Session passes,
  the session ID changes, or Shelfmark restarts.
- The detail cache prunes expired entries instead of growing for as long
  as Shelfmark runs.
This commit is contained in:
CaliBrain
2026-09-25 21:21:54 -04:00
committed by GitHub
parent efb1f66bc3
commit 893f7bdb92
6 changed files with 760 additions and 111 deletions
+5 -1
View File
@@ -35,4 +35,8 @@ Compare it with the IP listed next to the session on MAM's security page.
## How it works
After a Prowlarr search, Shelfmark sends the same query text to MAM's JSON search API (normally one request) and matches torrents back to Prowlarr's results by their MAM torrent ID. Lookups are cached for an hour, stay inside the Prowlarr search time budget, and never fail the search: if MAM errors or rejects the session, the release list simply shows without the extra columns filled in.
After a Prowlarr search, Shelfmark reruns the search Prowlarr sent to MyAnonamouse's JSON search API: the same cleaned-up query text, the same categories (audiobooks, or e-books), and your MyAnonamouse indexer's own options from Prowlarr (search type, search in description/series/filenames, languages). That normally returns the same torrents in one request per title. Shelfmark then matches them back to Prowlarr's results by their MAM torrent ID. Only results from the MyAnonamouse indexer are looked up, and the session ID is only ever sent to `https://www.myanonamouse.net`.
If some torrents are missing from the first page, Shelfmark reads further pages, up to 4 requests per search. Lookups are cached for an hour, stay inside the Prowlarr search time budget, and never fail the search: if MAM errors or rejects the session, the release list simply shows without the extra columns filled in.
After a failed request (a 403, a timeout, an unexpected reply), Shelfmark waits before trying MAM again: 1 minute, then 2, 4 and so on, up to 30 minutes. After 10 failures in a row it stops using MAM. **Test MAM Session** (once it succeeds), a new session ID, or a restart turns it back on. Each failure is logged with the reason.
+16 -10
View File
@@ -107,7 +107,8 @@ def _normalize_json_object_list(payload: object, *, context: str) -> list[dict[s
return [_normalize_json_object(item, context=context) for item in payload]
def _get_field_value(fields: object, name: str) -> object | None:
def get_indexer_field(fields: object, name: str) -> object | None:
"""Return a setting's value from a Prowlarr indexer's "fields" list."""
if not isinstance(fields, list):
return None
@@ -120,6 +121,17 @@ def _get_field_value(fields: object, name: str) -> object | None:
return None
def is_myanonamouse_indexer(indexer: Mapping[str, Any]) -> bool:
"""Whether a Prowlarr indexer record is the MyAnonamouse implementation."""
implementation = str(
indexer.get("implementation")
or indexer.get("implementationName")
or indexer.get("definitionName")
or ""
)
return implementation.strip().lower() == "myanonamouse"
class ProwlarrClient:
"""Client for interacting with the Prowlarr API."""
@@ -300,14 +312,8 @@ class ProwlarrClient:
if restrict_to is not None and idx_id_int not in restrict_to:
continue
impl = str(
idx.get("implementation")
or idx.get("implementationName")
or idx.get("definitionName")
or ""
)
# Currently only MyAnonamouse provides consistently rich Torznab metadata.
if impl.strip().lower() == "myanonamouse":
if is_myanonamouse_indexer(idx):
enriched_ids.append(idx_id_int)
return enriched_ids
@@ -339,9 +345,9 @@ class ProwlarrClient:
continue
fields = idx.get("fields")
ratio_limit = coerce_float_like(_get_field_value(fields, _INDEXER_FIELD_SEED_RATIO))
ratio_limit = coerce_float_like(get_indexer_field(fields, _INDEXER_FIELD_SEED_RATIO))
seeding_time_limit = coerce_int_like(
_get_field_value(fields, _INDEXER_FIELD_SEED_TIME_MINUTES)
get_indexer_field(fields, _INDEXER_FIELD_SEED_TIME_MINUTES)
)
settings: IndexerSeedSettings = {}
+260 -61
View File
@@ -3,41 +3,86 @@
Prowlarr's MyAnonamouse indexer keeps the title, author, language and filetype
of each torrent but drops the narrator and series MAM returns alongside them, and
Torznab has no field to carry them. With the user's MAM session ID (``mam_id``)
Shelfmark runs the same text search against MAM's JSON API and matches the
torrents back to Prowlarr's results by their MAM torrent ID.
Shelfmark reruns the search Prowlarr sent - same text, categories and indexer
options - against MAM's JSON API and matches the torrents back to Prowlarr's
results by their MAM torrent ID.
"""
import json
import re
import time
import unicodedata
from collections import deque
from dataclasses import dataclass
from threading import Lock
from typing import Any
from typing import TYPE_CHECKING, Any
from urllib.parse import urlsplit
import requests
if TYPE_CHECKING:
from collections.abc import Iterable, Mapping
from shelfmark.core.logger import setup_logger
from shelfmark.download.network import get_proxies, get_ssl_verify
from shelfmark.release_sources.prowlarr.api import get_indexer_field
from shelfmark.release_sources.prowlarr.utils import coerce_int_like
logger = setup_logger(__name__)
DEFAULT_MAM_BASE_URL = "https://www.myanonamouse.net"
# The only origin Prowlarr's MyAnonamouse indexer uses. The session cookie is never
# sent anywhere else, whatever URL a search result carries.
MAM_BASE_URL = "https://www.myanonamouse.net"
_MAM_DOMAIN = "myanonamouse.net"
_SEARCH_PATH = "/tor/js/loadSearchJSONbasic.php"
_USER_PATH = "/jsonLoad.php"
_REQUEST_TIMEOUT_SECONDS = 15
_RESULTS_PER_PAGE = 100
_MAX_SEARCHES = 3
# Rerunning Prowlarr's search normally takes one request per title variant; the rest
# page past torrents that moved (a new upload, an option Shelfmark can't mirror).
_MAX_REQUESTS = 4
_CACHE_TTL_SECONDS = 3600
_HTTP_FORBIDDEN = 403
# Prowlarr sets both infoUrl and guid to "{BaseUrl}t/{id}".
_TORRENT_URL_RE = re.compile(r"^https?://[^/]*myanonamouse\.net/t/(\d+)", re.IGNORECASE)
# Consecutive failures back off for 1, 2, 4 ... minutes, at most 30. The tenth in a
# row stops enrichment until the session ID changes or "Test MAM Session" succeeds.
_MAX_CONSECUTIVE_FAILURES = 10
_BACKOFF_BASE_SECONDS = 60
_BACKOFF_MAX_SECONDS = 1800
# Prowlarr sets both infoUrl and guid to "https://www.myanonamouse.net/t/{id}".
_TORRENT_PATH_RE = re.compile(r"/t/(\d+)/?")
# Uploaders put the bitrate in the free-text tags: "64 kbps", "128kbps", "64 kb/s".
_BITRATE_RE = re.compile(r"\b(\d{2,4}(?:\.\d+)?)\s*k(?:bps|b/s|bit/s)\b", re.IGNORECASE)
_MAM_REQUEST_ERRORS = (requests.exceptions.RequestException, ValueError)
# Prowlarr's clean-up of the query (SearchCriteriaBase.GetSanitizedTerm keeps these,
# after standardising dashes and quotes; MyAnonamouse's generator then turns every
# non-word run into a space). Anything else is dropped outright.
_PROWLARR_KEPT_PUNCTUATION = frozenset("-._()@/'[]+%`´‘’")
_NON_WORD_RE = re.compile(r"[^\w]+")
# Prowlarr's MyAnonamouse "Search Type" options, by their value in the indexer settings.
_SEARCH_TYPES = {0: "all", 1: "active", 2: "fl", 3: "fl-VIP", 4: "VIP", 5: "nVIP"}
# Optional "Search in ..." indexer settings; title, author and narrator always apply.
_OPTIONAL_SEARCH_FIELDS = {
"searchInDescription": "description",
"searchInSeries": "series",
"searchInFilenames": "filenames",
}
# The MAM main categories Prowlarr maps to the Torznab categories Shelfmark searches:
# AudioBooks, Musicology and Radio are Audio/Audiobook, E-Books is the Books tree.
_MAIN_CATEGORIES_BY_TORZNAB = {3030: ("13", "15", "16"), 7000: ("14",)}
class MamError(Exception):
"""MAM answered, but not with a usable result."""
class MamAuthError(MamError):
"""MAM rejected the session ID."""
_MAM_REQUEST_ERRORS = (MamError, requests.exceptions.RequestException, ValueError)
@dataclass(frozen=True)
@@ -50,24 +95,98 @@ class MamTorrentDetails:
bitrate_kbps: int | None = None
@dataclass(frozen=True)
class MamSearchOptions:
"""How Prowlarr searched MAM, so rerunning the search returns the same torrents."""
search_type: str = "all"
search_in: tuple[str, ...] = ("title", "author", "narrator")
languages: tuple[str, ...] = ()
main_categories: tuple[str, ...] = () # empty searches every category
@dataclass
class _Failures:
"""Consecutive failed MAM requests for one session ID."""
mam_id: str = ""
count: int = 0
retry_at: float = 0.0
_cache: dict[int, tuple[MamTorrentDetails, float]] = {}
_cache_lock = Lock()
_failures = _Failures()
_failures_lock = Lock()
def mam_torrent_id(url: object) -> int | None:
"""Return the MAM torrent ID from a Prowlarr infoUrl/guid, or None for other trackers."""
if not isinstance(url, str):
return None
match = _TORRENT_URL_RE.match(url.strip())
try:
parts = urlsplit(url.strip())
except ValueError:
return None
host = (parts.hostname or "").lower()
if (
parts.scheme not in {"http", "https"}
or "@" in parts.netloc
or "\\" in parts.netloc
or not (host == _MAM_DOMAIN or host.endswith(f".{_MAM_DOMAIN}"))
):
return None
match = _TORRENT_PATH_RE.fullmatch(parts.path)
return int(match.group(1)) if match else None
def mam_base_url(url: object) -> str:
"""Return the MAM origin a Prowlarr result points at (Prowlarr's configured base URL)."""
if isinstance(url, str) and mam_torrent_id(url) is not None:
parts = urlsplit(url.strip())
return f"{parts.scheme}://{parts.netloc}"
return DEFAULT_MAM_BASE_URL
def prowlarr_search_text(query: str) -> str:
"""Clean a query the way Prowlarr does before it sends the query to MAM."""
kept = "".join(
char
for char in query
if char.isalpha()
or char.isdecimal()
or char.isspace()
or char in _PROWLARR_KEPT_PUNCTUATION
or unicodedata.category(char) == "Pd"
)
return _NON_WORD_RE.sub(" ", kept).strip()
def mam_search_options(
indexer: Mapping[str, Any] | None, torznab_categories: Iterable[int] | None
) -> MamSearchOptions:
"""Mirror a Prowlarr MyAnonamouse indexer's search settings and the categories searched."""
fields = indexer.get("fields") if indexer else None
languages = get_indexer_field(fields, "searchLanguages")
return MamSearchOptions(
search_type=_SEARCH_TYPES.get(
coerce_int_like(get_indexer_field(fields, "searchType")) or 0, "all"
),
search_in=(
"title",
"author",
"narrator",
*(
name
for key, name in _OPTIONAL_SEARCH_FIELDS.items()
if get_indexer_field(fields, key) is True
),
),
languages=tuple(
str(language_id)
for language in (languages if isinstance(languages, list) else [])
if (language_id := coerce_int_like(language)) is not None
),
main_categories=tuple(
dict.fromkeys(
main_category
for category in torznab_categories or ()
for main_category in _MAIN_CATEGORIES_BY_TORZNAB.get(category, ())
)
),
)
def _decode_info(raw: object) -> dict[str, Any]:
@@ -128,28 +247,25 @@ def parse_torrent_details(item: dict[str, Any]) -> MamTorrentDetails:
)
class MamAuthError(Exception):
"""MAM rejected the session ID."""
class MamClient:
"""Minimal MyAnonamouse JSON API client authenticated by a ``mam_id`` cookie."""
def __init__(self, mam_id: str, base_url: str = DEFAULT_MAM_BASE_URL) -> None:
"""Create a client for the given session ID and MAM origin."""
self.base_url = base_url.rstrip("/")
self._session = requests.Session()
self._session.cookies.set("mam_id", mam_id.strip())
self._session.headers.update({"Accept": "application/json"})
def __init__(self, mam_id: str) -> None:
"""Create a client for the given session ID."""
# A plain header rather than a cookie jar entry: requests drops the header on a
# redirect, and redirects are not followed anyway.
self._headers = {"Accept": "application/json", "Cookie": f"mam_id={mam_id.strip()}"}
def _get(self, path: str, params: dict[str, str] | None = None) -> object:
url = self.base_url + path
response = self._session.get(
def _get(self, path: str, params: Mapping[str, str | list[str]] | None = None) -> object:
url = MAM_BASE_URL + path
response = requests.get(
url,
params=params,
headers=self._headers,
timeout=_REQUEST_TIMEOUT_SECONDS,
proxies=get_proxies(url),
verify=get_ssl_verify(url),
allow_redirects=False,
)
if response.status_code == _HTTP_FORBIDDEN:
# MAM explains itself in the body (bad cookie, IP/ASN mismatch, ...).
@@ -160,7 +276,9 @@ class MamClient:
+ (f". MAM said: {reply}" if reply else "")
)
raise MamAuthError(msg)
response.raise_for_status()
if response.status_code != requests.codes.ok:
msg = f"MyAnonamouse answered HTTP {response.status_code}"
raise requests.exceptions.HTTPError(msg, response=response)
return response.json()
def get_username(self) -> str | None:
@@ -171,29 +289,35 @@ class MamClient:
username = data.get("username")
return str(username) if username else None
def search(self, text: str) -> list[dict[str, Any]]:
"""Search torrents across all categories by title, author, narrator and series."""
params = {
def search(
self, text: str, options: MamSearchOptions, *, start: int = 0
) -> list[dict[str, Any]]:
"""Return one page of torrents; fewer than a full page means it was the last."""
params: dict[str, str | list[str]] = {
"tor[text]": text,
"tor[searchType]": "all",
"tor[searchType]": options.search_type,
"tor[searchIn]": "torrents",
"tor[srchIn][title]": "true",
"tor[srchIn][author]": "true",
"tor[srchIn][narrator]": "true",
"tor[srchIn][series]": "true",
**{f"tor[srchIn][{field}]": "true" for field in options.search_in},
"tor[cat][]": "0",
"tor[sortType]": "default",
"tor[startNumber]": "0",
"tor[startNumber]": str(start),
"perpage": str(_RESULTS_PER_PAGE),
}
if options.main_categories:
params["tor[main_cat][]"] = list(options.main_categories)
if options.languages:
params["tor[browse_lang][]"] = list(options.languages)
data = self._get(_SEARCH_PATH, params)
if not isinstance(data, dict):
return []
items = data.get("data") if isinstance(data, dict) else None
if isinstance(items, list):
return [item for item in items if isinstance(item, dict)]
error = data.get("error") if isinstance(data, dict) else None
# An empty search answers {"error": "Nothing returned, out of N"}.
items = data.get("data")
if not isinstance(items, list):
if isinstance(error, str) and error.startswith("Nothing returned"):
return []
return [item for item in items if isinstance(item, dict)]
msg = f"Unexpected MyAnonamouse search reply: {str(error or data)[:200]}"
raise MamError(msg)
def _cached(torrent_id: int) -> MamTorrentDetails | None:
@@ -211,23 +335,86 @@ def _cached(torrent_id: int) -> MamTorrentDetails | None:
def _store(details_by_id: dict[int, MamTorrentDetails]) -> None:
now = time.time()
with _cache_lock:
# Most IDs are never looked up again, so expiry on read alone would let the
# cache grow for as long as Shelfmark runs.
expired = [
torrent_id
for torrent_id, (_, cached_at) in _cache.items()
if now - cached_at > _CACHE_TTL_SECONDS
]
for torrent_id in expired:
del _cache[torrent_id]
for torrent_id, details in details_by_id.items():
_cache[torrent_id] = (details, now)
def _blocked_reason(mam_id: str) -> str | None:
"""Return why MAM must not be called right now, or None if it may be."""
with _failures_lock:
if _failures.mam_id != mam_id or not _failures.count:
return None
if _failures.count >= _MAX_CONSECUTIVE_FAILURES:
return f"stopped after {_failures.count} consecutive failures"
wait = _failures.retry_at - time.monotonic()
if wait > 0:
return f"backing off for {wait:.0f}s after {_failures.count} failure(s)"
return None
def _record_success(mam_id: str) -> None:
with _failures_lock:
if _failures.mam_id == mam_id:
_failures.count = 0
_failures.retry_at = 0.0
def _record_failure(mam_id: str, error: Exception) -> None:
with _failures_lock:
if _failures.mam_id != mam_id:
_failures.mam_id = mam_id
_failures.count = 0
_failures.count += 1
count = _failures.count
delay = min(_BACKOFF_BASE_SECONDS * 2 ** (count - 1), _BACKOFF_MAX_SECONDS)
_failures.retry_at = time.monotonic() + delay
if count >= _MAX_CONSECUTIVE_FAILURES:
logger.warning(
"MAM enrichment stopped after %s consecutive failures (last: %s). "
"Test the MAM session in the Prowlarr settings to turn it back on.",
count,
error,
)
else:
logger.warning(
"MAM enrichment failed (%s/%s), next try in %ss: %s",
count,
_MAX_CONSECUTIVE_FAILURES,
delay,
error,
)
def reset_failures() -> None:
"""Let enrichment call MAM again straight away (the session test succeeded)."""
with _failures_lock:
_failures.count = 0
_failures.retry_at = 0.0
def lookup_torrent_details(
mam_id: str,
torrent_ids: set[int],
queries: list[str],
*,
base_url: str = DEFAULT_MAM_BASE_URL,
options: MamSearchOptions | None = None,
deadline: float | None = None,
) -> dict[int, MamTorrentDetails]:
"""Fetch narrator/series/bitrate for the given MAM torrent IDs (best effort).
Reruns the queries Prowlarr was sent until every ID is found, capped at a few
requests. Any failure is logged and returns what was found so far; enrichment
must never break the Prowlarr search it decorates.
Reruns Prowlarr's search for each query, reading further pages only while IDs
are missing, capped at a few requests. A failure backs MAM off and returns what
was found so far; enrichment must never break the Prowlarr search it decorates.
"""
found: dict[int, MamTorrentDetails] = {}
for torrent_id in torrent_ids:
@@ -236,26 +423,36 @@ def lookup_torrent_details(
found[torrent_id] = details
missing = torrent_ids - found.keys()
if not missing or not mam_id.strip():
mam_id = mam_id.strip()
if not missing or not mam_id:
return found
client = MamClient(mam_id, base_url)
searched = 0
for query in dict.fromkeys(q.strip() for q in queries if q and q.strip()):
if not missing or searched >= _MAX_SEARCHES:
break
blocked = _blocked_reason(mam_id)
if blocked:
logger.debug("MAM enrichment skipped: %s", blocked)
return found
client = MamClient(mam_id)
options = options or MamSearchOptions()
# Page 1 of every query before page 2 of any: each query found its own torrents.
pages = deque(
(text, 0)
for text in dict.fromkeys(prowlarr_search_text(query) for query in queries)
if text
)
requests_made = 0
while pages and missing and requests_made < _MAX_REQUESTS:
if deadline is not None and time.monotonic() > deadline:
logger.debug("MAM enrichment: search budget spent, %s torrent(s) left", len(missing))
break
searched += 1
text, start = pages.popleft()
requests_made += 1
try:
items = client.search(query)
except MamAuthError as e:
logger.warning("MAM enrichment disabled for this search: %s", e)
break
items = client.search(text, options, start=start)
except _MAM_REQUEST_ERRORS as e:
logger.warning("MAM enrichment search failed for '%s': %s", query, e)
_record_failure(mam_id, e)
break
_record_success(mam_id)
fetched: dict[int, MamTorrentDetails] = {}
for item in items:
@@ -266,11 +463,13 @@ def lookup_torrent_details(
for torrent_id in missing & fetched.keys():
found[torrent_id] = fetched[torrent_id]
missing -= fetched.keys()
if len(items) >= _RESULTS_PER_PAGE:
pages.append((text, start + len(items)))
logger.debug(
"MAM enrichment: %s of %s torrent(s) matched in %s search(es)",
"MAM enrichment: %s of %s torrent(s) matched in %s request(s)",
len(found),
len(torrent_ids),
searched,
requests_made,
)
return found
@@ -135,7 +135,7 @@ def _test_prowlarr_connection(current_values: dict[str, Any] | None = None) -> d
def _test_mam_connection(current_values: dict[str, Any] | None = None) -> dict[str, Any]:
"""Check the MAM session ID by asking MAM which account it belongs to."""
from shelfmark.release_sources.prowlarr.mam import MamAuthError, MamClient
from shelfmark.release_sources.prowlarr.mam import MamClient, MamError, reset_failures
current_values = current_values or {}
mam_id = _resolve_setting_text(current_values, "PROWLARR_MAM_ID")
@@ -144,13 +144,15 @@ def _test_mam_connection(current_values: dict[str, Any] | None = None) -> dict[s
try:
username = MamClient(mam_id).get_username()
except MamAuthError as e:
except MamError as e:
return {"success": False, "message": str(e)}
except _PROWLARR_SETTINGS_ERRORS as e:
return {"success": False, "message": f"Connection failed: {e!s}"}
if not username:
return {"success": False, "message": "MyAnonamouse did not return an account for this ID"}
# A working session ends any back-off from earlier failed lookups.
reset_failures()
return {"success": True, "message": f"Connected to MyAnonamouse as {username}"}
@@ -280,7 +282,10 @@ def prowlarr_config_settings() -> list[SettingsField]:
ActionButton(
key="test_prowlarr_mam",
label="Test MAM Session",
description="Check that MyAnonamouse accepts the session ID",
description=(
"Check that MyAnonamouse accepts the session ID. A passing test also resumes "
"enrichment after it stopped on repeated failures."
),
style="primary",
callback=_test_mam_connection,
show_when={"field": "PROWLARR_ENABLED", "value": True},
+28 -14
View File
@@ -11,6 +11,7 @@ import requests
if TYPE_CHECKING:
from shelfmark.core.search_plan import ReleaseSearchPlan
from shelfmark.metadata_providers import BookMetadata
from shelfmark.release_sources.prowlarr.mam import MamSearchOptions
from shelfmark.core.author_match import AUTHOR_UNKNOWN, author_affinity
from shelfmark.core.config import config
@@ -39,11 +40,12 @@ from shelfmark.release_sources.prowlarr.api import (
IndexerSeedSettings,
ProwlarrClient,
ProwlarrSearchError,
is_myanonamouse_indexer,
)
from shelfmark.release_sources.prowlarr.cache import cache_release
from shelfmark.release_sources.prowlarr.mam import (
lookup_torrent_details,
mam_base_url,
mam_search_options,
mam_torrent_id,
)
from shelfmark.release_sources.prowlarr.utils import (
@@ -89,26 +91,23 @@ def _enrich_mam_releases(
releases: list[Release],
mam_id: str,
queries: list[str],
options: MamSearchOptions,
deadline: float | None,
) -> None:
"""Fill narrator/series/bitrate on MyAnonamouse releases from MAM's own API."""
ids_by_source_id: dict[str, int] = {}
base_url: str | None = None
for release in releases:
torrent_id = mam_torrent_id(release.info_url)
if torrent_id is None:
continue
ids_by_source_id[release.source_id] = torrent_id
base_url = base_url or mam_base_url(release.info_url)
if not ids_by_source_id or base_url is None:
ids_by_source_id = {
release.source_id: torrent_id
for release in releases
if (torrent_id := mam_torrent_id(release.info_url)) is not None
}
if not ids_by_source_id:
return
details_by_id = lookup_torrent_details(
mam_id,
set(ids_by_source_id.values()),
queries,
base_url=base_url,
options=options,
deadline=deadline,
)
for release in releases:
@@ -1126,6 +1125,12 @@ class ProwlarrSource(ReleaseSource):
restrict_to=indexer_ids, indexers=enabled_indexers
)
enriched_indexer_ids_set = set(enriched_indexer_ids)
mam_indexers = {
indexer_id: indexer
for indexer in enabled_indexers
if is_myanonamouse_indexer(indexer)
and (indexer_id := _coerce_indexer_id(indexer.get("id"))) is not None
}
indexer_seed_settings = (
_fetch_indexer_seed_settings(client, indexer_ids)
if config.get("PROWLARR_USE_SEED_PREFERENCES", False)
@@ -1266,6 +1271,8 @@ class ProwlarrSource(ReleaseSource):
results: list[Release] = []
enriched_source_ids: set[str] = set()
affinity_by_source_id: dict[str, int] = {}
mam_releases: list[Release] = []
mam_indexer: dict | None = None
# A manual query is the user's own words; ranking it against the
# metadata author would second-guess what they typed.
wanted_author = "" if plan.manual_query else plan.author
@@ -1294,14 +1301,21 @@ class ProwlarrSource(ReleaseSource):
if is_enriched:
enriched_source_ids.add(release.source_id)
if idx_id_int in mam_indexers:
mam_releases.append(release)
mam_indexer = mam_indexer or mam_indexers[idx_id_int]
mam_session_id = _get_mam_session_id()
if mam_session_id:
if mam_session_id and mam_releases:
# Rerun the search Prowlarr sent to MAM, so it returns the same torrents:
# the indexer's own options, and every category once the search expanded.
mam_categories = None if self.last_search_type == "expanded" else categories
try:
_enrich_mam_releases(
results,
mam_releases,
mam_session_id,
[variant.title for variant in variants],
mam_search_options(mam_indexer, mam_categories),
deadline,
)
except _PROWLARR_REQUEST_ERRORS:
+443 -22
View File
@@ -1,12 +1,14 @@
"""Tests for MyAnonamouse enrichment of Prowlarr results (narrator, series, bitrate)."""
import json
import time
import pytest
import requests
import shelfmark.release_sources.prowlarr.mam as mam
import shelfmark.release_sources.prowlarr.source as prowlarr_source
from shelfmark.metadata_providers import BookMetadata
from shelfmark.release_sources import (
ColumnSchema,
ReleaseColumnConfig,
@@ -15,11 +17,16 @@ from shelfmark.release_sources import (
)
from shelfmark.release_sources.prowlarr.mam import (
MamAuthError,
MamClient,
MamError,
MamSearchOptions,
lookup_torrent_details,
mam_base_url,
mam_search_options,
mam_torrent_id,
parse_torrent_details,
prowlarr_search_text,
)
from shelfmark.release_sources.prowlarr.settings import _test_mam_connection
from shelfmark.release_sources.prowlarr.source import (
ProwlarrSource,
_enrich_mam_releases,
@@ -28,8 +35,9 @@ from shelfmark.release_sources.prowlarr.source import (
@pytest.fixture(autouse=True)
def _clear_mam_cache():
def _reset_mam_state(monkeypatch):
mam._cache.clear()
monkeypatch.setattr(mam, "_failures", mam._Failures())
yield
mam._cache.clear()
@@ -44,16 +52,35 @@ def _mam_item(torrent_id: int, **fields) -> dict:
return item
def _items(first_id: int, last_id: int) -> list[dict]:
return [_mam_item(torrent_id) for torrent_id in range(first_id, last_id + 1)]
class TestParsing:
def test_torrent_id_from_prowlarr_info_url(self):
assert mam_torrent_id("https://www.myanonamouse.net/t/123456") == 123456
assert mam_torrent_id("https://cdn.myanonamouse.net/t/42") == 42
assert mam_torrent_id("https://cdn.myanonamouse.net/t/42/") == 42
assert mam_torrent_id("https://example.org/t/123") is None
assert mam_torrent_id(None) is None
def test_base_url_follows_the_result(self):
assert mam_base_url("https://cdn.myanonamouse.net/t/42") == "https://cdn.myanonamouse.net"
assert mam_base_url("https://example.org/t/42") == mam.DEFAULT_MAM_BASE_URL
@pytest.mark.parametrize(
"url",
[
"https://evil.example?myanonamouse.net/t/123",
"https://evil.example#myanonamouse.net/t/123",
"https://evil.example/myanonamouse.net/t/123",
"https://notmyanonamouse.net/t/123",
"https://www.myanonamouse.net.evil.example/t/123",
"https://evil.example\\@www.myanonamouse.net/t/123",
"https://user@www.myanonamouse.net/t/123",
"ftp://www.myanonamouse.net/t/123",
"https://www.myanonamouse.net/t/123abc",
"https://www.myanonamouse.net/tor/t/123",
"https://[::1/t/123",
],
)
def test_urls_that_only_mention_mam_are_not_mam(self, url):
assert mam_torrent_id(url) is None
def test_parses_narrator_series_and_bitrate(self):
details = parse_torrent_details(
@@ -83,22 +110,153 @@ class TestParsing:
assert details == mam.MamTorrentDetails()
class TestProwlarrSearchMirror:
@pytest.mark.parametrize(
("query", "expected"),
[
("Empire of Silence", "Empire of Silence"),
("Ender's Game", "Ender s Game"),
("Ender’s Game", "Ender s Game"),
("Dune: Part One", "Dune Part One"),
("Spider‑Man — Homecoming", "Spider Man Homecoming"),
("1,000 Years & More", "1000 Years More"),
("C++ Primer (5th ed.)", "C Primer 5th ed"),
("Les Misérables", "Les Misérables"),
("!!!", ""),
],
)
def test_query_is_cleaned_like_prowlarr(self, query, expected):
assert prowlarr_search_text(query) == expected
def test_defaults_match_prowlarr_defaults(self):
options = mam_search_options({"implementation": "MyAnonamouse"}, [3030])
assert options == MamSearchOptions(main_categories=("13", "15", "16"))
def test_indexer_settings_are_mirrored(self):
indexer = {
"fields": [
{"name": "searchType", "value": 1},
{"name": "searchInDescription", "value": False},
{"name": "searchInSeries", "value": True},
{"name": "searchLanguages", "value": [1, "37", "junk"]},
]
}
options = mam_search_options(indexer, [7000])
assert options == MamSearchOptions(
search_type="active",
search_in=("title", "author", "narrator", "series"),
languages=("1", "37"),
main_categories=("14",),
)
def test_expanded_search_covers_every_category(self):
assert mam_search_options(None, None).main_categories == ()
class _FakeResponse:
def __init__(self, status_code=200, payload=None, text=""):
self.status_code = status_code
self._payload = payload
self.text = text
def json(self):
if self._payload is None:
msg = "not JSON"
raise ValueError(msg)
return self._payload
@pytest.fixture
def mam_http(monkeypatch):
"""Record every HTTP request the MAM client makes and answer with `response`."""
class _Http:
def __init__(self):
self.response = _FakeResponse(payload={"data": []})
self.calls: list[tuple[str, dict]] = []
def get(self, url, **kwargs):
self.calls.append((url, kwargs))
return self.response
http = _Http()
monkeypatch.setattr(mam.requests, "get", http.get)
monkeypatch.setattr(mam, "get_proxies", lambda _url: {})
monkeypatch.setattr(mam, "get_ssl_verify", lambda _url: True)
return http
class TestMamClient:
def test_search_repeats_prowlarrs_request(self, mam_http):
options = MamSearchOptions(
search_in=("title", "author", "narrator", "series"),
languages=("1",),
main_categories=("13", "15", "16"),
)
MamClient(" session ").search("Empire of Silence", options, start=100)
[(url, kwargs)] = mam_http.calls
assert url == "https://www.myanonamouse.net/tor/js/loadSearchJSONbasic.php"
assert kwargs["headers"]["Cookie"] == "mam_id=session"
assert kwargs["allow_redirects"] is False
params = kwargs["params"]
assert params["tor[text]"] == "Empire of Silence"
assert params["tor[searchType]"] == "all"
assert params["tor[srchIn][series]"] == "true"
assert "tor[srchIn][description]" not in params
assert params["tor[main_cat][]"] == ["13", "15", "16"]
assert params["tor[browse_lang][]"] == ["1"]
assert params["tor[startNumber]"] == "100"
assert params["perpage"] == "100"
def test_nothing_returned_is_an_empty_page(self, mam_http):
mam_http.response = _FakeResponse(payload={"error": "Nothing returned, out of 12"})
assert MamClient("session").search("q", MamSearchOptions()) == []
def test_unexpected_reply_is_an_error(self, mam_http):
mam_http.response = _FakeResponse(payload={"error": "Invalid search parameters"})
with pytest.raises(MamError, match="Invalid search parameters"):
MamClient("session").search("q", MamSearchOptions())
def test_redirect_is_an_error_not_followed(self, mam_http):
mam_http.response = _FakeResponse(status_code=302)
with pytest.raises(requests.exceptions.HTTPError, match="302"):
MamClient("session").search("q", MamSearchOptions())
assert len(mam_http.calls) == 1
def test_forbidden_carries_mams_reply(self, mam_http):
mam_http.response = _FakeResponse(status_code=403, text="Bad cookie: ASN mismatch")
with pytest.raises(MamAuthError, match="ASN mismatch"):
MamClient("session").get_username()
class _FakeMamClient:
"""Serves pages of `items_by_query` and records each (query, start) requested."""
def __init__(self, items_by_query=None, error=None):
self.items_by_query = items_by_query or {}
self.error = error
self.queries: list[str] = []
self.requests: list[tuple[str, int]] = []
self.options: MamSearchOptions | None = None
def __call__(self, mam_id, base_url=mam.DEFAULT_MAM_BASE_URL):
def __call__(self, mam_id):
self.mam_id = mam_id
self.base_url = base_url
return self
def search(self, text):
self.queries.append(text)
def search(self, text, options, *, start=0):
self.requests.append((text, start))
self.options = options
if self.error:
raise self.error
return self.items_by_query.get(text, [])
return self.items_by_query.get(text, [])[start : start + mam._RESULTS_PER_PAGE]
class TestLookup:
@@ -111,7 +269,15 @@ class TestLookup:
found = lookup_torrent_details("session", {1}, ["Empire of Silence", "Other title"])
assert found[1].narrator == "Roukin"
assert fake.queries == ["Empire of Silence"]
assert fake.requests == [("Empire of Silence", 0)]
def test_queries_are_sent_as_prowlarr_sent_them(self, monkeypatch):
fake = _FakeMamClient()
monkeypatch.setattr(mam, "MamClient", fake)
lookup_torrent_details("session", {1}, ["Ender's Game", "Ender’s Game", "!!!"])
assert fake.requests == [("Ender s Game", 0)]
def test_uses_cache_on_repeat(self, monkeypatch):
fake = _FakeMamClient({"q": [_mam_item(1, narrator_info=json.dumps({"1": "Roukin"}))]})
@@ -120,25 +286,160 @@ class TestLookup:
lookup_torrent_details("session", {1}, ["q"])
lookup_torrent_details("session", {1}, ["q"])
assert fake.queries == ["q"]
assert fake.requests == [("q", 0)]
def test_expired_entries_are_pruned_when_storing(self):
mam._cache[5] = (mam.MamTorrentDetails(), time.time() - 2 * mam._CACHE_TTL_SECONDS)
mam._store({1: mam.MamTorrentDetails()})
assert set(mam._cache) == {1}
def test_pages_until_every_id_is_found(self, monkeypatch):
fake = _FakeMamClient({"q": _items(1, 150)})
monkeypatch.setattr(mam, "MamClient", fake)
found = lookup_torrent_details("session", {5, 140}, ["q"])
assert set(found) == {5, 140}
assert fake.requests == [("q", 0), ("q", 100)]
def test_first_page_of_every_query_comes_before_further_pages(self, monkeypatch):
fake = _FakeMamClient({"a": [*_items(1, 100), _mam_item(500)], "b": [_mam_item(900)]})
monkeypatch.setattr(mam, "MamClient", fake)
found = lookup_torrent_details("session", {500, 900}, ["a", "b"])
assert set(found) == {500, 900}
assert fake.requests == [("a", 0), ("b", 0), ("a", 100)]
def test_a_short_page_is_the_last(self, monkeypatch):
fake = _FakeMamClient({"q": _items(1, 50)})
monkeypatch.setattr(mam, "MamClient", fake)
assert lookup_torrent_details("session", {999}, ["q"]) == {}
assert fake.requests == [("q", 0)]
def test_requests_are_capped(self, monkeypatch):
fake = _FakeMamClient({"q": _items(1, 1000)})
monkeypatch.setattr(mam, "MamClient", fake)
lookup_torrent_details("session", {999}, ["q"])
assert len(fake.requests) == mam._MAX_REQUESTS
@pytest.mark.parametrize(
"error",
[MamAuthError("rejected"), requests.exceptions.ConnectionError("down")],
[
MamAuthError("rejected"),
MamError("odd reply"),
requests.exceptions.ConnectionError("down"),
ValueError("not JSON"),
],
)
def test_failures_return_empty_instead_of_raising(self, monkeypatch, error):
fake = _FakeMamClient(error=error)
monkeypatch.setattr(mam, "MamClient", fake)
assert lookup_torrent_details("session", {1}, ["q", "r"]) == {}
assert fake.queries == ["q"]
assert fake.requests == [("q", 0)]
def test_expired_deadline_skips_the_request(self, monkeypatch):
fake = _FakeMamClient()
monkeypatch.setattr(mam, "MamClient", fake)
assert lookup_torrent_details("session", {1}, ["q"], deadline=0.0) == {}
assert fake.queries == []
assert fake.requests == []
class TestFailureBackoff:
@staticmethod
def _failing(monkeypatch, error=None):
fake = _FakeMamClient(error=error or requests.exceptions.ConnectionError("down"))
monkeypatch.setattr(mam, "MamClient", fake)
return fake
@staticmethod
def _retry_now():
mam._failures.retry_at = 0.0
def test_a_failure_holds_the_next_lookup_back(self, monkeypatch):
fake = self._failing(monkeypatch)
lookup_torrent_details("session", {1}, ["q"])
lookup_torrent_details("session", {1}, ["q"])
assert fake.requests == [("q", 0)]
assert mam._failures.count == 1
assert mam._failures.retry_at - time.monotonic() == pytest.approx(60, abs=5)
def test_backoff_doubles_up_to_half_an_hour(self, monkeypatch):
self._failing(monkeypatch)
delays = []
for _ in range(mam._MAX_CONSECUTIVE_FAILURES - 1):
self._retry_now()
before = time.monotonic()
lookup_torrent_details("session", {1}, ["q"])
delays.append(round(mam._failures.retry_at - before))
assert delays == [60, 120, 240, 480, 960, 1800, 1800, 1800, 1800]
def test_ten_failures_in_a_row_stop_enrichment(self, monkeypatch):
fake = self._failing(monkeypatch, MamAuthError("rejected"))
for _ in range(mam._MAX_CONSECUTIVE_FAILURES + 3):
self._retry_now()
lookup_torrent_details("session", {1}, ["q"])
assert len(fake.requests) == mam._MAX_CONSECUTIVE_FAILURES
assert mam._blocked_reason("session") == "stopped after 10 consecutive failures"
def test_a_success_resets_the_count(self, monkeypatch):
fake = self._failing(monkeypatch)
for _ in range(3):
self._retry_now()
lookup_torrent_details("session", {1}, ["q"])
fake.error = None
self._retry_now()
lookup_torrent_details("session", {1}, ["q"])
assert mam._failures.count == 0
assert mam._blocked_reason("session") is None
def test_a_new_session_id_starts_over(self, monkeypatch):
fake = self._failing(monkeypatch)
for _ in range(mam._MAX_CONSECUTIVE_FAILURES):
self._retry_now()
lookup_torrent_details("old", {1}, ["q"])
fake.error = None
lookup_torrent_details("new", {1}, ["q"])
assert fake.requests[-1] == ("q", 0)
assert len(fake.requests) == mam._MAX_CONSECUTIVE_FAILURES + 1
def test_a_passing_session_test_turns_enrichment_back_on(self, monkeypatch):
self._failing(monkeypatch)
for _ in range(mam._MAX_CONSECUTIVE_FAILURES):
self._retry_now()
lookup_torrent_details("session", {1}, ["q"])
assert mam._blocked_reason("session") is not None
monkeypatch.setattr(mam, "MamClient", _UsernameClient)
result = _test_mam_connection({"PROWLARR_MAM_ID": "session"})
assert result["success"] is True
assert mam._blocked_reason("session") is None
class _UsernameClient:
def __init__(self, mam_id):
self.mam_id = mam_id
def get_username(self):
return "reader"
class TestReleaseEnrichment:
@@ -156,7 +457,7 @@ class TestReleaseEnrichment:
"audiobook",
)
def test_only_mam_releases_are_enriched(self, monkeypatch):
def test_mam_details_fill_the_release(self, monkeypatch):
fake = _FakeMamClient(
{
"Empire of Silence": [
@@ -170,19 +471,28 @@ class TestReleaseEnrichment:
}
)
monkeypatch.setattr(mam, "MamClient", fake)
options = MamSearchOptions(main_categories=("13", "15", "16"))
mam_release = self._release("https://www.myanonamouse.net/t/99", "mam-guid")
other_release = self._release("https://tracker.example/t/99", "other-guid")
_enrich_mam_releases(
[mam_release, other_release], "session", ["Empire of Silence"], deadline=None
[mam_release], "session", ["Empire of Silence"], options, deadline=None
)
assert mam_release.extra["narrator"] == "Samuel Roukin"
assert mam_release.extra["series"] == "The Sun Eater #1"
assert mam_release.extra["bitrate"] == "64 Kbps"
assert mam_release.extra["bitrate_value"] == 64
assert "narrator" not in other_release.extra
assert fake.base_url == "https://www.myanonamouse.net"
assert fake.options == options
def test_a_url_that_only_mentions_mam_is_never_looked_up(self, mam_http):
spoofed = self._release("https://evil.example?myanonamouse.net/t/99", "guid")
_enrich_mam_releases(
[spoofed], "session", ["Empire of Silence"], MamSearchOptions(), deadline=None
)
assert mam_http.calls == []
assert "narrator" not in spoofed.extra
def test_torznab_bitrate_attribute_is_used_for_other_indexers(self):
release = _prowlarr_result_to_release(
@@ -198,6 +508,117 @@ class TestReleaseEnrichment:
assert release.extra["bitrate_value"] == 320
MAM_INDEXER_ID = 1
OTHER_INDEXER_ID = 2
class _ProwlarrWithMam:
"""A Prowlarr with MyAnonamouse and one other indexer enabled."""
indexer_timeout = 90
def __init__(self, mam_fields=None):
self.mam_fields = mam_fields or []
def get_enabled_indexers_detailed(self, *, raise_on_error=False):
del raise_on_error
capabilities = {"categories": [{"id": 7000, "subCategories": []}]}
return [
{
"id": MAM_INDEXER_ID,
"enable": True,
"implementation": "MyAnonamouse",
"fields": self.mam_fields,
"capabilities": capabilities,
},
{
"id": OTHER_INDEXER_ID,
"enable": True,
"implementation": "Torznab",
"capabilities": capabilities,
},
]
def get_enriched_indexer_ids(self, restrict_to=None, indexers=None):
del restrict_to, indexers
return [MAM_INDEXER_ID]
def torznab_search(
self, *, indexer_id, query, categories=None, search_type="book", limit=100, offset=0
):
del query, categories, search_type, limit, offset
# The other indexer's result also claims a MyAnonamouse URL.
torrent_id = 99 if indexer_id == MAM_INDEXER_ID else 77
return [
{
"guid": f"guid-{indexer_id}",
"title": "Empire of Silence",
"indexerId": indexer_id,
"indexer": f"Indexer {indexer_id}",
"protocol": "torrent",
"categories": [{"id": 7020}],
"infoUrl": f"https://www.myanonamouse.net/t/{torrent_id}",
}
]
class TestSearchEnrichment:
def _search(self, monkeypatch, client, *, expand_search=False):
from shelfmark.core.search_plan import build_release_search_plan
values = {
"PROWLARR_INDEXERS": "",
"PROWLARR_AUTO_EXPAND": False,
"PROWLARR_MAM_ID": "session",
}
monkeypatch.setattr(
prowlarr_source.config, "get", lambda key, default=None: values.get(key, default)
)
lookups = []
def fake_lookup(mam_id, torrent_ids, queries, *, options, deadline):
del deadline
lookups.append((mam_id, torrent_ids, queries, options))
return {99: mam.MamTorrentDetails(series="The Sun Eater #1")}
monkeypatch.setattr(prowlarr_source, "lookup_torrent_details", fake_lookup)
source = ProwlarrSource()
monkeypatch.setattr(source, "_get_client", lambda: client)
book = BookMetadata(
provider="hardcover",
provider_id="1",
title="Empire of Silence",
authors=["Christopher Ruocchio"],
)
plan = build_release_search_plan(book, languages=["en"])
results = source.search(book, plan, expand_search=expand_search, content_type="ebook")
return results, lookups
def test_only_the_mam_indexer_is_enriched(self, monkeypatch):
results, lookups = self._search(monkeypatch, _ProwlarrWithMam())
[(mam_id, torrent_ids, queries, _options)] = lookups
assert (mam_id, torrent_ids, queries) == ("session", {99}, ["Empire of Silence"])
series = {release.indexer: release.extra.get("series") for release in results}
assert series == {"Indexer 1": "The Sun Eater #1", "Indexer 2": None}
def test_lookup_mirrors_the_mam_indexer_settings(self, monkeypatch):
client = _ProwlarrWithMam(
[{"name": "searchType", "value": 2}, {"name": "searchLanguages", "value": [1]}]
)
_results, lookups = self._search(monkeypatch, client)
assert lookups[0][3] == MamSearchOptions(
search_type="fl", languages=("1",), main_categories=("14",)
)
def test_expanded_search_looks_in_every_category(self, monkeypatch):
_results, lookups = self._search(monkeypatch, _ProwlarrWithMam(), expand_search=True)
assert lookups[0][3].main_categories == ()
class TestColumnConfig:
def _source(self, monkeypatch, mam_id: str) -> ProwlarrSource:
monkeypatch.setattr(prowlarr_source, "_get_mam_session_id", lambda: mam_id)