diff --git a/docs/myanonamouse-enrichment.md b/docs/myanonamouse-enrichment.md index f78bb804..7585f9bb 100644 --- a/docs/myanonamouse-enrichment.md +++ b/docs/myanonamouse-enrichment.md @@ -35,4 +35,8 @@ Compare it with the IP listed next to the session on MAM's security page. ## How it works -After a Prowlarr search, Shelfmark sends the same query text to MAM's JSON search API (normally one request) and matches torrents back to Prowlarr's results by their MAM torrent ID. Lookups are cached for an hour, stay inside the Prowlarr search time budget, and never fail the search: if MAM errors or rejects the session, the release list simply shows without the extra columns filled in. +After a Prowlarr search, Shelfmark reruns the search Prowlarr sent to MyAnonamouse's JSON search API: the same cleaned-up query text, the same categories (audiobooks, or e-books), and your MyAnonamouse indexer's own options from Prowlarr (search type, search in description/series/filenames, languages). That normally returns the same torrents in one request per title. Shelfmark then matches them back to Prowlarr's results by their MAM torrent ID. Only results from the MyAnonamouse indexer are looked up, and the session ID is only ever sent to `https://www.myanonamouse.net`. + +If some torrents are missing from the first page, Shelfmark reads further pages, up to 4 requests per search. Lookups are cached for an hour, stay inside the Prowlarr search time budget, and never fail the search: if MAM errors or rejects the session, the release list simply shows without the extra columns filled in. + +After a failed request (a 403, a timeout, an unexpected reply), Shelfmark waits before trying MAM again: 1 minute, then 2, 4 and so on, up to 30 minutes. After 10 failures in a row it stops using MAM. **Test MAM Session** (once it succeeds), a new session ID, or a restart turns it back on. Each failure is logged with the reason. diff --git a/shelfmark/release_sources/prowlarr/api.py b/shelfmark/release_sources/prowlarr/api.py index 5b9bb92a..34678c97 100644 --- a/shelfmark/release_sources/prowlarr/api.py +++ b/shelfmark/release_sources/prowlarr/api.py @@ -107,7 +107,8 @@ def _normalize_json_object_list(payload: object, *, context: str) -> list[dict[s return [_normalize_json_object(item, context=context) for item in payload] -def _get_field_value(fields: object, name: str) -> object | None: +def get_indexer_field(fields: object, name: str) -> object | None: + """Return a setting's value from a Prowlarr indexer's "fields" list.""" if not isinstance(fields, list): return None @@ -120,6 +121,17 @@ def _get_field_value(fields: object, name: str) -> object | None: return None +def is_myanonamouse_indexer(indexer: Mapping[str, Any]) -> bool: + """Whether a Prowlarr indexer record is the MyAnonamouse implementation.""" + implementation = str( + indexer.get("implementation") + or indexer.get("implementationName") + or indexer.get("definitionName") + or "" + ) + return implementation.strip().lower() == "myanonamouse" + + class ProwlarrClient: """Client for interacting with the Prowlarr API.""" @@ -300,14 +312,8 @@ class ProwlarrClient: if restrict_to is not None and idx_id_int not in restrict_to: continue - impl = str( - idx.get("implementation") - or idx.get("implementationName") - or idx.get("definitionName") - or "" - ) # Currently only MyAnonamouse provides consistently rich Torznab metadata. - if impl.strip().lower() == "myanonamouse": + if is_myanonamouse_indexer(idx): enriched_ids.append(idx_id_int) return enriched_ids @@ -339,9 +345,9 @@ class ProwlarrClient: continue fields = idx.get("fields") - ratio_limit = coerce_float_like(_get_field_value(fields, _INDEXER_FIELD_SEED_RATIO)) + ratio_limit = coerce_float_like(get_indexer_field(fields, _INDEXER_FIELD_SEED_RATIO)) seeding_time_limit = coerce_int_like( - _get_field_value(fields, _INDEXER_FIELD_SEED_TIME_MINUTES) + get_indexer_field(fields, _INDEXER_FIELD_SEED_TIME_MINUTES) ) settings: IndexerSeedSettings = {} diff --git a/shelfmark/release_sources/prowlarr/mam.py b/shelfmark/release_sources/prowlarr/mam.py index a9571794..c6745208 100644 --- a/shelfmark/release_sources/prowlarr/mam.py +++ b/shelfmark/release_sources/prowlarr/mam.py @@ -3,41 +3,86 @@ Prowlarr's MyAnonamouse indexer keeps the title, author, language and filetype of each torrent but drops the narrator and series MAM returns alongside them, and Torznab has no field to carry them. With the user's MAM session ID (``mam_id``) -Shelfmark runs the same text search against MAM's JSON API and matches the -torrents back to Prowlarr's results by their MAM torrent ID. +Shelfmark reruns the search Prowlarr sent - same text, categories and indexer +options - against MAM's JSON API and matches the torrents back to Prowlarr's +results by their MAM torrent ID. """ import json import re import time +import unicodedata +from collections import deque from dataclasses import dataclass from threading import Lock -from typing import Any +from typing import TYPE_CHECKING, Any from urllib.parse import urlsplit import requests +if TYPE_CHECKING: + from collections.abc import Iterable, Mapping + from shelfmark.core.logger import setup_logger from shelfmark.download.network import get_proxies, get_ssl_verify +from shelfmark.release_sources.prowlarr.api import get_indexer_field from shelfmark.release_sources.prowlarr.utils import coerce_int_like logger = setup_logger(__name__) -DEFAULT_MAM_BASE_URL = "https://www.myanonamouse.net" +# The only origin Prowlarr's MyAnonamouse indexer uses. The session cookie is never +# sent anywhere else, whatever URL a search result carries. +MAM_BASE_URL = "https://www.myanonamouse.net" +_MAM_DOMAIN = "myanonamouse.net" _SEARCH_PATH = "/tor/js/loadSearchJSONbasic.php" _USER_PATH = "/jsonLoad.php" _REQUEST_TIMEOUT_SECONDS = 15 _RESULTS_PER_PAGE = 100 -_MAX_SEARCHES = 3 +# Rerunning Prowlarr's search normally takes one request per title variant; the rest +# page past torrents that moved (a new upload, an option Shelfmark can't mirror). +_MAX_REQUESTS = 4 _CACHE_TTL_SECONDS = 3600 _HTTP_FORBIDDEN = 403 -# Prowlarr sets both infoUrl and guid to "{BaseUrl}t/{id}". -_TORRENT_URL_RE = re.compile(r"^https?://[^/]*myanonamouse\.net/t/(\d+)", re.IGNORECASE) +# Consecutive failures back off for 1, 2, 4 ... minutes, at most 30. The tenth in a +# row stops enrichment until the session ID changes or "Test MAM Session" succeeds. +_MAX_CONSECUTIVE_FAILURES = 10 +_BACKOFF_BASE_SECONDS = 60 +_BACKOFF_MAX_SECONDS = 1800 + +# Prowlarr sets both infoUrl and guid to "https://www.myanonamouse.net/t/{id}". +_TORRENT_PATH_RE = re.compile(r"/t/(\d+)/?") # Uploaders put the bitrate in the free-text tags: "64 kbps", "128kbps", "64 kb/s". _BITRATE_RE = re.compile(r"\b(\d{2,4}(?:\.\d+)?)\s*k(?:bps|b/s|bit/s)\b", re.IGNORECASE) -_MAM_REQUEST_ERRORS = (requests.exceptions.RequestException, ValueError) +# Prowlarr's clean-up of the query (SearchCriteriaBase.GetSanitizedTerm keeps these, +# after standardising dashes and quotes; MyAnonamouse's generator then turns every +# non-word run into a space). Anything else is dropped outright. +_PROWLARR_KEPT_PUNCTUATION = frozenset("-._()@/'[]+%`´‘’") +_NON_WORD_RE = re.compile(r"[^\w]+") + +# Prowlarr's MyAnonamouse "Search Type" options, by their value in the indexer settings. +_SEARCH_TYPES = {0: "all", 1: "active", 2: "fl", 3: "fl-VIP", 4: "VIP", 5: "nVIP"} +# Optional "Search in ..." indexer settings; title, author and narrator always apply. +_OPTIONAL_SEARCH_FIELDS = { + "searchInDescription": "description", + "searchInSeries": "series", + "searchInFilenames": "filenames", +} +# The MAM main categories Prowlarr maps to the Torznab categories Shelfmark searches: +# AudioBooks, Musicology and Radio are Audio/Audiobook, E-Books is the Books tree. +_MAIN_CATEGORIES_BY_TORZNAB = {3030: ("13", "15", "16"), 7000: ("14",)} + + +class MamError(Exception): + """MAM answered, but not with a usable result.""" + + +class MamAuthError(MamError): + """MAM rejected the session ID.""" + + +_MAM_REQUEST_ERRORS = (MamError, requests.exceptions.RequestException, ValueError) @dataclass(frozen=True) @@ -50,24 +95,98 @@ class MamTorrentDetails: bitrate_kbps: int | None = None +@dataclass(frozen=True) +class MamSearchOptions: + """How Prowlarr searched MAM, so rerunning the search returns the same torrents.""" + + search_type: str = "all" + search_in: tuple[str, ...] = ("title", "author", "narrator") + languages: tuple[str, ...] = () + main_categories: tuple[str, ...] = () # empty searches every category + + +@dataclass +class _Failures: + """Consecutive failed MAM requests for one session ID.""" + + mam_id: str = "" + count: int = 0 + retry_at: float = 0.0 + + _cache: dict[int, tuple[MamTorrentDetails, float]] = {} _cache_lock = Lock() +_failures = _Failures() +_failures_lock = Lock() def mam_torrent_id(url: object) -> int | None: """Return the MAM torrent ID from a Prowlarr infoUrl/guid, or None for other trackers.""" if not isinstance(url, str): return None - match = _TORRENT_URL_RE.match(url.strip()) + try: + parts = urlsplit(url.strip()) + except ValueError: + return None + host = (parts.hostname or "").lower() + if ( + parts.scheme not in {"http", "https"} + or "@" in parts.netloc + or "\\" in parts.netloc + or not (host == _MAM_DOMAIN or host.endswith(f".{_MAM_DOMAIN}")) + ): + return None + match = _TORRENT_PATH_RE.fullmatch(parts.path) return int(match.group(1)) if match else None -def mam_base_url(url: object) -> str: - """Return the MAM origin a Prowlarr result points at (Prowlarr's configured base URL).""" - if isinstance(url, str) and mam_torrent_id(url) is not None: - parts = urlsplit(url.strip()) - return f"{parts.scheme}://{parts.netloc}" - return DEFAULT_MAM_BASE_URL +def prowlarr_search_text(query: str) -> str: + """Clean a query the way Prowlarr does before it sends the query to MAM.""" + kept = "".join( + char + for char in query + if char.isalpha() + or char.isdecimal() + or char.isspace() + or char in _PROWLARR_KEPT_PUNCTUATION + or unicodedata.category(char) == "Pd" + ) + return _NON_WORD_RE.sub(" ", kept).strip() + + +def mam_search_options( + indexer: Mapping[str, Any] | None, torznab_categories: Iterable[int] | None +) -> MamSearchOptions: + """Mirror a Prowlarr MyAnonamouse indexer's search settings and the categories searched.""" + fields = indexer.get("fields") if indexer else None + languages = get_indexer_field(fields, "searchLanguages") + return MamSearchOptions( + search_type=_SEARCH_TYPES.get( + coerce_int_like(get_indexer_field(fields, "searchType")) or 0, "all" + ), + search_in=( + "title", + "author", + "narrator", + *( + name + for key, name in _OPTIONAL_SEARCH_FIELDS.items() + if get_indexer_field(fields, key) is True + ), + ), + languages=tuple( + str(language_id) + for language in (languages if isinstance(languages, list) else []) + if (language_id := coerce_int_like(language)) is not None + ), + main_categories=tuple( + dict.fromkeys( + main_category + for category in torznab_categories or () + for main_category in _MAIN_CATEGORIES_BY_TORZNAB.get(category, ()) + ) + ), + ) def _decode_info(raw: object) -> dict[str, Any]: @@ -128,28 +247,25 @@ def parse_torrent_details(item: dict[str, Any]) -> MamTorrentDetails: ) -class MamAuthError(Exception): - """MAM rejected the session ID.""" - - class MamClient: """Minimal MyAnonamouse JSON API client authenticated by a ``mam_id`` cookie.""" - def __init__(self, mam_id: str, base_url: str = DEFAULT_MAM_BASE_URL) -> None: - """Create a client for the given session ID and MAM origin.""" - self.base_url = base_url.rstrip("/") - self._session = requests.Session() - self._session.cookies.set("mam_id", mam_id.strip()) - self._session.headers.update({"Accept": "application/json"}) + def __init__(self, mam_id: str) -> None: + """Create a client for the given session ID.""" + # A plain header rather than a cookie jar entry: requests drops the header on a + # redirect, and redirects are not followed anyway. + self._headers = {"Accept": "application/json", "Cookie": f"mam_id={mam_id.strip()}"} - def _get(self, path: str, params: dict[str, str] | None = None) -> object: - url = self.base_url + path - response = self._session.get( + def _get(self, path: str, params: Mapping[str, str | list[str]] | None = None) -> object: + url = MAM_BASE_URL + path + response = requests.get( url, params=params, + headers=self._headers, timeout=_REQUEST_TIMEOUT_SECONDS, proxies=get_proxies(url), verify=get_ssl_verify(url), + allow_redirects=False, ) if response.status_code == _HTTP_FORBIDDEN: # MAM explains itself in the body (bad cookie, IP/ASN mismatch, ...). @@ -160,7 +276,9 @@ class MamClient: + (f". MAM said: {reply}" if reply else "") ) raise MamAuthError(msg) - response.raise_for_status() + if response.status_code != requests.codes.ok: + msg = f"MyAnonamouse answered HTTP {response.status_code}" + raise requests.exceptions.HTTPError(msg, response=response) return response.json() def get_username(self) -> str | None: @@ -171,29 +289,35 @@ class MamClient: username = data.get("username") return str(username) if username else None - def search(self, text: str) -> list[dict[str, Any]]: - """Search torrents across all categories by title, author, narrator and series.""" - params = { + def search( + self, text: str, options: MamSearchOptions, *, start: int = 0 + ) -> list[dict[str, Any]]: + """Return one page of torrents; fewer than a full page means it was the last.""" + params: dict[str, str | list[str]] = { "tor[text]": text, - "tor[searchType]": "all", + "tor[searchType]": options.search_type, "tor[searchIn]": "torrents", - "tor[srchIn][title]": "true", - "tor[srchIn][author]": "true", - "tor[srchIn][narrator]": "true", - "tor[srchIn][series]": "true", + **{f"tor[srchIn][{field}]": "true" for field in options.search_in}, "tor[cat][]": "0", "tor[sortType]": "default", - "tor[startNumber]": "0", + "tor[startNumber]": str(start), "perpage": str(_RESULTS_PER_PAGE), } + if options.main_categories: + params["tor[main_cat][]"] = list(options.main_categories) + if options.languages: + params["tor[browse_lang][]"] = list(options.languages) + data = self._get(_SEARCH_PATH, params) - if not isinstance(data, dict): - return [] + items = data.get("data") if isinstance(data, dict) else None + if isinstance(items, list): + return [item for item in items if isinstance(item, dict)] + error = data.get("error") if isinstance(data, dict) else None # An empty search answers {"error": "Nothing returned, out of N"}. - items = data.get("data") - if not isinstance(items, list): + if isinstance(error, str) and error.startswith("Nothing returned"): return [] - return [item for item in items if isinstance(item, dict)] + msg = f"Unexpected MyAnonamouse search reply: {str(error or data)[:200]}" + raise MamError(msg) def _cached(torrent_id: int) -> MamTorrentDetails | None: @@ -211,23 +335,86 @@ def _cached(torrent_id: int) -> MamTorrentDetails | None: def _store(details_by_id: dict[int, MamTorrentDetails]) -> None: now = time.time() with _cache_lock: + # Most IDs are never looked up again, so expiry on read alone would let the + # cache grow for as long as Shelfmark runs. + expired = [ + torrent_id + for torrent_id, (_, cached_at) in _cache.items() + if now - cached_at > _CACHE_TTL_SECONDS + ] + for torrent_id in expired: + del _cache[torrent_id] for torrent_id, details in details_by_id.items(): _cache[torrent_id] = (details, now) +def _blocked_reason(mam_id: str) -> str | None: + """Return why MAM must not be called right now, or None if it may be.""" + with _failures_lock: + if _failures.mam_id != mam_id or not _failures.count: + return None + if _failures.count >= _MAX_CONSECUTIVE_FAILURES: + return f"stopped after {_failures.count} consecutive failures" + wait = _failures.retry_at - time.monotonic() + if wait > 0: + return f"backing off for {wait:.0f}s after {_failures.count} failure(s)" + return None + + +def _record_success(mam_id: str) -> None: + with _failures_lock: + if _failures.mam_id == mam_id: + _failures.count = 0 + _failures.retry_at = 0.0 + + +def _record_failure(mam_id: str, error: Exception) -> None: + with _failures_lock: + if _failures.mam_id != mam_id: + _failures.mam_id = mam_id + _failures.count = 0 + _failures.count += 1 + count = _failures.count + delay = min(_BACKOFF_BASE_SECONDS * 2 ** (count - 1), _BACKOFF_MAX_SECONDS) + _failures.retry_at = time.monotonic() + delay + + if count >= _MAX_CONSECUTIVE_FAILURES: + logger.warning( + "MAM enrichment stopped after %s consecutive failures (last: %s). " + "Test the MAM session in the Prowlarr settings to turn it back on.", + count, + error, + ) + else: + logger.warning( + "MAM enrichment failed (%s/%s), next try in %ss: %s", + count, + _MAX_CONSECUTIVE_FAILURES, + delay, + error, + ) + + +def reset_failures() -> None: + """Let enrichment call MAM again straight away (the session test succeeded).""" + with _failures_lock: + _failures.count = 0 + _failures.retry_at = 0.0 + + def lookup_torrent_details( mam_id: str, torrent_ids: set[int], queries: list[str], *, - base_url: str = DEFAULT_MAM_BASE_URL, + options: MamSearchOptions | None = None, deadline: float | None = None, ) -> dict[int, MamTorrentDetails]: """Fetch narrator/series/bitrate for the given MAM torrent IDs (best effort). - Reruns the queries Prowlarr was sent until every ID is found, capped at a few - requests. Any failure is logged and returns what was found so far; enrichment - must never break the Prowlarr search it decorates. + Reruns Prowlarr's search for each query, reading further pages only while IDs + are missing, capped at a few requests. A failure backs MAM off and returns what + was found so far; enrichment must never break the Prowlarr search it decorates. """ found: dict[int, MamTorrentDetails] = {} for torrent_id in torrent_ids: @@ -236,26 +423,36 @@ def lookup_torrent_details( found[torrent_id] = details missing = torrent_ids - found.keys() - if not missing or not mam_id.strip(): + mam_id = mam_id.strip() + if not missing or not mam_id: return found - client = MamClient(mam_id, base_url) - searched = 0 - for query in dict.fromkeys(q.strip() for q in queries if q and q.strip()): - if not missing or searched >= _MAX_SEARCHES: - break + blocked = _blocked_reason(mam_id) + if blocked: + logger.debug("MAM enrichment skipped: %s", blocked) + return found + + client = MamClient(mam_id) + options = options or MamSearchOptions() + # Page 1 of every query before page 2 of any: each query found its own torrents. + pages = deque( + (text, 0) + for text in dict.fromkeys(prowlarr_search_text(query) for query in queries) + if text + ) + requests_made = 0 + while pages and missing and requests_made < _MAX_REQUESTS: if deadline is not None and time.monotonic() > deadline: logger.debug("MAM enrichment: search budget spent, %s torrent(s) left", len(missing)) break - searched += 1 + text, start = pages.popleft() + requests_made += 1 try: - items = client.search(query) - except MamAuthError as e: - logger.warning("MAM enrichment disabled for this search: %s", e) - break + items = client.search(text, options, start=start) except _MAM_REQUEST_ERRORS as e: - logger.warning("MAM enrichment search failed for '%s': %s", query, e) + _record_failure(mam_id, e) break + _record_success(mam_id) fetched: dict[int, MamTorrentDetails] = {} for item in items: @@ -266,11 +463,13 @@ def lookup_torrent_details( for torrent_id in missing & fetched.keys(): found[torrent_id] = fetched[torrent_id] missing -= fetched.keys() + if len(items) >= _RESULTS_PER_PAGE: + pages.append((text, start + len(items))) logger.debug( - "MAM enrichment: %s of %s torrent(s) matched in %s search(es)", + "MAM enrichment: %s of %s torrent(s) matched in %s request(s)", len(found), len(torrent_ids), - searched, + requests_made, ) return found diff --git a/shelfmark/release_sources/prowlarr/settings.py b/shelfmark/release_sources/prowlarr/settings.py index 8519336c..d4f74780 100644 --- a/shelfmark/release_sources/prowlarr/settings.py +++ b/shelfmark/release_sources/prowlarr/settings.py @@ -135,7 +135,7 @@ def _test_prowlarr_connection(current_values: dict[str, Any] | None = None) -> d def _test_mam_connection(current_values: dict[str, Any] | None = None) -> dict[str, Any]: """Check the MAM session ID by asking MAM which account it belongs to.""" - from shelfmark.release_sources.prowlarr.mam import MamAuthError, MamClient + from shelfmark.release_sources.prowlarr.mam import MamClient, MamError, reset_failures current_values = current_values or {} mam_id = _resolve_setting_text(current_values, "PROWLARR_MAM_ID") @@ -144,13 +144,15 @@ def _test_mam_connection(current_values: dict[str, Any] | None = None) -> dict[s try: username = MamClient(mam_id).get_username() - except MamAuthError as e: + except MamError as e: return {"success": False, "message": str(e)} except _PROWLARR_SETTINGS_ERRORS as e: return {"success": False, "message": f"Connection failed: {e!s}"} if not username: return {"success": False, "message": "MyAnonamouse did not return an account for this ID"} + # A working session ends any back-off from earlier failed lookups. + reset_failures() return {"success": True, "message": f"Connected to MyAnonamouse as {username}"} @@ -280,7 +282,10 @@ def prowlarr_config_settings() -> list[SettingsField]: ActionButton( key="test_prowlarr_mam", label="Test MAM Session", - description="Check that MyAnonamouse accepts the session ID", + description=( + "Check that MyAnonamouse accepts the session ID. A passing test also resumes " + "enrichment after it stopped on repeated failures." + ), style="primary", callback=_test_mam_connection, show_when={"field": "PROWLARR_ENABLED", "value": True}, diff --git a/shelfmark/release_sources/prowlarr/source.py b/shelfmark/release_sources/prowlarr/source.py index e4491e7f..1c1f68cf 100644 --- a/shelfmark/release_sources/prowlarr/source.py +++ b/shelfmark/release_sources/prowlarr/source.py @@ -11,6 +11,7 @@ import requests if TYPE_CHECKING: from shelfmark.core.search_plan import ReleaseSearchPlan from shelfmark.metadata_providers import BookMetadata + from shelfmark.release_sources.prowlarr.mam import MamSearchOptions from shelfmark.core.author_match import AUTHOR_UNKNOWN, author_affinity from shelfmark.core.config import config @@ -39,11 +40,12 @@ from shelfmark.release_sources.prowlarr.api import ( IndexerSeedSettings, ProwlarrClient, ProwlarrSearchError, + is_myanonamouse_indexer, ) from shelfmark.release_sources.prowlarr.cache import cache_release from shelfmark.release_sources.prowlarr.mam import ( lookup_torrent_details, - mam_base_url, + mam_search_options, mam_torrent_id, ) from shelfmark.release_sources.prowlarr.utils import ( @@ -89,26 +91,23 @@ def _enrich_mam_releases( releases: list[Release], mam_id: str, queries: list[str], + options: MamSearchOptions, deadline: float | None, ) -> None: """Fill narrator/series/bitrate on MyAnonamouse releases from MAM's own API.""" - ids_by_source_id: dict[str, int] = {} - base_url: str | None = None - for release in releases: - torrent_id = mam_torrent_id(release.info_url) - if torrent_id is None: - continue - ids_by_source_id[release.source_id] = torrent_id - base_url = base_url or mam_base_url(release.info_url) - - if not ids_by_source_id or base_url is None: + ids_by_source_id = { + release.source_id: torrent_id + for release in releases + if (torrent_id := mam_torrent_id(release.info_url)) is not None + } + if not ids_by_source_id: return details_by_id = lookup_torrent_details( mam_id, set(ids_by_source_id.values()), queries, - base_url=base_url, + options=options, deadline=deadline, ) for release in releases: @@ -1126,6 +1125,12 @@ class ProwlarrSource(ReleaseSource): restrict_to=indexer_ids, indexers=enabled_indexers ) enriched_indexer_ids_set = set(enriched_indexer_ids) + mam_indexers = { + indexer_id: indexer + for indexer in enabled_indexers + if is_myanonamouse_indexer(indexer) + and (indexer_id := _coerce_indexer_id(indexer.get("id"))) is not None + } indexer_seed_settings = ( _fetch_indexer_seed_settings(client, indexer_ids) if config.get("PROWLARR_USE_SEED_PREFERENCES", False) @@ -1266,6 +1271,8 @@ class ProwlarrSource(ReleaseSource): results: list[Release] = [] enriched_source_ids: set[str] = set() affinity_by_source_id: dict[str, int] = {} + mam_releases: list[Release] = [] + mam_indexer: dict | None = None # A manual query is the user's own words; ranking it against the # metadata author would second-guess what they typed. wanted_author = "" if plan.manual_query else plan.author @@ -1294,14 +1301,21 @@ class ProwlarrSource(ReleaseSource): if is_enriched: enriched_source_ids.add(release.source_id) + if idx_id_int in mam_indexers: + mam_releases.append(release) + mam_indexer = mam_indexer or mam_indexers[idx_id_int] mam_session_id = _get_mam_session_id() - if mam_session_id: + if mam_session_id and mam_releases: + # Rerun the search Prowlarr sent to MAM, so it returns the same torrents: + # the indexer's own options, and every category once the search expanded. + mam_categories = None if self.last_search_type == "expanded" else categories try: _enrich_mam_releases( - results, + mam_releases, mam_session_id, [variant.title for variant in variants], + mam_search_options(mam_indexer, mam_categories), deadline, ) except _PROWLARR_REQUEST_ERRORS: diff --git a/tests/prowlarr/test_mam_enrichment.py b/tests/prowlarr/test_mam_enrichment.py index 781be1fb..cfbd4589 100644 --- a/tests/prowlarr/test_mam_enrichment.py +++ b/tests/prowlarr/test_mam_enrichment.py @@ -1,12 +1,14 @@ """Tests for MyAnonamouse enrichment of Prowlarr results (narrator, series, bitrate).""" import json +import time import pytest import requests import shelfmark.release_sources.prowlarr.mam as mam import shelfmark.release_sources.prowlarr.source as prowlarr_source +from shelfmark.metadata_providers import BookMetadata from shelfmark.release_sources import ( ColumnSchema, ReleaseColumnConfig, @@ -15,11 +17,16 @@ from shelfmark.release_sources import ( ) from shelfmark.release_sources.prowlarr.mam import ( MamAuthError, + MamClient, + MamError, + MamSearchOptions, lookup_torrent_details, - mam_base_url, + mam_search_options, mam_torrent_id, parse_torrent_details, + prowlarr_search_text, ) +from shelfmark.release_sources.prowlarr.settings import _test_mam_connection from shelfmark.release_sources.prowlarr.source import ( ProwlarrSource, _enrich_mam_releases, @@ -28,8 +35,9 @@ from shelfmark.release_sources.prowlarr.source import ( @pytest.fixture(autouse=True) -def _clear_mam_cache(): +def _reset_mam_state(monkeypatch): mam._cache.clear() + monkeypatch.setattr(mam, "_failures", mam._Failures()) yield mam._cache.clear() @@ -44,16 +52,35 @@ def _mam_item(torrent_id: int, **fields) -> dict: return item +def _items(first_id: int, last_id: int) -> list[dict]: + return [_mam_item(torrent_id) for torrent_id in range(first_id, last_id + 1)] + + class TestParsing: def test_torrent_id_from_prowlarr_info_url(self): assert mam_torrent_id("https://www.myanonamouse.net/t/123456") == 123456 - assert mam_torrent_id("https://cdn.myanonamouse.net/t/42") == 42 + assert mam_torrent_id("https://cdn.myanonamouse.net/t/42/") == 42 assert mam_torrent_id("https://example.org/t/123") is None assert mam_torrent_id(None) is None - def test_base_url_follows_the_result(self): - assert mam_base_url("https://cdn.myanonamouse.net/t/42") == "https://cdn.myanonamouse.net" - assert mam_base_url("https://example.org/t/42") == mam.DEFAULT_MAM_BASE_URL + @pytest.mark.parametrize( + "url", + [ + "https://evil.example?myanonamouse.net/t/123", + "https://evil.example#myanonamouse.net/t/123", + "https://evil.example/myanonamouse.net/t/123", + "https://notmyanonamouse.net/t/123", + "https://www.myanonamouse.net.evil.example/t/123", + "https://evil.example\\@www.myanonamouse.net/t/123", + "https://user@www.myanonamouse.net/t/123", + "ftp://www.myanonamouse.net/t/123", + "https://www.myanonamouse.net/t/123abc", + "https://www.myanonamouse.net/tor/t/123", + "https://[::1/t/123", + ], + ) + def test_urls_that_only_mention_mam_are_not_mam(self, url): + assert mam_torrent_id(url) is None def test_parses_narrator_series_and_bitrate(self): details = parse_torrent_details( @@ -83,22 +110,153 @@ class TestParsing: assert details == mam.MamTorrentDetails() +class TestProwlarrSearchMirror: + @pytest.mark.parametrize( + ("query", "expected"), + [ + ("Empire of Silence", "Empire of Silence"), + ("Ender's Game", "Ender s Game"), + ("Ender’s Game", "Ender s Game"), + ("Dune: Part One", "Dune Part One"), + ("Spider‑Man — Homecoming", "Spider Man Homecoming"), + ("1,000 Years & More", "1000 Years More"), + ("C++ Primer (5th ed.)", "C Primer 5th ed"), + ("Les Misérables", "Les Misérables"), + ("!!!", ""), + ], + ) + def test_query_is_cleaned_like_prowlarr(self, query, expected): + assert prowlarr_search_text(query) == expected + + def test_defaults_match_prowlarr_defaults(self): + options = mam_search_options({"implementation": "MyAnonamouse"}, [3030]) + + assert options == MamSearchOptions(main_categories=("13", "15", "16")) + + def test_indexer_settings_are_mirrored(self): + indexer = { + "fields": [ + {"name": "searchType", "value": 1}, + {"name": "searchInDescription", "value": False}, + {"name": "searchInSeries", "value": True}, + {"name": "searchLanguages", "value": [1, "37", "junk"]}, + ] + } + + options = mam_search_options(indexer, [7000]) + + assert options == MamSearchOptions( + search_type="active", + search_in=("title", "author", "narrator", "series"), + languages=("1", "37"), + main_categories=("14",), + ) + + def test_expanded_search_covers_every_category(self): + assert mam_search_options(None, None).main_categories == () + + +class _FakeResponse: + def __init__(self, status_code=200, payload=None, text=""): + self.status_code = status_code + self._payload = payload + self.text = text + + def json(self): + if self._payload is None: + msg = "not JSON" + raise ValueError(msg) + return self._payload + + +@pytest.fixture +def mam_http(monkeypatch): + """Record every HTTP request the MAM client makes and answer with `response`.""" + + class _Http: + def __init__(self): + self.response = _FakeResponse(payload={"data": []}) + self.calls: list[tuple[str, dict]] = [] + + def get(self, url, **kwargs): + self.calls.append((url, kwargs)) + return self.response + + http = _Http() + monkeypatch.setattr(mam.requests, "get", http.get) + monkeypatch.setattr(mam, "get_proxies", lambda _url: {}) + monkeypatch.setattr(mam, "get_ssl_verify", lambda _url: True) + return http + + +class TestMamClient: + def test_search_repeats_prowlarrs_request(self, mam_http): + options = MamSearchOptions( + search_in=("title", "author", "narrator", "series"), + languages=("1",), + main_categories=("13", "15", "16"), + ) + + MamClient(" session ").search("Empire of Silence", options, start=100) + + [(url, kwargs)] = mam_http.calls + assert url == "https://www.myanonamouse.net/tor/js/loadSearchJSONbasic.php" + assert kwargs["headers"]["Cookie"] == "mam_id=session" + assert kwargs["allow_redirects"] is False + params = kwargs["params"] + assert params["tor[text]"] == "Empire of Silence" + assert params["tor[searchType]"] == "all" + assert params["tor[srchIn][series]"] == "true" + assert "tor[srchIn][description]" not in params + assert params["tor[main_cat][]"] == ["13", "15", "16"] + assert params["tor[browse_lang][]"] == ["1"] + assert params["tor[startNumber]"] == "100" + assert params["perpage"] == "100" + + def test_nothing_returned_is_an_empty_page(self, mam_http): + mam_http.response = _FakeResponse(payload={"error": "Nothing returned, out of 12"}) + + assert MamClient("session").search("q", MamSearchOptions()) == [] + + def test_unexpected_reply_is_an_error(self, mam_http): + mam_http.response = _FakeResponse(payload={"error": "Invalid search parameters"}) + + with pytest.raises(MamError, match="Invalid search parameters"): + MamClient("session").search("q", MamSearchOptions()) + + def test_redirect_is_an_error_not_followed(self, mam_http): + mam_http.response = _FakeResponse(status_code=302) + + with pytest.raises(requests.exceptions.HTTPError, match="302"): + MamClient("session").search("q", MamSearchOptions()) + assert len(mam_http.calls) == 1 + + def test_forbidden_carries_mams_reply(self, mam_http): + mam_http.response = _FakeResponse(status_code=403, text="Bad cookie: ASN mismatch") + + with pytest.raises(MamAuthError, match="ASN mismatch"): + MamClient("session").get_username() + + class _FakeMamClient: + """Serves pages of `items_by_query` and records each (query, start) requested.""" + def __init__(self, items_by_query=None, error=None): self.items_by_query = items_by_query or {} self.error = error - self.queries: list[str] = [] + self.requests: list[tuple[str, int]] = [] + self.options: MamSearchOptions | None = None - def __call__(self, mam_id, base_url=mam.DEFAULT_MAM_BASE_URL): + def __call__(self, mam_id): self.mam_id = mam_id - self.base_url = base_url return self - def search(self, text): - self.queries.append(text) + def search(self, text, options, *, start=0): + self.requests.append((text, start)) + self.options = options if self.error: raise self.error - return self.items_by_query.get(text, []) + return self.items_by_query.get(text, [])[start : start + mam._RESULTS_PER_PAGE] class TestLookup: @@ -111,7 +269,15 @@ class TestLookup: found = lookup_torrent_details("session", {1}, ["Empire of Silence", "Other title"]) assert found[1].narrator == "Roukin" - assert fake.queries == ["Empire of Silence"] + assert fake.requests == [("Empire of Silence", 0)] + + def test_queries_are_sent_as_prowlarr_sent_them(self, monkeypatch): + fake = _FakeMamClient() + monkeypatch.setattr(mam, "MamClient", fake) + + lookup_torrent_details("session", {1}, ["Ender's Game", "Ender’s Game", "!!!"]) + + assert fake.requests == [("Ender s Game", 0)] def test_uses_cache_on_repeat(self, monkeypatch): fake = _FakeMamClient({"q": [_mam_item(1, narrator_info=json.dumps({"1": "Roukin"}))]}) @@ -120,25 +286,160 @@ class TestLookup: lookup_torrent_details("session", {1}, ["q"]) lookup_torrent_details("session", {1}, ["q"]) - assert fake.queries == ["q"] + assert fake.requests == [("q", 0)] + + def test_expired_entries_are_pruned_when_storing(self): + mam._cache[5] = (mam.MamTorrentDetails(), time.time() - 2 * mam._CACHE_TTL_SECONDS) + + mam._store({1: mam.MamTorrentDetails()}) + + assert set(mam._cache) == {1} + + def test_pages_until_every_id_is_found(self, monkeypatch): + fake = _FakeMamClient({"q": _items(1, 150)}) + monkeypatch.setattr(mam, "MamClient", fake) + + found = lookup_torrent_details("session", {5, 140}, ["q"]) + + assert set(found) == {5, 140} + assert fake.requests == [("q", 0), ("q", 100)] + + def test_first_page_of_every_query_comes_before_further_pages(self, monkeypatch): + fake = _FakeMamClient({"a": [*_items(1, 100), _mam_item(500)], "b": [_mam_item(900)]}) + monkeypatch.setattr(mam, "MamClient", fake) + + found = lookup_torrent_details("session", {500, 900}, ["a", "b"]) + + assert set(found) == {500, 900} + assert fake.requests == [("a", 0), ("b", 0), ("a", 100)] + + def test_a_short_page_is_the_last(self, monkeypatch): + fake = _FakeMamClient({"q": _items(1, 50)}) + monkeypatch.setattr(mam, "MamClient", fake) + + assert lookup_torrent_details("session", {999}, ["q"]) == {} + assert fake.requests == [("q", 0)] + + def test_requests_are_capped(self, monkeypatch): + fake = _FakeMamClient({"q": _items(1, 1000)}) + monkeypatch.setattr(mam, "MamClient", fake) + + lookup_torrent_details("session", {999}, ["q"]) + + assert len(fake.requests) == mam._MAX_REQUESTS @pytest.mark.parametrize( "error", - [MamAuthError("rejected"), requests.exceptions.ConnectionError("down")], + [ + MamAuthError("rejected"), + MamError("odd reply"), + requests.exceptions.ConnectionError("down"), + ValueError("not JSON"), + ], ) def test_failures_return_empty_instead_of_raising(self, monkeypatch, error): fake = _FakeMamClient(error=error) monkeypatch.setattr(mam, "MamClient", fake) assert lookup_torrent_details("session", {1}, ["q", "r"]) == {} - assert fake.queries == ["q"] + assert fake.requests == [("q", 0)] def test_expired_deadline_skips_the_request(self, monkeypatch): fake = _FakeMamClient() monkeypatch.setattr(mam, "MamClient", fake) assert lookup_torrent_details("session", {1}, ["q"], deadline=0.0) == {} - assert fake.queries == [] + assert fake.requests == [] + + +class TestFailureBackoff: + @staticmethod + def _failing(monkeypatch, error=None): + fake = _FakeMamClient(error=error or requests.exceptions.ConnectionError("down")) + monkeypatch.setattr(mam, "MamClient", fake) + return fake + + @staticmethod + def _retry_now(): + mam._failures.retry_at = 0.0 + + def test_a_failure_holds_the_next_lookup_back(self, monkeypatch): + fake = self._failing(monkeypatch) + + lookup_torrent_details("session", {1}, ["q"]) + lookup_torrent_details("session", {1}, ["q"]) + + assert fake.requests == [("q", 0)] + assert mam._failures.count == 1 + assert mam._failures.retry_at - time.monotonic() == pytest.approx(60, abs=5) + + def test_backoff_doubles_up_to_half_an_hour(self, monkeypatch): + self._failing(monkeypatch) + delays = [] + + for _ in range(mam._MAX_CONSECUTIVE_FAILURES - 1): + self._retry_now() + before = time.monotonic() + lookup_torrent_details("session", {1}, ["q"]) + delays.append(round(mam._failures.retry_at - before)) + + assert delays == [60, 120, 240, 480, 960, 1800, 1800, 1800, 1800] + + def test_ten_failures_in_a_row_stop_enrichment(self, monkeypatch): + fake = self._failing(monkeypatch, MamAuthError("rejected")) + + for _ in range(mam._MAX_CONSECUTIVE_FAILURES + 3): + self._retry_now() + lookup_torrent_details("session", {1}, ["q"]) + + assert len(fake.requests) == mam._MAX_CONSECUTIVE_FAILURES + assert mam._blocked_reason("session") == "stopped after 10 consecutive failures" + + def test_a_success_resets_the_count(self, monkeypatch): + fake = self._failing(monkeypatch) + for _ in range(3): + self._retry_now() + lookup_torrent_details("session", {1}, ["q"]) + + fake.error = None + self._retry_now() + lookup_torrent_details("session", {1}, ["q"]) + + assert mam._failures.count == 0 + assert mam._blocked_reason("session") is None + + def test_a_new_session_id_starts_over(self, monkeypatch): + fake = self._failing(monkeypatch) + for _ in range(mam._MAX_CONSECUTIVE_FAILURES): + self._retry_now() + lookup_torrent_details("old", {1}, ["q"]) + + fake.error = None + lookup_torrent_details("new", {1}, ["q"]) + + assert fake.requests[-1] == ("q", 0) + assert len(fake.requests) == mam._MAX_CONSECUTIVE_FAILURES + 1 + + def test_a_passing_session_test_turns_enrichment_back_on(self, monkeypatch): + self._failing(monkeypatch) + for _ in range(mam._MAX_CONSECUTIVE_FAILURES): + self._retry_now() + lookup_torrent_details("session", {1}, ["q"]) + assert mam._blocked_reason("session") is not None + + monkeypatch.setattr(mam, "MamClient", _UsernameClient) + result = _test_mam_connection({"PROWLARR_MAM_ID": "session"}) + + assert result["success"] is True + assert mam._blocked_reason("session") is None + + +class _UsernameClient: + def __init__(self, mam_id): + self.mam_id = mam_id + + def get_username(self): + return "reader" class TestReleaseEnrichment: @@ -156,7 +457,7 @@ class TestReleaseEnrichment: "audiobook", ) - def test_only_mam_releases_are_enriched(self, monkeypatch): + def test_mam_details_fill_the_release(self, monkeypatch): fake = _FakeMamClient( { "Empire of Silence": [ @@ -170,19 +471,28 @@ class TestReleaseEnrichment: } ) monkeypatch.setattr(mam, "MamClient", fake) + options = MamSearchOptions(main_categories=("13", "15", "16")) mam_release = self._release("https://www.myanonamouse.net/t/99", "mam-guid") - other_release = self._release("https://tracker.example/t/99", "other-guid") _enrich_mam_releases( - [mam_release, other_release], "session", ["Empire of Silence"], deadline=None + [mam_release], "session", ["Empire of Silence"], options, deadline=None ) assert mam_release.extra["narrator"] == "Samuel Roukin" assert mam_release.extra["series"] == "The Sun Eater #1" assert mam_release.extra["bitrate"] == "64 Kbps" assert mam_release.extra["bitrate_value"] == 64 - assert "narrator" not in other_release.extra - assert fake.base_url == "https://www.myanonamouse.net" + assert fake.options == options + + def test_a_url_that_only_mentions_mam_is_never_looked_up(self, mam_http): + spoofed = self._release("https://evil.example?myanonamouse.net/t/99", "guid") + + _enrich_mam_releases( + [spoofed], "session", ["Empire of Silence"], MamSearchOptions(), deadline=None + ) + + assert mam_http.calls == [] + assert "narrator" not in spoofed.extra def test_torznab_bitrate_attribute_is_used_for_other_indexers(self): release = _prowlarr_result_to_release( @@ -198,6 +508,117 @@ class TestReleaseEnrichment: assert release.extra["bitrate_value"] == 320 +MAM_INDEXER_ID = 1 +OTHER_INDEXER_ID = 2 + + +class _ProwlarrWithMam: + """A Prowlarr with MyAnonamouse and one other indexer enabled.""" + + indexer_timeout = 90 + + def __init__(self, mam_fields=None): + self.mam_fields = mam_fields or [] + + def get_enabled_indexers_detailed(self, *, raise_on_error=False): + del raise_on_error + capabilities = {"categories": [{"id": 7000, "subCategories": []}]} + return [ + { + "id": MAM_INDEXER_ID, + "enable": True, + "implementation": "MyAnonamouse", + "fields": self.mam_fields, + "capabilities": capabilities, + }, + { + "id": OTHER_INDEXER_ID, + "enable": True, + "implementation": "Torznab", + "capabilities": capabilities, + }, + ] + + def get_enriched_indexer_ids(self, restrict_to=None, indexers=None): + del restrict_to, indexers + return [MAM_INDEXER_ID] + + def torznab_search( + self, *, indexer_id, query, categories=None, search_type="book", limit=100, offset=0 + ): + del query, categories, search_type, limit, offset + # The other indexer's result also claims a MyAnonamouse URL. + torrent_id = 99 if indexer_id == MAM_INDEXER_ID else 77 + return [ + { + "guid": f"guid-{indexer_id}", + "title": "Empire of Silence", + "indexerId": indexer_id, + "indexer": f"Indexer {indexer_id}", + "protocol": "torrent", + "categories": [{"id": 7020}], + "infoUrl": f"https://www.myanonamouse.net/t/{torrent_id}", + } + ] + + +class TestSearchEnrichment: + def _search(self, monkeypatch, client, *, expand_search=False): + from shelfmark.core.search_plan import build_release_search_plan + + values = { + "PROWLARR_INDEXERS": "", + "PROWLARR_AUTO_EXPAND": False, + "PROWLARR_MAM_ID": "session", + } + monkeypatch.setattr( + prowlarr_source.config, "get", lambda key, default=None: values.get(key, default) + ) + lookups = [] + + def fake_lookup(mam_id, torrent_ids, queries, *, options, deadline): + del deadline + lookups.append((mam_id, torrent_ids, queries, options)) + return {99: mam.MamTorrentDetails(series="The Sun Eater #1")} + + monkeypatch.setattr(prowlarr_source, "lookup_torrent_details", fake_lookup) + source = ProwlarrSource() + monkeypatch.setattr(source, "_get_client", lambda: client) + book = BookMetadata( + provider="hardcover", + provider_id="1", + title="Empire of Silence", + authors=["Christopher Ruocchio"], + ) + plan = build_release_search_plan(book, languages=["en"]) + results = source.search(book, plan, expand_search=expand_search, content_type="ebook") + return results, lookups + + def test_only_the_mam_indexer_is_enriched(self, monkeypatch): + results, lookups = self._search(monkeypatch, _ProwlarrWithMam()) + + [(mam_id, torrent_ids, queries, _options)] = lookups + assert (mam_id, torrent_ids, queries) == ("session", {99}, ["Empire of Silence"]) + series = {release.indexer: release.extra.get("series") for release in results} + assert series == {"Indexer 1": "The Sun Eater #1", "Indexer 2": None} + + def test_lookup_mirrors_the_mam_indexer_settings(self, monkeypatch): + client = _ProwlarrWithMam( + [{"name": "searchType", "value": 2}, {"name": "searchLanguages", "value": [1]}] + ) + + _results, lookups = self._search(monkeypatch, client) + + assert lookups[0][3] == MamSearchOptions( + search_type="fl", languages=("1",), main_categories=("14",) + ) + + def test_expanded_search_looks_in_every_category(self, monkeypatch): + _results, lookups = self._search(monkeypatch, _ProwlarrWithMam(), expand_search=True) + + assert lookups[0][3].main_categories == () + + class TestColumnConfig: def _source(self, monkeypatch, mam_id: str) -> ProwlarrSource: monkeypatch.setattr(prowlarr_source, "_get_mam_session_id", lambda: mam_id)