fix(mam): keep the session ID on MAM and rerun Prowlarr's exact search (#1399)

Follow-up to #1390.

- The origin came from each result's infoUrl, matched by a regex that
did
  not check the host, and the mam_id cookie had no domain. Any Prowlarr
  indexer returning a URL like https://evil.example?myanonamouse.net/t/1
  sent the session ID to evil.example. Requests now always go to
https://www.myanonamouse.net (Prowlarr's only MAM URL), with the cookie
  as a header and redirects off. Only results from Prowlarr's
  MyAnonamouse indexer are looked up, and their URLs must be on
  myanonamouse.net.
- The lookup searched every category and read one page, so for common
  titles most of Prowlarr's results were missed (a "Dune" audiobook
  search: 56 audiobooks on the first 100 of 328 matches). It now reruns
  Prowlarr's exact search: the same query clean-up, the MAM main
  categories behind the Torznab categories searched (13/15/16 for
  audiobooks, 14 for e-books, all once expanded), and the MAM indexer's
  own search type, search-in options and languages. Further pages are
  read while IDs are missing, page 1 of every title first, at most 4
  requests per search.
- Failed requests back off for 1, 2, 4 ... up to 30 minutes. The 10th
  consecutive failure stops enrichment until Test MAM Session passes,
  the session ID changes, or Shelfmark restarts.
- The detail cache prunes expired entries instead of growing for as long
  as Shelfmark runs.
This commit is contained in:
CaliBrain
2026-09-25 21:21:54 -04:00
committed by GitHub
parent efb1f66bc3
commit 893f7bdb92
6 changed files with 760 additions and 111 deletions
+443 -22
View File
@@ -1,12 +1,14 @@
"""Tests for MyAnonamouse enrichment of Prowlarr results (narrator, series, bitrate)."""
import json
import time
import pytest
import requests
import shelfmark.release_sources.prowlarr.mam as mam
import shelfmark.release_sources.prowlarr.source as prowlarr_source
from shelfmark.metadata_providers import BookMetadata
from shelfmark.release_sources import (
ColumnSchema,
ReleaseColumnConfig,
@@ -15,11 +17,16 @@ from shelfmark.release_sources import (
)
from shelfmark.release_sources.prowlarr.mam import (
MamAuthError,
MamClient,
MamError,
MamSearchOptions,
lookup_torrent_details,
mam_base_url,
mam_search_options,
mam_torrent_id,
parse_torrent_details,
prowlarr_search_text,
)
from shelfmark.release_sources.prowlarr.settings import _test_mam_connection
from shelfmark.release_sources.prowlarr.source import (
ProwlarrSource,
_enrich_mam_releases,
@@ -28,8 +35,9 @@ from shelfmark.release_sources.prowlarr.source import (
@pytest.fixture(autouse=True)
def _clear_mam_cache():
def _reset_mam_state(monkeypatch):
mam._cache.clear()
monkeypatch.setattr(mam, "_failures", mam._Failures())
yield
mam._cache.clear()
@@ -44,16 +52,35 @@ def _mam_item(torrent_id: int, **fields) -> dict:
return item
def _items(first_id: int, last_id: int) -> list[dict]:
return [_mam_item(torrent_id) for torrent_id in range(first_id, last_id + 1)]
class TestParsing:
def test_torrent_id_from_prowlarr_info_url(self):
assert mam_torrent_id("https://www.myanonamouse.net/t/123456") == 123456
assert mam_torrent_id("https://cdn.myanonamouse.net/t/42") == 42
assert mam_torrent_id("https://cdn.myanonamouse.net/t/42/") == 42
assert mam_torrent_id("https://example.org/t/123") is None
assert mam_torrent_id(None) is None
def test_base_url_follows_the_result(self):
assert mam_base_url("https://cdn.myanonamouse.net/t/42") == "https://cdn.myanonamouse.net"
assert mam_base_url("https://example.org/t/42") == mam.DEFAULT_MAM_BASE_URL
@pytest.mark.parametrize(
"url",
[
"https://evil.example?myanonamouse.net/t/123",
"https://evil.example#myanonamouse.net/t/123",
"https://evil.example/myanonamouse.net/t/123",
"https://notmyanonamouse.net/t/123",
"https://www.myanonamouse.net.evil.example/t/123",
"https://evil.example\\@www.myanonamouse.net/t/123",
"https://user@www.myanonamouse.net/t/123",
"ftp://www.myanonamouse.net/t/123",
"https://www.myanonamouse.net/t/123abc",
"https://www.myanonamouse.net/tor/t/123",
"https://[::1/t/123",
],
)
def test_urls_that_only_mention_mam_are_not_mam(self, url):
assert mam_torrent_id(url) is None
def test_parses_narrator_series_and_bitrate(self):
details = parse_torrent_details(
@@ -83,22 +110,153 @@ class TestParsing:
assert details == mam.MamTorrentDetails()
class TestProwlarrSearchMirror:
@pytest.mark.parametrize(
("query", "expected"),
[
("Empire of Silence", "Empire of Silence"),
("Ender's Game", "Ender s Game"),
("Ender’s Game", "Ender s Game"),
("Dune: Part One", "Dune Part One"),
("Spider‑Man — Homecoming", "Spider Man Homecoming"),
("1,000 Years & More", "1000 Years More"),
("C++ Primer (5th ed.)", "C Primer 5th ed"),
("Les Misérables", "Les Misérables"),
("!!!", ""),
],
)
def test_query_is_cleaned_like_prowlarr(self, query, expected):
assert prowlarr_search_text(query) == expected
def test_defaults_match_prowlarr_defaults(self):
options = mam_search_options({"implementation": "MyAnonamouse"}, [3030])
assert options == MamSearchOptions(main_categories=("13", "15", "16"))
def test_indexer_settings_are_mirrored(self):
indexer = {
"fields": [
{"name": "searchType", "value": 1},
{"name": "searchInDescription", "value": False},
{"name": "searchInSeries", "value": True},
{"name": "searchLanguages", "value": [1, "37", "junk"]},
]
}
options = mam_search_options(indexer, [7000])
assert options == MamSearchOptions(
search_type="active",
search_in=("title", "author", "narrator", "series"),
languages=("1", "37"),
main_categories=("14",),
)
def test_expanded_search_covers_every_category(self):
assert mam_search_options(None, None).main_categories == ()
class _FakeResponse:
def __init__(self, status_code=200, payload=None, text=""):
self.status_code = status_code
self._payload = payload
self.text = text
def json(self):
if self._payload is None:
msg = "not JSON"
raise ValueError(msg)
return self._payload
@pytest.fixture
def mam_http(monkeypatch):
"""Record every HTTP request the MAM client makes and answer with `response`."""
class _Http:
def __init__(self):
self.response = _FakeResponse(payload={"data": []})
self.calls: list[tuple[str, dict]] = []
def get(self, url, **kwargs):
self.calls.append((url, kwargs))
return self.response
http = _Http()
monkeypatch.setattr(mam.requests, "get", http.get)
monkeypatch.setattr(mam, "get_proxies", lambda _url: {})
monkeypatch.setattr(mam, "get_ssl_verify", lambda _url: True)
return http
class TestMamClient:
def test_search_repeats_prowlarrs_request(self, mam_http):
options = MamSearchOptions(
search_in=("title", "author", "narrator", "series"),
languages=("1",),
main_categories=("13", "15", "16"),
)
MamClient(" session ").search("Empire of Silence", options, start=100)
[(url, kwargs)] = mam_http.calls
assert url == "https://www.myanonamouse.net/tor/js/loadSearchJSONbasic.php"
assert kwargs["headers"]["Cookie"] == "mam_id=session"
assert kwargs["allow_redirects"] is False
params = kwargs["params"]
assert params["tor[text]"] == "Empire of Silence"
assert params["tor[searchType]"] == "all"
assert params["tor[srchIn][series]"] == "true"
assert "tor[srchIn][description]" not in params
assert params["tor[main_cat][]"] == ["13", "15", "16"]
assert params["tor[browse_lang][]"] == ["1"]
assert params["tor[startNumber]"] == "100"
assert params["perpage"] == "100"
def test_nothing_returned_is_an_empty_page(self, mam_http):
mam_http.response = _FakeResponse(payload={"error": "Nothing returned, out of 12"})
assert MamClient("session").search("q", MamSearchOptions()) == []
def test_unexpected_reply_is_an_error(self, mam_http):
mam_http.response = _FakeResponse(payload={"error": "Invalid search parameters"})
with pytest.raises(MamError, match="Invalid search parameters"):
MamClient("session").search("q", MamSearchOptions())
def test_redirect_is_an_error_not_followed(self, mam_http):
mam_http.response = _FakeResponse(status_code=302)
with pytest.raises(requests.exceptions.HTTPError, match="302"):
MamClient("session").search("q", MamSearchOptions())
assert len(mam_http.calls) == 1
def test_forbidden_carries_mams_reply(self, mam_http):
mam_http.response = _FakeResponse(status_code=403, text="Bad cookie: ASN mismatch")
with pytest.raises(MamAuthError, match="ASN mismatch"):
MamClient("session").get_username()
class _FakeMamClient:
"""Serves pages of `items_by_query` and records each (query, start) requested."""
def __init__(self, items_by_query=None, error=None):
self.items_by_query = items_by_query or {}
self.error = error
self.queries: list[str] = []
self.requests: list[tuple[str, int]] = []
self.options: MamSearchOptions | None = None
def __call__(self, mam_id, base_url=mam.DEFAULT_MAM_BASE_URL):
def __call__(self, mam_id):
self.mam_id = mam_id
self.base_url = base_url
return self
def search(self, text):
self.queries.append(text)
def search(self, text, options, *, start=0):
self.requests.append((text, start))
self.options = options
if self.error:
raise self.error
return self.items_by_query.get(text, [])
return self.items_by_query.get(text, [])[start : start + mam._RESULTS_PER_PAGE]
class TestLookup:
@@ -111,7 +269,15 @@ class TestLookup:
found = lookup_torrent_details("session", {1}, ["Empire of Silence", "Other title"])
assert found[1].narrator == "Roukin"
assert fake.queries == ["Empire of Silence"]
assert fake.requests == [("Empire of Silence", 0)]
def test_queries_are_sent_as_prowlarr_sent_them(self, monkeypatch):
fake = _FakeMamClient()
monkeypatch.setattr(mam, "MamClient", fake)
lookup_torrent_details("session", {1}, ["Ender's Game", "Ender’s Game", "!!!"])
assert fake.requests == [("Ender s Game", 0)]
def test_uses_cache_on_repeat(self, monkeypatch):
fake = _FakeMamClient({"q": [_mam_item(1, narrator_info=json.dumps({"1": "Roukin"}))]})
@@ -120,25 +286,160 @@ class TestLookup:
lookup_torrent_details("session", {1}, ["q"])
lookup_torrent_details("session", {1}, ["q"])
assert fake.queries == ["q"]
assert fake.requests == [("q", 0)]
def test_expired_entries_are_pruned_when_storing(self):
mam._cache[5] = (mam.MamTorrentDetails(), time.time() - 2 * mam._CACHE_TTL_SECONDS)
mam._store({1: mam.MamTorrentDetails()})
assert set(mam._cache) == {1}
def test_pages_until_every_id_is_found(self, monkeypatch):
fake = _FakeMamClient({"q": _items(1, 150)})
monkeypatch.setattr(mam, "MamClient", fake)
found = lookup_torrent_details("session", {5, 140}, ["q"])
assert set(found) == {5, 140}
assert fake.requests == [("q", 0), ("q", 100)]
def test_first_page_of_every_query_comes_before_further_pages(self, monkeypatch):
fake = _FakeMamClient({"a": [*_items(1, 100), _mam_item(500)], "b": [_mam_item(900)]})
monkeypatch.setattr(mam, "MamClient", fake)
found = lookup_torrent_details("session", {500, 900}, ["a", "b"])
assert set(found) == {500, 900}
assert fake.requests == [("a", 0), ("b", 0), ("a", 100)]
def test_a_short_page_is_the_last(self, monkeypatch):
fake = _FakeMamClient({"q": _items(1, 50)})
monkeypatch.setattr(mam, "MamClient", fake)
assert lookup_torrent_details("session", {999}, ["q"]) == {}
assert fake.requests == [("q", 0)]
def test_requests_are_capped(self, monkeypatch):
fake = _FakeMamClient({"q": _items(1, 1000)})
monkeypatch.setattr(mam, "MamClient", fake)
lookup_torrent_details("session", {999}, ["q"])
assert len(fake.requests) == mam._MAX_REQUESTS
@pytest.mark.parametrize(
"error",
[MamAuthError("rejected"), requests.exceptions.ConnectionError("down")],
[
MamAuthError("rejected"),
MamError("odd reply"),
requests.exceptions.ConnectionError("down"),
ValueError("not JSON"),
],
)
def test_failures_return_empty_instead_of_raising(self, monkeypatch, error):
fake = _FakeMamClient(error=error)
monkeypatch.setattr(mam, "MamClient", fake)
assert lookup_torrent_details("session", {1}, ["q", "r"]) == {}
assert fake.queries == ["q"]
assert fake.requests == [("q", 0)]
def test_expired_deadline_skips_the_request(self, monkeypatch):
fake = _FakeMamClient()
monkeypatch.setattr(mam, "MamClient", fake)
assert lookup_torrent_details("session", {1}, ["q"], deadline=0.0) == {}
assert fake.queries == []
assert fake.requests == []
class TestFailureBackoff:
@staticmethod
def _failing(monkeypatch, error=None):
fake = _FakeMamClient(error=error or requests.exceptions.ConnectionError("down"))
monkeypatch.setattr(mam, "MamClient", fake)
return fake
@staticmethod
def _retry_now():
mam._failures.retry_at = 0.0
def test_a_failure_holds_the_next_lookup_back(self, monkeypatch):
fake = self._failing(monkeypatch)
lookup_torrent_details("session", {1}, ["q"])
lookup_torrent_details("session", {1}, ["q"])
assert fake.requests == [("q", 0)]
assert mam._failures.count == 1
assert mam._failures.retry_at - time.monotonic() == pytest.approx(60, abs=5)
def test_backoff_doubles_up_to_half_an_hour(self, monkeypatch):
self._failing(monkeypatch)
delays = []
for _ in range(mam._MAX_CONSECUTIVE_FAILURES - 1):
self._retry_now()
before = time.monotonic()
lookup_torrent_details("session", {1}, ["q"])
delays.append(round(mam._failures.retry_at - before))
assert delays == [60, 120, 240, 480, 960, 1800, 1800, 1800, 1800]
def test_ten_failures_in_a_row_stop_enrichment(self, monkeypatch):
fake = self._failing(monkeypatch, MamAuthError("rejected"))
for _ in range(mam._MAX_CONSECUTIVE_FAILURES + 3):
self._retry_now()
lookup_torrent_details("session", {1}, ["q"])
assert len(fake.requests) == mam._MAX_CONSECUTIVE_FAILURES
assert mam._blocked_reason("session") == "stopped after 10 consecutive failures"
def test_a_success_resets_the_count(self, monkeypatch):
fake = self._failing(monkeypatch)
for _ in range(3):
self._retry_now()
lookup_torrent_details("session", {1}, ["q"])
fake.error = None
self._retry_now()
lookup_torrent_details("session", {1}, ["q"])
assert mam._failures.count == 0
assert mam._blocked_reason("session") is None
def test_a_new_session_id_starts_over(self, monkeypatch):
fake = self._failing(monkeypatch)
for _ in range(mam._MAX_CONSECUTIVE_FAILURES):
self._retry_now()
lookup_torrent_details("old", {1}, ["q"])
fake.error = None
lookup_torrent_details("new", {1}, ["q"])
assert fake.requests[-1] == ("q", 0)
assert len(fake.requests) == mam._MAX_CONSECUTIVE_FAILURES + 1
def test_a_passing_session_test_turns_enrichment_back_on(self, monkeypatch):
self._failing(monkeypatch)
for _ in range(mam._MAX_CONSECUTIVE_FAILURES):
self._retry_now()
lookup_torrent_details("session", {1}, ["q"])
assert mam._blocked_reason("session") is not None
monkeypatch.setattr(mam, "MamClient", _UsernameClient)
result = _test_mam_connection({"PROWLARR_MAM_ID": "session"})
assert result["success"] is True
assert mam._blocked_reason("session") is None
class _UsernameClient:
def __init__(self, mam_id):
self.mam_id = mam_id
def get_username(self):
return "reader"
class TestReleaseEnrichment:
@@ -156,7 +457,7 @@ class TestReleaseEnrichment:
"audiobook",
)
def test_only_mam_releases_are_enriched(self, monkeypatch):
def test_mam_details_fill_the_release(self, monkeypatch):
fake = _FakeMamClient(
{
"Empire of Silence": [
@@ -170,19 +471,28 @@ class TestReleaseEnrichment:
}
)
monkeypatch.setattr(mam, "MamClient", fake)
options = MamSearchOptions(main_categories=("13", "15", "16"))
mam_release = self._release("https://www.myanonamouse.net/t/99", "mam-guid")
other_release = self._release("https://tracker.example/t/99", "other-guid")
_enrich_mam_releases(
[mam_release, other_release], "session", ["Empire of Silence"], deadline=None
[mam_release], "session", ["Empire of Silence"], options, deadline=None
)
assert mam_release.extra["narrator"] == "Samuel Roukin"
assert mam_release.extra["series"] == "The Sun Eater #1"
assert mam_release.extra["bitrate"] == "64 Kbps"
assert mam_release.extra["bitrate_value"] == 64
assert "narrator" not in other_release.extra
assert fake.base_url == "https://www.myanonamouse.net"
assert fake.options == options
def test_a_url_that_only_mentions_mam_is_never_looked_up(self, mam_http):
spoofed = self._release("https://evil.example?myanonamouse.net/t/99", "guid")
_enrich_mam_releases(
[spoofed], "session", ["Empire of Silence"], MamSearchOptions(), deadline=None
)
assert mam_http.calls == []
assert "narrator" not in spoofed.extra
def test_torznab_bitrate_attribute_is_used_for_other_indexers(self):
release = _prowlarr_result_to_release(
@@ -198,6 +508,117 @@ class TestReleaseEnrichment:
assert release.extra["bitrate_value"] == 320
MAM_INDEXER_ID = 1
OTHER_INDEXER_ID = 2
class _ProwlarrWithMam:
"""A Prowlarr with MyAnonamouse and one other indexer enabled."""
indexer_timeout = 90
def __init__(self, mam_fields=None):
self.mam_fields = mam_fields or []
def get_enabled_indexers_detailed(self, *, raise_on_error=False):
del raise_on_error
capabilities = {"categories": [{"id": 7000, "subCategories": []}]}
return [
{
"id": MAM_INDEXER_ID,
"enable": True,
"implementation": "MyAnonamouse",
"fields": self.mam_fields,
"capabilities": capabilities,
},
{
"id": OTHER_INDEXER_ID,
"enable": True,
"implementation": "Torznab",
"capabilities": capabilities,
},
]
def get_enriched_indexer_ids(self, restrict_to=None, indexers=None):
del restrict_to, indexers
return [MAM_INDEXER_ID]
def torznab_search(
self, *, indexer_id, query, categories=None, search_type="book", limit=100, offset=0
):
del query, categories, search_type, limit, offset
# The other indexer's result also claims a MyAnonamouse URL.
torrent_id = 99 if indexer_id == MAM_INDEXER_ID else 77
return [
{
"guid": f"guid-{indexer_id}",
"title": "Empire of Silence",
"indexerId": indexer_id,
"indexer": f"Indexer {indexer_id}",
"protocol": "torrent",
"categories": [{"id": 7020}],
"infoUrl": f"https://www.myanonamouse.net/t/{torrent_id}",
}
]
class TestSearchEnrichment:
def _search(self, monkeypatch, client, *, expand_search=False):
from shelfmark.core.search_plan import build_release_search_plan
values = {
"PROWLARR_INDEXERS": "",
"PROWLARR_AUTO_EXPAND": False,
"PROWLARR_MAM_ID": "session",
}
monkeypatch.setattr(
prowlarr_source.config, "get", lambda key, default=None: values.get(key, default)
)
lookups = []
def fake_lookup(mam_id, torrent_ids, queries, *, options, deadline):
del deadline
lookups.append((mam_id, torrent_ids, queries, options))
return {99: mam.MamTorrentDetails(series="The Sun Eater #1")}
monkeypatch.setattr(prowlarr_source, "lookup_torrent_details", fake_lookup)
source = ProwlarrSource()
monkeypatch.setattr(source, "_get_client", lambda: client)
book = BookMetadata(
provider="hardcover",
provider_id="1",
title="Empire of Silence",
authors=["Christopher Ruocchio"],
)
plan = build_release_search_plan(book, languages=["en"])
results = source.search(book, plan, expand_search=expand_search, content_type="ebook")
return results, lookups
def test_only_the_mam_indexer_is_enriched(self, monkeypatch):
results, lookups = self._search(monkeypatch, _ProwlarrWithMam())
[(mam_id, torrent_ids, queries, _options)] = lookups
assert (mam_id, torrent_ids, queries) == ("session", {99}, ["Empire of Silence"])
series = {release.indexer: release.extra.get("series") for release in results}
assert series == {"Indexer 1": "The Sun Eater #1", "Indexer 2": None}
def test_lookup_mirrors_the_mam_indexer_settings(self, monkeypatch):
client = _ProwlarrWithMam(
[{"name": "searchType", "value": 2}, {"name": "searchLanguages", "value": [1]}]
)
_results, lookups = self._search(monkeypatch, client)
assert lookups[0][3] == MamSearchOptions(
search_type="fl", languages=("1",), main_categories=("14",)
)
def test_expanded_search_looks_in_every_category(self, monkeypatch):
_results, lookups = self._search(monkeypatch, _ProwlarrWithMam(), expand_search=True)
assert lookups[0][3].main_categories == ()
class TestColumnConfig:
def _source(self, monkeypatch, mam_id: str) -> ProwlarrSource:
monkeypatch.setattr(prowlarr_source, "_get_mam_session_id", lambda: mam_id)