mirror of
https://github.com/calibrain/shelfmark.git
synced 2026-10-05 22:05:50 +01:00
MyAnonamouse is the only indexer Shelfmark treats as enriched, and it
alone was sent {title} {author} while every other indexer got the title
on its own. MAM matches all search terms conjunctively, so whenever the
metadata provider spelled the author differently to the tracker -
Hardcover says Timothy Ferriss, MAM lists Tim Ferriss - the search came
back empty and the UI reported No releases found for this book, with the
release sitting on the tracker the whole time.
The enriched flag is a statement about responses: MAM returns clean
author and bookTitle attributes, which is why it earns format detection
and preferential ordering. Using that same flag to shape the request is
the actual defect, and it is why turning the flag off recovers the
search but takes format detection down with it.
So the query is title-only for every indexer now, and the author orders
the results rather than narrowing them. MAM already hands us its author
field, so agreement is judged on data we hold instead of by an AND we
cannot control. The ranking is three-way on purpose - agrees, no
metadata, disagrees - so an indexer reporting no author does not sort
below one reporting the wrong author.
A wrong verdict costs a release its position, never its visibility: a
transliteration such as Dostoevsky against Dostoyevsky sorts last
instead of vanishing. That is what makes the loose token comparison safe
to ship without a tuning knob.
Falling back to a title-only query on zero results was the alternative.
It only rescues total failure - if two of six editions happen to use the
provider's spelling, the search returns those two, no fallback fires,
and the user quietly gets a truncated list. It also spends a round trip
inside the search deadline and stacks a retry on an indexer that may
still be solving a challenge (#1249).
Manual queries skip author ranking: they are the user's own words and
should not be reordered against the metadata they were typed to
override.
This commit is contained in:
@@ -0,0 +1,186 @@
|
||||
"""Author agreement ranks Prowlarr results; it never narrows the query (#1293).
|
||||
|
||||
MyAnonamouse - the only indexer Shelfmark treats as enriched - used to receive
|
||||
"{title} {author}". MAM ANDs its search terms, so any difference between the
|
||||
metadata provider's author spelling and the tracker's ("Timothy Ferriss" vs
|
||||
"Tim Ferriss") returned nothing at all and the UI reported the book as missing.
|
||||
"""
|
||||
|
||||
import pytest
|
||||
|
||||
from shelfmark.metadata_providers import BookMetadata
|
||||
from shelfmark.release_sources.prowlarr.source import ProwlarrSource
|
||||
from shelfmark.release_sources.prowlarr.utils import (
|
||||
AUTHOR_MATCH,
|
||||
AUTHOR_MISMATCH,
|
||||
AUTHOR_UNKNOWN,
|
||||
author_affinity,
|
||||
)
|
||||
|
||||
MAM_INDEXER_ID = 1
|
||||
|
||||
|
||||
class TestAuthorAffinity:
|
||||
@pytest.mark.parametrize(
|
||||
("wanted", "offered"),
|
||||
[
|
||||
("Timothy Ferriss", "Tim Ferriss"),
|
||||
("Tim Ferriss", "Timothy Ferriss"),
|
||||
("T. Ferriss", "Timothy Ferriss"),
|
||||
("Ursula K. Le Guin", "Ursula Le Guin"),
|
||||
("Iain M. Banks", "Iain Banks"),
|
||||
("Frank Herbert", "Frank Herbert, Brian Herbert"),
|
||||
("Robert Jordan Jr.", "Robert Jordan"),
|
||||
("homer", "Homer"),
|
||||
],
|
||||
)
|
||||
def test_same_author_spelled_differently_agrees(self, wanted, offered):
|
||||
assert author_affinity(wanted, offered) == AUTHOR_MATCH
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("wanted", "offered"),
|
||||
[
|
||||
("Timothy Ferriss", "Frank Herbert"),
|
||||
("Frank Herbert", "Brian Herbert"),
|
||||
("Homer", "Virgil"),
|
||||
],
|
||||
)
|
||||
def test_different_author_disagrees(self, wanted, offered):
|
||||
assert author_affinity(wanted, offered) == AUTHOR_MISMATCH
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("wanted", "offered"),
|
||||
[
|
||||
("Timothy Ferriss", None),
|
||||
("Timothy Ferriss", ""),
|
||||
("", "Tim Ferriss"),
|
||||
(None, "Tim Ferriss"),
|
||||
("Timothy Ferriss", {"name": "Tim Ferriss"}),
|
||||
],
|
||||
)
|
||||
def test_missing_metadata_is_neither_agreement_nor_disagreement(self, wanted, offered):
|
||||
# An indexer that reports no author must not sort below one that reports
|
||||
# the wrong author, so this tier sits between the two.
|
||||
assert author_affinity(wanted, offered) == AUTHOR_UNKNOWN
|
||||
assert AUTHOR_MATCH < AUTHOR_UNKNOWN < AUTHOR_MISMATCH
|
||||
|
||||
def test_a_surname_alone_is_not_enough_for_a_full_name(self):
|
||||
# "Ferriss" appearing under some other given name is a different person.
|
||||
assert author_affinity("Timothy Ferriss", "Bruce Ferriss") == AUTHOR_MISMATCH
|
||||
|
||||
|
||||
class _EnrichedIndexerClient:
|
||||
"""Stands in for a Prowlarr with MyAnonamouse enabled."""
|
||||
|
||||
def __init__(self, search_results=None):
|
||||
self.queries: list[str] = []
|
||||
self.search_results = search_results or []
|
||||
self.indexer_timeout = 90
|
||||
|
||||
def get_enabled_indexers_detailed(self, *, raise_on_error=False):
|
||||
del raise_on_error
|
||||
return [
|
||||
{
|
||||
"id": MAM_INDEXER_ID,
|
||||
"enable": True,
|
||||
"implementation": "MyAnonamouse",
|
||||
"capabilities": {
|
||||
"categories": [
|
||||
{"id": 7000, "subCategories": []},
|
||||
{"id": 3030, "subCategories": []},
|
||||
]
|
||||
},
|
||||
}
|
||||
]
|
||||
|
||||
def torznab_search(
|
||||
self, *, indexer_id, query, categories=None, search_type="book", limit=100, offset=0
|
||||
):
|
||||
del indexer_id, categories, search_type, limit, offset
|
||||
self.queries.append(query)
|
||||
return self.search_results
|
||||
|
||||
def get_enriched_indexer_ids(self, restrict_to=None, indexers=None):
|
||||
del restrict_to, indexers
|
||||
return [MAM_INDEXER_ID]
|
||||
|
||||
def get_indexer_seed_settings(self, restrict_to=None):
|
||||
del restrict_to
|
||||
return {}
|
||||
|
||||
|
||||
def _mam_result(guid: str, author: str | None) -> dict:
|
||||
return {
|
||||
"guid": guid,
|
||||
"title": "The Tao of Seneca",
|
||||
"author": author,
|
||||
"indexerId": MAM_INDEXER_ID,
|
||||
"indexer": "MyAnonamouse",
|
||||
"protocol": "torrent",
|
||||
"size": 1048576,
|
||||
"seeders": 10,
|
||||
"leechers": 1,
|
||||
"categories": [{"id": 7020}],
|
||||
"infoUrl": f"https://tracker.example/{guid}",
|
||||
}
|
||||
|
||||
|
||||
def _search(monkeypatch, client, *, manual_query=None):
|
||||
import shelfmark.release_sources.prowlarr.source as prowlarr_source
|
||||
from shelfmark.core.search_plan import build_release_search_plan
|
||||
|
||||
values = {"PROWLARR_INDEXERS": "", "PROWLARR_AUTO_EXPAND": False}
|
||||
monkeypatch.setattr(
|
||||
prowlarr_source.config, "get", lambda key, default=None: values.get(key, default)
|
||||
)
|
||||
|
||||
source = ProwlarrSource()
|
||||
monkeypatch.setattr(source, "_get_client", lambda: client)
|
||||
|
||||
book = BookMetadata(
|
||||
provider="hardcover",
|
||||
provider_id="123",
|
||||
title="The Tao of Seneca",
|
||||
authors=["Timothy Ferriss"],
|
||||
)
|
||||
plan = build_release_search_plan(book, languages=["en"], manual_query=manual_query)
|
||||
return source.search(book, plan, content_type="ebook")
|
||||
|
||||
|
||||
class TestEnrichedIndexerQuery:
|
||||
def test_enriched_indexer_is_queried_without_the_author(self, monkeypatch):
|
||||
client = _EnrichedIndexerClient()
|
||||
|
||||
_search(monkeypatch, client)
|
||||
|
||||
assert client.queries == ["The Tao of Seneca"]
|
||||
assert not any("Ferriss" in query for query in client.queries)
|
||||
|
||||
|
||||
class TestAuthorOrdering:
|
||||
def test_matching_author_leads_and_mismatch_stays_visible(self, monkeypatch):
|
||||
client = _EnrichedIndexerClient(
|
||||
search_results=[
|
||||
_mam_result("other-author", "Frank Herbert"),
|
||||
_mam_result("no-author", None),
|
||||
_mam_result("right-author", "Tim Ferriss"),
|
||||
]
|
||||
)
|
||||
|
||||
results = _search(monkeypatch, client)
|
||||
|
||||
# Ranked, not filtered: the wrong author is last but still reachable.
|
||||
assert [r.extra["author"] for r in results] == ["Tim Ferriss", None, "Frank Herbert"]
|
||||
|
||||
def test_manual_query_is_not_reordered_against_the_metadata_author(self, monkeypatch):
|
||||
client = _EnrichedIndexerClient(
|
||||
search_results=[
|
||||
_mam_result("other-author", "Frank Herbert"),
|
||||
_mam_result("right-author", "Tim Ferriss"),
|
||||
]
|
||||
)
|
||||
|
||||
results = _search(monkeypatch, client, manual_query="tao seneca")
|
||||
|
||||
assert client.queries == ["tao seneca"]
|
||||
assert [r.extra["author"] for r in results] == ["Frank Herbert", "Tim Ferriss"]
|
||||
Reference in New Issue
Block a user