mirror of
https://github.com/calibrain/shelfmark.git
synced 2026-10-05 10:51:16 +01:00
Advanced title search, advanced title+author search, and the title typeahead returned zero results every time, and the sort fallback added in #1183 blamed the sort value for it. Hardcover turns the `fields` search parameter into Typesense's `query_by` but keeps `num_typos` and `query_by_weights` as fixed-length presets per query_type. For query_type=Book the preset expects exactly five fields, so a shorter list is not searched loosely - the whole search is rejected with a null results body. Confirmed against the live API: 1, 2, 3, 4 and 6 fields are all rejected, only 5 works, and weights must match one-for-one when sent. Every Book-type list we sent was the wrong length - the title typeahead and advanced title search sent 2, title+author sent 3. - Send BOOK_SEARCH_FIELDS (the full five) for every narrowed Book search and express the intent through weights instead. Weights only bias ranking - a field weighted 0 still matches - so a title search now ranks titles first rather than restricting to them. That is the closest behaviour Hardcover still allows, and there is no client-side filter to restore the old precision. - Pin the field and weight counts in tests, since the failure mode is a silent zero results rather than an error. The sort fallback from #1183 also misread these rejections: - Select the `error` field on every search and log Hardcover's own explanation. The reason is only ever in that sibling field, so a rejection surfaced as "returned no result body" with nothing to act on. Reading it is what made the field-count rule findable. - Drop `sort` entirely on the retry instead of sending an empty string. An empty sort is a value like any other and can be rejected too. - Arm the 900s sticky window only after the sortless retry succeeds. It was armed before the retry and never rolled back, so one rejected typeahead disabled sorting process-wide for 15 minutes whatever the actual cause. Verified against the live Hardcover API: advanced title search 0 -> 84 results, title+author 0 -> 139, title typeahead 0 -> 84 with the exact title top. 2566 unit tests pass; ruff, basedpyright and vulture clean. Refs #1183. The sort_by regression #1183 was written for is gone from Hardcover's side - every sort value it rejected, including the one in the report, is accepted again today. Two plain-search rejections in that report (fields=None) remain unexplained: they could not be reproduced under any per_page, page depth, sort value or query shape, and are most likely transient upstream. They now self-report the reason if they recur.
83 lines
2.7 KiB
Python
83 lines
2.7 KiB
Python
"""Guards on the shape of Hardcover's `fields`/`weights` search parameters.
|
|
|
|
Hardcover turns `fields` into Typesense's `query_by` but keeps `num_typos` and
|
|
`query_by_weights` as fixed-length presets per query_type. A field list of the
|
|
wrong length is not searched loosely -- the whole search is rejected with a null
|
|
results body, which used to surface as "0 results". These tests pin the counts
|
|
so a narrower field list cannot silently ship again.
|
|
"""
|
|
|
|
import pytest
|
|
|
|
from shelfmark.metadata_providers.hardcover import (
|
|
AUTHOR_SUGGESTION_FIELDS,
|
|
AUTHOR_SUGGESTION_WEIGHTS,
|
|
BOOK_SEARCH_FIELD_COUNT,
|
|
BOOK_SEARCH_FIELDS,
|
|
BOOK_TITLE_AUTHOR_WEIGHTS,
|
|
BOOK_TITLE_WEIGHTS,
|
|
SERIES_SEARCH_FIELDS,
|
|
SERIES_SEARCH_WEIGHTS,
|
|
TITLE_SUGGESTION_FIELDS,
|
|
TITLE_SUGGESTION_WEIGHTS,
|
|
HardcoverProvider,
|
|
)
|
|
|
|
|
|
def _count(value: str) -> int:
|
|
return len([part for part in value.split(",") if part.strip()])
|
|
|
|
|
|
class TestBookSearchFieldCounts:
|
|
def test_book_field_list_matches_hardcovers_preset_length(self):
|
|
assert _count(BOOK_SEARCH_FIELDS) == BOOK_SEARCH_FIELD_COUNT
|
|
|
|
@pytest.mark.parametrize(
|
|
("label", "weights"),
|
|
[
|
|
("title", BOOK_TITLE_WEIGHTS),
|
|
("title+author", BOOK_TITLE_AUTHOR_WEIGHTS),
|
|
("title typeahead", TITLE_SUGGESTION_WEIGHTS),
|
|
],
|
|
)
|
|
def test_book_weights_line_up_with_the_field_list(self, label, weights):
|
|
assert _count(weights) == BOOK_SEARCH_FIELD_COUNT, label
|
|
|
|
def test_title_typeahead_uses_the_full_book_field_list(self):
|
|
assert TITLE_SUGGESTION_FIELDS == BOOK_SEARCH_FIELDS
|
|
|
|
|
|
class TestNonBookSearchFieldCounts:
|
|
@pytest.mark.parametrize(
|
|
("fields", "weights"),
|
|
[
|
|
(AUTHOR_SUGGESTION_FIELDS, AUTHOR_SUGGESTION_WEIGHTS),
|
|
(SERIES_SEARCH_FIELDS, SERIES_SEARCH_WEIGHTS),
|
|
],
|
|
)
|
|
def test_weights_line_up_with_their_field_list(self, fields, weights):
|
|
assert _count(fields) == _count(weights)
|
|
|
|
|
|
class TestBuildSearchParams:
|
|
@pytest.mark.parametrize(
|
|
("author", "title", "series"),
|
|
[
|
|
("", "Dune", ""),
|
|
("Herbert", "Dune", ""),
|
|
("Herbert", "", ""),
|
|
("", "", ""),
|
|
],
|
|
)
|
|
def test_every_branch_sends_a_usable_field_weight_pair(self, author, title, series):
|
|
provider = HardcoverProvider(api_key="test-token")
|
|
|
|
_query, fields, weights = provider._build_search_params("dune", author, title, series)
|
|
|
|
if fields is None:
|
|
# No override: Hardcover applies its own preset, so weights must be absent too.
|
|
assert weights is None
|
|
return
|
|
assert _count(fields) == BOOK_SEARCH_FIELD_COUNT
|
|
assert _count(weights) == BOOK_SEARCH_FIELD_COUNT
|