From 37a77e95620d4136d0cba80d3ce34c3345926759 Mon Sep 17 00:00:00 2001 From: splitsec2 <35583321+splitsec2@users.noreply.github.com> Date: Fri, 25 Sep 2026 16:18:10 -0600 Subject: [PATCH] feat(library): mark search results already in a Calibre library (#1377) Per discussion #1372, where you said you were fine with this specific implementation: check whether metadata.db exists, read it if so, and show a check mark saying the book is already there. Searching for a book you already own gives no hint that you own it, so the easiest way to end up with a second copy is to not remember you have the first. This reads a Calibre `metadata.db`, read only, and marks matching results with an **In library** badge in the card, list and compact views and in the details dialog. Off by default. It sits in Settings, General beside the existing Library URL, with a test button that reports how many books it indexed. No HTTP call, no token, nothing written back. Matching runs most to least confident: a shared external id, then an ISBN compared in both ISBN-10 and ISBN-13 form, then fuzzy title tokens plus the author surname. The check fails open, so an unreadable database degrades the badge and never blocks a search, and entries are cached for ten minutes with an early refresh when the file changes, so a large library costs one read rather than one per search. `text_match.py` is new and shared by the index and the provider, so title, author and ISBN matching stays consistent in one place. ## On the provider interface `library_index` talks only to a `LibraryProvider` protocol and knows nothing about Calibre. That is deliberate but it is not speculative generality, it is what let me send you the Calibre half on its own: I run an Audiobookshelf provider on the same interface in my fork, which is where the audiobook side of the badge comes from. I have left that out because it is a new service integration rather than something already in the codebase, which is the line your non-goals draw. Happy to send it separately if you ever want it, and equally happy for the answer to be no. Adding a library is a module with the `LibraryProvider` shape plus one line in `all_providers()`. ## Verification - `tests/core/test_library_index.py`: id, ISBN and fuzzy matching, per-content-type provider selection, fail-open on provider errors, stale-cache reuse, TTL and fingerprint refresh, per-provider cache isolation, and the test-connection path including unsaved form values. - `tests/core/test_text_match.py`: ISBN variants and token matching. - `src/frontend/src/tests/libraryBadge.test.ts` and the added cases in `bookTransformers.test.ts`. - Python suite (3269) and frontend suite (206) green, plus ruff, ruff format, basedpyright, vulture, tsc, oxlint, oxfmt and the production build. --- docs/index.md | 1 + docs/library-check.md | 43 ++ shelfmark/config/settings.py | 32 ++ shelfmark/core/library_index.py | 192 ++++++++ shelfmark/core/library_providers/__init__.py | 73 +++ shelfmark/core/library_providers/calibre.py | 161 +++++++ shelfmark/core/text_match.py | 171 +++++++ shelfmark/main.py | 13 + src/frontend/src/components/DetailsModal.tsx | 8 + .../src/components/resultsViews/CardView.tsx | 3 +- .../components/resultsViews/CompactView.tsx | 3 +- .../src/components/resultsViews/ListView.tsx | 3 +- .../src/components/shared/LibraryBadge.tsx | 50 +++ src/frontend/src/components/shared/index.ts | 1 + .../src/tests/bookTransformers.test.ts | 20 + src/frontend/src/tests/libraryBadge.test.ts | 39 ++ src/frontend/src/types/index.ts | 9 + src/frontend/src/utils/bookTransformers.ts | 4 +- tests/core/test_library_index.py | 421 ++++++++++++++++++ tests/core/test_library_provider_calibre.py | 318 +++++++++++++ tests/core/test_metadata_search_library.py | 71 +++ tests/core/test_text_match.py | 77 ++++ 22 files changed, 1709 insertions(+), 4 deletions(-) create mode 100644 docs/library-check.md create mode 100644 shelfmark/core/library_index.py create mode 100644 shelfmark/core/library_providers/__init__.py create mode 100644 shelfmark/core/library_providers/calibre.py create mode 100644 shelfmark/core/text_match.py create mode 100644 src/frontend/src/components/shared/LibraryBadge.tsx create mode 100644 src/frontend/src/tests/libraryBadge.test.ts create mode 100644 tests/core/test_library_index.py create mode 100644 tests/core/test_library_provider_calibre.py create mode 100644 tests/core/test_metadata_search_library.py create mode 100644 tests/core/test_text_match.py diff --git a/docs/index.md b/docs/index.md index 3dd79d1e..3498746d 100644 --- a/docs/index.md +++ b/docs/index.md @@ -17,6 +17,7 @@ Use the guides below to set up the app, connect your library tools, and understa - [OIDC](oidc.md) - [API Access](api-access.md) - [URL Search Parameters](url-search-parameters.md) +- [Library Check](library-check.md) - [Custom Scripts](custom-scripts.md) ## Help diff --git a/docs/library-check.md b/docs/library-check.md new file mode 100644 index 00000000..33a8f573 --- /dev/null +++ b/docs/library-check.md @@ -0,0 +1,43 @@ +# Library Check + +Shelfmark can mark search results you already own, so you do not download a second copy +of a book that is already on your shelf. The check is read only and off by default. + +## Calibre + +Point Shelfmark at the `metadata.db` of a Calibre library (Calibre, Calibre-Web, +Calibre-Web-Automated, anything that keeps the standard Calibre format) and it reads the +database directly. No HTTP call, no API token, and nothing is ever written back. + +1. Mount the library folder into the container read only, for example + `/path/to/calibre-library:/calibre-library:ro`. Mount the folder rather than the file + so the `-wal` and `-shm` sidecars are visible, otherwise a library that is being + written to can read as out of date. +2. In **Settings, General**, turn on **Mark books already in your Calibre library**. +3. Leave **Calibre metadata.db path** at `/calibre-library/metadata.db` unless you mounted + it somewhere else. +4. Press **Test Calibre library**. It reports how many books it indexed. + +A result that matches the library then carries an **In library** badge in the card, list +and compact views, and in the details dialog. + +## How a match is decided + +In order of confidence: + +1. An external id the metadata provider and the library agree on. +2. An ISBN, compared in both ISBN-10 and ISBN-13 form. +3. Fuzzy title tokens plus the author surname, the same rule the rest of the app uses for + book matching. + +## Behaviour worth knowing + +- **It fails open.** If the database cannot be read, Shelfmark logs a warning, reuses the + last successful read if it has one, and otherwise treats the book as not owned. A broken + path degrades the badge, it never blocks a search. +- **Results are cached** for ten minutes, and refreshed early when the database file + changes, so a large library costs one read rather than one per search. +- **Ebooks only.** A Calibre library holds ebooks, so the badge answers for ebooks. The + provider interface in `shelfmark/core/library_providers/` takes more libraries: add a + module with the `LibraryProvider` shape and list it in `all_providers()`. Nothing above + that function knows which libraries exist. diff --git a/shelfmark/config/settings.py b/shelfmark/config/settings.py index 476e9d54..4b70b14d 100644 --- a/shelfmark/config/settings.py +++ b/shelfmark/config/settings.py @@ -389,6 +389,13 @@ def _clear_metadata_cache(current_values: dict) -> dict: } +def _test_calibre_library(current_values: dict[str, Any] | None = None) -> dict[str, Any]: + """Action-button callback: read the Calibre database and count the books.""" + from shelfmark.core import library_index + + return library_index.test_connection("calibre", current_values) + + @register_settings("general", "General", icon="settings", order=0) def general_settings() -> list[SettingsField]: """Core application settings.""" @@ -412,6 +419,31 @@ def general_settings() -> list[SettingsField]: description="Adds a separate navigation button for your audiobook library (Audiobookshelf, Plex, etc). When both URLs are set, icons are shown instead of text.", placeholder="http://audiobookshelf:8080", ), + CheckboxField( + key="LIBRARY_CHECK_CALIBRE_ENABLED", + label="Mark books already in your Calibre library", + description=( + "Read the Calibre metadata.db and mark search results you already own, so you " + "do not download a second copy. Read only, nothing is written to the library." + ), + default=False, + ), + TextField( + key="CALIBRE_LIBRARY_DB_PATH", + label="Calibre metadata.db path", + description=( + "Path to metadata.db as seen from inside the Shelfmark container. Mount the " + "Calibre library folder read-only, e.g. /path/to/calibre-library:/calibre-library:ro." + ), + default="/calibre-library/metadata.db", + placeholder="/calibre-library/metadata.db", + ), + ActionButton( + key="test_calibre_library", + label="Test Calibre library", + description="Check that Shelfmark can read the Calibre database and count the books.", + callback=_test_calibre_library, + ), HeadingField( key="search_defaults_heading", title="Default Search Filters", diff --git a/shelfmark/core/library_index.py b/shelfmark/core/library_index.py new file mode 100644 index 00000000..26d25db8 --- /dev/null +++ b/shelfmark/core/library_index.py @@ -0,0 +1,192 @@ +"""Library ownership check: is this book already in one of the user's libraries? + +Each ``LibraryProvider`` (see ``library_providers``) indexes its library into +``LibraryEntry`` rows. This module caches those rows per provider, routes a lookup to +the providers that hold the requested content type, and matches by identifier first +(the book's id in the same metadata provider, ISBN in either 10 or 13 form), otherwise +fuzzy title-token overlap plus the author surname (``text_match``) - the same rule the +release matcher in ``auto_download`` uses. + +Fail-open by design: a disabled or unreachable library never stalls the pipeline. +``is_in_library`` answers False (or from the stale cache) and logs a warning. +""" + +from __future__ import annotations + +import threading +import time +from dataclasses import dataclass +from typing import TYPE_CHECKING, Any + +from shelfmark.core.library_providers import all_providers +from shelfmark.core.logger import setup_logger +from shelfmark.core.text_match import ( + COLLECTION_MARKERS, + author_surname, + extra_work_tokens, + isbn_variants, + title_tokens_match, +) + +if TYPE_CHECKING: + from collections.abc import Mapping + + from shelfmark.core.library_providers import LibraryEntry, LibraryProvider + from shelfmark.metadata_providers import BookMetadata + +logger = setup_logger(__name__) + +_CACHE_TTL_SECONDS = 600 # Re-index a library at most every 10 minutes unless it changed. + + +@dataclass +class _CacheSlot: + entries: list[LibraryEntry] | None = None + fetched_at: float = 0.0 + fingerprint: object | None = None + + +_lock = threading.Lock() +_cache: dict[str, _CacheSlot] = {} + + +def _slot(provider_name: str) -> _CacheSlot: + with _lock: + return _cache.setdefault(provider_name, _CacheSlot()) + + +def _store(provider_name: str, entries: list[LibraryEntry], fingerprint: object | None) -> None: + slot = _slot(provider_name) + with _lock: + slot.entries = entries + slot.fetched_at = time.monotonic() + slot.fingerprint = fingerprint + + +def _entries_for(provider: LibraryProvider) -> list[LibraryEntry]: + """Cached entries for one provider, re-indexed past the TTL or when the library changed.""" + slot = _slot(provider.name) + try: + fingerprint = provider.fingerprint() + with _lock: + cached = slot.entries + fresh = time.monotonic() - slot.fetched_at < _CACHE_TTL_SECONDS + unchanged = fingerprint == slot.fingerprint + if cached is not None and fresh and unchanged: + return cached + entries = provider.fetch_entries() + except Exception as exc: # noqa: BLE001 - any failure must fail open + logger.warning("library check: %s unavailable (%s); failing open", provider.describe(), exc) + with _lock: + return slot.entries or [] # Use the stale cache if we have one. + + _store(provider.name, entries, fingerprint) + logger.info("library check: indexed %d %s item(s)", len(entries), provider.display_name) + return entries + + +def _enabled_providers(content_type: str | None) -> list[LibraryProvider]: + return [ + provider + for provider in all_providers() + if provider.is_enabled() + and (content_type is None or content_type in provider.content_types) + ] + + +def any_provider_enabled() -> bool: + return any(provider.is_enabled() for provider in all_providers()) + + +def match_entries(book: BookMetadata, entries: list[LibraryEntry]) -> str | None: + """How ``book`` is held: ``"owned"``, ``"collection"``, or None when it is not. + + ``"collection"`` means a shelf title that bundles several works matched, e.g. an + omnibus. The reader has the book, but saying so plainly would misdescribe what is + on the shelf. + """ + if not entries: + return None + + external_id = (book.provider, str(book.provider_id)) if book.provider_id else None + book_isbns = isbn_variants(book.isbn_13) | isbn_variants(book.isbn_10) + for entry in entries: + if external_id in entry.external_ids or entry.isbns & book_isbns: + return "owned" + + title = book.search_title or book.title + surname = author_surname(book.search_author or (book.authors[0] if book.authors else "")) + collection: str | None = None + for entry in entries: + if not title_tokens_match(title, set(entry.tokens)): + continue + if surname is not None and surname not in entry.tokens: + continue + # A shorter search title is a subset of every longer shelf title sharing its + # words, so "Dune" matched "Dune Messiah". Words the shelf adds that name + # another work disqualify it; packaging words do not. + if extra_work_tokens(set(entry.title_tokens), title, set(entry.context_tokens)): + continue + if entry.title_tokens & COLLECTION_MARKERS: + collection = "collection" + continue + return "owned" + + return collection + + +def book_matches_entries(book: BookMetadata, entries: list[LibraryEntry]) -> bool: + """True if ``book`` is on the shelf at all, however it is packaged.""" + return match_entries(book, entries) is not None + + +def is_in_library(book: BookMetadata, content_type: str | None = None) -> bool: + """True if an enabled library holding ``content_type`` already has ``book`` (fail-open). + + ``content_type`` None consults every enabled provider. + """ + return any( + book_matches_entries(book, _entries_for(provider)) + for provider in _enabled_providers(content_type) + ) + + +def _holding(book: BookMetadata, content_type: str) -> str | None: + """Strongest holding across the enabled libraries for one content type.""" + kinds = { + match_entries(book, _entries_for(provider)) for provider in _enabled_providers(content_type) + } + if "owned" in kinds: + return "owned" + return "collection" if "collection" in kinds else None + + +def ownership(book: BookMetadata) -> dict[str, str | None] | None: + """Per-format holding for the UI: ``{"ebook": "owned" | "collection" | None}``. + + Only content types with an enabled library are reported; None when no library + check is enabled at all. Fail-open like ``is_in_library``. + """ + result = { + content_type: _holding(book, content_type) + for content_type in ("ebook", "audiobook") + if _enabled_providers(content_type) + } + return result or None + + +def test_connection(provider_name: str, current_values: Mapping[str, Any] | None) -> dict[str, Any]: + """Settings action: index one library now and report the item count. + + ``current_values`` are the unsaved form values, so the button works before Save. + """ + provider = next((p for p in all_providers(current_values) if p.name == provider_name), None) + if provider is None: + return {"success": False, "message": f"Unknown library provider: {provider_name}"} + try: + fingerprint = provider.fingerprint() + entries = provider.fetch_entries() + except Exception as exc: # noqa: BLE001 - surface any error to the user + return {"success": False, "message": f"{provider.describe()}: {exc}"} + _store(provider.name, entries, fingerprint) + return {"success": True, "message": f"{provider.describe()}: indexed {len(entries)} item(s)."} diff --git a/shelfmark/core/library_providers/__init__.py b/shelfmark/core/library_providers/__init__.py new file mode 100644 index 00000000..16a12789 --- /dev/null +++ b/shelfmark/core/library_providers/__init__.py @@ -0,0 +1,73 @@ +"""Library ownership providers. + +Each provider indexes one library the user already owns into matchable +``LibraryEntry`` rows. ``shelfmark.core.library_index`` caches those rows per provider +and answers whether a requested book is already on the shelf. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import TYPE_CHECKING, Any, Protocol + +if TYPE_CHECKING: + from collections.abc import Mapping + + +@dataclass(frozen=True) +class LibraryEntry: + """Normalized, matchable representation of one library item.""" + + tokens: frozenset[str] + isbns: frozenset[str] + asins: frozenset[str] + # (metadata provider name, id in that provider) pairs, e.g. ("hardcover", "446681"). + external_ids: frozenset[tuple[str, str]] = frozenset() + # Title tokens alone, without the author and series that `tokens` also carries. + # The matcher needs them to tell "the shelf title says more than the search did" + # from "the shelf entry merely has an author", which decides whether a shorter + # search title is the same book or a different one in the same series. + title_tokens: frozenset[str] = frozenset() + # Series and author words. Shelf titles often bake these in ("Alex Cross 25: + # Cross Kill"), so the matcher must not read them as naming a different work. + context_tokens: frozenset[str] = frozenset() + + +class LibraryProvider(Protocol): + """One owned-book library the ownership check can consult.""" + + name: str + display_name: str + content_types: frozenset[str] + + def is_enabled(self) -> bool: ... + + def fetch_entries(self) -> list[LibraryEntry]: ... + + def describe(self) -> str: ... + + def fingerprint(self) -> object | None: + """Cheap change token (e.g. a file mtime); None when the library has none.""" + ... + + +def setting( + config: Any, key: str, default: object = "", overrides: Mapping[str, Any] | None = None +) -> object: + """Read ``key`` from ``config``, preferring a non-empty unsaved form value in ``overrides``.""" + if overrides: + value = overrides.get(key) + if value not in (None, ""): + return value + return config.get(key, default) + + +def all_providers(overrides: Mapping[str, Any] | None = None) -> list[LibraryProvider]: + """Concrete providers, in a fixed order. ``overrides`` are unsaved form values. + + Adding a library means writing a module with the :class:`LibraryProvider` shape and + listing it here. Nothing above this function knows which libraries exist. + """ + from shelfmark.core.library_providers.calibre import CalibreLibrary + + return [CalibreLibrary(overrides)] diff --git a/shelfmark/core/library_providers/calibre.py b/shelfmark/core/library_providers/calibre.py new file mode 100644 index 00000000..1c0611af --- /dev/null +++ b/shelfmark/core/library_providers/calibre.py @@ -0,0 +1,161 @@ +"""Calibre library provider. + +Reads a Calibre ``metadata.db`` (plain Calibre, Calibre-Web or Calibre-Web Automated) +read-only and indexes each book's title, authors, series and identifiers. + +The database is opened with ``mode=ro`` first: on a WAL-mode library that sees the +un-checkpointed writes as long as the ``-wal``/``-shm`` files beside it are visible. +When that open fails (typically a read-only mount where SQLite cannot create those +files, or a locked database) it is retried once with ``immutable=1``, a snapshot that +can lag until Calibre next checkpoints. The file mtime is the change fingerprint, so a +write or checkpoint refreshes the index ahead of the cache TTL. +""" + +from __future__ import annotations + +import re +import sqlite3 +from contextlib import closing +from pathlib import Path +from typing import TYPE_CHECKING, Any + +from shelfmark.core.config import config as app_config +from shelfmark.core.library_providers import LibraryEntry, setting +from shelfmark.core.logger import setup_logger +from shelfmark.core.text_match import isbn_variants, tokens + +if TYPE_CHECKING: + from collections.abc import Mapping + +logger = setup_logger(__name__) + +# "(Alex Cross Series #11)", "[Illustrated Edition]" and friends. +_PARENTHETICAL = re.compile(r"[(\[][^)\]]*[)\]]") + +_DEFAULT_DB_PATH = "/calibre-library/metadata.db" +_BUSY_TIMEOUT_SECONDS = 5 + +_ISBN_TYPES = frozenset({"isbn", "isbn13", "isbn-13", "isbn10", "isbn-10"}) +_ASIN_TYPES = frozenset({"amazon", "mobi-asin", "asin"}) +# Calibre identifier type -> Shelfmark metadata provider name. +_EXTERNAL_ID_TYPES = { + "hardcover-id": "hardcover", + "google": "googlebooks", + "openlibrary": "openlibrary", + "olid": "openlibrary", +} + +_BOOKS_SQL = "SELECT id, title FROM books" +_AUTHORS_SQL = "SELECT l.book, a.name FROM books_authors_link l JOIN authors a ON a.id = l.author" +_SERIES_SQL = "SELECT l.book, s.name FROM books_series_link l JOIN series s ON s.id = l.series" +_IDENTIFIERS_SQL = "SELECT book, type, val FROM identifiers" + + +def _db_path(overrides: Mapping[str, Any] | None = None) -> Path: + return Path( + str( + setting(app_config, "CALIBRE_LIBRARY_DB_PATH", _DEFAULT_DB_PATH, overrides) + or _DEFAULT_DB_PATH + ) + ) + + +def _probe(conn: sqlite3.Connection) -> sqlite3.Connection: + """Force the first read so open failures surface here rather than mid-query.""" + try: + conn.execute("PRAGMA schema_version").fetchone() + except sqlite3.Error: + conn.close() + raise + return conn + + +def _connect(path: Path) -> sqlite3.Connection: + uri = f"{path.absolute().as_uri()}?mode=ro" + try: + conn = _probe(sqlite3.connect(uri, uri=True, timeout=_BUSY_TIMEOUT_SECONDS)) + except sqlite3.OperationalError as exc: + logger.info( + "library check: %s cannot be read in place (%s); using an immutable snapshot, " + "which may lag until Calibre checkpoints the library", + path, + exc, + ) + conn = _probe(sqlite3.connect(f"{uri}&immutable=1", uri=True)) + return conn + + +def _read_entries(conn: sqlite3.Connection) -> list[LibraryEntry]: + titles: dict[int, str] = dict(conn.execute(_BOOKS_SQL).fetchall()) + series: dict[int, str] = dict(conn.execute(_SERIES_SQL).fetchall()) + + authors: dict[int, list[str]] = {} + for book_id, name in conn.execute(_AUTHORS_SQL): + authors.setdefault(book_id, []).append(name) + + isbns: dict[int, set[str]] = {} + asins: dict[int, set[str]] = {} + external_ids: dict[int, set[tuple[str, str]]] = {} + for book_id, id_type, raw in conn.execute(_IDENTIFIERS_SQL): + value = str(raw or "").strip() + kind = str(id_type or "").strip().lower() + if not value: + continue + if kind in _ISBN_TYPES: + isbns.setdefault(book_id, set()).update(isbn_variants(value)) + elif kind in _ASIN_TYPES: + asins.setdefault(book_id, set()).add(value.upper()) + elif provider := _EXTERNAL_ID_TYPES.get(kind): + external_ids.setdefault(book_id, set()).add((provider, value)) + + entries: list[LibraryEntry] = [] + for book_id, title in titles.items(): + # A trailing parenthetical is series or edition metadata by convention rather + # than part of the work name, and Calibre users often put it there instead of + # in the series field. It stays in the recall set below and is dropped from the + # title set, which is what decides whether the shelf holds a different book. + title_tok = set(tokens(_PARENTHETICAL.sub(" ", title))) + context_tok = set(tokens(series.get(book_id))) + for name in authors.get(book_id, ()): + context_tok |= set(tokens(name)) + tok = title_tok | context_tok + entries.append( + LibraryEntry( + frozenset(tok), + frozenset(isbns.get(book_id, ())), + frozenset(asins.get(book_id, ())), + frozenset(external_ids.get(book_id, ())), + frozenset(title_tok), + frozenset(context_tok), + ) + ) + return entries + + +class CalibreLibrary: + """Ebook ownership via a read-only Calibre ``metadata.db``.""" + + name = "calibre" + display_name = "Calibre" + content_types = frozenset({"ebook"}) + + def __init__(self, overrides: Mapping[str, Any] | None = None) -> None: + self._overrides = overrides + + def is_enabled(self) -> bool: + return bool(setting(app_config, "LIBRARY_CHECK_CALIBRE_ENABLED", False, self._overrides)) + + def describe(self) -> str: + return f"Calibre library at {_db_path(self._overrides)}" + + def fingerprint(self) -> float | None: + path = _db_path(self._overrides) + candidates = (path, path.with_name(f"{path.name}-wal")) + return max((p.stat().st_mtime for p in candidates if p.exists()), default=None) + + def fetch_entries(self) -> list[LibraryEntry]: + path = _db_path(self._overrides) + if not path.is_file(): + raise FileNotFoundError(f"No Calibre database at {path}") + with closing(_connect(path)) as conn: + return _read_entries(conn) diff --git a/shelfmark/core/text_match.py b/shelfmark/core/text_match.py new file mode 100644 index 00000000..4c66fa07 --- /dev/null +++ b/shelfmark/core/text_match.py @@ -0,0 +1,171 @@ +"""Shared text-normalization + fuzzy token-matching helpers for book matching. + +Used by the library ownership check (``library_index``) and its providers, so every +library matches titles, authors and ISBNs the same way. +""" + +from __future__ import annotations + +import re + +DEFAULT_TITLE_MATCH_THRESHOLD = 0.85 + +# Short/common words that add noise to title token matching. +STOPWORDS = frozenset( + { + "a", + "an", + "the", + "of", + "and", + "or", + "to", + "in", + "on", + "for", + "with", + "is", + "by", + } +) + + +def tokens(text: str | None) -> list[str]: + """Lowercase alphanumeric tokens from arbitrary text.""" + if not text: + return [] + return [tok for tok in re.split(r"[^a-z0-9]+", text.lower()) if tok] + + +def significant_tokens(text: str | None) -> list[str]: + """Tokens with stopwords and 1-char noise removed.""" + return [tok for tok in tokens(text) if len(tok) >= 2 and tok not in STOPWORDS] + + +def author_surname(author: str | None) -> str | None: + """Return the most distinctive author token (the surname), or None.""" + value = author or "" + if "," in value: # "Last, First" -> keep the "Last" portion + value = value.split(",")[0] + toks = significant_tokens(value) + return toks[-1] if toks else None + + +def title_tokens_match( + title: str | None, + haystack_tokens: set[str], + threshold: float = DEFAULT_TITLE_MATCH_THRESHOLD, +) -> bool: + """True when enough significant title tokens appear in ``haystack_tokens``.""" + title_toks = significant_tokens(title) + if not title_toks: + return False + present = sum(1 for tok in title_toks if tok in haystack_tokens) + return (present / len(title_toks)) >= threshold + + +# Words marking a title that bundles several works. A shelf entry carrying one of +# these holds the searched book, but as part of something larger, which is worth +# telling the reader apart from owning it on its own. +COLLECTION_MARKERS = frozenset( + { + "omnibus", + "collection", + "complete", + "boxed", + "boxset", + "box", + "set", + "bundle", + "anthology", + "compendium", + "trilogy", + "duology", + "books", + "volumes", + "vols", + } +) + +# Words marking a different printing of the same work, which changes nothing about +# whether the reader owns it. +EDITION_MARKERS = frozenset( + { + "edition", + "editions", + "series", + "book", + "vol", + "volume", + "illustrated", + "annotated", + "unabridged", + "abridged", + "deluxe", + "revised", + "reissue", + "anniversary", + } +) + +# Neither kind is evidence that the shelf holds a different book. +PACKAGING_MARKERS = COLLECTION_MARKERS | EDITION_MARKERS + + +def extra_work_tokens( + shelf_title_tokens: set[str], + search_title: str | None, + context_tokens: set[str] | None = None, +) -> set[str]: + """Significant words the shelf title adds that suggest a different work. + + Packaging words are ignored, since "Illustrated Edition" and "Books 1-6" describe + the same work differently wrapped, while "Messiah" or "Chapter Two" name another + Only alphabetic words of three or more characters count, so volume numbers and + structural words do not make a shelf entry look like a different book. + + ``context_tokens`` are the entry's series and author words, which shelf titles + often repeat ("Alex Cross 25: Cross Kill") without meaning another book. + """ + search_toks = set(tokens(search_title)) | (context_tokens or set()) + return { + tok + for tok in shelf_title_tokens + # Only a real word counts. Numbers are left alone deliberately: a shelf title + # carries volume numbers far more often than it names a numbered sequel, and + # wrongly claiming ownership is the costlier mistake, since the sync then + # never fetches the book. "Persepolis 2" is the case this concedes. + if tok.isalpha() + and len(tok) >= 3 + and tok not in STOPWORDS + and tok not in PACKAGING_MARKERS + and tok not in search_toks + } + + +def normalize_isbn(value: object) -> str: + """Normalize an ISBN to comparable form (digits + trailing X, uppercased).""" + if not value: + return "" + return re.sub(r"[^0-9xX]", "", str(value)).upper() + + +def _isbn10_to_isbn13(isbn10: str) -> str | None: + """ISBN-13 form of a normalized ISBN-10: 978 prefix plus a recomputed check digit.""" + core = isbn10[:9] + if len(isbn10) != 10 or not core.isdigit(): + return None + digits = f"978{core}" + total = sum(int(d) * (1 if i % 2 == 0 else 3) for i, d in enumerate(digits)) + return f"{digits}{(10 - total % 10) % 10}" + + +def isbn_variants(value: object) -> frozenset[str]: + """Comparable forms of an ISBN: normalized, plus the ISBN-13 form of an ISBN-10.""" + isbn = normalize_isbn(value) + if not isbn: + return frozenset() + variants = {isbn} + if len(isbn) == 10 and (isbn13 := _isbn10_to_isbn13(isbn)): + variants.add(isbn13) + return frozenset(variants) diff --git a/shelfmark/main.py b/shelfmark/main.py index e1936cfa..e52cc1f8 100644 --- a/shelfmark/main.py +++ b/shelfmark/main.py @@ -2700,6 +2700,8 @@ def api_metadata_search() -> Response | tuple[Response, int]: if book_dict.get("cover_url"): cache_id = f"{book_dict['provider']}_{book_dict['provider_id']}" book_dict["cover_url"] = transform_cover_url(book_dict["cover_url"], cache_id) + for book, book_dict in zip(search_result.books, books_data, strict=True): + _attach_library_ownership(book, book_dict) response_data = { "books": books_data, @@ -2760,6 +2762,15 @@ def api_metadata_field_options() -> Response: return jsonify({"options": []}) +def _attach_library_ownership(book: Any, book_dict: dict[str, Any]) -> None: + """Add per-format ownership under ``library`` when a library check is enabled.""" + from shelfmark.core import library_index + + owned = library_index.ownership(book) + if owned is not None: + book_dict["library"] = owned + + def _resolve_metadata_provider(provider_name: str) -> MetadataProvider: """Validate, instantiate and return a ready metadata provider. @@ -2808,6 +2819,7 @@ def api_metadata_book(provider: str, book_id: str) -> Response | tuple[Response, return jsonify({"error": "Book not found"}), 404 book_dict = asdict(book) + _attach_library_ownership(book, book_dict) # Transform cover_url to local proxy URL when caching is enabled from shelfmark.core.utils import transform_cover_url @@ -3114,6 +3126,7 @@ def api_releases() -> Response | tuple[Response, int]: # Convert book to dict and transform cover_url book_dict = asdict(book) + _attach_library_ownership(book, book_dict) from shelfmark.core.utils import transform_cover_url if book_dict.get("cover_url"): diff --git a/src/frontend/src/components/DetailsModal.tsx b/src/frontend/src/components/DetailsModal.tsx index ad5b4af2..d5d598a2 100644 --- a/src/frontend/src/components/DetailsModal.tsx +++ b/src/frontend/src/components/DetailsModal.tsx @@ -9,6 +9,7 @@ import { isMetadataBook } from '../types'; import { bookSupportsTargets } from '../utils/bookTargetLoader'; import { isUserCancelledError } from '../utils/errors'; import { BookTargetDropdown } from './BookTargetDropdown'; +import { LibraryBadge, isInLibrary } from './shared'; interface DetailsModalProps { book: Book | null; @@ -292,6 +293,13 @@ export const DetailsModal = ({ ))} + {isMetadata && isInLibrary(book.library) && ( +
+

Library

+ +
+ )} + {/* ISBN - Universal mode only */} {isMetadata && (book.isbn_13 || book.isbn_10) && (
diff --git a/src/frontend/src/components/resultsViews/CardView.tsx b/src/frontend/src/components/resultsViews/CardView.tsx index e20f5917..9ecbd7b5 100644 --- a/src/frontend/src/components/resultsViews/CardView.tsx +++ b/src/frontend/src/components/resultsViews/CardView.tsx @@ -6,7 +6,7 @@ import { getDownloadsCount } from '../../types'; import { bookSupportsTargets } from '../../utils/bookTargetLoader'; import { BookActionButton } from '../BookActionButton'; import { BookTargetDropdown } from '../BookTargetDropdown'; -import { DisplayFieldBadges } from '../shared'; +import { DisplayFieldBadges, LibraryBadge } from '../shared'; const SkeletonLoader = () => (
@@ -99,6 +99,7 @@ export const CardView = ({ #{book.series_position}
)} + {book.preview && !imageError ? ( <> {!imageLoaded && ( diff --git a/src/frontend/src/components/resultsViews/CompactView.tsx b/src/frontend/src/components/resultsViews/CompactView.tsx index 6d074c6d..217d9ecb 100644 --- a/src/frontend/src/components/resultsViews/CompactView.tsx +++ b/src/frontend/src/components/resultsViews/CompactView.tsx @@ -6,7 +6,7 @@ import { getDownloadsCount } from '../../types'; import { bookSupportsTargets } from '../../utils/bookTargetLoader'; import { BookActionButton } from '../BookActionButton'; import { BookTargetDropdown } from '../BookTargetDropdown'; -import { DisplayFieldBadges, DisplayFieldIcon } from '../shared'; +import { DisplayFieldBadges, DisplayFieldIcon, LibraryBadge } from '../shared'; const SkeletonLoader = () => (
@@ -99,6 +99,7 @@ export const CompactView = ({ #{book.series_position}
)} + {book.preview && !imageError ? ( <> {!imageLoaded && ( diff --git a/src/frontend/src/components/resultsViews/ListView.tsx b/src/frontend/src/components/resultsViews/ListView.tsx index e457dcf7..f2e68ebe 100644 --- a/src/frontend/src/components/resultsViews/ListView.tsx +++ b/src/frontend/src/components/resultsViews/ListView.tsx @@ -7,7 +7,7 @@ import { bookSupportsTargets } from '../../utils/bookTargetLoader'; import { getFormatColor, getLanguageColor } from '../../utils/colorMaps'; import { BookActionButton } from '../BookActionButton'; import { BookTargetDropdown } from '../BookTargetDropdown'; -import { DisplayFieldIcon, DisplayFieldBadge } from '../shared'; +import { DisplayFieldIcon, DisplayFieldBadge, LibraryBadge } from '../shared'; interface ListViewProps { books: Book[]; @@ -203,6 +203,7 @@ export const ListView = ({ {book.author || 'Unknown author'} {book.year && • {book.year}}

+
{/* Mobile universal mode info */} diff --git a/src/frontend/src/components/shared/LibraryBadge.tsx b/src/frontend/src/components/shared/LibraryBadge.tsx new file mode 100644 index 00000000..104b8bda --- /dev/null +++ b/src/frontend/src/components/shared/LibraryBadge.tsx @@ -0,0 +1,50 @@ +import type { LibraryOwnership } from '../../types'; + +/** True when the library check reports this book held in any format. */ +export function isInLibrary(library?: LibraryOwnership | null): boolean { + if (!library) return false; + return Object.values(library).some((holding) => holding === 'owned' || holding === 'collection'); +} + +/** True when every format that holds it holds it inside a larger volume. */ +export function isCollectionOnly(library?: LibraryOwnership | null): boolean { + if (!isInLibrary(library)) return false; + return !Object.values(library ?? {}).includes('owned'); +} + +interface LibraryBadgeProps { + library?: LibraryOwnership | null; + /** Solid pill for use over cover art; otherwise a tinted inline pill. */ + overlay?: boolean; + className?: string; +} + +/** "Already in your library" badge. Renders nothing when the book is not held. */ +export function LibraryBadge({ library, overlay = false, className = '' }: LibraryBadgeProps) { + if (!isInLibrary(library)) return null; + + const collectionOnly = isCollectionOnly(library); + const label = collectionOnly ? 'In your library, inside a collection' : 'Already in your library'; + const text = collectionOnly ? 'In a collection' : 'In library'; + + const pill = overlay + ? 'rounded-md border border-sky-700 bg-sky-600 px-1.5 py-0.5 text-[10px] font-bold text-white' + : 'rounded-md bg-sky-600/15 px-1.5 py-0.5 text-[10px] font-semibold text-sky-700 dark:text-sky-300'; + const style = overlay + ? { boxShadow: '0 2px 8px rgba(0, 0, 0, 0.4), 0 1px 3px rgba(0, 0, 0, 0.3)' } + : undefined; + + return ( + + + + + {text} + + ); +} diff --git a/src/frontend/src/components/shared/index.ts b/src/frontend/src/components/shared/index.ts index b282eb67..c6983181 100644 --- a/src/frontend/src/components/shared/index.ts +++ b/src/frontend/src/components/shared/index.ts @@ -1,3 +1,4 @@ export { DisplayFieldIcon, DisplayFieldBadge, DisplayFieldBadges } from './DisplayFieldIcon'; export { CircularProgress } from './CircularProgress'; export { ToggleSwitch } from './ToggleSwitch'; +export { LibraryBadge, isInLibrary, isCollectionOnly } from './LibraryBadge'; diff --git a/src/frontend/src/tests/bookTransformers.test.ts b/src/frontend/src/tests/bookTransformers.test.ts index 1fe23079..cf1dc002 100644 --- a/src/frontend/src/tests/bookTransformers.test.ts +++ b/src/frontend/src/tests/bookTransformers.test.ts @@ -2,6 +2,7 @@ import { describe, it, expect } from 'vitest'; import { isMetadataBook, type Release } from '../types/index'; import { + transformMetadataToBook, transformReleaseToDirectBook, transformSourceRecordToBook, } from '../utils/bookTransformers'; @@ -84,3 +85,22 @@ describe('bookTransformers.transformSourceRecordToBook', () => { expect(isMetadataBook(book)).toBe(false); }); }); + +describe('bookTransformers.transformMetadataToBook', () => { + const metadata = { + provider: 'hardcover', + provider_id: '446681', + title: 'Dungeon Crawler Carl', + authors: ['Matt Dinniman'], + }; + + it('carries the library ownership flags through to the Book', () => { + const book = transformMetadataToBook({ ...metadata, library: { ebook: 'owned' as const } }); + + expect(book.library).toEqual({ ebook: 'owned' }); + }); + + it('leaves library undefined when the API sends none', () => { + expect(transformMetadataToBook(metadata).library).toBeUndefined(); + }); +}); diff --git a/src/frontend/src/tests/libraryBadge.test.ts b/src/frontend/src/tests/libraryBadge.test.ts new file mode 100644 index 00000000..08975d2b --- /dev/null +++ b/src/frontend/src/tests/libraryBadge.test.ts @@ -0,0 +1,39 @@ +import { describe, expect, it } from 'vitest'; + +import { isCollectionOnly, isInLibrary } from '../components/shared/LibraryBadge'; + +describe('isInLibrary', () => { + it('is false without a library result', () => { + expect(isInLibrary(undefined)).toBe(false); + expect(isInLibrary(null)).toBe(false); + expect(isInLibrary({})).toBe(false); + }); + + it('is false when no format holds it', () => { + expect(isInLibrary({ ebook: null })).toBe(false); + expect(isInLibrary({ ebook: null, audiobook: null })).toBe(false); + }); + + it('is true for a book held outright or inside a collection', () => { + expect(isInLibrary({ ebook: 'owned' })).toBe(true); + expect(isInLibrary({ ebook: 'collection' })).toBe(true); + expect(isInLibrary({ ebook: null, audiobook: 'owned' })).toBe(true); + }); +}); + +describe('isCollectionOnly', () => { + it('is true only when every holding is inside a collection', () => { + expect(isCollectionOnly({ ebook: 'collection' })).toBe(true); + expect(isCollectionOnly({ ebook: 'collection', audiobook: null })).toBe(true); + }); + + it('is false when any format holds the book on its own', () => { + expect(isCollectionOnly({ ebook: 'owned' })).toBe(false); + expect(isCollectionOnly({ ebook: 'collection', audiobook: 'owned' })).toBe(false); + }); + + it('is false when the book is not held at all', () => { + expect(isCollectionOnly({ ebook: null })).toBe(false); + expect(isCollectionOnly(undefined)).toBe(false); + }); +}); diff --git a/src/frontend/src/types/index.ts b/src/frontend/src/types/index.ts index 3dfb6f44..38f1a8eb 100644 --- a/src/frontend/src/types/index.ts +++ b/src/frontend/src/types/index.ts @@ -32,6 +32,7 @@ export interface Book { status_message?: string; // Detailed status message (e.g., "Trying Libgen (2/5)") added_time?: number; // Timestamp when added to queue content_type?: string; // "ebook", "audiobook", or related book subtype + library?: LibraryOwnership; // "already in your library" flags from the library check source?: string; // Release source handler (e.g., "direct_download", "prowlarr") source_display_name?: string; // Human-readable source name (e.g., "Direct Download") // Metadata provider fields (used in universal search mode) @@ -277,6 +278,14 @@ export interface QueuedDownloadResult { export type RequestSubmissionResult = RequestRecord | QueuedDownloadResult; +/** How the library holds a book, per format: owned outright, or inside a collection. */ +export type LibraryHolding = 'owned' | 'collection'; + +export interface LibraryOwnership { + ebook?: LibraryHolding | null; + audiobook?: LibraryHolding | null; +} + export type BooksOutputMode = 'folder' | 'booklore' | 'email'; export interface AppConfig { diff --git a/src/frontend/src/utils/bookTransformers.ts b/src/frontend/src/utils/bookTransformers.ts index 1279f732..546f5e9d 100644 --- a/src/frontend/src/utils/bookTransformers.ts +++ b/src/frontend/src/utils/bookTransformers.ts @@ -1,4 +1,4 @@ -import type { Book, Release } from '../types'; +import type { Book, LibraryOwnership, Release } from '../types'; import { isRecord, isStringArray } from './objectHelpers'; /** @@ -6,6 +6,7 @@ import { isRecord, isStringArray } from './objectHelpers'; * Used by both search and single-book endpoints. */ export interface MetadataBookData { + library?: LibraryOwnership; // ownership flags from the library check provider: string; provider_display_name?: string; provider_id: string; @@ -102,6 +103,7 @@ export function transformMetadataToBook(data: MetadataBookData): Book { search_author: data.search_author, authors: data.authors, titles_by_language: data.titles_by_language, + library: data.library, info: { ...(data.isbn_13 && { ISBN: data.isbn_13 }), ...(data.isbn_10 && !data.isbn_13 && { ISBN: data.isbn_10 }), diff --git a/tests/core/test_library_index.py b/tests/core/test_library_index.py new file mode 100644 index 00000000..74c86fbf --- /dev/null +++ b/tests/core/test_library_index.py @@ -0,0 +1,421 @@ +"""Tests for the library ownership facade: matcher, routing, caching and fail-open.""" + +from __future__ import annotations + +from typing import Any + +import pytest + +from shelfmark.core import library_index +from shelfmark.core.library_providers import LibraryEntry +from shelfmark.core.text_match import isbn_variants +from shelfmark.metadata_providers import BookMetadata + + +def _book(**overrides: Any) -> BookMetadata: + fields: dict[str, Any] = { + "provider": "hardcover", + "provider_id": "446681", + "title": "Dungeon Crawler Carl", + "authors": ["Matt Dinniman"], + } + fields.update(overrides) + return BookMetadata(**fields) + + +def _entry( + *words: str, + isbns: frozenset[str] = frozenset(), + external_ids: frozenset[tuple[str, str]] = frozenset(), +) -> LibraryEntry: + return LibraryEntry(frozenset(words), isbns, frozenset(), external_ids) + + +_DCC_ENTRY = _entry("dungeon", "crawler", "carl", "matt", "dinniman") + + +class _Provider: + def __init__( + self, + name: str, + content_types: set[str], + entries: list[LibraryEntry] | None = None, + *, + enabled: bool = True, + error: Exception | None = None, + token: object | None = None, + ) -> None: + self.name = name + self.display_name = name.title() + self.content_types = frozenset(content_types) + self.entries = entries or [] + self.enabled = enabled + self.error = error + self.token = token + self.fetches = 0 + + def is_enabled(self) -> bool: + return self.enabled + + def describe(self) -> str: + return f"{self.display_name} (test)" + + def fingerprint(self) -> object | None: + return self.token + + def fetch_entries(self) -> list[LibraryEntry]: + self.fetches += 1 + if self.error is not None: + raise self.error + return list(self.entries) + + +@pytest.fixture +def providers(monkeypatch: pytest.MonkeyPatch) -> list[_Provider]: + registered: list[_Provider] = [] + monkeypatch.setattr(library_index, "all_providers", lambda overrides=None: list(registered)) + monkeypatch.setattr(library_index, "_cache", {}) + return registered + + +@pytest.fixture +def clock(monkeypatch: pytest.MonkeyPatch) -> list[float]: + now = [1000.0] + monkeypatch.setattr(library_index.time, "monotonic", lambda: now[0]) + return now + + +def _warnings(monkeypatch: pytest.MonkeyPatch) -> list[str]: + messages: list[str] = [] + monkeypatch.setattr( + library_index.logger, "warning", lambda msg, *args: messages.append(msg % args) + ) + return messages + + +# --------------------------------------------------------------------------- # +# ISBN forms +# --------------------------------------------------------------------------- # +def test_isbn_variants_add_the_isbn13_form_of_an_isbn10() -> None: + assert isbn_variants("0-593-82024-X") == {"059382024X", "9780593820247"} + assert isbn_variants("978-0-593-82024-7") == {"9780593820247"} + assert isbn_variants(None) == frozenset() + assert isbn_variants("n/a") == frozenset() + + +# --------------------------------------------------------------------------- # +# Pure matcher +# --------------------------------------------------------------------------- # +def test_no_entries_never_match() -> None: + assert library_index.book_matches_entries(_book(), []) is False + + +def test_isbn13_exact_match() -> None: + entries = [_entry("unrelated", isbns=frozenset({"9780593820247"}))] + book = _book(title="Different Title", authors=["Someone Else"], isbn_13="978-0-593-82024-7") + + assert library_index.book_matches_entries(book, entries) is True + + +def test_isbn10_book_matches_isbn13_entry() -> None: + entries = [_entry("unrelated", isbns=frozenset({"9780593820247"}))] + book = _book(title="Different Title", authors=["Someone Else"], isbn_10="059382024X") + + assert library_index.book_matches_entries(book, entries) is True + + +def test_isbn13_book_matches_entry_indexed_from_an_isbn10() -> None: + entries = [_entry("unrelated", isbns=isbn_variants("059382024X"))] + book = _book(title="Different Title", authors=["Someone Else"], isbn_13="9780593820247") + + assert library_index.book_matches_entries(book, entries) is True + + +def test_external_id_match_requires_the_same_provider() -> None: + entries = [_entry("unrelated", external_ids=frozenset({("hardcover", "446681")}))] + book = _book(title="Different Title", authors=["Someone Else"]) + + assert library_index.book_matches_entries(book, entries) is True + assert library_index.book_matches_entries(book, [_entry("x")]) is False + other_provider = _book(provider="googlebooks", title="Different", authors=["Else"]) + assert library_index.book_matches_entries(other_provider, entries) is False + + +def test_fuzzy_title_and_author_surname_match() -> None: + assert library_index.book_matches_entries(_book(), [_DCC_ENTRY]) is True + + +def test_near_miss_title_fails() -> None: + sequel = _book(title="Dungeon Crawler Carl 2: Carl's Doomsday Scenario") + + assert library_index.book_matches_entries(sequel, [_DCC_ENTRY]) is False + + +def test_right_title_wrong_author_fails() -> None: + assert library_index.book_matches_entries(_book(authors=["Andy Weir"]), [_DCC_ENTRY]) is False + + +def test_search_fields_take_precedence_over_display_fields() -> None: + book = _book( + title="Dungeon Crawler Carl: A LitRPG Adventure", search_title="Dungeon Crawler Carl" + ) + + assert library_index.book_matches_entries(book, [_DCC_ENTRY]) is True + + +# --------------------------------------------------------------------------- # +# Routing +# --------------------------------------------------------------------------- # +def test_ebook_request_consults_only_the_ebook_provider(providers: list[_Provider]) -> None: + audio = _Provider("audio", {"audiobook"}, [_DCC_ENTRY]) + ebook = _Provider("ebook", {"ebook"}, []) + providers.extend([audio, ebook]) + + assert library_index.is_in_library(_book(), "ebook") is False + assert (audio.fetches, ebook.fetches) == (0, 1) + + assert library_index.is_in_library(_book(), "audiobook") is True + assert (audio.fetches, ebook.fetches) == (1, 1) + + +def test_no_content_type_consults_every_enabled_provider(providers: list[_Provider]) -> None: + audio = _Provider("audio", {"audiobook"}, []) + ebook = _Provider("ebook", {"ebook"}, [_DCC_ENTRY]) + providers.extend([audio, ebook]) + + assert library_index.is_in_library(_book()) is True + assert (audio.fetches, ebook.fetches) == (1, 1) + + +def test_disabled_provider_is_not_consulted(providers: list[_Provider]) -> None: + disabled = _Provider("ebook", {"ebook"}, [_DCC_ENTRY], enabled=False) + providers.append(disabled) + + assert library_index.is_in_library(_book(), "ebook") is False + assert disabled.fetches == 0 + + +def test_any_provider_enabled(providers: list[_Provider]) -> None: + assert library_index.any_provider_enabled() is False + + providers.append(_Provider("ebook", {"ebook"}, enabled=False)) + assert library_index.any_provider_enabled() is False + + providers.append(_Provider("audio", {"audiobook"})) + assert library_index.any_provider_enabled() is True + + +# --------------------------------------------------------------------------- # +# Fail-open + cache +# --------------------------------------------------------------------------- # +def test_provider_error_fails_open_without_a_cache( + providers: list[_Provider], monkeypatch: pytest.MonkeyPatch +) -> None: + providers.append(_Provider("ebook", {"ebook"}, error=OSError("no such file"))) + warnings = _warnings(monkeypatch) + + assert library_index.is_in_library(_book(), "ebook") is False + assert warnings == ["library check: Ebook (test) unavailable (no such file); failing open"] + + +def test_provider_error_keeps_answering_from_the_stale_cache( + providers: list[_Provider], monkeypatch: pytest.MonkeyPatch, clock: list[float] +) -> None: + provider = _Provider("ebook", {"ebook"}, [_DCC_ENTRY]) + providers.append(provider) + assert library_index.is_in_library(_book(), "ebook") is True + + provider.error = RuntimeError("boom") + clock[0] += library_index._CACHE_TTL_SECONDS + 1 + warnings = _warnings(monkeypatch) + + assert library_index.is_in_library(_book(), "ebook") is True + assert provider.fetches == 2 + assert len(warnings) == 1 + + +def test_entries_are_cached_until_the_ttl_expires( + providers: list[_Provider], clock: list[float] +) -> None: + provider = _Provider("ebook", {"ebook"}, [_DCC_ENTRY]) + providers.append(provider) + + library_index.is_in_library(_book(), "ebook") + library_index.is_in_library(_book(), "ebook") + assert provider.fetches == 1 + + clock[0] += library_index._CACHE_TTL_SECONDS - 1 + library_index.is_in_library(_book(), "ebook") + assert provider.fetches == 1 + + clock[0] += 2 + library_index.is_in_library(_book(), "ebook") + assert provider.fetches == 2 + + +def test_fingerprint_change_refreshes_before_the_ttl( + providers: list[_Provider], clock: list[float] +) -> None: + provider = _Provider("ebook", {"ebook"}, [], token=1.0) + providers.append(provider) + + assert library_index.is_in_library(_book(), "ebook") is False + clock[0] += 5 + library_index.is_in_library(_book(), "ebook") + assert provider.fetches == 1 + + provider.entries = [_DCC_ENTRY] + provider.token = 2.0 + assert library_index.is_in_library(_book(), "ebook") is True + assert provider.fetches == 2 + + +def test_cache_is_kept_per_provider(providers: list[_Provider]) -> None: + audio = _Provider("audio", {"audiobook"}, []) + ebook = _Provider("ebook", {"ebook"}, [_DCC_ENTRY]) + providers.extend([audio, ebook]) + + assert library_index.is_in_library(_book(), "audiobook") is False + assert library_index.is_in_library(_book(), "ebook") is True + assert library_index.is_in_library(_book(), "audiobook") is False + assert (audio.fetches, ebook.fetches) == (1, 1) + + +# --------------------------------------------------------------------------- # +# Settings test button +# --------------------------------------------------------------------------- # +def test_connection_reports_the_count_and_primes_the_cache(providers: list[_Provider]) -> None: + provider = _Provider("ebook", {"ebook"}, [_DCC_ENTRY, _entry("other")]) + providers.append(provider) + + result = library_index.test_connection("ebook", None) + + assert result == {"success": True, "message": "Ebook (test): indexed 2 item(s)."} + assert library_index.is_in_library(_book(), "ebook") is True + assert provider.fetches == 1 + + +def test_connection_reports_provider_errors(providers: list[_Provider]) -> None: + providers.append(_Provider("ebook", {"ebook"}, error=OSError("no such file"))) + + result = library_index.test_connection("ebook", None) + + assert result == {"success": False, "message": "Ebook (test): no such file"} + + +def test_connection_rejects_unknown_providers(providers: list[_Provider]) -> None: + result = library_index.test_connection("nope", None) + + assert result == {"success": False, "message": "Unknown library provider: nope"} + + +def test_unsaved_form_values_override_saved_settings_for_test_connection(tmp_path): + from shelfmark.core.library_providers import setting + from shelfmark.core.library_providers.calibre import CalibreLibrary + + class _Saved: + def get(self, key, default=None, user_id=None): + return {"CALIBRE_LIBRARY_DB_PATH": "/saved/metadata.db"}.get(key, default) + + assert ( + setting(_Saved(), "CALIBRE_LIBRARY_DB_PATH", "", {"CALIBRE_LIBRARY_DB_PATH": "/form/x.db"}) + == "/form/x.db" + ) + assert ( + setting(_Saved(), "CALIBRE_LIBRARY_DB_PATH", "", {"CALIBRE_LIBRARY_DB_PATH": ""}) + == "/saved/metadata.db" + ) + assert setting(_Saved(), "CALIBRE_LIBRARY_DB_PATH", "", None) == "/saved/metadata.db" + + provider = CalibreLibrary({"CALIBRE_LIBRARY_DB_PATH": str(tmp_path / "form.db")}) + assert provider.describe().endswith("form.db") + + +def test_ownership_reports_only_formats_with_an_enabled_library(providers): + owned = LibraryEntry( + frozenset({"dungeon", "crawler", "carl"}), + frozenset({"9780593820247"}), + frozenset(), + frozenset(), + ) + providers.append(_Provider("calibre", {"ebook"}, [owned])) + providers.append(_Provider("audiobooks", {"audiobook"}, [], enabled=False)) + + assert library_index.ownership(_book(isbn_13="9780593820247")) == {"ebook": "owned"} + assert library_index.ownership(_book(isbn_13="9999999999999", title="Something Else")) == { + "ebook": None + } + + +def test_ownership_is_none_when_no_library_check_is_enabled(providers): + providers.append(_Provider("calibre", {"ebook"}, [], enabled=False)) + + assert library_index.ownership(_book()) is None + + +def test_ownership_covers_both_formats_when_both_libraries_are_enabled(providers): + entry = LibraryEntry( + frozenset({"dungeon", "crawler", "carl"}), + frozenset({"9780593820247"}), + frozenset(), + frozenset(), + ) + providers.append(_Provider("calibre", {"ebook"}, [])) + providers.append(_Provider("audiobooks", {"audiobook"}, [entry])) + + assert library_index.ownership(_book(isbn_13="9780593820247")) == { + "ebook": None, + "audiobook": "owned", + } + + +def _shelf_entry(title_tokens: set[str], context: set[str] | None = None) -> LibraryEntry: + context = context or set() + return LibraryEntry( + frozenset(title_tokens | context), + frozenset(), + frozenset(), + frozenset(), + frozenset(title_tokens), + frozenset(context), + ) + + +def test_a_longer_shelf_title_naming_another_work_is_not_a_match() -> None: + # "Dune" is a subset of "Dune Messiah", which is a different book. + shelf = [_shelf_entry({"dune", "messiah"}, {"frank", "herbert"})] + book = _book(title="Dune", authors=["Frank Herbert"]) + + assert library_index.match_entries(book, shelf) is None + + +def test_the_series_and_volume_a_shelf_title_repeats_do_not_block_a_match() -> None: + # Calibre titles often carry the series and its number: "Alex Cross 25: Cross Kill". + shelf = [_shelf_entry({"alex", "cross", "25", "kill"}, {"alex", "cross", "james", "patterson"})] + book = _book(title="Cross Kill", authors=["James Patterson"]) + + assert library_index.match_entries(book, shelf) == "owned" + + +def test_an_edition_word_does_not_block_a_match() -> None: + shelf = [_shelf_entry({"dune", "illustrated", "edition"}, {"frank", "herbert"})] + + assert ( + library_index.match_entries(_book(title="Dune", authors=["Frank Herbert"]), shelf) + == "owned" + ) + + +def test_a_collection_reports_itself_as_a_collection() -> None: + shelf = [_shelf_entry({"dungeon", "crawler", "carl", "books", "1", "6"}, {"matt", "dinniman"})] + book = _book(title="Dungeon Crawler Carl", authors=["Matt Dinniman"]) + + assert library_index.match_entries(book, shelf) == "collection" + + +def test_an_exact_shelf_title_outranks_a_collection_holding_the_same_book() -> None: + collection = _shelf_entry({"dungeon", "crawler", "carl", "omnibus"}, {"matt", "dinniman"}) + single = _shelf_entry({"dungeon", "crawler", "carl"}, {"matt", "dinniman"}) + book = _book(title="Dungeon Crawler Carl", authors=["Matt Dinniman"]) + + assert library_index.match_entries(book, [collection, single]) == "owned" diff --git a/tests/core/test_library_provider_calibre.py b/tests/core/test_library_provider_calibre.py new file mode 100644 index 00000000..6894c84f --- /dev/null +++ b/tests/core/test_library_provider_calibre.py @@ -0,0 +1,318 @@ +"""Tests for the read-only Calibre metadata.db library provider.""" + +from __future__ import annotations + +import os +import sqlite3 +from pathlib import Path +from typing import Any + +import pytest + +from shelfmark.core import library_index +from shelfmark.core.library_providers import calibre +from shelfmark.metadata_providers import BookMetadata + +_SCHEMA = """ +CREATE TABLE books (id INTEGER PRIMARY KEY, title TEXT NOT NULL, uuid TEXT); +CREATE TABLE authors (id INTEGER PRIMARY KEY, name TEXT NOT NULL); +CREATE TABLE books_authors_link (id INTEGER PRIMARY KEY, book INTEGER NOT NULL, author INTEGER NOT NULL); +CREATE TABLE series (id INTEGER PRIMARY KEY, name TEXT NOT NULL); +CREATE TABLE books_series_link (id INTEGER PRIMARY KEY, book INTEGER NOT NULL, series INTEGER NOT NULL); +CREATE TABLE identifiers (id INTEGER PRIMARY KEY, book INTEGER NOT NULL, type TEXT NOT NULL, val TEXT NOT NULL); +""" + +_BOOKS = [ + (1, "Dungeon Crawler Carl", "u1"), + (2, "Good Omens", "u2"), + (3, "Untitled Draft", "u3"), + (4, "The Martian", "u4"), +] +_AUTHORS = [(1, "Matt Dinniman"), (2, "Terry Pratchett"), (3, "Neil Gaiman"), (4, "Andy Weir")] +_AUTHOR_LINKS = [(1, 1), (2, 2), (2, 3), (4, 4)] # book 3 has no author +_SERIES = [(1, "Dungeon Crawler Carl")] +_SERIES_LINKS = [(1, 1)] +_IDENTIFIERS = [ + (1, "isbn", "9780593820247"), + (1, "mobi-asin", "b08bkgyqxw"), + (1, "hardcover-id", "446681"), + (1, "google", "yEem0QEACAAJ"), + (2, "isbn-10", "0060853980"), + (2, "amazon", "B000FC0V3W"), + (2, "goodreads", "12067"), + (4, "isbn-13", "978-0-8041-3902-1"), + (4, "openlibrary", "OL17091839W"), +] + + +def _make_library(path: Path, *, wal: bool = False) -> None: + conn = sqlite3.connect(path) + if wal: + conn.execute("PRAGMA journal_mode=wal") + conn.executescript(_SCHEMA) + conn.executemany("INSERT INTO books(id, title, uuid) VALUES (?, ?, ?)", _BOOKS) + conn.executemany("INSERT INTO authors(id, name) VALUES (?, ?)", _AUTHORS) + conn.executemany("INSERT INTO books_authors_link(book, author) VALUES (?, ?)", _AUTHOR_LINKS) + conn.executemany("INSERT INTO series(id, name) VALUES (?, ?)", _SERIES) + conn.executemany("INSERT INTO books_series_link(book, series) VALUES (?, ?)", _SERIES_LINKS) + conn.executemany("INSERT INTO identifiers(book, type, val) VALUES (?, ?, ?)", _IDENTIFIERS) + conn.commit() + conn.close() + + +def _configure(monkeypatch: pytest.MonkeyPatch, path: Path, *, enabled: bool = True) -> None: + values: dict[str, Any] = { + "CALIBRE_LIBRARY_DB_PATH": str(path), + "LIBRARY_CHECK_CALIBRE_ENABLED": enabled, + } + monkeypatch.setattr( + calibre.app_config, + "get", + lambda key, default=None, user_id=None: values.get(key, default), + ) + monkeypatch.setattr(library_index, "_cache", {}) + + +def _infos(monkeypatch: pytest.MonkeyPatch) -> list[str]: + messages: list[str] = [] + monkeypatch.setattr(calibre.logger, "info", lambda msg, *args: messages.append(msg % args)) + return messages + + +def _by_token(entries: list[Any], token: str) -> Any: + matches = [entry for entry in entries if token in entry.tokens] + assert len(matches) == 1, token + return matches[0] + + +def _dcc() -> BookMetadata: + return BookMetadata( + provider="hardcover", + provider_id="446681", + title="Dungeon Crawler Carl", + authors=["Matt Dinniman"], + ) + + +def test_indexes_titles_authors_series_and_identifiers( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + path = tmp_path / "metadata.db" + _make_library(path) + _configure(monkeypatch, path) + + entries = calibre.CalibreLibrary().fetch_entries() + + assert len(entries) == 4 + + dcc = _by_token(entries, "dinniman") + assert {"dungeon", "crawler", "carl", "matt"} <= dcc.tokens + assert dcc.isbns == {"9780593820247"} + assert dcc.asins == {"B08BKGYQXW"} + assert dcc.external_ids == {("hardcover", "446681"), ("googlebooks", "yEem0QEACAAJ")} + + omens = _by_token(entries, "omens") + assert {"pratchett", "gaiman"} <= omens.tokens + assert omens.isbns == {"0060853980", "9780060853983"} + assert omens.asins == {"B000FC0V3W"} + assert omens.external_ids == frozenset() + + martian = _by_token(entries, "martian") + assert martian.isbns == {"9780804139021"} + assert martian.external_ids == {("openlibrary", "OL17091839W")} + + draft = _by_token(entries, "draft") + assert draft.tokens == {"untitled", "draft"} + assert (draft.isbns, draft.asins, draft.external_ids) == (frozenset(), frozenset(), frozenset()) + + +def test_reads_a_read_only_file(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + path = tmp_path / "metadata.db" + _make_library(path) + path.chmod(0o444) + _configure(monkeypatch, path) + + assert len(calibre.CalibreLibrary().fetch_entries()) == 4 + + +def test_wal_reads_see_uncheckpointed_writes( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + path = tmp_path / "metadata.db" + _make_library(path, wal=True) + _configure(monkeypatch, path) + infos = _infos(monkeypatch) + + writer = sqlite3.connect(path) + writer.execute("PRAGMA wal_autocheckpoint=0") + writer.execute("INSERT INTO books(id, title, uuid) VALUES (5, 'Fresh Arrival', 'u5')") + writer.commit() + try: + assert (tmp_path / "metadata.db-wal").exists() + entries = calibre.CalibreLibrary().fetch_entries() + finally: + writer.close() + + assert len(entries) == 5 + assert infos == [] + + +@pytest.mark.skipif(os.geteuid() == 0, reason="root is not bound by directory permissions") +def test_wal_without_side_files_on_a_read_only_mount_uses_a_snapshot( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + lib_dir = tmp_path / "library" + lib_dir.mkdir() + path = lib_dir / "metadata.db" + _make_library(path, wal=True) + for suffix in ("-wal", "-shm"): + (lib_dir / f"metadata.db{suffix}").unlink(missing_ok=True) + _configure(monkeypatch, path) + infos = _infos(monkeypatch) + + lib_dir.chmod(0o555) + try: + entries = calibre.CalibreLibrary().fetch_entries() + finally: + lib_dir.chmod(0o755) + + assert len(entries) == 4 + assert not (lib_dir / "metadata.db-wal").exists() + assert len(infos) == 1 + assert "immutable snapshot" in infos[0] + + +def test_in_place_open_failure_falls_back_to_an_immutable_open( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + path = tmp_path / "metadata.db" + _make_library(path) + _configure(monkeypatch, path) + infos = _infos(monkeypatch) + + real_connect = sqlite3.connect + uris: list[str] = [] + + def fake_connect(database: str, **kwargs: Any) -> sqlite3.Connection: + uris.append(database) + if "immutable=1" not in database: + raise sqlite3.OperationalError("attempt to write a readonly database") + return real_connect(database, **kwargs) + + monkeypatch.setattr(calibre.sqlite3, "connect", fake_connect) + + assert len(calibre.CalibreLibrary().fetch_entries()) == 4 + assert [uri.endswith("?mode=ro") for uri in uris] == [True, False] + assert uris[1].endswith("?mode=ro&immutable=1") + assert "attempt to write a readonly database" in infos[0] + + +def test_locked_database_falls_back_to_a_snapshot( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + path = tmp_path / "metadata.db" + _make_library(path) + _configure(monkeypatch, path) + monkeypatch.setattr(calibre, "_BUSY_TIMEOUT_SECONDS", 0.01) + infos = _infos(monkeypatch) + + writer = sqlite3.connect(path, isolation_level=None) + writer.execute("BEGIN EXCLUSIVE") + writer.execute("INSERT INTO books(id, title, uuid) VALUES (5, 'Pending', 'u5')") + try: + entries = calibre.CalibreLibrary().fetch_entries() + finally: + writer.rollback() + writer.close() + + assert len(entries) == 4 + assert "database is locked" in infos[0] + + +def test_unreadable_database_raises_and_the_facade_fails_open( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + path = tmp_path / "metadata.db" + _make_library(path) + _configure(monkeypatch, path) + + def always_fail(database: str, **kwargs: Any) -> sqlite3.Connection: + raise sqlite3.OperationalError("disk I/O error") + + monkeypatch.setattr(calibre.sqlite3, "connect", always_fail) + warnings: list[str] = [] + monkeypatch.setattr( + library_index.logger, "warning", lambda msg, *args: warnings.append(msg % args) + ) + + with pytest.raises(sqlite3.OperationalError): + calibre.CalibreLibrary().fetch_entries() + assert library_index.is_in_library(_dcc(), "ebook") is False + assert warnings == [ + f"library check: Calibre library at {path} unavailable (disk I/O error); failing open" + ] + + +def test_missing_file_is_enabled_but_fetch_raises_cleanly( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + path = tmp_path / "missing.db" + _configure(monkeypatch, path) + provider = calibre.CalibreLibrary() + + assert provider.is_enabled() is True + assert provider.describe() == f"Calibre library at {path}" + assert provider.fingerprint() is None + with pytest.raises(FileNotFoundError): + provider.fetch_entries() + + result = library_index.test_connection("calibre", None) + assert result["success"] is False + assert str(path) in result["message"] + assert library_index.is_in_library(_dcc(), "ebook") is False + + +def test_disabled_provider_is_not_consulted( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + path = tmp_path / "metadata.db" + _make_library(path) + _configure(monkeypatch, path, enabled=False) + + assert calibre.CalibreLibrary().is_enabled() is False + assert library_index.any_provider_enabled() is False + assert library_index.is_in_library(_dcc(), "ebook") is False + + +def test_enabled_provider_answers_ebook_lookups_only( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + path = tmp_path / "metadata.db" + _make_library(path) + _configure(monkeypatch, path) + + assert library_index.any_provider_enabled() is True + assert library_index.is_in_library(_dcc(), "ebook") is True + assert library_index.is_in_library(_dcc(), "audiobook") is False + assert library_index.test_connection("calibre", None) == { + "success": True, + "message": f"Calibre library at {path}: indexed 4 item(s).", + } + + +def test_fingerprint_follows_the_database_and_wal_mtime( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + path = tmp_path / "metadata.db" + _make_library(path) + _configure(monkeypatch, path) + provider = calibre.CalibreLibrary() + + assert provider.fingerprint() == path.stat().st_mtime + + wal = tmp_path / "metadata.db-wal" + wal.write_bytes(b"") + later = path.stat().st_mtime + 60 + os.utime(wal, (later, later)) + + assert provider.fingerprint() == later diff --git a/tests/core/test_metadata_search_library.py b/tests/core/test_metadata_search_library.py new file mode 100644 index 00000000..17bba7f4 --- /dev/null +++ b/tests/core/test_metadata_search_library.py @@ -0,0 +1,71 @@ +"""The metadata search and book endpoints carry per-format library ownership.""" + +from __future__ import annotations + +import importlib +from unittest.mock import patch + +import pytest + +from shelfmark.metadata_providers import BookMetadata, SearchResult + + +@pytest.fixture(scope="module") +def main_module(): + with patch("shelfmark.download.orchestrator.start"): + import shelfmark.main as main + + importlib.reload(main) + return main + + +@pytest.fixture +def client(main_module): + return main_module.app.test_client() + + +class _StubProvider: + name = "stub" + display_name = "Stub" + search_fields: list[object] = [] + + def __init__(self, books: list[BookMetadata]) -> None: + self._books = books + + def is_available(self) -> bool: + return True + + def search_paginated(self, options): + return SearchResult(books=self._books, page=1, total_found=len(self._books), has_more=False) + + def get_book(self, book_id: str) -> BookMetadata | None: + return next((b for b in self._books if b.provider_id == book_id), None) + + +def _search(main_module, client, ownership): + book = BookMetadata(provider="stub", provider_id="42", title="Dungeon Crawler Carl") + stub = _StubProvider([book]) + with ( + patch.object(main_module, "get_auth_mode", return_value="none"), + patch("shelfmark.metadata_providers.is_provider_registered", return_value=True), + patch("shelfmark.metadata_providers.is_provider_enabled", return_value=True), + patch("shelfmark.metadata_providers.get_provider_kwargs", return_value={}), + patch("shelfmark.metadata_providers.get_provider", return_value=stub), + patch("shelfmark.metadata_providers.get_configured_provider", return_value=stub), + patch("shelfmark.core.library_index.ownership", ownership), + ): + response = client.get("/api/metadata/search?query=carl&provider=stub") + assert response.status_code == 200, response.get_json() + return response.get_json()["books"][0] + + +def test_search_results_carry_library_ownership(main_module, client): + result = _search(main_module, client, lambda book: {"ebook": True, "audiobook": False}) + + assert result["library"] == {"ebook": True, "audiobook": False} + + +def test_search_results_omit_library_when_no_check_is_enabled(main_module, client): + result = _search(main_module, client, lambda book: None) + + assert "library" not in result diff --git a/tests/core/test_text_match.py b/tests/core/test_text_match.py new file mode 100644 index 00000000..cf2fcd34 --- /dev/null +++ b/tests/core/test_text_match.py @@ -0,0 +1,77 @@ +"""Tests for the shared token/ISBN/surname matching helpers.""" + +import pytest + +from shelfmark.core import text_match + +_TWENTY_WORDS = " ".join(f"word{i:02d}" for i in range(20)) + + +def test_tokens_lowercases_and_splits_on_non_alphanumerics(): + assert text_match.tokens("Dungeon Crawler Carl: Book 1!") == [ + "dungeon", + "crawler", + "carl", + "book", + "1", + ] + + +@pytest.mark.parametrize("value", [None, "", " --- "]) +def test_tokens_empty_input(value): + assert text_match.tokens(value) == [] + + +def test_significant_tokens_drops_stopwords_and_single_characters(): + assert text_match.significant_tokens("The Way of Kings, Part A 2") == ["way", "kings", "part"] + + +@pytest.mark.parametrize( + ("author", "expected"), + [ + ("Sanderson, Brandon", "sanderson"), + ("Brandon Sanderson", "sanderson"), + ("Dinniman, Matt J.", "dinniman"), + ("", None), + (None, None), + ("A", None), + ], +) +def test_author_surname(author, expected): + assert text_match.author_surname(author) == expected + + +def test_title_tokens_match_default_threshold_edges(): + words = _TWENTY_WORDS.split() + + assert text_match.title_tokens_match(_TWENTY_WORDS, set(words[:17])) is True # 17/20 == 0.85 + assert text_match.title_tokens_match(_TWENTY_WORDS, set(words[:16])) is False + + +def test_title_tokens_match_explicit_threshold_is_inclusive(): + assert text_match.title_tokens_match("alpha beta", {"alpha"}, threshold=0.5) is True + assert text_match.title_tokens_match("alpha beta", {"alpha"}, threshold=0.51) is False + + +def test_title_tokens_match_ignores_stopwords_in_title(): + assert text_match.title_tokens_match("The Name of the Wind", {"name", "wind"}) is True + + +@pytest.mark.parametrize("title", [None, "", "the of a"]) +def test_title_tokens_match_without_significant_tokens_is_false(title): + assert text_match.title_tokens_match(title, {"the", "of"}) is False + + +@pytest.mark.parametrize( + ("value", "expected"), + [ + ("978-0-59-382024-7", "9780593820247"), + ("0-306-40615-x", "030640615X"), + (" 0306406152 ", "0306406152"), + (9780593820247, "9780593820247"), + (None, ""), + ("", ""), + ], +) +def test_normalize_isbn(value, expected): + assert text_match.normalize_isbn(value) == expected