diff --git a/shelfmark/metadata_providers/hardcover.py b/shelfmark/metadata_providers/hardcover.py deleted file mode 100644 index f8fd32f..0000000 --- a/shelfmark/metadata_providers/hardcover.py +++ /dev/null @@ -1,2906 +0,0 @@ -"""Hardcover.app metadata provider. Requires API key.""" - -import re -from contextlib import suppress -from dataclasses import dataclass -from datetime import UTC, datetime -from http import HTTPStatus -from typing import Any, ClassVar -from urllib.parse import urlparse - -import requests - -from shelfmark.core.cache import cache_key, cacheable, get_metadata_cache -from shelfmark.core.config import config as app_config -from shelfmark.core.logger import setup_logger -from shelfmark.core.request_helpers import coerce_bool, coerce_int, normalize_optional_text -from shelfmark.core.settings_registry import ( - ActionButton, - CheckboxField, - HeadingField, - PasswordField, - SelectField, - SettingsField, - register_settings, -) -from shelfmark.download.network import get_ssl_verify -from shelfmark.metadata_providers import ( - BookMetadata, - DisplayField, - DynamicSelectSearchField, - MetadataCapability, - MetadataProvider, - MetadataSearchOptions, - SearchField, - SearchResult, - SearchType, - SortOrder, - TextSearchField, - register_provider, - register_provider_kwargs, -) - -logger = setup_logger(__name__) - -HARDCOVER_API_URL = "https://api.hardcover.app/v1/graphql" -HARDCOVER_PAGE_SIZE = 25 # Hardcover API returns max 25 results per page -HARDCOVER_MIN_AUTHOR_PARTS = 2 -HARDCOVER_MIN_TYPEAHEAD_QUERY_LENGTH = 2 -HARDCOVER_MAX_SERIES_OPTIONS = 7 -HARDCOVER_API_KEY_MIN_LENGTH = 100 -HARDCOVER_LIST_URL_PATTERN = re.compile( - r"^/(?:@([\w.-]+)/)?lists?/([\w-]+)/?$", - re.IGNORECASE, -) - -LIST_LOOKUP_QUERY = """ -query LookupListsBySlug($slug: String!) { - lists(where: {slug: {_eq: $slug}}, limit: 20) { - id - slug - user { - username - } - } -} -""" - -LIST_BOOKS_BY_ID_QUERY = """ -query GetListBooksById($id: Int!, $limit: Int!, $offset: Int!) { - lists(where: {id: {_eq: $id}}, limit: 1) { - name - slug - user { - username - } - books_count - list_books(order_by: {position: asc}, limit: $limit, offset: $offset) { - book { - id - title - subtitle - slug - release_date - headline - description - pages - rating - ratings_count - users_count - cached_image - cached_contributors - contributions(where: {contribution: {_eq: "Author"}}) { - author { - name - } - } - featured_book_series { - position - series { - id - name - primary_books_count - } - } - } - } - } -} -""" - -USER_LISTS_QUERY = """ -query GetUserLists { - me { - id - username - want_to_read_count: user_books_aggregate(where: {status_id: {_eq: 1}}) { - aggregate { - count(columns: [book_id], distinct: true) - } - } - currently_reading_count: user_books_aggregate(where: {status_id: {_eq: 2}}) { - aggregate { - count(columns: [book_id], distinct: true) - } - } - read_count: user_books_aggregate(where: {status_id: {_eq: 3}}) { - aggregate { - count(columns: [book_id], distinct: true) - } - } - did_not_finish_count: user_books_aggregate(where: {status_id: {_eq: 5}}) { - aggregate { - count(columns: [book_id], distinct: true) - } - } - lists(order_by: {name: asc}) { - id - name - slug - books_count - } - followed_lists(order_by: {created_at: desc}) { - list { - id - name - slug - books_count - user { - username - } - } - } - } -} -""" - -USER_BOOKS_BY_STATUS_QUERY = """ -query GetCurrentUserBooksByStatus($statusId: Int!, $limit: Int!, $offset: Int!) { - me { - status_books: user_books( - where: {status_id: {_eq: $statusId}} - distinct_on: [book_id] - order_by: [{book_id: asc}, {created_at: desc}] - limit: $limit - offset: $offset - ) { - book { - id - title - subtitle - slug - release_date - headline - description - pages - rating - ratings_count - users_count - cached_image - cached_contributors - contributions(where: {contribution: {_eq: "Author"}}) { - author { - name - } - } - featured_book_series { - position - series { - id - name - primary_books_count - } - } - } - } - status_books_aggregate: user_books_aggregate(where: {status_id: {_eq: $statusId}}) { - aggregate { - count(columns: [book_id], distinct: true) - } - } - } -} -""" - -BOOK_TARGET_MEMBERSHIP_QUERY = """ -query GetBookTargetMembership($bookId: Int!) { - me { - user_books(where: {book_id: {_eq: $bookId}}, limit: 1, order_by: [{created_at: desc}]) { - id - status_id - } - lists { - id - list_books(where: {book_id: {_eq: $bookId}}, limit: 1) { - id - } - } - } -} -""" - -BOOK_TARGET_MEMBERSHIP_BATCH_QUERY = """ -query GetBookTargetMembershipBatch($bookIds: [Int!]!) { - me { - user_books(where: {book_id: {_in: $bookIds}}, order_by: [{created_at: desc}]) { - id - book_id - status_id - } - lists { - id - list_books(where: {book_id: {_in: $bookIds}}) { - id - book_id - } - } - } -} -""" - -INSERT_USER_BOOK_MUTATION = """ -mutation AddBookToStatus($bookId: Int!, $statusId: Int!) { - insert_user_book(object: {book_id: $bookId, status_id: $statusId}) { - id - error - user_book { - id - book_id - status_id - } - } -} -""" - -UPDATE_USER_BOOK_MUTATION = """ -mutation UpdateBookStatus($userBookId: Int!, $statusId: Int!) { - update_user_book(id: $userBookId, object: {status_id: $statusId}) { - id - error - user_book { - id - book_id - status_id - } - } -} -""" - -DELETE_USER_BOOK_MUTATION = """ -mutation RemoveBookStatus($userBookId: Int!) { - delete_user_book(id: $userBookId) { - id - book_id - user_id - } -} -""" - -INSERT_LIST_BOOK_MUTATION = """ -mutation AddBookToList($bookId: Int!, $listId: Int!) { - insert_list_book(object: {book_id: $bookId, list_id: $listId}) { - id - list_book { - id - book_id - list_id - } - } -} -""" - -DELETE_LIST_BOOK_MUTATION = """ -mutation RemoveBookFromList($listBookId: Int!) { - delete_list_book(id: $listBookId) { - id - list_id - } -} -""" - -SEARCH_FIELD_OPTIONS_QUERY = """ -query SearchFieldOptions( - $query: String!, - $queryType: String!, - $limit: Int!, - $page: Int!, - $sort: String, - $fields: String, - $weights: String -) { - search( - query: $query, - query_type: $queryType, - per_page: $limit, - page: $page, - sort: $sort, - fields: $fields, - weights: $weights - ) { - results - } -} -""" - -SERIES_BY_AUTHOR_IDS_QUERY = """ -query SeriesByAuthorIds($authorIds: [Int!], $limit: Int!) { - series( - where: { - author_id: {_in: $authorIds}, - canonical_id: {_is_null: true}, - state: {_eq: "active"} - }, - limit: $limit, - order_by: [{primary_books_count: desc_nulls_last}, {books_count: desc}, {name: asc}] - ) { - id - name - primary_books_count - books_count - author { - name - } - } -} -""" - -SERIES_BOOKS_BY_ID_QUERY = """ -query GetSeriesBooks($seriesId: Int!) { - series(where: {id: {_eq: $seriesId}}, limit: 1) { - id - name - primary_books_count - book_series( - where: { - book: { - canonical_id: {_is_null: true}, - state: {_in: ["normalized", "normalizing"]} - } - } - order_by: [{position: asc_nulls_last}, {book_id: asc}] - ) { - position - book { - id - title - subtitle - slug - release_date - headline - description - pages - rating - ratings_count - users_count - compilation - editions_count - cached_image - cached_contributors - contributions(where: {contribution: {_eq: "Author"}}) { - author { - name - } - } - featured_book_series { - position - series { - id - name - primary_books_count - } - } - } - } - } -} -""" - -HARDCOVER_STATUS_PREFIX = "status:" -HARDCOVER_STATUSES: list[dict] = [ - {"id": 1, "label": "Want to Read", "slug": "want-to-read", "query_key": "want_to_read_count"}, - { - "id": 2, - "label": "Currently Reading", - "slug": "currently-reading", - "query_key": "currently_reading_count", - }, - {"id": 3, "label": "Read", "slug": "read", "query_key": "read_count"}, - { - "id": 5, - "label": "Did Not Finish", - "slug": "did-not-finish", - "query_key": "did_not_finish_count", - }, -] -HARDCOVER_STATUS_URL_SLUGS: dict[int, str] = {s["id"]: s["slug"] for s in HARDCOVER_STATUSES} -HARDCOVER_STATUS_GROUP = "Reading Status" -HARDCOVER_LIST_ID_PREFIX = "id:" -HARDCOVER_WRITABLE_TARGET_GROUPS = {HARDCOVER_STATUS_GROUP, "My Lists"} - - -@dataclass(frozen=True) -class HardcoverBookTargetState: - """Current Hardcover target state for a specific book.""" - - user_book_id: int | None - status_id: int | None - list_book_ids: dict[int, int] - - -class HardcoverGraphQLError(ValueError): - """GraphQL request was rejected by Hardcover.""" - - -class HardcoverTargetPayloadError(RuntimeError): - """Hardcover returned an invalid payload while loading book targets.""" - - -def _extract_graphql_error_message(payload: Any) -> str: - """Extract a readable message from a GraphQL error payload.""" - if not isinstance(payload, dict): - return "" - - errors = payload.get("errors", []) - if not isinstance(errors, list): - return "" - - messages: list[str] = [] - for error in errors: - if not isinstance(error, dict): - continue - message = str(error.get("message") or "").strip() - if message: - messages.append(message) - - return "; ".join(messages) - - -# Mapping from abstract sort order to Hardcover sort parameter -# Note: release_year is more consistently populated than release_date_i -SORT_MAPPING: dict[SortOrder, str] = { - SortOrder.RELEVANCE: "_text_match:desc,users_count:desc", - SortOrder.POPULARITY: "users_count:desc", - SortOrder.RATING: "rating:desc", - SortOrder.NEWEST: "release_year:desc", - SortOrder.OLDEST: "release_year:asc", -} - -# Mapping from abstract search type to Hardcover fields parameter -SEARCH_TYPE_FIELDS: dict[SearchType, str] = { - SearchType.GENERAL: "title,isbns,series_names,author_names,alternative_titles", - SearchType.TITLE: "title,alternative_titles", - SearchType.AUTHOR: "author_names", - # ISBN is handled separately via search_by_isbn() -} - -SERIES_SEARCH_FIELDS = "name,books,author_name" -SERIES_SEARCH_WEIGHTS = "2,1,1" -SERIES_SEARCH_SORT = "_text_match:desc,readers_count:desc" -AUTHOR_SUGGESTION_FIELDS = "name,name_personal,alternate_names" -AUTHOR_SUGGESTION_WEIGHTS = "4,3,2" -AUTHOR_SUGGESTION_SORT = "_text_match:desc,books_count:desc" -TITLE_SUGGESTION_FIELDS = "title,alternative_titles" -TITLE_SUGGESTION_WEIGHTS = "5,2" -TITLE_SUGGESTION_SORT = "_text_match:desc,users_count:desc" - - -def _combine_headline_description(headline: str | None, description: str | None) -> str | None: - """Combine headline (tagline) and description into a single description.""" - if headline and description: - return f"{headline}\n\n{description}" - return headline or description - - -def _extract_cover_url(data: dict, *keys: str) -> str | None: - """Extract cover URL from data dict, trying multiple keys. - - Handles both string URLs and dict with 'url' key. - """ - for key in keys: - value = data.get(key) - if value: - if isinstance(value, str): - return value - if isinstance(value, dict): - return value.get("url") - return None - - -def _extract_publish_year(data: dict) -> int | None: - """Extract publish year from release_year or release_date fields.""" - if data.get("release_year"): - try: - return int(data["release_year"]) - except ValueError, TypeError: - pass - if data.get("release_date"): - try: - return int(str(data["release_date"])[:4]) - except ValueError, TypeError: - pass - return None - - -def _parse_release_date(value: Any) -> datetime | None: - """Parse Hardcover release dates stored as YYYY-MM-DD strings.""" - if not value: - return None - - normalized_value = str(value).strip() - if not normalized_value: - return None - - try: - return datetime.fromisoformat(normalized_value[:10]) - except ValueError: - return None - - -def _normalize_series_position(value: Any) -> float | None: - """Normalize a series position to a float for sorting and grouping.""" - if value is None: - return None - - try: - return float(value) - except TypeError, ValueError: - return None - - -def _normalize_hardcover_api_key(value: object) -> str: - """Normalize Hardcover API keys, stripping copied auth-header prefixes.""" - normalized_value = normalize_optional_text(value) or "" - return normalized_value.removeprefix("Bearer ").strip() - - -def _normalize_search_text(value: str) -> str: - """Normalize free-text search input for matching and caching.""" - return " ".join(value.split()).strip() - - -def _unwrap_hit_document(hit: Any) -> dict[str, Any] | None: - """Extract the document dict from a Typesense hit, or return None.""" - if not isinstance(hit, dict): - return None - item = hit.get("document", hit) - return item if isinstance(item, dict) else None - - -def _search_tokens(value: str) -> list[str]: - """Tokenize search text for lightweight prefix matching.""" - return re.findall(r"[a-z0-9']+", value.casefold()) - - -def _query_matches_author_name(query: str, author_name: str) -> bool: - """Return True when the query looks like an author-name search.""" - normalized_query = _normalize_search_text(query) - normalized_author_name = _normalize_search_text(author_name) - if not normalized_query or not normalized_author_name: - return False - - query_folded = normalized_query.casefold() - author_folded = normalized_author_name.casefold() - if query_folded in author_folded: - return True - - query_tokens = _search_tokens(normalized_query) - author_tokens = _search_tokens(normalized_author_name) - if not query_tokens or not author_tokens: - return False - - return all( - any(author_token.startswith(query_token) for author_token in author_tokens) - for query_token in query_tokens - ) - - -def _split_part_base_title(title: str) -> str | None: - """Extract the base title from segmented part releases like ', Part 2'.""" - normalized_title = _normalize_search_text(title) - if not normalized_title: - return None - - match = re.match(r"^(?P.+?),\s*Part\s+\d+$", normalized_title, re.IGNORECASE) - if not match: - return None - - base_title = str(match.group("base") or "").strip() - return base_title or None - - -def _series_allows_split_parts(series_name: str) -> bool: - """Return True for series that intentionally organize split-part releases.""" - normalized_name = _normalize_search_text(series_name).casefold() - if not normalized_name: - return False - - markers = ( - "dramatized adaptation", - "graphicaudio", - "graphic audio", - "(3 parts)", - "(2 parts)", - "(4 parts)", - ) - return any(marker in normalized_name for marker in markers) - - -def _extract_typesense_hits(result: dict[str, Any]) -> tuple[list[dict[str, Any]], int]: - """Extract hit documents + total count from Hardcover search output.""" - root = result.get("search", result) if isinstance(result, dict) else {} - results_obj = root.get("results", {}) if isinstance(root, dict) else {} - if isinstance(results_obj, dict): - hits = results_obj.get("hits", []) - found_count = results_obj.get("found", 0) - else: - hits = results_obj if isinstance(results_obj, list) else [] - found_count = 0 - return hits, found_count - - -def _build_source_url(slug: str) -> str | None: - """Build Hardcover source URL from book slug.""" - return f"https://hardcover.app/books/{slug}" if slug else None - - -def _is_probably_series_position(subtitle: str) -> bool: - normalized = subtitle.strip().lower() - - # Common patterns: "Book One", "Book 1", "Part 2", "Volume III", etc. - if re.match( - r"^(book|part|volume|vol\.?|episode)\s+([0-9]+|[ivxlcdm]+|one|two|three|four|five|six|seven|eight|nine|ten)\b", - normalized, - ): - return True - - # e.g. "A Novel", "An Epic Fantasy", etc. These add noise to indexer queries. - if normalized in {"a novel", "a novella", "a story", "a memoir"}: - return True - - # Descriptive subtitles like "A [Name] Novel", "An [Name] Mystery", etc. - genre_words = ( - "novel", - "novella", - "story", - "memoir", - "tale", - "thriller", - "mystery", - "romance", - "adventure", - "epic", - "saga", - "chronicle", - "fantasy", - "novel-in-stories", - ) - genre_pattern = "|".join(re.escape(w) for w in genre_words) - return bool(re.match(rf"^an?\s+.+\s+({genre_pattern})$", normalized)) - - -def _strip_parenthetical_suffix(title: str) -> str: - # Drop trailing qualifiers like "(Unabridged)", "(Illustrated Edition)", etc. - return re.sub(r"\s*\([^)]*\)\s*$", "", title).strip() - - -def _simplify_author_for_search(author: str) -> str | None: - """Return a looser author string for indexer searches. - - Primary goal: reduce mismatch between metadata providers and indexers. - Indexers store author names inconsistently ("R.A.", "R. A.", "Salvatore, R.A.") - so initials add noise and hurt recall. - - Heuristics: - - Strip all initials (single or compound), keeping only full names - e.g. "R. A. Salvatore" -> "Salvatore", "George R.R. Martin" -> "George Martin" - - Preserve suffixes like "Jr."/"Sr."/"III" as they sometimes matter - """ - if not author: - return None - - normalized = " ".join(author.split()).strip() - if not normalized: - return None - - # Handle "Last, First ..." -> "First ... Last" - if "," in normalized: - parts = [p.strip() for p in normalized.split(",") if p.strip()] - if len(parts) >= HARDCOVER_MIN_AUTHOR_PARTS: - normalized = " ".join([*parts[1:], parts[0]]).strip() - - tokens = normalized.split(" ") - if len(tokens) < HARDCOVER_MIN_AUTHOR_PARTS: - return None - - keep_suffixes = {"jr", "jr.", "sr", "sr.", "ii", "iii", "iv", "v"} - - simplified: list[str] = [] - for idx, token in enumerate(tokens): - t = token.strip() - if not t: - continue - - t_lower = t.lower() - is_suffix = (idx == len(tokens) - 1) and (t_lower in keep_suffixes) - if is_suffix: - simplified.append(t) - continue - - # Drop all initials: "R.", "R", "R.R.", "J.K.", etc. - if re.match(r"^[A-Za-z]$|^([A-Za-z]\.)+[A-Za-z]?$", t): - continue - - simplified.append(t) - - if not simplified: - return None - - candidate = " ".join(simplified).strip() - if candidate.lower() == normalized.lower(): - return None - - return candidate - - -def _compute_search_title( - title: str, - subtitle: str | None, - *, - series_name: str | None = None, -) -> str | None: - """Compute a provider-specific, *looser* title for indexer searching. - - Goal: produce a string that maximizes recall in downstream sources (Prowlarr, - IRC bots, etc.). Being too detailed is counterproductive. - - Hardcover often stores titles in a "Series: Book Title" format and places the - standalone book title in `subtitle`. When this appears to be the case, prefer - the subtitle (unless it looks like a series position or other noise). - - Additional heuristics: - - If Hardcover prefixes the series in the title, remove it. - - Drop trailing parenthetical qualifiers. - """ - if not title: - return None - - original_title = " ".join(title.split()).strip() - - normalized_title = _strip_parenthetical_suffix(original_title) - - normalized_subtitle = " ".join(subtitle.split()).strip() if subtitle else "" - normalized_subtitle = ( - _strip_parenthetical_suffix(normalized_subtitle) if normalized_subtitle else "" - ) - - if normalized_subtitle and normalized_subtitle.lower() == normalized_title.lower(): - normalized_subtitle = "" - - # If subtitle is noise, strip it from the title and use just the prefix. - if normalized_subtitle and _is_probably_series_position(normalized_subtitle): - match = re.match(r"^(.+?)\s*:\s*(.+)$", normalized_title) - if match: - suffix = _strip_parenthetical_suffix(match.group(2).strip()) - if ( - normalized_subtitle.lower() == suffix.lower() - or normalized_subtitle.lower() in suffix.lower() - ): - return None - - # Prefer subtitle when it looks like the real title. - if normalized_subtitle and not _is_probably_series_position(normalized_subtitle): - match = re.match(r"^(.+?)\s*:\s*(.+)$", normalized_title) - if match: - prefix = match.group(1).strip() - suffix = _strip_parenthetical_suffix(match.group(2).strip()) - - prefix_words = len(prefix.split()) if prefix else 0 - subtitle_words = len(normalized_subtitle.split()) - - series_normalized = " ".join(series_name.split()).strip() if series_name else "" - if series_normalized and prefix.lower() == series_normalized.lower(): - return normalized_subtitle - - # If the subtitle is much longer than the prefix, treat it as a descriptive subtitle. - if prefix and subtitle_words >= (prefix_words + 4): - return prefix - - # Otherwise assume "Series: Book Title" and prefer the subtitle. - if ( - normalized_subtitle.lower() == suffix.lower() - or normalized_subtitle.lower() in suffix.lower() - ): - return normalized_subtitle - - # Fallback: if title contains the subtitle, this is likely "Series: Subtitle". - if normalized_subtitle.lower() in normalized_title.lower(): - return normalized_subtitle - - # If we know the series name (from full book fetch), strip it. - if series_name: - series_normalized = " ".join(series_name.split()).strip() - if series_normalized: - # Common Hardcover format: "Series: Book Title". - prefix = f"{series_normalized}:" - if normalized_title.lower().startswith(prefix.lower()): - candidate = normalized_title[len(prefix) :].strip() - candidate = _strip_parenthetical_suffix(candidate) - if candidate and candidate.lower() != normalized_title.lower(): - return candidate - - # Last resort: return a cleaned version of the title if we removed noise. - if normalized_title and normalized_title.lower() != original_title.lower(): - return normalized_title - - return None - - -@register_provider_kwargs("hardcover") -def _hardcover_kwargs() -> dict[str, Any]: - """Provide Hardcover-specific constructor kwargs.""" - return {"api_key": app_config.get("HARDCOVER_API_KEY", "")} - - -@register_provider("hardcover") -class HardcoverProvider(MetadataProvider): - """Hardcover.app metadata provider using GraphQL API.""" - - name = "hardcover" - display_name = "Hardcover" - requires_auth = True - supported_sorts: ClassVar[tuple[SortOrder, ...]] = ( - SortOrder.RELEVANCE, - SortOrder.POPULARITY, - SortOrder.RATING, - SortOrder.NEWEST, - SortOrder.OLDEST, - SortOrder.SERIES_ORDER, - ) - capabilities: ClassVar[tuple[MetadataCapability, ...]] = ( - MetadataCapability( - key="view_series", - field_key="series", - sort=SortOrder.SERIES_ORDER, - ), - ) - search_fields: ClassVar[tuple[SearchField, ...]] = ( - TextSearchField( - key="author", - label="Author", - placeholder="Search author...", - description="Search by author name", - ), - TextSearchField( - key="title", - label="Title", - placeholder="Search title...", - description="Search by book title", - ), - TextSearchField( - key="series", - label="Series", - placeholder="Search series...", - description="Search by series name", - suggestions_endpoint="/api/metadata/field-options?provider=hardcover&field=series", - ), - DynamicSelectSearchField( - key="hardcover_list", - label="List", - options_endpoint="/api/metadata/field-options?provider=hardcover&field=hardcover_list", - placeholder="Browse a list...", - description="Browse books from a Hardcover list", - ), - ) - - def __init__(self, api_key: str | None = None) -> None: - """Initialize provider with optional API key (falls back to config).""" - raw_key = api_key or app_config.get("HARDCOVER_API_KEY", "") - self.api_key = _normalize_hardcover_api_key(raw_key) - self.session = requests.Session() - if self.api_key: - self.session.headers.update( - { - "Authorization": f"Bearer {self.api_key}", - "Content-Type": "application/json", - } - ) - - def is_available(self) -> bool: - """Check if provider is configured with an API key.""" - return bool(self.api_key) - - def _build_search_params( - self, default_query: str, author: str, title: str, series: str - ) -> tuple[str, str | None, str | None]: - """Build search query, fields, and weights based on provided values. - - Returns (query, fields, weights) tuple. Fields/weights are None for general search. - """ - if author and not title and not series: - return author, "author_names", "1" - if title and not author and not series: - return title, "title,alternative_titles", "5,1" - if author and title and not series: - return f"{title} {author}", "title,alternative_titles,author_names", "5,1,3" - return default_query, None, None - - def _detect_list_url(self, query: str) -> tuple[str | None, str] | None: - """Detect and extract optional owner username + list slug from a URL string.""" - candidate = query.strip() - if not candidate: - return None - - parsed = urlparse(candidate) - if parsed.scheme not in {"http", "https"}: - return None - - hostname = (parsed.hostname or "").lower() - if hostname not in {"hardcover.app", "www.hardcover.app"}: - return None - - match = HARDCOVER_LIST_URL_PATTERN.match(parsed.path or "") - if not match: - return None - - owner_username = match.group(1).strip() if match.group(1) else None - slug = match.group(2).strip() - if not slug: - return None - - return owner_username, slug - - @cacheable(ttl_key="METADATA_CACHE_SEARCH_TTL", ttl_default=300, key_prefix="hardcover:list:id") - def _fetch_list_books_by_id(self, list_id: int, page: int, limit: int) -> SearchResult: - """Fetch list books by unique Hardcover list ID.""" - if not self.api_key: - return SearchResult(books=[], page=page, total_found=0, has_more=False) - - offset = (page - 1) * limit - - result = self._execute_query( - LIST_BOOKS_BY_ID_QUERY, - { - "id": list_id, - "limit": limit, - "offset": offset, - }, - ) - if not result: - return SearchResult(books=[], page=page, total_found=0, has_more=False) - - lists = result.get("lists", []) - if not lists: - return SearchResult(books=[], page=page, total_found=0, has_more=False) - - list_data = lists[0] if isinstance(lists[0], dict) else {} - list_books = list_data.get("list_books", []) if isinstance(list_data, dict) else [] - books_count_raw = list_data.get("books_count", 0) if isinstance(list_data, dict) else 0 - - # Build source URL and title from list metadata - source_url = None - source_title = str(list_data.get("name") or "").strip() or None - list_slug = str(list_data.get("slug") or "").strip() - user_data = list_data.get("user", {}) - owner_username = ( - str(user_data.get("username") or "").strip() if isinstance(user_data, dict) else "" - ) - if list_slug and owner_username: - source_url = f"https://hardcover.app/@{owner_username}/lists/{list_slug}" - - try: - books_count = int(books_count_raw) - except TypeError, ValueError: - books_count = 0 - - books: list[BookMetadata] = [] - for item in list_books: - if not isinstance(item, dict): - continue - book_data = item.get("book", {}) - if not isinstance(book_data, dict) or not book_data: - continue - try: - parsed_book = self._parse_book(book_data) - if parsed_book: - books.append(parsed_book) - except (AttributeError, IndexError, KeyError, TypeError, ValueError) as exc: - logger.debug("Failed to parse Hardcover list book for list_id=%s: %s", list_id, exc) - - has_more = offset + len(list_books) < books_count - return SearchResult( - books=books, - page=page, - total_found=books_count, - has_more=has_more, - source_url=source_url, - source_title=source_title, - ) - - @cacheable( - ttl_key="METADATA_CACHE_SEARCH_TTL", ttl_default=300, key_prefix="hardcover:list:slug" - ) - def _fetch_list_books( - self, slug: str, owner_username: str | None, page: int, limit: int - ) -> SearchResult: - """Fetch list books by slug, optionally disambiguating by owner username.""" - if not self.api_key: - return SearchResult(books=[], page=page, total_found=0, has_more=False) - - lookup = self._execute_query(LIST_LOOKUP_QUERY, {"slug": slug}) - if not lookup: - return SearchResult(books=[], page=page, total_found=0, has_more=False) - - lists = lookup.get("lists", []) - if not isinstance(lists, list) or not lists: - return SearchResult(books=[], page=page, total_found=0, has_more=False) - - selected: dict[str, Any] | None = None - normalized_owner = owner_username.lower() if owner_username else None - if normalized_owner: - for item in lists: - if not isinstance(item, dict): - continue - owner_data = item.get("user", {}) - if not isinstance(owner_data, dict): - continue - candidate_owner = str(owner_data.get("username") or "").strip().lower() - if candidate_owner == normalized_owner: - selected = item - break - - if selected is None: - first_item = lists[0] - selected = first_item if isinstance(first_item, dict) else None - - if not selected: - return SearchResult(books=[], page=page, total_found=0, has_more=False) - - list_id = coerce_int(selected.get("id"), 0) - if list_id < 1: - return SearchResult(books=[], page=page, total_found=0, has_more=False) - - return self._fetch_list_books_by_id(list_id, page, limit) - - def _resolve_current_user_id(self) -> str | None: - """Resolve current Hardcover user id from saved settings or API me query.""" - connected_user_id = _get_connected_user_id() - if connected_user_id: - return connected_user_id - - result = self._execute_query("query { me { id, username } }", {}) - if not result: - return None - - me_data = result.get("me", {}) - if isinstance(me_data, list) and me_data: - me_data = me_data[0] - if not isinstance(me_data, dict): - return None - - user_id_raw = me_data.get("id") - if user_id_raw is None: - return None - - user_id = str(user_id_raw) - username_raw = me_data.get("username") - username = str(username_raw).strip() if username_raw else _get_connected_username() - _save_connected_user(user_id, username) - return user_id - - def get_user_lists(self) -> list[dict[str, str]]: - """Get authenticated user's own and followed Hardcover lists.""" - if not self.api_key: - return [] - - connected_user_id = self._resolve_current_user_id() - if not connected_user_id: - return self._fetch_user_lists() - - return self._get_user_lists_cached(connected_user_id) - - def get_search_field_options( - self, - field_key: str, - query: str | None = None, - ) -> list[dict[str, str]]: - """Provide dynamic options for Hardcover-specific advanced fields.""" - if field_key == "author": - return self._search_author_options(query or "") - if field_key == "title": - return self._search_title_options(query or "") - if field_key == "series": - return self._search_series_options(query or "") - if field_key == "hardcover_list": - return self.get_user_lists() - return [] - - def _search_field_hits( - self, - *, - query: str, - query_type: str, - limit: int, - sort: str | None, - fields: str | None, - weights: str | None, - ) -> list[dict[str, Any]]: - """Run a Hardcover search request for field-level typeahead options.""" - normalized_query = _normalize_search_text(query) - if not self.api_key or len(normalized_query) < HARDCOVER_MIN_TYPEAHEAD_QUERY_LENGTH: - return [] - - result = self._execute_query( - SEARCH_FIELD_OPTIONS_QUERY, - { - "query": normalized_query, - "queryType": query_type, - "limit": limit, - "page": 1, - "sort": sort, - "fields": fields, - "weights": weights, - }, - ) - if not result: - return [] - - hits, _found_count = _extract_typesense_hits(result) - return hits - - def _search_series_by_matching_author(self, query: str) -> list[dict[str, Any]]: - """Return direct series rows when the query clearly matches an author.""" - author_hits = self._search_field_hits( - query=query, - query_type="Author", - limit=2, - sort=AUTHOR_SUGGESTION_SORT, - fields=AUTHOR_SUGGESTION_FIELDS, - weights=AUTHOR_SUGGESTION_WEIGHTS, - ) - - author_ids: list[int] = [] - for hit in author_hits: - item = _unwrap_hit_document(hit) - if item is None: - continue - - author_name = str(item.get("name") or "").strip() - if not _query_matches_author_name(query, author_name): - continue - - author_id = coerce_int(item.get("id"), 0) - if author_id < 1: - continue - - if author_id not in author_ids: - author_ids.append(author_id) - - if not author_ids: - return [] - - result = self._execute_query( - SERIES_BY_AUTHOR_IDS_QUERY, - { - "authorIds": author_ids, - "limit": 7, - }, - ) - if not result: - return [] - - series_rows = result.get("series", []) - return [row for row in series_rows if isinstance(row, dict)] - - @cacheable(ttl=120, key_prefix="hardcover:author:options") - def _search_author_options(self, query: str) -> list[dict[str, str]]: - """Return typeahead options for Hardcover author search.""" - hits = self._search_field_hits( - query=query, - query_type="Author", - limit=7, - sort=AUTHOR_SUGGESTION_SORT, - fields=AUTHOR_SUGGESTION_FIELDS, - weights=AUTHOR_SUGGESTION_WEIGHTS, - ) - options: list[dict[str, str]] = [] - seen_labels: set[str] = set() - - for hit in hits: - item = _unwrap_hit_document(hit) - if item is None: - continue - - label = str(item.get("name") or "").strip() - normalized_label = label.casefold() - if not label or normalized_label in seen_labels: - continue - - seen_labels.add(normalized_label) - options.append({"value": label, "label": label}) - - return options - - @cacheable(ttl=120, key_prefix="hardcover:title:options") - def _search_title_options(self, query: str) -> list[dict[str, str]]: - """Return typeahead options for Hardcover title search.""" - hits = self._search_field_hits( - query=query, - query_type="Book", - limit=7, - sort=TITLE_SUGGESTION_SORT, - fields=TITLE_SUGGESTION_FIELDS, - weights=TITLE_SUGGESTION_WEIGHTS, - ) - - exclude_compilations = coerce_bool( - app_config.get("HARDCOVER_EXCLUDE_COMPILATIONS", False), - default=False, - ) - exclude_unreleased = coerce_bool( - app_config.get("HARDCOVER_EXCLUDE_UNRELEASED", False), - default=False, - ) - current_year = datetime.now(UTC).year - - options: list[dict[str, str]] = [] - seen_labels: set[str] = set() - - for hit in hits: - item = _unwrap_hit_document(hit) - if item is None: - continue - - if exclude_compilations and item.get("compilation"): - continue - - if exclude_unreleased: - release_year = item.get("release_year") - try: - if release_year is not None and int(release_year) > current_year: - continue - except TypeError, ValueError: - pass - - label = str(item.get("title") or "").strip() - normalized_label = label.casefold() - if not label or normalized_label in seen_labels: - continue - - seen_labels.add(normalized_label) - options.append({"value": label, "label": label}) - - return options - - def _format_series_option_description(self, item: dict[str, Any]) -> str | None: - """Build a short description for a series suggestion option.""" - author_name = item.get("author_name") - if not author_name: - author_data = item.get("author") - if isinstance(author_data, dict): - author_name = author_data.get("name") - - parts: list[str] = [] - if author_name: - parts.append(f"by {author_name}") - - books_count = item.get("primary_books_count") - if books_count is None: - books_count = item.get("books_count") - - try: - if books_count is not None: - books_count_int = int(books_count) - parts.append(f"{books_count_int} book{'s' if books_count_int != 1 else ''}") - except TypeError, ValueError: - pass - - return " • ".join(parts) if parts else None - - @cacheable(ttl=120, key_prefix="hardcover:series:options") - def _search_series_options(self, query: str) -> list[dict[str, str]]: - """Return typeahead options for Hardcover series search.""" - from concurrent.futures import ThreadPoolExecutor - - with ThreadPoolExecutor(max_workers=2) as executor: - author_future = executor.submit(self._search_series_by_matching_author, query) - series_future = executor.submit( - self._search_field_hits, - query=query, - query_type="Series", - limit=7, - sort=SERIES_SEARCH_SORT, - fields=SERIES_SEARCH_FIELDS, - weights=SERIES_SEARCH_WEIGHTS, - ) - - author_series = author_future.result() - hits = series_future.result() - options: list[dict[str, str]] = [] - seen_values: set[str] = set() - - series_items: list[dict[str, Any]] = [] - series_items.extend(author_series) - series_items.extend(doc for hit in hits if (doc := _unwrap_hit_document(hit)) is not None) - - for item in series_items: - series_id = item.get("id") - name = str(item.get("name") or "").strip() - if series_id is None or not name: - continue - - value = f"id:{series_id}" - if value in seen_values: - continue - seen_values.add(value) - - option: dict[str, str] = { - "value": value, - "label": name, - } - description = self._format_series_option_description(item) - if description: - option["description"] = description - options.append(option) - if len(options) >= HARDCOVER_MAX_SERIES_OPTIONS: - break - - return options - - def _resolve_series_search_value(self, series_value: str) -> dict[str, Any] | None: - """Resolve a series field value to a canonical Hardcover series.""" - normalized_value = _normalize_search_text(series_value) - if not normalized_value: - return None - - if normalized_value.startswith(HARDCOVER_LIST_ID_PREFIX): - try: - return {"id": self._parse_prefixed_int(normalized_value, "series id")} - except ValueError: - logger.debug("Invalid Hardcover series id field value: %s", normalized_value) - return None - - result = self._execute_query( - SEARCH_FIELD_OPTIONS_QUERY, - { - "query": normalized_value, - "queryType": "Series", - "limit": 10, - "page": 1, - "sort": SERIES_SEARCH_SORT, - "fields": SERIES_SEARCH_FIELDS, - "weights": SERIES_SEARCH_WEIGHTS, - }, - ) - if not result: - return None - - hits, _found_count = _extract_typesense_hits(result) - if not hits: - return None - - normalized_lookup = normalized_value.lower() - candidates: list[dict[str, Any]] = [] - for hit in hits: - item = _unwrap_hit_document(hit) - if item is None: - continue - series_id = coerce_int(item.get("id"), 0) - if series_id < 1: - continue - name = str(item.get("name") or "").strip() - if not name: - continue - candidates.append({"id": series_id, "name": name}) - - if not candidates: - return None - - exact_match = next( - ( - candidate - for candidate in candidates - if candidate["name"].lower() == normalized_lookup - ), - None, - ) - return exact_match or candidates[0] - - @cacheable( - ttl_key="METADATA_CACHE_SEARCH_TTL", ttl_default=300, key_prefix="hardcover:series:rows:v4" - ) - def _fetch_series_ordered_rows( - self, - series_id: int, - *, - exclude_compilations: bool, - exclude_unreleased: bool, - ) -> dict[str, Any]: - """Fetch and process all books for a series (cached independently of page).""" - empty: dict[str, Any] = {"rows": [], "series_name": "", "total": 0} - if not self.api_key: - return empty - - result = self._execute_query( - SERIES_BOOKS_BY_ID_QUERY, - {"seriesId": series_id}, - ) - if not result: - return empty - - series_items = result.get("series", []) - if not isinstance(series_items, list) or not series_items: - return empty - - series_data = series_items[0] if isinstance(series_items[0], dict) else {} - series_name = ( - str(series_data.get("name") or "").strip() if isinstance(series_data, dict) else "" - ) - allow_split_parts = _series_allows_split_parts(series_name) - today = datetime.now(UTC).date() - - book_series_rows = ( - series_data.get("book_series", []) if isinstance(series_data, dict) else [] - ) - rows_by_position: dict[float, dict[str, Any]] = {} - for row in book_series_rows: - if not isinstance(row, dict): - continue - book_data = row.get("book", {}) - if not isinstance(book_data, dict) or not book_data: - continue - if exclude_compilations and book_data.get("compilation"): - continue - if not allow_split_parts and _split_part_base_title(str(book_data.get("title") or "")): - continue - - position = _normalize_series_position(row.get("position")) - if position is None: - continue - - release_date = _parse_release_date(book_data.get("release_date")) - if exclude_unreleased and (release_date is None or release_date.date() > today): - continue - - sort_key = ( - 1 if release_date and release_date.date() <= today else 0, - 0 if book_data.get("compilation") else 1, - coerce_int(book_data.get("users_count"), 0), - coerce_int(book_data.get("ratings_count"), 0), - coerce_int(book_data.get("editions_count"), 0), - -coerce_int(book_data.get("id"), 0), - ) - existing_row = rows_by_position.get(position) - if existing_row is None: - rows_by_position[position] = {"row": row, "sort_key": sort_key} - continue - if sort_key > existing_row["sort_key"]: - rows_by_position[position] = {"row": row, "sort_key": sort_key} - - ordered_rows = [ - entry["row"] - for _position, entry in sorted(rows_by_position.items(), key=lambda item: item[0]) - ] - return {"rows": ordered_rows, "series_name": series_name, "total": len(ordered_rows)} - - def _fetch_series_books_by_id( - self, - series_id: int, - page: int, - limit: int, - *, - exclude_compilations: bool, - exclude_unreleased: bool, - ) -> SearchResult: - """Fetch books for a Hardcover series in canonical series order.""" - cached = self._fetch_series_ordered_rows( - series_id, - exclude_compilations=exclude_compilations, - exclude_unreleased=exclude_unreleased, - ) - ordered_rows = cached["rows"] - series_name = cached["series_name"] - total_found = cached["total"] - - offset = (page - 1) * limit - page_rows = ordered_rows[offset : offset + limit] - - books: list[BookMetadata] = [] - for row in page_rows: - book_data = row.get("book", {}) - if not isinstance(book_data, dict) or not book_data: - continue - try: - parsed_book = self._parse_book(book_data) - if not parsed_book: - continue - parsed_book.series_id = str(series_id) - if series_name: - parsed_book.series_name = series_name - parsed_book.series_position = row.get("position") - parsed_book.series_count = total_found - books.append(parsed_book) - except (AttributeError, IndexError, KeyError, TypeError, ValueError) as exc: - logger.debug( - "Failed to parse Hardcover series book for series_id=%s: %s", series_id, exc - ) - - has_more = offset + len(page_rows) < total_found - return SearchResult(books=books, page=page, total_found=total_found, has_more=has_more) - - @cacheable(ttl=120, key_prefix="hardcover:user_lists") - def _get_user_lists_cached(self, _cache_user_id: str) -> list[dict[str, str]]: - """Return cached user lists keyed by Hardcover user id.""" - return self._fetch_user_lists() - - def _fetch_current_user_books_by_status( - self, status_id: int, page: int, limit: int - ) -> SearchResult: - """Fetch the current user's Hardcover books for a specific status shelf.""" - if not self.api_key: - return SearchResult(books=[], page=page, total_found=0, has_more=False) - - connected_user_id = self._resolve_current_user_id() - if not connected_user_id: - return SearchResult(books=[], page=page, total_found=0, has_more=False) - - return self._fetch_user_books_by_status_cached(connected_user_id, status_id, page, limit) - - @cacheable( - ttl_key="METADATA_CACHE_SEARCH_TTL", - ttl_default=300, - key_prefix="hardcover:user_books:status", - ) - def _fetch_user_books_by_status_cached( - self, - _cache_user_id: str, - status_id: int, - page: int, - limit: int, - ) -> SearchResult: - """Return cached status-shelf books keyed by user id and shelf.""" - return self._fetch_user_books_by_status(status_id, page, limit) - - def _fetch_user_books_by_status(self, status_id: int, page: int, limit: int) -> SearchResult: - """Fetch books from the current user's Hardcover status shelf.""" - if not self.api_key: - return SearchResult(books=[], page=page, total_found=0, has_more=False) - - offset = (page - 1) * limit - result = self._execute_query( - USER_BOOKS_BY_STATUS_QUERY, - { - "statusId": status_id, - "limit": limit, - "offset": offset, - }, - ) - if not result: - return SearchResult(books=[], page=page, total_found=0, has_more=False) - - me_data = result.get("me", {}) - if isinstance(me_data, list) and me_data: - me_data = me_data[0] - if not isinstance(me_data, dict): - return SearchResult(books=[], page=page, total_found=0, has_more=False) - - status_books = me_data.get("status_books", []) - aggregate_data = me_data.get("status_books_aggregate", {}) - aggregate = aggregate_data.get("aggregate", {}) if isinstance(aggregate_data, dict) else {} - count_raw = aggregate.get("count", 0) if isinstance(aggregate, dict) else 0 - - try: - total_found = int(count_raw) - except TypeError, ValueError: - total_found = 0 - - books: list[BookMetadata] = [] - for item in status_books: - if not isinstance(item, dict): - continue - book_data = item.get("book", {}) - if not isinstance(book_data, dict) or not book_data: - continue - try: - parsed_book = self._parse_book(book_data) - if parsed_book: - books.append(parsed_book) - except (AttributeError, KeyError, TypeError, ValueError) as exc: - logger.debug( - "Failed to parse Hardcover status book for status_id=%s: %s", status_id, exc - ) - - has_more = offset + len(status_books) < total_found - - # Build source URL for the status shelf - source_url = None - url_slug = HARDCOVER_STATUS_URL_SLUGS.get(status_id) - username = _get_connected_username() - if url_slug and username: - source_url = f"https://hardcover.app/@{username}/books/{url_slug}" - - return SearchResult( - books=books, - page=page, - total_found=total_found, - has_more=has_more, - source_url=source_url, - ) - - def _fetch_user_lists(self) -> list[dict[str, str]]: - """Fetch raw list options from Hardcover me query.""" - result = self._execute_query(USER_LISTS_QUERY, {}) - if not result: - return [] - - me_data = result.get("me", {}) - if isinstance(me_data, list) and me_data: - me_data = me_data[0] - if not isinstance(me_data, dict): - return [] - - options: list[dict[str, str]] = [] - seen_values: set[str] = set() - current_username = str(me_data.get("username") or "").strip() - - def _format_label(name: str, books_count: Any) -> str: - try: - return f"{name} ({int(books_count)})" - except TypeError, ValueError: - return name - - for status in HARDCOVER_STATUSES: - count_data = me_data.get(status["query_key"], {}) - aggregate = count_data.get("aggregate", {}) if isinstance(count_data, dict) else {} - count = aggregate.get("count") if isinstance(aggregate, dict) else None - value = f"{HARDCOVER_STATUS_PREFIX}{status['id']}" - seen_values.add(value) - options.append( - { - "value": value, - "label": _format_label(status["label"], count), - "group": HARDCOVER_STATUS_GROUP, - } - ) - - for list_item in me_data.get("lists", []): - if not isinstance(list_item, dict): - continue - list_id = list_item.get("id") - slug = str(list_item.get("slug") or "").strip() - name = str(list_item.get("name") or "").strip() - value = f"id:{list_id}" if list_id is not None else slug - if not value or not name or value in seen_values: - continue - seen_values.add(value) - options.append( - { - "value": value, - "label": _format_label(name, list_item.get("books_count")), - "group": "My Lists", - } - ) - - for followed_item in me_data.get("followed_lists", []): - if not isinstance(followed_item, dict): - continue - - list_item = followed_item.get("list", {}) - if not isinstance(list_item, dict): - continue - - list_id = list_item.get("id") - slug = str(list_item.get("slug") or "").strip() - name = str(list_item.get("name") or "").strip() - value = f"id:{list_id}" if list_id is not None else slug - if not value or not name or value in seen_values: - continue - seen_values.add(value) - - option: dict[str, str] = { - "value": value, - "label": _format_label(name, list_item.get("books_count")), - "group": "Followed Lists", - } - owner_data = list_item.get("user", {}) - if isinstance(owner_data, dict): - owner_username = str(owner_data.get("username") or "").strip() - if owner_username: - option["description"] = f"by @{owner_username}" - elif current_username: - option["description"] = f"by @{current_username}" - options.append(option) - - return options - - def get_book_targets(self, book_id: str) -> list[dict[str, Any]]: - """Get writable Hardcover list/status targets for a specific book.""" - if not self.api_key: - return [] - - book_id_int = coerce_int(book_id, 0) - if book_id_int < 1: - msg = "book_id must be a valid Hardcover book id" - raise ValueError(msg) - - state = self._fetch_book_target_state(book_id_int) - options: list[dict[str, Any]] = [ - dict(option) - for option in self.get_user_lists() - if option.get("group") in HARDCOVER_WRITABLE_TARGET_GROUPS - ] - - for option in options: - value = str(option.get("value") or "").strip() - option["checked"] = self._is_target_checked(value, state) - option["writable"] = True - - return options - - def set_book_target_state( - self, - book_id: str, - target: str, - *, - selected: bool, - ) -> dict[str, Any]: - """Set whether a Hardcover book belongs to a status shelf or user list.""" - if not self.api_key: - msg = "Hardcover is not configured" - raise ValueError(msg) - - book_id_int = coerce_int(book_id, 0) - if book_id_int < 1: - msg = "book_id must be a valid Hardcover book id" - raise ValueError(msg) - - selected_target = str(target or "").strip() - if not selected_target: - msg = "target is required" - raise ValueError(msg) - - if selected_target not in self._get_writable_targets(): - msg = "Unsupported Hardcover target" - raise ValueError(msg) - - state = self._fetch_book_target_state(book_id_int) - status_ids_to_invalidate: set[int] = set() - list_ids_to_invalidate: set[int] = set() - deselected_target: str | None = None - - if selected_target.startswith(HARDCOVER_STATUS_PREFIX): - status_id = self._parse_prefixed_int(selected_target, "status target") - previous_status_id = state.status_id - changed = self._set_status_target_state( - book_id_int, - status_id, - selected=selected, - state=state, - ) - if changed: - if previous_status_id is not None: - status_ids_to_invalidate.add(previous_status_id) - if selected and previous_status_id != status_id: - deselected_target = f"{HARDCOVER_STATUS_PREFIX}{previous_status_id}" - status_ids_to_invalidate.add(status_id) - elif selected_target.startswith(HARDCOVER_LIST_ID_PREFIX): - list_id = self._parse_prefixed_int(selected_target, "list target") - changed = self._set_list_target_state( - book_id_int, - list_id, - selected=selected, - state=state, - ) - if changed: - list_ids_to_invalidate.add(list_id) - else: - msg = "Unsupported Hardcover target" - raise ValueError(msg) - - if changed: - self._invalidate_book_target_caches( - connected_user_id=self._resolve_current_user_id(), - status_ids=status_ids_to_invalidate, - list_ids=list_ids_to_invalidate, - ) - - result_data: dict[str, Any] = {"changed": changed} - if deselected_target: - result_data["deselected_target"] = deselected_target - return result_data - - @staticmethod - def _unwrap_me_data(result: dict | None) -> dict: - """Extract and validate the ``me`` payload from a GraphQL result.""" - if not isinstance(result, dict): - msg = "Hardcover could not load book targets" - raise HardcoverTargetPayloadError(msg) - - me_data = result.get("me", {}) - if isinstance(me_data, list) and me_data: - me_data = me_data[0] - if not isinstance(me_data, dict): - msg = "Hardcover returned an invalid target payload" - raise HardcoverTargetPayloadError(msg) - return me_data - - def _fetch_book_target_state(self, book_id: int) -> HardcoverBookTargetState: - """Load current Hardcover membership state for a specific book.""" - result = self._execute_query( - BOOK_TARGET_MEMBERSHIP_QUERY, - {"bookId": book_id}, - raise_on_error=True, - ) - me_data = self._unwrap_me_data(result) - - user_book_id: int | None = None - status_id: int | None = None - user_books = me_data.get("user_books", []) - if isinstance(user_books, list) and user_books: - latest_user_book = user_books[0] if isinstance(user_books[0], dict) else {} - user_book_id = coerce_int(latest_user_book.get("id"), 0) or None - status_id = coerce_int(latest_user_book.get("status_id"), 0) or None - - list_book_ids: dict[int, int] = {} - for user_list in me_data.get("lists", []): - if not isinstance(user_list, dict): - continue - list_id = coerce_int(user_list.get("id"), 0) - if list_id < 1: - continue - - list_books = user_list.get("list_books", []) - if not isinstance(list_books, list) or not list_books: - continue - - list_book = list_books[0] if isinstance(list_books[0], dict) else {} - list_book_id = coerce_int(list_book.get("id"), 0) - if list_book_id > 0: - list_book_ids[list_id] = list_book_id - - return HardcoverBookTargetState( - user_book_id=user_book_id, - status_id=status_id, - list_book_ids=list_book_ids, - ) - - def _fetch_book_target_states_batch( - self, - book_ids: list[int], - ) -> dict[int, HardcoverBookTargetState]: - """Load Hardcover membership state for multiple books in one query.""" - result = self._execute_query( - BOOK_TARGET_MEMBERSHIP_BATCH_QUERY, - {"bookIds": book_ids}, - raise_on_error=True, - ) - me_data = self._unwrap_me_data(result) - - # Group user_books by book_id (keep only the latest per book) - user_book_by_book: dict[int, dict] = {} - for ub in me_data.get("user_books", []): - if not isinstance(ub, dict): - continue - bid = coerce_int(ub.get("book_id"), 0) - if bid > 0 and bid not in user_book_by_book: - user_book_by_book[bid] = ub - - # Group list_book memberships by book_id - list_book_ids_by_book: dict[int, dict[int, int]] = {} - for user_list in me_data.get("lists", []): - if not isinstance(user_list, dict): - continue - list_id = coerce_int(user_list.get("id"), 0) - if list_id < 1: - continue - for lb in user_list.get("list_books", []): - if not isinstance(lb, dict): - continue - bid = coerce_int(lb.get("book_id"), 0) - lb_id = coerce_int(lb.get("id"), 0) - if bid > 0 and lb_id > 0: - list_book_ids_by_book.setdefault(bid, {})[list_id] = lb_id - - states: dict[int, HardcoverBookTargetState] = {} - for bid in book_ids: - ub = user_book_by_book.get(bid) - states[bid] = HardcoverBookTargetState( - user_book_id=coerce_int(ub.get("id"), 0) or None if ub else None, - status_id=coerce_int(ub.get("status_id"), 0) or None if ub else None, - list_book_ids=list_book_ids_by_book.get(bid, {}), - ) - return states - - def get_book_targets_batch(self, book_ids: list[str]) -> dict[str, list[dict[str, Any]]]: - """Get writable Hardcover list/status targets for multiple books.""" - if not self.api_key or not book_ids: - return {bid: [] for bid in book_ids} - - int_ids = [] - id_map: dict[int, str] = {} - for bid in book_ids: - int_id = coerce_int(bid, 0) - if int_id > 0: - int_ids.append(int_id) - id_map[int_id] = bid - - if not int_ids: - return {bid: [] for bid in book_ids} - - states = self._fetch_book_target_states_batch(int_ids) - writable_options: list[dict[str, Any]] = [ - dict(option) - for option in self.get_user_lists() - if option.get("group") in HARDCOVER_WRITABLE_TARGET_GROUPS - ] - - results: dict[str, list[dict[str, Any]]] = {} - for int_id, str_id in id_map.items(): - state = states.get( - int_id, - HardcoverBookTargetState( - user_book_id=None, - status_id=None, - list_book_ids={}, - ), - ) - options = [dict(opt) for opt in writable_options] - for option in options: - value = str(option.get("value") or "").strip() - option["checked"] = self._is_target_checked(value, state) - option["writable"] = True - results[str_id] = options - - # Fill in any book_ids that didn't parse as valid ints - for bid in book_ids: - if bid not in results: - results[bid] = [] - - return results - - def _get_writable_targets(self) -> set[str]: - """Return the set of writable Hardcover targets for the current user.""" - writable_targets: set[str] = set() - for option in self.get_user_lists(): - value = str(option.get("value") or "").strip() - if ( - option.get("group") in HARDCOVER_WRITABLE_TARGET_GROUPS - and value - and value.startswith((HARDCOVER_STATUS_PREFIX, HARDCOVER_LIST_ID_PREFIX)) - ): - writable_targets.add(value) - return writable_targets - - def _is_target_checked(self, target: str, state: HardcoverBookTargetState) -> bool: - """Return whether a target is currently selected for the book.""" - if target.startswith(HARDCOVER_STATUS_PREFIX): - return state.status_id == self._parse_prefixed_int(target) - if target.startswith(HARDCOVER_LIST_ID_PREFIX): - return self._parse_prefixed_int(target) in state.list_book_ids - return False - - def _set_status_target_state( - self, - book_id: int, - status_id: int, - *, - selected: bool, - state: HardcoverBookTargetState, - ) -> bool: - """Set whether the book belongs to a Hardcover status shelf.""" - if selected: - if state.user_book_id is None: - result = self._execute_query( - INSERT_USER_BOOK_MUTATION, - {"bookId": book_id, "statusId": status_id}, - raise_on_error=True, - ) - self._check_mutation_result(result, "insert_user_book") - return True - - if state.status_id == status_id: - return False - - result = self._execute_query( - UPDATE_USER_BOOK_MUTATION, - {"userBookId": state.user_book_id, "statusId": status_id}, - raise_on_error=True, - ) - self._check_mutation_result(result, "update_user_book") - return True - - if state.user_book_id is None or state.status_id != status_id: - return False - - result = self._execute_query( - DELETE_USER_BOOK_MUTATION, - {"userBookId": state.user_book_id}, - raise_on_error=True, - ) - self._check_mutation_result(result, "delete_user_book", check_error=False) - return True - - def _set_list_target_state( - self, - book_id: int, - list_id: int, - *, - selected: bool, - state: HardcoverBookTargetState, - ) -> bool: - """Set whether the book belongs to a Hardcover list.""" - list_book_id = state.list_book_ids.get(list_id) - - if selected: - if list_book_id is not None: - return False - - result = self._execute_query( - INSERT_LIST_BOOK_MUTATION, - {"bookId": book_id, "listId": list_id}, - raise_on_error=True, - ) - self._check_mutation_result(result, "insert_list_book") - return True - - if list_book_id is None: - return False - - result = self._execute_query( - DELETE_LIST_BOOK_MUTATION, - {"listBookId": list_book_id}, - raise_on_error=True, - ) - self._check_mutation_result(result, "delete_list_book", check_error=False) - return True - - def _invalidate_book_target_caches( - self, - *, - connected_user_id: str | None, - status_ids: set[int], - list_ids: set[int], - ) -> None: - """Invalidate caches affected by a target membership change.""" - metadata_cache = get_metadata_cache() - - if connected_user_id: - metadata_cache.invalidate(cache_key("hardcover:user_lists", connected_user_id)) - for status_id in status_ids: - metadata_cache.invalidate_prefix( - cache_key("hardcover:user_books:status", connected_user_id, status_id) - ) - - for list_id in list_ids: - metadata_cache.invalidate_prefix(cache_key("hardcover:list:id", list_id)) - - @staticmethod - def _parse_prefixed_int(value: str, label: str = "target") -> int: - """Parse an integer from a colon-prefixed value like 'status:1' or 'id:42'.""" - try: - return int(value.split(":", 1)[1]) - except (IndexError, ValueError) as exc: - msg = f"Invalid Hardcover {label}" - raise ValueError(msg) from exc - - @staticmethod - def _check_mutation_result(result: Any, key: str, *, check_error: bool = True) -> None: - """Raise if a Hardcover mutation failed. - - When *check_error* is True (the default) the ``error`` field inside - the payload is inspected and surfaced as a ``ValueError``. Pass - ``check_error=False`` for delete mutations that don't return an - error field. - """ - payload = result.get(key, {}) if isinstance(result, dict) else {} - if isinstance(payload, dict): - if check_error: - error_text = str(payload.get("error") or "").strip() - if error_text: - raise ValueError(error_text) - if payload.get("id") is not None: - return - msg = "Hardcover could not complete this action" - raise RuntimeError(msg) - - def search(self, options: MetadataSearchOptions) -> list[BookMetadata]: - """Search for books using Hardcover's search API.""" - return self.search_paginated(options).books - - def search_paginated(self, options: MetadataSearchOptions) -> SearchResult: - """Search for books with pagination info.""" - if not self.api_key: - logger.warning("Hardcover API key not configured") - return SearchResult(books=[], page=options.page, total_found=0, has_more=False) - - # Allow pasting a Hardcover list URL directly in the search input - list_url_parts = self._detect_list_url(options.query) - if list_url_parts: - owner_username, list_slug = list_url_parts - return self._fetch_list_books(list_slug, owner_username, options.page, options.limit) - - # Advanced filter list selector (shared fetch path with URL detection) - list_value_from_field = str(options.fields.get("hardcover_list", "")).strip() - if list_value_from_field: - if list_value_from_field.startswith(HARDCOVER_STATUS_PREFIX): - try: - status_id = self._parse_prefixed_int(list_value_from_field, "status") - return self._fetch_current_user_books_by_status( - status_id, options.page, options.limit - ) - except ValueError: - logger.debug("Invalid Hardcover status field value: %s", list_value_from_field) - return SearchResult(books=[], page=options.page, total_found=0, has_more=False) - if list_value_from_field.startswith(HARDCOVER_LIST_ID_PREFIX): - try: - list_id = self._parse_prefixed_int(list_value_from_field, "list") - return self._fetch_list_books_by_id(list_id, options.page, options.limit) - except ValueError: - logger.debug("Invalid hardcover_list field value: %s", list_value_from_field) - return SearchResult(books=[], page=options.page, total_found=0, has_more=False) - return self._fetch_list_books(list_value_from_field, None, options.page, options.limit) - - series_value_from_field = str(options.fields.get("series", "")).strip() - if series_value_from_field: - resolved_series = self._resolve_series_search_value(series_value_from_field) - if not resolved_series: - return SearchResult(books=[], page=options.page, total_found=0, has_more=False) - exclude_compilations = coerce_bool( - app_config.get("HARDCOVER_EXCLUDE_COMPILATIONS", False), - default=False, - ) - exclude_unreleased = coerce_bool( - app_config.get("HARDCOVER_EXCLUDE_UNRELEASED", False), - default=False, - ) - return self._fetch_series_books_by_id( - int(resolved_series["id"]), - options.page, - options.limit, - exclude_compilations=exclude_compilations, - exclude_unreleased=exclude_unreleased, - ) - - # Handle ISBN search separately - if options.search_type == SearchType.ISBN: - result = self.search_by_isbn(options.query) - books = [result] if result else [] - return SearchResult(books=books, page=1, total_found=len(books), has_more=False) - - # Build cache key from options (include fields and settings for cache differentiation) - fields_key = ":".join(f"{k}={v}" for k, v in sorted(options.fields.items())) - exclude_compilations = coerce_bool( - app_config.get("HARDCOVER_EXCLUDE_COMPILATIONS", False), - default=False, - ) - exclude_unreleased = coerce_bool( - app_config.get("HARDCOVER_EXCLUDE_UNRELEASED", False), - default=False, - ) - cache_key = f"{options.query}:{options.search_type.value}:{options.sort.value}:{options.limit}:{options.page}:{fields_key}:excl_comp={exclude_compilations}:excl_unrel={exclude_unreleased}" - return self._search_cached(cache_key, options) - - @cacheable(ttl_key="METADATA_CACHE_SEARCH_TTL", ttl_default=300, key_prefix="hardcover:search") - def _search_cached(self, cache_key: str, options: MetadataSearchOptions) -> SearchResult: - """Return cached Hardcover search results.""" - # Determine query and fields based on custom search fields - # Note: Hardcover API requires 'weights' when using 'fields' parameter - author_value = options.fields.get("author", "").strip() - title_value = options.fields.get("title", "").strip() - - # Build query and field configuration based on which fields are provided - query, search_fields, search_weights = self._build_search_params( - options.query, author_value, title_value, "" - ) - - # Build GraphQL query - include fields/weights parameters only when needed - if search_fields: - graphql_query = """ - query SearchBooks($query: String!, $limit: Int!, $page: Int!, $sort: String, $fields: String, $weights: String) { - search(query: $query, query_type: "Book", per_page: $limit, page: $page, sort: $sort, fields: $fields, weights: $weights) { - results - } - } - """ - else: - graphql_query = """ - query SearchBooks($query: String!, $limit: Int!, $page: Int!, $sort: String) { - search(query: $query, query_type: "Book", per_page: $limit, page: $page, sort: $sort) { - results - } - } - """ - - # Map abstract sort order to Hardcover's sort parameter - sort_param = SORT_MAPPING.get(options.sort, SORT_MAPPING[SortOrder.RELEVANCE]) - - variables = { - "query": query, - "limit": options.limit, - "page": options.page, - "sort": sort_param, - } - - if search_fields: - variables["fields"] = search_fields - variables["weights"] = search_weights - - try: - result = self._execute_query(graphql_query, variables) - if not result: - logger.debug("Hardcover search: No result from API") - return SearchResult(books=[], page=options.page, total_found=0, has_more=False) - - # Extract hits from Typesense response - hits, found_count = _extract_typesense_hits(result) - - # Parse hits, filtering compilations and unreleased books if enabled - exclude_compilations = coerce_bool( - app_config.get("HARDCOVER_EXCLUDE_COMPILATIONS", False), - default=False, - ) - exclude_unreleased = coerce_bool( - app_config.get("HARDCOVER_EXCLUDE_UNRELEASED", False), - default=False, - ) - current_year = datetime.now(UTC).year - books = [] - for hit in hits: - item = _unwrap_hit_document(hit) - if item is None: - continue - if exclude_compilations and item.get("compilation"): - continue - if exclude_unreleased: - release_year = item.get("release_year") - if release_year is not None and release_year > current_year: - continue - book = self._parse_search_result(item) - if book: - books.append(book) - - logger.info( - "Hardcover search '%s' (fields=%s) returned %s results", - query, - search_fields, - len(books), - ) - - # Calculate if there are more results - results_so_far = (options.page - 1) * HARDCOVER_PAGE_SIZE + len(hits) - has_more = results_so_far < found_count - - return SearchResult( - books=books, page=options.page, total_found=found_count, has_more=has_more - ) - - except AttributeError, KeyError, TypeError, ValueError: - logger.exception("Hardcover search error") - return SearchResult(books=[], page=options.page, total_found=0, has_more=False) - - @cacheable(ttl_key="METADATA_CACHE_BOOK_TTL", ttl_default=600, key_prefix="hardcover:book") - def get_book(self, book_id: str) -> BookMetadata | None: - """Get book details by Hardcover ID.""" - if not self.api_key: - logger.warning("Hardcover API key not configured") - return None - - # Query for specific book by ID - # Use contributions with filter to get only primary authors (not translators/narrators) - # Also include cached_contributors as fallback if contributions is empty - # Include featured_book_series for series info - # Include editions with titles and languages for localized search support - graphql_query = """ - query GetBook($id: Int!) { - books(where: {id: {_eq: $id}}, limit: 1) { - id - title - subtitle - slug - release_date - headline - description - pages - cached_image - cached_tags - cached_contributors - contributions(where: {contribution: {_eq: "Author"}}) { - author { - name - } - } - default_physical_edition { - isbn_10 - isbn_13 - } - featured_book_series { - position - series { - id - name - primary_books_count - } - } - editions( - distinct_on: language_id - order_by: [{language_id: asc}, {users_count: desc}] - limit: 200 - ) { - title - language { - language - code2 - code3 - } - } - } - } - """ - - try: - book_id_int = int(book_id) - result = self._execute_query(graphql_query, {"id": book_id_int}) - if not result: - return None - - books = result.get("books", []) - if not books: - return None - - return self._parse_book(books[0]) - - except ValueError: - logger.exception("Invalid book ID: %s", book_id) - return None - except AttributeError, KeyError, TypeError: - logger.exception("Hardcover get_book error") - return None - - @cacheable(ttl_key="METADATA_CACHE_BOOK_TTL", ttl_default=600, key_prefix="hardcover:isbn") - def search_by_isbn(self, isbn: str) -> BookMetadata | None: - """Search for a book by ISBN-10 or ISBN-13.""" - if not self.api_key: - logger.warning("Hardcover API key not configured") - return None - - # Clean ISBN (remove hyphens) - clean_isbn = isbn.replace("-", "").strip() - - # Search for editions with matching ISBN - # Use contributions with filter to get only primary authors (not translators/narrators) - graphql_query = """ - query SearchByISBN($isbn: String!) { - editions( - where: { - _or: [ - {isbn_10: {_eq: $isbn}}, - {isbn_13: {_eq: $isbn}} - ] - }, - limit: 1 - ) { - isbn_10 - isbn_13 - book { - id - title - subtitle - slug - release_date - headline - description - pages - cached_image - cached_tags - contributions(where: {contribution: {_eq: "Author"}}) { - author { - name - } - } - } - } - } - """ - - try: - result = self._execute_query(graphql_query, {"isbn": clean_isbn}) - if not result: - return None - - editions = result.get("editions", []) - if not editions: - logger.debug("No Hardcover book found for ISBN: %s", isbn) - return None - - edition = editions[0] - book_data = edition.get("book", {}) - if not book_data: - return None - - # Add ISBN data from edition to book data - book_data["isbn_10"] = edition.get("isbn_10") - book_data["isbn_13"] = edition.get("isbn_13") - - return self._parse_book(book_data) - - except AttributeError, IndexError, KeyError, TypeError, ValueError: - logger.exception("Hardcover ISBN search error") - return None - - def _execute_query( - self, - query: str, - variables: dict[str, Any], - *, - raise_on_error: bool = False, - ) -> dict | None: - """Execute a GraphQL query and return data or None on error.""" - - def _raise_graphql_error(message: str) -> None: - raise HardcoverGraphQLError(message) - - try: - response = self.session.post( - HARDCOVER_API_URL, - json={"query": query, "variables": variables}, - timeout=15, - verify=get_ssl_verify(HARDCOVER_API_URL), - ) - response.raise_for_status() - - data = response.json() - - if "errors" in data: - logger.error("GraphQL errors: %s", data["errors"]) - if raise_on_error: - message = ( - _extract_graphql_error_message(data) or "Hardcover rejected this request" - ) - _raise_graphql_error(message) - return None - - return data.get("data") - - except requests.Timeout as e: - logger.warning("Hardcover API request timed out") - if raise_on_error: - msg = "Hardcover API request timed out" - raise RuntimeError(msg) from e - return None - except requests.HTTPError as e: - if e.response.status_code == HTTPStatus.UNAUTHORIZED: - logger.exception("Hardcover API key is invalid") - if raise_on_error: - msg = "Hardcover API key is invalid" - raise RuntimeError(msg) from e - else: - logger.exception("Hardcover API HTTP error") - if raise_on_error: - msg = f"Hardcover API HTTP error: {e}" - raise RuntimeError(msg) from e - return None - except HardcoverGraphQLError: - raise - except ValueError as e: - logger.exception("Hardcover API returned invalid JSON") - if raise_on_error: - msg = "Hardcover API returned an invalid response" - raise RuntimeError(msg) from e - return None - except (TypeError, requests.RequestException) as e: - logger.exception("Hardcover API request failed") - if raise_on_error: - msg = "Hardcover API request failed" - raise RuntimeError(msg) from e - return None - - def _parse_search_result(self, item: dict) -> BookMetadata | None: - """Parse a search result item into BookMetadata.""" - try: - book_id = item.get("id") or item.get("document", {}).get("id") - title = item.get("title") or item.get("document", {}).get("title") - - if not book_id or not title: - return None - - # Extract authors - use contribution_types to filter author_names if available - authors = [] - - author_names = item.get("author_names", []) - if isinstance(author_names, str): - author_names = [author_names] - - contribution_types = item.get("contribution_types", []) - - # If we have parallel arrays, filter to only "Author" contributions - if contribution_types and len(contribution_types) == len(author_names): - for name, contrib_type in zip(author_names, contribution_types, strict=True): - if contrib_type == "Author": - authors.append(name) - elif author_names: - # No contribution_types or length mismatch - use all names as fallback - authors = author_names - - # Normalize whitespace in author names (some API data has multiple spaces) - authors = [" ".join(name.split()) for name in authors] - - search_author = _simplify_author_for_search(authors[0]) if authors else None - - cover_url = _extract_cover_url(item, "image") - publish_year = _extract_publish_year(item) - source_url = _build_source_url(item.get("slug", "")) - - # Build display fields from Hardcover-specific data - display_fields = [] - - # Rating (e.g., "4.5 (3,764)") - rating = item.get("rating") - ratings_count = item.get("ratings_count") - if rating is not None: - rating_str = f"{rating:.1f}" - if ratings_count: - rating_str += f" ({ratings_count:,})" - display_fields.append(DisplayField(label="Rating", value=rating_str, icon="star")) - - # Readers (users who have this book) - users_count = item.get("users_count") - if users_count: - display_fields.append( - DisplayField(label="Readers", value=f"{users_count:,}", icon="users") - ) - - # Combine headline and description if both present - headline = item.get("headline") - description = item.get("description") - full_description = _combine_headline_description(headline, description) - - # Extract subtitle if available in search results - subtitle = item.get("subtitle") - - return BookMetadata( - provider="hardcover", - provider_id=str(book_id), - title=title, - subtitle=subtitle, - search_title=_compute_search_title(title, subtitle), - search_author=search_author, - provider_display_name="Hardcover", - authors=authors, - cover_url=cover_url, - description=full_description, - publish_year=publish_year, - source_url=source_url, - display_fields=display_fields, - ) - - except (AttributeError, KeyError, TypeError, ValueError) as e: - logger.debug("Failed to parse Hardcover search result: %s", e) - return None - - def _parse_book(self, book: dict) -> BookMetadata: - """Parse a book object into BookMetadata.""" - title = str(book.get("title") or "") - subtitle = book.get("subtitle") - - # Extract authors - try contributions first (filtered), fall back to cached_contributors - authors = [] - contributions = book.get("contributions") or [] - cached_contributors = book.get("cached_contributors") or [] - - # Try contributions first (filtered to "Author" role only - cleaner data) - for contrib in contributions: - author = contrib.get("author", {}) - if author and author.get("name"): - authors.append(author["name"]) - - # Fallback to cached_contributors if no authors found - if not authors: - for contrib in cached_contributors: - if isinstance(contrib, dict): - # Handle nested structure: {"author": {"name": "..."}, "contribution": ...} - if contrib.get("author", {}).get("name"): - authors.append(contrib["author"]["name"]) - # Handle flat structure: {"name": "..."} - elif contrib.get("name"): - authors.append(contrib["name"]) - elif isinstance(contrib, str): - authors.append(contrib) - - # Normalize whitespace in author names (some API data has multiple spaces) - authors = [" ".join(name.split()) for name in authors] - - search_author = _simplify_author_for_search(authors[0]) if authors else None - - cover_url = _extract_cover_url(book, "cached_image", "image") - publish_year = _extract_publish_year(book) - - # Extract genres from cached_tags - genres = [] - for tag in book.get("cached_tags", []): - if isinstance(tag, dict) and tag.get("tag"): - genres.append(tag["tag"]) - elif isinstance(tag, str): - genres.append(tag) - - # Get ISBN from direct fields, default_physical_edition, or editions - isbn_10 = book.get("isbn_10") - isbn_13 = book.get("isbn_13") - - if not isbn_10 and not isbn_13: - # Try default_physical_edition first - edition = book.get("default_physical_edition") - if edition: - isbn_10 = edition.get("isbn_10") - isbn_13 = edition.get("isbn_13") - - # Fallback to editions array - if not isbn_10 and not isbn_13 and book.get("editions"): - for ed in book["editions"]: - if not isbn_10 and ed.get("isbn_10"): - isbn_10 = ed["isbn_10"] - if not isbn_13 and ed.get("isbn_13"): - isbn_13 = ed["isbn_13"] - if isbn_10 and isbn_13: - break - - source_url = _build_source_url(book.get("slug", "")) - - # Combine headline and description if both present - headline = book.get("headline") - description = book.get("description") - full_description = _combine_headline_description(headline, description) - - # Extract series info from featured_book_series - series_id = None - series_name = None - series_position = None - series_count = None - featured_series = book.get("featured_book_series") - if featured_series: - series_position = featured_series.get("position") - series_data = featured_series.get("series") - if series_data: - if series_data.get("id") is not None: - series_id = str(series_data.get("id")) - series_name = series_data.get("name") - series_count = series_data.get("primary_books_count") - - # Extract titles by language from editions - # This allows searching with localized titles when language filter is active - titles_by_language: dict[str, str] = {} - editions = book.get("editions", []) - for edition in editions: - edition_title = edition.get("title") - lang_data = edition.get("language") - if edition_title and lang_data: - # Store by various language identifiers for flexible matching - # Language name (e.g., "German", "English") - lang_name = lang_data.get("language") - # 2-letter code (e.g., "de", "en") - code2 = lang_data.get("code2") - # 3-letter code (e.g., "deu", "eng") - code3 = lang_data.get("code3") - - # Store with all available keys (first title wins for each language) - if lang_name and lang_name not in titles_by_language: - titles_by_language[lang_name] = edition_title - if code2 and code2 not in titles_by_language: - titles_by_language[code2] = edition_title - if code3 and code3 not in titles_by_language: - titles_by_language[code3] = edition_title - - # Build display fields from Hardcover-specific metrics - display_fields: list[DisplayField] = [] - - rating = book.get("rating") - ratings_count = book.get("ratings_count") - if rating is not None: - try: - rating_str = f"{float(rating):.1f}" - except TypeError, ValueError: - rating_str = str(rating) - - if ratings_count: - with suppress(TypeError, ValueError): - rating_str += f" ({int(ratings_count):,})" - - display_fields.append(DisplayField(label="Rating", value=rating_str, icon="star")) - - users_count = book.get("users_count") - if users_count: - try: - readers_value = f"{int(users_count):,}" - except TypeError, ValueError: - readers_value = str(users_count) - display_fields.append(DisplayField(label="Readers", value=readers_value, icon="users")) - - return BookMetadata( - provider="hardcover", - provider_id=str(book["id"]), - title=title, - subtitle=subtitle, - search_title=_compute_search_title(title, subtitle, series_name=series_name), - search_author=search_author, - provider_display_name="Hardcover", - authors=authors, - isbn_10=isbn_10, - isbn_13=isbn_13, - cover_url=cover_url, - description=full_description, - publish_year=publish_year, - genres=genres, - source_url=source_url, - series_id=series_id, - series_name=series_name, - series_position=series_position, - series_count=series_count, - titles_by_language=titles_by_language, - display_fields=display_fields, - ) - - -def _test_hardcover_connection(current_values: dict[str, Any] | None = None) -> dict[str, Any]: - """Test the Hardcover API connection using current form values.""" - from shelfmark.core.config import config as app_config - - current_values = current_values or {} - - # Use current form values first, fall back to saved config - raw_key = current_values.get("HARDCOVER_API_KEY") or app_config.get("HARDCOVER_API_KEY", "") - api_key = _normalize_hardcover_api_key(raw_key) - - key_len = len(api_key) if api_key else 0 - logger.debug("Hardcover test: key length=%s", key_len) - - if not api_key: - # Clear any stored connection metadata since there's no key - _save_connected_user(None, None) - return {"success": False, "message": "API key is required"} - - if key_len < HARDCOVER_API_KEY_MIN_LENGTH: - return { - "success": False, - "message": ( - f"API key seems too short ({key_len} chars). " - f"Expected {HARDCOVER_API_KEY_MIN_LENGTH}+ chars." - ), - } - - connection_result = {"success": False, "message": "API request failed - check your API key"} - try: - provider = HardcoverProvider(api_key=api_key) - # Use the 'me' query to test connection (recommended by API docs) - result = provider._execute_query("query { me { id, username } }", {}) - if result is not None: - # Handle both single object and array response formats - me_data = result.get("me", {}) - if isinstance(me_data, list) and me_data: - me_data = me_data[0] - user_id = ( - str(me_data.get("id")) - if isinstance(me_data, dict) and me_data.get("id") is not None - else None - ) - username = ( - me_data.get("username", "Unknown") if isinstance(me_data, dict) else "Unknown" - ) - - # Save connected user metadata for persistent display + per-user list caching - _save_connected_user(user_id, username) - connection_result = {"success": True, "message": f"Connected as: {username}"} - else: - _save_connected_user(None, None) - except (AttributeError, KeyError, requests.RequestException, TypeError, ValueError) as e: - logger.exception("Hardcover connection test failed") - _save_connected_user(None, None) - return {"success": False, "message": f"Connection failed: {e!s}"} - - return connection_result - - -def _save_connected_user(user_id: str | None, username: str | None) -> None: - """Save or clear connected user metadata in config.""" - from shelfmark.core.settings_registry import load_config_file, save_config_file - - config = load_config_file("hardcover") - if user_id: - config["_connected_user_id"] = user_id - else: - config.pop("_connected_user_id", None) - - if username: - config["_connected_username"] = username - else: - config.pop("_connected_username", None) - - save_config_file("hardcover", config) - - -def _get_connected_username() -> str | None: - """Get the stored connected username.""" - from shelfmark.core.settings_registry import load_config_file - - config = load_config_file("hardcover") - return config.get("_connected_username") - - -def _get_connected_user_id() -> str | None: - """Get the stored connected Hardcover user id.""" - from shelfmark.core.settings_registry import load_config_file - - config = load_config_file("hardcover") - value = config.get("_connected_user_id") - return str(value) if value is not None else None - - -# Hardcover sort options for settings UI -_HARDCOVER_SORT_OPTIONS = [ - {"value": "relevance", "label": "Most relevant"}, - {"value": "popularity", "label": "Most popular"}, - {"value": "rating", "label": "Highest rated"}, - {"value": "newest", "label": "Newest"}, - {"value": "oldest", "label": "Oldest"}, -] - - -@register_settings("hardcover", "Hardcover", icon="book", order=51, group="metadata_providers") -def hardcover_settings() -> list[SettingsField]: - """Hardcover metadata provider settings.""" - # Check for connected username to show status - connected_user = _get_connected_username() - test_button_description = ( - f"Connected as: {connected_user}" if connected_user else "Verify your API key works" - ) - - return [ - HeadingField( - key="hardcover_heading", - title="Hardcover", - description="A modern book tracking and discovery platform with a comprehensive API.", - link_url="https://hardcover.app", - link_text="hardcover.app", - ), - CheckboxField( - key="HARDCOVER_ENABLED", - label="Enable Hardcover", - description="Enable Hardcover as a metadata provider for book searches", - default=False, - ), - PasswordField( - key="HARDCOVER_API_KEY", - label="API Key", - description="Get your API key from hardcover.app/account/api", - required=True, - ), - ActionButton( - key="test_connection", - label="Test Connection", - description=test_button_description, - style="primary", - callback=_test_hardcover_connection, - ), - SelectField( - key="HARDCOVER_DEFAULT_SORT", - label="Default Sort Order", - description="Default sort order for Hardcover search results.", - options=_HARDCOVER_SORT_OPTIONS, - default="relevance", - ), - CheckboxField( - key="HARDCOVER_EXCLUDE_COMPILATIONS", - label="Exclude Compilations", - description="Filter out compilations, anthologies, and omnibus editions from search results", - default=False, - ), - CheckboxField( - key="HARDCOVER_EXCLUDE_UNRELEASED", - label="Exclude Unreleased Books", - description="Filter out books with a release year in the future", - default=False, - ), - CheckboxField( - key="HARDCOVER_AUTO_REMOVE_ON_DOWNLOAD", - label="Auto-Remove from List on Download", - description="Automatically remove a book from the active Hardcover list when you download it", - default=True, - ), - ] diff --git a/shelfmark/metadata_providers/hardcover/__init__.py b/shelfmark/metadata_providers/hardcover/__init__.py new file mode 100644 index 0000000..b64fabd --- /dev/null +++ b/shelfmark/metadata_providers/hardcover/__init__.py @@ -0,0 +1,35 @@ +"""Hardcover metadata provider package.""" + +from shelfmark.core.cache import get_metadata_cache +from shelfmark.core.config import config as app_config + +from .auth import _get_connected_user_id, _get_connected_username, _save_connected_user +from .constants import ( + HARDCOVER_LIST_ID_PREFIX, + HARDCOVER_STATUS_GROUP, + HARDCOVER_STATUS_PREFIX, + HARDCOVER_WRITABLE_TARGET_GROUPS, +) +from .models import HardcoverBookTargetState, HardcoverGraphQLError, HardcoverTargetPayloadError +from .parsing import _compute_search_title, _simplify_author_for_search +from .provider import HardcoverProvider +from .settings import hardcover_settings + +__all__ = [ + "HARDCOVER_LIST_ID_PREFIX", + "HARDCOVER_STATUS_GROUP", + "HARDCOVER_STATUS_PREFIX", + "HARDCOVER_WRITABLE_TARGET_GROUPS", + "HardcoverBookTargetState", + "HardcoverGraphQLError", + "HardcoverProvider", + "HardcoverTargetPayloadError", + "_compute_search_title", + "_get_connected_user_id", + "_get_connected_username", + "_save_connected_user", + "_simplify_author_for_search", + "app_config", + "get_metadata_cache", + "hardcover_settings", +] diff --git a/shelfmark/metadata_providers/hardcover/auth.py b/shelfmark/metadata_providers/hardcover/auth.py new file mode 100644 index 0000000..e4c62f7 --- /dev/null +++ b/shelfmark/metadata_providers/hardcover/auth.py @@ -0,0 +1,36 @@ +"""Persistence helpers for the connected Hardcover account.""" + + +def _save_connected_user(user_id: str | None, username: str | None) -> None: + """Save or clear connected user metadata in config.""" + from shelfmark.core.settings_registry import load_config_file, save_config_file + + config = load_config_file("hardcover") + if user_id: + config["_connected_user_id"] = user_id + else: + config.pop("_connected_user_id", None) + + if username: + config["_connected_username"] = username + else: + config.pop("_connected_username", None) + + save_config_file("hardcover", config) + + +def _get_connected_username() -> str | None: + """Get the stored connected username.""" + from shelfmark.core.settings_registry import load_config_file + + config = load_config_file("hardcover") + return config.get("_connected_username") + + +def _get_connected_user_id() -> str | None: + """Get the stored connected Hardcover user id.""" + from shelfmark.core.settings_registry import load_config_file + + config = load_config_file("hardcover") + value = config.get("_connected_user_id") + return str(value) if value is not None else None diff --git a/shelfmark/metadata_providers/hardcover/client.py b/shelfmark/metadata_providers/hardcover/client.py new file mode 100644 index 0000000..01ebea9 --- /dev/null +++ b/shelfmark/metadata_providers/hardcover/client.py @@ -0,0 +1,105 @@ +"""GraphQL transport helpers for Hardcover.""" + +from http import HTTPStatus +from typing import Any + +import requests + +from shelfmark.core.logger import setup_logger +from shelfmark.download.network import get_ssl_verify + +from .constants import HARDCOVER_API_URL +from .models import HardcoverGraphQLError + +logger = setup_logger(__name__) + + +def _extract_graphql_error_message(payload: Any) -> str: + """Extract a readable message from a GraphQL error payload.""" + if not isinstance(payload, dict): + return "" + + errors = payload.get("errors", []) + if not isinstance(errors, list): + return "" + + messages: list[str] = [] + for error in errors: + if not isinstance(error, dict): + continue + message = str(error.get("message") or "").strip() + if message: + messages.append(message) + + return "; ".join(messages) + + +class HardcoverClientMixin: + session: requests.Session + + def _execute_query( + self, + query: str, + variables: dict[str, Any], + *, + raise_on_error: bool = False, + ) -> dict | None: + """Execute a GraphQL query and return data or None on error.""" + + def _raise_graphql_error(message: str) -> None: + raise HardcoverGraphQLError(message) + + try: + response = self.session.post( + HARDCOVER_API_URL, + json={"query": query, "variables": variables}, + timeout=15, + verify=get_ssl_verify(HARDCOVER_API_URL), + ) + response.raise_for_status() + + data = response.json() + + if "errors" in data: + logger.error("GraphQL errors: %s", data["errors"]) + if raise_on_error: + message = ( + _extract_graphql_error_message(data) or "Hardcover rejected this request" + ) + _raise_graphql_error(message) + return None + + return data.get("data") + + except requests.Timeout as e: + logger.warning("Hardcover API request timed out") + if raise_on_error: + msg = "Hardcover API request timed out" + raise RuntimeError(msg) from e + return None + except requests.HTTPError as e: + if e.response.status_code == HTTPStatus.UNAUTHORIZED: + logger.exception("Hardcover API key is invalid") + if raise_on_error: + msg = "Hardcover API key is invalid" + raise RuntimeError(msg) from e + else: + logger.exception("Hardcover API HTTP error") + if raise_on_error: + msg = f"Hardcover API HTTP error: {e}" + raise RuntimeError(msg) from e + return None + except HardcoverGraphQLError: + raise + except ValueError as e: + logger.exception("Hardcover API returned invalid JSON") + if raise_on_error: + msg = "Hardcover API returned an invalid response" + raise RuntimeError(msg) from e + return None + except (TypeError, requests.RequestException) as e: + logger.exception("Hardcover API request failed") + if raise_on_error: + msg = "Hardcover API request failed" + raise RuntimeError(msg) from e + return None diff --git a/shelfmark/metadata_providers/hardcover/constants.py b/shelfmark/metadata_providers/hardcover/constants.py new file mode 100644 index 0000000..b680881 --- /dev/null +++ b/shelfmark/metadata_providers/hardcover/constants.py @@ -0,0 +1,61 @@ +"""Constants for the Hardcover metadata provider.""" + +import re + +from shelfmark.metadata_providers import SearchType, SortOrder + +HARDCOVER_API_URL = "https://api.hardcover.app/v1/graphql" +HARDCOVER_PAGE_SIZE = 25 # Hardcover API returns max 25 results per page +HARDCOVER_MIN_AUTHOR_PARTS = 2 +HARDCOVER_MIN_TYPEAHEAD_QUERY_LENGTH = 2 +HARDCOVER_MAX_SERIES_OPTIONS = 7 +HARDCOVER_API_KEY_MIN_LENGTH = 100 +HARDCOVER_LIST_URL_PATTERN = re.compile( + r"^/(?:@([\w.-]+)/)?lists?/([\w-]+)/?$", + re.IGNORECASE, +) + +HARDCOVER_STATUS_PREFIX = "status:" +HARDCOVER_STATUSES: list[dict] = [ + {"id": 1, "label": "Want to Read", "slug": "want-to-read", "query_key": "want_to_read_count"}, + { + "id": 2, + "label": "Currently Reading", + "slug": "currently-reading", + "query_key": "currently_reading_count", + }, + {"id": 3, "label": "Read", "slug": "read", "query_key": "read_count"}, + { + "id": 5, + "label": "Did Not Finish", + "slug": "did-not-finish", + "query_key": "did_not_finish_count", + }, +] +HARDCOVER_STATUS_URL_SLUGS: dict[int, str] = {s["id"]: s["slug"] for s in HARDCOVER_STATUSES} +HARDCOVER_STATUS_GROUP = "Reading Status" +HARDCOVER_LIST_ID_PREFIX = "id:" +HARDCOVER_WRITABLE_TARGET_GROUPS = {HARDCOVER_STATUS_GROUP, "My Lists"} + +SORT_MAPPING: dict[SortOrder, str] = { + SortOrder.RELEVANCE: "_text_match:desc,users_count:desc", + SortOrder.POPULARITY: "users_count:desc", + SortOrder.RATING: "rating:desc", + SortOrder.NEWEST: "release_year:desc", + SortOrder.OLDEST: "release_year:asc", +} +SEARCH_TYPE_FIELDS: dict[SearchType, str] = { + SearchType.GENERAL: "title,isbns,series_names,author_names,alternative_titles", + SearchType.TITLE: "title,alternative_titles", + SearchType.AUTHOR: "author_names", + # ISBN is handled separately via search_by_isbn() +} +SERIES_SEARCH_FIELDS = "name,books,author_name" +SERIES_SEARCH_WEIGHTS = "2,1,1" +SERIES_SEARCH_SORT = "_text_match:desc,readers_count:desc" +AUTHOR_SUGGESTION_FIELDS = "name,name_personal,alternate_names" +AUTHOR_SUGGESTION_WEIGHTS = "4,3,2" +AUTHOR_SUGGESTION_SORT = "_text_match:desc,books_count:desc" +TITLE_SUGGESTION_FIELDS = "title,alternative_titles" +TITLE_SUGGESTION_WEIGHTS = "5,2" +TITLE_SUGGESTION_SORT = "_text_match:desc,users_count:desc" diff --git a/shelfmark/metadata_providers/hardcover/lists.py b/shelfmark/metadata_providers/hardcover/lists.py new file mode 100644 index 0000000..e4cd1a1 --- /dev/null +++ b/shelfmark/metadata_providers/hardcover/lists.py @@ -0,0 +1,400 @@ +"""Hardcover list and status-shelf workflows.""" + +from typing import TYPE_CHECKING, Any +from urllib.parse import urlparse + +from shelfmark.core.cache import cacheable +from shelfmark.core.logger import setup_logger +from shelfmark.core.request_helpers import coerce_int +from shelfmark.metadata_providers import BookMetadata, SearchResult + +from .auth import _get_connected_user_id, _get_connected_username, _save_connected_user +from .constants import ( + HARDCOVER_LIST_URL_PATTERN, + HARDCOVER_STATUS_GROUP, + HARDCOVER_STATUS_PREFIX, + HARDCOVER_STATUS_URL_SLUGS, + HARDCOVER_STATUSES, +) +from .queries import ( + LIST_BOOKS_BY_ID_QUERY, + LIST_LOOKUP_QUERY, + USER_BOOKS_BY_STATUS_QUERY, + USER_LISTS_QUERY, +) + +logger = setup_logger(__name__) + + +class HardcoverListsMixin: + if TYPE_CHECKING: + api_key: str + + def _execute_query( + self, + query: str, + variables: dict[str, Any], + *, + raise_on_error: bool = False, + ) -> dict[str, Any] | None: ... + + def _parse_book(self, book: dict[str, Any]) -> BookMetadata: ... + + def _detect_list_url(self, query: str) -> tuple[str | None, str] | None: + """Detect and extract optional owner username + list slug from a URL string.""" + candidate = query.strip() + if not candidate: + return None + + parsed = urlparse(candidate) + if parsed.scheme not in {"http", "https"}: + return None + + hostname = (parsed.hostname or "").lower() + if hostname not in {"hardcover.app", "www.hardcover.app"}: + return None + + match = HARDCOVER_LIST_URL_PATTERN.match(parsed.path or "") + if not match: + return None + + owner_username = match.group(1).strip() if match.group(1) else None + slug = match.group(2).strip() + if not slug: + return None + + return owner_username, slug + + @cacheable(ttl_key="METADATA_CACHE_SEARCH_TTL", ttl_default=300, key_prefix="hardcover:list:id") + def _fetch_list_books_by_id(self, list_id: int, page: int, limit: int) -> SearchResult: + """Fetch list books by unique Hardcover list ID.""" + if not self.api_key: + return SearchResult(books=[], page=page, total_found=0, has_more=False) + + offset = (page - 1) * limit + + result = self._execute_query( + LIST_BOOKS_BY_ID_QUERY, + { + "id": list_id, + "limit": limit, + "offset": offset, + }, + ) + if not result: + return SearchResult(books=[], page=page, total_found=0, has_more=False) + + lists = result.get("lists", []) + if not lists: + return SearchResult(books=[], page=page, total_found=0, has_more=False) + + list_data = lists[0] if isinstance(lists[0], dict) else {} + list_books = list_data.get("list_books", []) if isinstance(list_data, dict) else [] + books_count_raw = list_data.get("books_count", 0) if isinstance(list_data, dict) else 0 + + # Build source URL and title from list metadata + source_url = None + source_title = str(list_data.get("name") or "").strip() or None + list_slug = str(list_data.get("slug") or "").strip() + user_data = list_data.get("user", {}) + owner_username = ( + str(user_data.get("username") or "").strip() if isinstance(user_data, dict) else "" + ) + if list_slug and owner_username: + source_url = f"https://hardcover.app/@{owner_username}/lists/{list_slug}" + + try: + books_count = int(books_count_raw) + except TypeError, ValueError: + books_count = 0 + + books: list[BookMetadata] = [] + for item in list_books: + if not isinstance(item, dict): + continue + book_data = item.get("book", {}) + if not isinstance(book_data, dict) or not book_data: + continue + try: + parsed_book = self._parse_book(book_data) + if parsed_book: + books.append(parsed_book) + except (AttributeError, IndexError, KeyError, TypeError, ValueError) as exc: + logger.debug("Failed to parse Hardcover list book for list_id=%s: %s", list_id, exc) + + has_more = offset + len(list_books) < books_count + return SearchResult( + books=books, + page=page, + total_found=books_count, + has_more=has_more, + source_url=source_url, + source_title=source_title, + ) + + @cacheable( + ttl_key="METADATA_CACHE_SEARCH_TTL", ttl_default=300, key_prefix="hardcover:list:slug" + ) + def _fetch_list_books( + self, slug: str, owner_username: str | None, page: int, limit: int + ) -> SearchResult: + """Fetch list books by slug, optionally disambiguating by owner username.""" + if not self.api_key: + return SearchResult(books=[], page=page, total_found=0, has_more=False) + + lookup = self._execute_query(LIST_LOOKUP_QUERY, {"slug": slug}) + if not lookup: + return SearchResult(books=[], page=page, total_found=0, has_more=False) + + lists = lookup.get("lists", []) + if not isinstance(lists, list) or not lists: + return SearchResult(books=[], page=page, total_found=0, has_more=False) + + selected: dict[str, Any] | None = None + normalized_owner = owner_username.lower() if owner_username else None + if normalized_owner: + for item in lists: + if not isinstance(item, dict): + continue + owner_data = item.get("user", {}) + if not isinstance(owner_data, dict): + continue + candidate_owner = str(owner_data.get("username") or "").strip().lower() + if candidate_owner == normalized_owner: + selected = item + break + + if selected is None: + first_item = lists[0] + selected = first_item if isinstance(first_item, dict) else None + + if not selected: + return SearchResult(books=[], page=page, total_found=0, has_more=False) + + list_id = coerce_int(selected.get("id"), 0) + if list_id < 1: + return SearchResult(books=[], page=page, total_found=0, has_more=False) + + return self._fetch_list_books_by_id(list_id, page, limit) + + def _resolve_current_user_id(self) -> str | None: + """Resolve current Hardcover user id from saved settings or API me query.""" + connected_user_id = _get_connected_user_id() + if connected_user_id: + return connected_user_id + + result = self._execute_query("query { me { id, username } }", {}) + if not result: + return None + + me_data = result.get("me", {}) + if isinstance(me_data, list) and me_data: + me_data = me_data[0] + if not isinstance(me_data, dict): + return None + + user_id_raw = me_data.get("id") + if user_id_raw is None: + return None + + user_id = str(user_id_raw) + username_raw = me_data.get("username") + username = str(username_raw).strip() if username_raw else _get_connected_username() + _save_connected_user(user_id, username) + return user_id + + def get_user_lists(self) -> list[dict[str, str]]: + """Get authenticated user's own and followed Hardcover lists.""" + if not self.api_key: + return [] + + connected_user_id = self._resolve_current_user_id() + if not connected_user_id: + return self._fetch_user_lists() + + return self._get_user_lists_cached(connected_user_id) + + @cacheable(ttl=120, key_prefix="hardcover:user_lists") + def _get_user_lists_cached(self, _cache_user_id: str) -> list[dict[str, str]]: + """Return cached user lists keyed by Hardcover user id.""" + return self._fetch_user_lists() + + def _fetch_current_user_books_by_status( + self, status_id: int, page: int, limit: int + ) -> SearchResult: + """Fetch the current user's Hardcover books for a specific status shelf.""" + if not self.api_key: + return SearchResult(books=[], page=page, total_found=0, has_more=False) + + connected_user_id = self._resolve_current_user_id() + if not connected_user_id: + return SearchResult(books=[], page=page, total_found=0, has_more=False) + + return self._fetch_user_books_by_status_cached(connected_user_id, status_id, page, limit) + + @cacheable( + ttl_key="METADATA_CACHE_SEARCH_TTL", + ttl_default=300, + key_prefix="hardcover:user_books:status", + ) + def _fetch_user_books_by_status_cached( + self, + _cache_user_id: str, + status_id: int, + page: int, + limit: int, + ) -> SearchResult: + """Return cached status-shelf books keyed by user id and shelf.""" + return self._fetch_user_books_by_status(status_id, page, limit) + + def _fetch_user_books_by_status(self, status_id: int, page: int, limit: int) -> SearchResult: + """Fetch books from the current user's Hardcover status shelf.""" + if not self.api_key: + return SearchResult(books=[], page=page, total_found=0, has_more=False) + + offset = (page - 1) * limit + result = self._execute_query( + USER_BOOKS_BY_STATUS_QUERY, + { + "statusId": status_id, + "limit": limit, + "offset": offset, + }, + ) + if not result: + return SearchResult(books=[], page=page, total_found=0, has_more=False) + + me_data = result.get("me", {}) + if isinstance(me_data, list) and me_data: + me_data = me_data[0] + if not isinstance(me_data, dict): + return SearchResult(books=[], page=page, total_found=0, has_more=False) + + status_books = me_data.get("status_books", []) + aggregate_data = me_data.get("status_books_aggregate", {}) + aggregate = aggregate_data.get("aggregate", {}) if isinstance(aggregate_data, dict) else {} + count_raw = aggregate.get("count", 0) if isinstance(aggregate, dict) else 0 + + try: + total_found = int(count_raw) + except TypeError, ValueError: + total_found = 0 + + books: list[BookMetadata] = [] + for item in status_books: + if not isinstance(item, dict): + continue + book_data = item.get("book", {}) + if not isinstance(book_data, dict) or not book_data: + continue + try: + parsed_book = self._parse_book(book_data) + if parsed_book: + books.append(parsed_book) + except (AttributeError, KeyError, TypeError, ValueError) as exc: + logger.debug( + "Failed to parse Hardcover status book for status_id=%s: %s", status_id, exc + ) + + has_more = offset + len(status_books) < total_found + + # Build source URL for the status shelf + source_url = None + url_slug = HARDCOVER_STATUS_URL_SLUGS.get(status_id) + username = _get_connected_username() + if url_slug and username: + source_url = f"https://hardcover.app/@{username}/books/{url_slug}" + + return SearchResult( + books=books, + page=page, + total_found=total_found, + has_more=has_more, + source_url=source_url, + ) + + def _fetch_user_lists(self) -> list[dict[str, str]]: + """Fetch raw list options from Hardcover me query.""" + result = self._execute_query(USER_LISTS_QUERY, {}) + if not result: + return [] + + me_data = result.get("me", {}) + if isinstance(me_data, list) and me_data: + me_data = me_data[0] + if not isinstance(me_data, dict): + return [] + + options: list[dict[str, str]] = [] + seen_values: set[str] = set() + current_username = str(me_data.get("username") or "").strip() + + def _format_label(name: str, books_count: Any) -> str: + try: + return f"{name} ({int(books_count)})" + except TypeError, ValueError: + return name + + for status in HARDCOVER_STATUSES: + count_data = me_data.get(status["query_key"], {}) + aggregate = count_data.get("aggregate", {}) if isinstance(count_data, dict) else {} + count = aggregate.get("count") if isinstance(aggregate, dict) else None + value = f"{HARDCOVER_STATUS_PREFIX}{status['id']}" + seen_values.add(value) + options.append( + { + "value": value, + "label": _format_label(status["label"], count), + "group": HARDCOVER_STATUS_GROUP, + } + ) + + for list_item in me_data.get("lists", []): + if not isinstance(list_item, dict): + continue + list_id = list_item.get("id") + slug = str(list_item.get("slug") or "").strip() + name = str(list_item.get("name") or "").strip() + value = f"id:{list_id}" if list_id is not None else slug + if not value or not name or value in seen_values: + continue + seen_values.add(value) + options.append( + { + "value": value, + "label": _format_label(name, list_item.get("books_count")), + "group": "My Lists", + } + ) + + for followed_item in me_data.get("followed_lists", []): + if not isinstance(followed_item, dict): + continue + + list_item = followed_item.get("list", {}) + if not isinstance(list_item, dict): + continue + + list_id = list_item.get("id") + slug = str(list_item.get("slug") or "").strip() + name = str(list_item.get("name") or "").strip() + value = f"id:{list_id}" if list_id is not None else slug + if not value or not name or value in seen_values: + continue + seen_values.add(value) + + option: dict[str, str] = { + "value": value, + "label": _format_label(name, list_item.get("books_count")), + "group": "Followed Lists", + } + owner_data = list_item.get("user", {}) + if isinstance(owner_data, dict): + owner_username = str(owner_data.get("username") or "").strip() + if owner_username: + option["description"] = f"by @{owner_username}" + elif current_username: + option["description"] = f"by @{current_username}" + options.append(option) + + return options diff --git a/shelfmark/metadata_providers/hardcover/models.py b/shelfmark/metadata_providers/hardcover/models.py new file mode 100644 index 0000000..c8ece65 --- /dev/null +++ b/shelfmark/metadata_providers/hardcover/models.py @@ -0,0 +1,20 @@ +"""Small Hardcover-specific models and errors.""" + +from dataclasses import dataclass + + +@dataclass(frozen=True) +class HardcoverBookTargetState: + """Current Hardcover target state for a specific book.""" + + user_book_id: int | None + status_id: int | None + list_book_ids: dict[int, int] + + +class HardcoverGraphQLError(ValueError): + """GraphQL request was rejected by Hardcover.""" + + +class HardcoverTargetPayloadError(RuntimeError): + """Hardcover returned an invalid payload while loading book targets.""" diff --git a/shelfmark/metadata_providers/hardcover/parsing.py b/shelfmark/metadata_providers/hardcover/parsing.py new file mode 100644 index 0000000..321ee44 --- /dev/null +++ b/shelfmark/metadata_providers/hardcover/parsing.py @@ -0,0 +1,611 @@ +"""Parsing and search-normalization helpers for Hardcover payloads.""" + +import re +from contextlib import suppress +from datetime import datetime +from typing import Any + +from shelfmark.core.logger import setup_logger +from shelfmark.core.request_helpers import normalize_optional_text +from shelfmark.metadata_providers import BookMetadata, DisplayField + +from .constants import HARDCOVER_MIN_AUTHOR_PARTS + +logger = setup_logger(__name__) + + +def _combine_headline_description(headline: str | None, description: str | None) -> str | None: + """Combine headline (tagline) and description into a single description.""" + if headline and description: + return f"{headline}\n\n{description}" + return headline or description + + +def _extract_cover_url(data: dict, *keys: str) -> str | None: + """Extract cover URL from data dict, trying multiple keys. + + Handles both string URLs and dict with 'url' key. + """ + for key in keys: + value = data.get(key) + if value: + if isinstance(value, str): + return value + if isinstance(value, dict): + return value.get("url") + return None + + +def _extract_publish_year(data: dict) -> int | None: + """Extract publish year from release_year or release_date fields.""" + if data.get("release_year"): + try: + return int(data["release_year"]) + except ValueError, TypeError: + pass + if data.get("release_date"): + try: + return int(str(data["release_date"])[:4]) + except ValueError, TypeError: + pass + return None + + +def _parse_release_date(value: Any) -> datetime | None: + """Parse Hardcover release dates stored as YYYY-MM-DD strings.""" + if not value: + return None + + normalized_value = str(value).strip() + if not normalized_value: + return None + + try: + return datetime.fromisoformat(normalized_value[:10]) + except ValueError: + return None + + +def _normalize_series_position(value: Any) -> float | None: + """Normalize a series position to a float for sorting and grouping.""" + if value is None: + return None + + try: + return float(value) + except TypeError, ValueError: + return None + + +def _normalize_hardcover_api_key(value: object) -> str: + """Normalize Hardcover API keys, stripping copied auth-header prefixes.""" + normalized_value = normalize_optional_text(value) or "" + return normalized_value.removeprefix("Bearer ").strip() + + +def _normalize_search_text(value: str) -> str: + """Normalize free-text search input for matching and caching.""" + return " ".join(value.split()).strip() + + +def _unwrap_hit_document(hit: Any) -> dict[str, Any] | None: + """Extract the document dict from a Typesense hit, or return None.""" + if not isinstance(hit, dict): + return None + item = hit.get("document", hit) + return item if isinstance(item, dict) else None + + +def _search_tokens(value: str) -> list[str]: + """Tokenize search text for lightweight prefix matching.""" + return re.findall(r"[a-z0-9']+", value.casefold()) + + +def _query_matches_author_name(query: str, author_name: str) -> bool: + """Return True when the query looks like an author-name search.""" + normalized_query = _normalize_search_text(query) + normalized_author_name = _normalize_search_text(author_name) + if not normalized_query or not normalized_author_name: + return False + + query_folded = normalized_query.casefold() + author_folded = normalized_author_name.casefold() + if query_folded in author_folded: + return True + + query_tokens = _search_tokens(normalized_query) + author_tokens = _search_tokens(normalized_author_name) + if not query_tokens or not author_tokens: + return False + + return all( + any(author_token.startswith(query_token) for author_token in author_tokens) + for query_token in query_tokens + ) + + +def _split_part_base_title(title: str) -> str | None: + """Extract the base title from segmented part releases like ', Part 2'.""" + normalized_title = _normalize_search_text(title) + if not normalized_title: + return None + + match = re.match(r"^(?P.+?),\s*Part\s+\d+$", normalized_title, re.IGNORECASE) + if not match: + return None + + base_title = str(match.group("base") or "").strip() + return base_title or None + + +def _series_allows_split_parts(series_name: str) -> bool: + """Return True for series that intentionally organize split-part releases.""" + normalized_name = _normalize_search_text(series_name).casefold() + if not normalized_name: + return False + + markers = ( + "dramatized adaptation", + "graphicaudio", + "graphic audio", + "(3 parts)", + "(2 parts)", + "(4 parts)", + ) + return any(marker in normalized_name for marker in markers) + + +def _extract_typesense_hits(result: dict[str, Any]) -> tuple[list[dict[str, Any]], int]: + """Extract hit documents + total count from Hardcover search output.""" + root = result.get("search", result) if isinstance(result, dict) else {} + results_obj = root.get("results", {}) if isinstance(root, dict) else {} + if isinstance(results_obj, dict): + hits = results_obj.get("hits", []) + found_count = results_obj.get("found", 0) + else: + hits = results_obj if isinstance(results_obj, list) else [] + found_count = 0 + return hits, found_count + + +def _build_source_url(slug: str) -> str | None: + """Build Hardcover source URL from book slug.""" + return f"https://hardcover.app/books/{slug}" if slug else None + + +def _is_probably_series_position(subtitle: str) -> bool: + normalized = subtitle.strip().lower() + + # Common patterns: "Book One", "Book 1", "Part 2", "Volume III", etc. + if re.match( + r"^(book|part|volume|vol\.?|episode)\s+([0-9]+|[ivxlcdm]+|one|two|three|four|five|six|seven|eight|nine|ten)\b", + normalized, + ): + return True + + # e.g. "A Novel", "An Epic Fantasy", etc. These add noise to indexer queries. + if normalized in {"a novel", "a novella", "a story", "a memoir"}: + return True + + # Descriptive subtitles like "A [Name] Novel", "An [Name] Mystery", etc. + genre_words = ( + "novel", + "novella", + "story", + "memoir", + "tale", + "thriller", + "mystery", + "romance", + "adventure", + "epic", + "saga", + "chronicle", + "fantasy", + "novel-in-stories", + ) + genre_pattern = "|".join(re.escape(w) for w in genre_words) + return bool(re.match(rf"^an?\s+.+\s+({genre_pattern})$", normalized)) + + +def _strip_parenthetical_suffix(title: str) -> str: + # Drop trailing qualifiers like "(Unabridged)", "(Illustrated Edition)", etc. + return re.sub(r"\s*\([^)]*\)\s*$", "", title).strip() + + +def _simplify_author_for_search(author: str) -> str | None: + """Return a looser author string for indexer searches. + + Primary goal: reduce mismatch between metadata providers and indexers. + Indexers store author names inconsistently ("R.A.", "R. A.", "Salvatore, R.A.") + so initials add noise and hurt recall. + + Heuristics: + - Strip all initials (single or compound), keeping only full names + e.g. "R. A. Salvatore" -> "Salvatore", "George R.R. Martin" -> "George Martin" + - Preserve suffixes like "Jr."/"Sr."/"III" as they sometimes matter + """ + if not author: + return None + + normalized = " ".join(author.split()).strip() + if not normalized: + return None + + # Handle "Last, First ..." -> "First ... Last" + if "," in normalized: + parts = [p.strip() for p in normalized.split(",") if p.strip()] + if len(parts) >= HARDCOVER_MIN_AUTHOR_PARTS: + normalized = " ".join([*parts[1:], parts[0]]).strip() + + tokens = normalized.split(" ") + if len(tokens) < HARDCOVER_MIN_AUTHOR_PARTS: + return None + + keep_suffixes = {"jr", "jr.", "sr", "sr.", "ii", "iii", "iv", "v"} + + simplified: list[str] = [] + for idx, token in enumerate(tokens): + t = token.strip() + if not t: + continue + + t_lower = t.lower() + is_suffix = (idx == len(tokens) - 1) and (t_lower in keep_suffixes) + if is_suffix: + simplified.append(t) + continue + + # Drop all initials: "R.", "R", "R.R.", "J.K.", etc. + if re.match(r"^[A-Za-z]$|^([A-Za-z]\.)+[A-Za-z]?$", t): + continue + + simplified.append(t) + + if not simplified: + return None + + candidate = " ".join(simplified).strip() + if candidate.lower() == normalized.lower(): + return None + + return candidate + + +def _compute_search_title( + title: str, + subtitle: str | None, + *, + series_name: str | None = None, +) -> str | None: + """Compute a provider-specific, *looser* title for indexer searching. + + Goal: produce a string that maximizes recall in downstream sources (Prowlarr, + IRC bots, etc.). Being too detailed is counterproductive. + + Hardcover often stores titles in a "Series: Book Title" format and places the + standalone book title in `subtitle`. When this appears to be the case, prefer + the subtitle (unless it looks like a series position or other noise). + + Additional heuristics: + - If Hardcover prefixes the series in the title, remove it. + - Drop trailing parenthetical qualifiers. + """ + if not title: + return None + + original_title = " ".join(title.split()).strip() + + normalized_title = _strip_parenthetical_suffix(original_title) + + normalized_subtitle = " ".join(subtitle.split()).strip() if subtitle else "" + normalized_subtitle = ( + _strip_parenthetical_suffix(normalized_subtitle) if normalized_subtitle else "" + ) + + if normalized_subtitle and normalized_subtitle.lower() == normalized_title.lower(): + normalized_subtitle = "" + + # If subtitle is noise, strip it from the title and use just the prefix. + if normalized_subtitle and _is_probably_series_position(normalized_subtitle): + match = re.match(r"^(.+?)\s*:\s*(.+)$", normalized_title) + if match: + suffix = _strip_parenthetical_suffix(match.group(2).strip()) + if ( + normalized_subtitle.lower() == suffix.lower() + or normalized_subtitle.lower() in suffix.lower() + ): + return None + + # Prefer subtitle when it looks like the real title. + if normalized_subtitle and not _is_probably_series_position(normalized_subtitle): + match = re.match(r"^(.+?)\s*:\s*(.+)$", normalized_title) + if match: + prefix = match.group(1).strip() + suffix = _strip_parenthetical_suffix(match.group(2).strip()) + + prefix_words = len(prefix.split()) if prefix else 0 + subtitle_words = len(normalized_subtitle.split()) + + series_normalized = " ".join(series_name.split()).strip() if series_name else "" + if series_normalized and prefix.lower() == series_normalized.lower(): + return normalized_subtitle + + # If the subtitle is much longer than the prefix, treat it as a descriptive subtitle. + if prefix and subtitle_words >= (prefix_words + 4): + return prefix + + # Otherwise assume "Series: Book Title" and prefer the subtitle. + if ( + normalized_subtitle.lower() == suffix.lower() + or normalized_subtitle.lower() in suffix.lower() + ): + return normalized_subtitle + + # Fallback: if title contains the subtitle, this is likely "Series: Subtitle". + if normalized_subtitle.lower() in normalized_title.lower(): + return normalized_subtitle + + # If we know the series name (from full book fetch), strip it. + if series_name: + series_normalized = " ".join(series_name.split()).strip() + if series_normalized: + # Common Hardcover format: "Series: Book Title". + prefix = f"{series_normalized}:" + if normalized_title.lower().startswith(prefix.lower()): + candidate = normalized_title[len(prefix) :].strip() + candidate = _strip_parenthetical_suffix(candidate) + if candidate and candidate.lower() != normalized_title.lower(): + return candidate + + # Last resort: return a cleaned version of the title if we removed noise. + if normalized_title and normalized_title.lower() != original_title.lower(): + return normalized_title + + return None + + +class HardcoverParsingMixin: + def _parse_search_result(self, item: dict) -> BookMetadata | None: + """Parse a search result item into BookMetadata.""" + try: + book_id = item.get("id") or item.get("document", {}).get("id") + title = item.get("title") or item.get("document", {}).get("title") + + if not book_id or not title: + return None + + # Extract authors - use contribution_types to filter author_names if available + authors = [] + + author_names = item.get("author_names", []) + if isinstance(author_names, str): + author_names = [author_names] + + contribution_types = item.get("contribution_types", []) + + # If we have parallel arrays, filter to only "Author" contributions + if contribution_types and len(contribution_types) == len(author_names): + for name, contrib_type in zip(author_names, contribution_types, strict=True): + if contrib_type == "Author": + authors.append(name) + elif author_names: + # No contribution_types or length mismatch - use all names as fallback + authors = author_names + + # Normalize whitespace in author names (some API data has multiple spaces) + authors = [" ".join(name.split()) for name in authors] + + search_author = _simplify_author_for_search(authors[0]) if authors else None + + cover_url = _extract_cover_url(item, "image") + publish_year = _extract_publish_year(item) + source_url = _build_source_url(item.get("slug", "")) + + # Build display fields from Hardcover-specific data + display_fields = [] + + # Rating (e.g., "4.5 (3,764)") + rating = item.get("rating") + ratings_count = item.get("ratings_count") + if rating is not None: + rating_str = f"{rating:.1f}" + if ratings_count: + rating_str += f" ({ratings_count:,})" + display_fields.append(DisplayField(label="Rating", value=rating_str, icon="star")) + + # Readers (users who have this book) + users_count = item.get("users_count") + if users_count: + display_fields.append( + DisplayField(label="Readers", value=f"{users_count:,}", icon="users") + ) + + # Combine headline and description if both present + headline = item.get("headline") + description = item.get("description") + full_description = _combine_headline_description(headline, description) + + # Extract subtitle if available in search results + subtitle = item.get("subtitle") + + return BookMetadata( + provider="hardcover", + provider_id=str(book_id), + title=title, + subtitle=subtitle, + search_title=_compute_search_title(title, subtitle), + search_author=search_author, + provider_display_name="Hardcover", + authors=authors, + cover_url=cover_url, + description=full_description, + publish_year=publish_year, + source_url=source_url, + display_fields=display_fields, + ) + + except (AttributeError, KeyError, TypeError, ValueError) as e: + logger.debug("Failed to parse Hardcover search result: %s", e) + return None + + def _parse_book(self, book: dict) -> BookMetadata: + """Parse a book object into BookMetadata.""" + title = str(book.get("title") or "") + subtitle = book.get("subtitle") + + # Extract authors - try contributions first (filtered), fall back to cached_contributors + authors = [] + contributions = book.get("contributions") or [] + cached_contributors = book.get("cached_contributors") or [] + + # Try contributions first (filtered to "Author" role only - cleaner data) + for contrib in contributions: + author = contrib.get("author", {}) + if author and author.get("name"): + authors.append(author["name"]) + + # Fallback to cached_contributors if no authors found + if not authors: + for contrib in cached_contributors: + if isinstance(contrib, dict): + # Handle nested structure: {"author": {"name": "..."}, "contribution": ...} + if contrib.get("author", {}).get("name"): + authors.append(contrib["author"]["name"]) + # Handle flat structure: {"name": "..."} + elif contrib.get("name"): + authors.append(contrib["name"]) + elif isinstance(contrib, str): + authors.append(contrib) + + # Normalize whitespace in author names (some API data has multiple spaces) + authors = [" ".join(name.split()) for name in authors] + + search_author = _simplify_author_for_search(authors[0]) if authors else None + + cover_url = _extract_cover_url(book, "cached_image", "image") + publish_year = _extract_publish_year(book) + + # Extract genres from cached_tags + genres = [] + for tag in book.get("cached_tags", []): + if isinstance(tag, dict) and tag.get("tag"): + genres.append(tag["tag"]) + elif isinstance(tag, str): + genres.append(tag) + + # Get ISBN from direct fields, default_physical_edition, or editions + isbn_10 = book.get("isbn_10") + isbn_13 = book.get("isbn_13") + + if not isbn_10 and not isbn_13: + # Try default_physical_edition first + edition = book.get("default_physical_edition") + if edition: + isbn_10 = edition.get("isbn_10") + isbn_13 = edition.get("isbn_13") + + # Fallback to editions array + if not isbn_10 and not isbn_13 and book.get("editions"): + for ed in book["editions"]: + if not isbn_10 and ed.get("isbn_10"): + isbn_10 = ed["isbn_10"] + if not isbn_13 and ed.get("isbn_13"): + isbn_13 = ed["isbn_13"] + if isbn_10 and isbn_13: + break + + source_url = _build_source_url(book.get("slug", "")) + + # Combine headline and description if both present + headline = book.get("headline") + description = book.get("description") + full_description = _combine_headline_description(headline, description) + + # Extract series info from featured_book_series + series_id = None + series_name = None + series_position = None + series_count = None + featured_series = book.get("featured_book_series") + if featured_series: + series_position = featured_series.get("position") + series_data = featured_series.get("series") + if series_data: + if series_data.get("id") is not None: + series_id = str(series_data.get("id")) + series_name = series_data.get("name") + series_count = series_data.get("primary_books_count") + + # Extract titles by language from editions + # This allows searching with localized titles when language filter is active + titles_by_language: dict[str, str] = {} + editions = book.get("editions", []) + for edition in editions: + edition_title = edition.get("title") + lang_data = edition.get("language") + if edition_title and lang_data: + # Store by various language identifiers for flexible matching + # Language name (e.g., "German", "English") + lang_name = lang_data.get("language") + # 2-letter code (e.g., "de", "en") + code2 = lang_data.get("code2") + # 3-letter code (e.g., "deu", "eng") + code3 = lang_data.get("code3") + + # Store with all available keys (first title wins for each language) + if lang_name and lang_name not in titles_by_language: + titles_by_language[lang_name] = edition_title + if code2 and code2 not in titles_by_language: + titles_by_language[code2] = edition_title + if code3 and code3 not in titles_by_language: + titles_by_language[code3] = edition_title + + # Build display fields from Hardcover-specific metrics + display_fields: list[DisplayField] = [] + + rating = book.get("rating") + ratings_count = book.get("ratings_count") + if rating is not None: + try: + rating_str = f"{float(rating):.1f}" + except TypeError, ValueError: + rating_str = str(rating) + + if ratings_count: + with suppress(TypeError, ValueError): + rating_str += f" ({int(ratings_count):,})" + + display_fields.append(DisplayField(label="Rating", value=rating_str, icon="star")) + + users_count = book.get("users_count") + if users_count: + try: + readers_value = f"{int(users_count):,}" + except TypeError, ValueError: + readers_value = str(users_count) + display_fields.append(DisplayField(label="Readers", value=readers_value, icon="users")) + + return BookMetadata( + provider="hardcover", + provider_id=str(book["id"]), + title=title, + subtitle=subtitle, + search_title=_compute_search_title(title, subtitle, series_name=series_name), + search_author=search_author, + provider_display_name="Hardcover", + authors=authors, + isbn_10=isbn_10, + isbn_13=isbn_13, + cover_url=cover_url, + description=full_description, + publish_year=publish_year, + genres=genres, + source_url=source_url, + series_id=series_id, + series_name=series_name, + series_position=series_position, + series_count=series_count, + titles_by_language=titles_by_language, + display_fields=display_fields, + ) diff --git a/shelfmark/metadata_providers/hardcover/provider.py b/shelfmark/metadata_providers/hardcover/provider.py new file mode 100644 index 0000000..445e267 --- /dev/null +++ b/shelfmark/metadata_providers/hardcover/provider.py @@ -0,0 +1,106 @@ +"""Hardcover.app metadata provider. Requires API key.""" + +from typing import Any, ClassVar + +import requests + +from shelfmark.core.config import config as app_config +from shelfmark.metadata_providers import ( + DynamicSelectSearchField, + MetadataCapability, + MetadataProvider, + SearchField, + SortOrder, + TextSearchField, + register_provider, + register_provider_kwargs, +) + +from .client import HardcoverClientMixin +from .lists import HardcoverListsMixin +from .parsing import HardcoverParsingMixin, _normalize_hardcover_api_key +from .search import HardcoverSearchMixin +from .targets import HardcoverTargetsMixin + + +@register_provider_kwargs("hardcover") +def _hardcover_kwargs() -> dict[str, Any]: + """Provide Hardcover-specific constructor kwargs.""" + return {"api_key": app_config.get("HARDCOVER_API_KEY", "")} + + +@register_provider("hardcover") +class HardcoverProvider( + HardcoverSearchMixin, + HardcoverListsMixin, + HardcoverTargetsMixin, + HardcoverClientMixin, + HardcoverParsingMixin, + MetadataProvider, +): + """Hardcover.app metadata provider using GraphQL API.""" + + name = "hardcover" + display_name = "Hardcover" + requires_auth = True + supported_sorts: ClassVar[tuple[SortOrder, ...]] = ( + SortOrder.RELEVANCE, + SortOrder.POPULARITY, + SortOrder.RATING, + SortOrder.NEWEST, + SortOrder.OLDEST, + SortOrder.SERIES_ORDER, + ) + capabilities: ClassVar[tuple[MetadataCapability, ...]] = ( + MetadataCapability( + key="view_series", + field_key="series", + sort=SortOrder.SERIES_ORDER, + ), + ) + search_fields: ClassVar[tuple[SearchField, ...]] = ( + TextSearchField( + key="author", + label="Author", + placeholder="Search author...", + description="Search by author name", + suggestions_endpoint="/api/metadata/field-options?provider=hardcover&field=author", + ), + TextSearchField( + key="title", + label="Title", + placeholder="Search title...", + description="Search by book title", + ), + TextSearchField( + key="series", + label="Series", + placeholder="Search series...", + description="Search by series name", + suggestions_endpoint="/api/metadata/field-options?provider=hardcover&field=series", + ), + DynamicSelectSearchField( + key="hardcover_list", + label="List", + options_endpoint="/api/metadata/field-options?provider=hardcover&field=hardcover_list", + placeholder="Browse a list...", + description="Browse books from a Hardcover list", + ), + ) + + def __init__(self, api_key: str | None = None) -> None: + """Initialize provider with optional API key (falls back to config).""" + raw_key = api_key or app_config.get("HARDCOVER_API_KEY", "") + self.api_key = _normalize_hardcover_api_key(raw_key) + self.session = requests.Session() + if self.api_key: + self.session.headers.update( + { + "Authorization": f"Bearer {self.api_key}", + "Content-Type": "application/json", + } + ) + + def is_available(self) -> bool: + """Check if provider is configured with an API key.""" + return bool(self.api_key) diff --git a/shelfmark/metadata_providers/hardcover/queries.py b/shelfmark/metadata_providers/hardcover/queries.py new file mode 100644 index 0000000..a0cc4af --- /dev/null +++ b/shelfmark/metadata_providers/hardcover/queries.py @@ -0,0 +1,525 @@ +"""GraphQL operations used by the Hardcover metadata provider.""" + +LIST_LOOKUP_QUERY = """ +query LookupListsBySlug($slug: String!) { + lists(where: {slug: {_eq: $slug}}, limit: 20) { + id + slug + user { + username + } + } +} +""" + +LIST_BOOKS_BY_ID_QUERY = """ +query GetListBooksById($id: Int!, $limit: Int!, $offset: Int!) { + lists(where: {id: {_eq: $id}}, limit: 1) { + name + slug + user { + username + } + books_count + list_books(order_by: {position: asc}, limit: $limit, offset: $offset) { + book { + id + title + subtitle + slug + release_date + headline + description + pages + rating + ratings_count + users_count + cached_image + cached_contributors + contributions(where: {contribution: {_eq: "Author"}}) { + author { + name + } + } + featured_book_series { + position + series { + id + name + primary_books_count + } + } + } + } + } +} +""" + +USER_LISTS_QUERY = """ +query GetUserLists { + me { + id + username + want_to_read_count: user_books_aggregate(where: {status_id: {_eq: 1}}) { + aggregate { + count(columns: [book_id], distinct: true) + } + } + currently_reading_count: user_books_aggregate(where: {status_id: {_eq: 2}}) { + aggregate { + count(columns: [book_id], distinct: true) + } + } + read_count: user_books_aggregate(where: {status_id: {_eq: 3}}) { + aggregate { + count(columns: [book_id], distinct: true) + } + } + did_not_finish_count: user_books_aggregate(where: {status_id: {_eq: 5}}) { + aggregate { + count(columns: [book_id], distinct: true) + } + } + lists(order_by: {name: asc}) { + id + name + slug + books_count + } + followed_lists(order_by: {created_at: desc}) { + list { + id + name + slug + books_count + user { + username + } + } + } + } +} +""" + +USER_BOOKS_BY_STATUS_QUERY = """ +query GetCurrentUserBooksByStatus($statusId: Int!, $limit: Int!, $offset: Int!) { + me { + status_books: user_books( + where: {status_id: {_eq: $statusId}} + distinct_on: [book_id] + order_by: [{book_id: asc}, {created_at: desc}] + limit: $limit + offset: $offset + ) { + book { + id + title + subtitle + slug + release_date + headline + description + pages + rating + ratings_count + users_count + cached_image + cached_contributors + contributions(where: {contribution: {_eq: "Author"}}) { + author { + name + } + } + featured_book_series { + position + series { + id + name + primary_books_count + } + } + } + } + status_books_aggregate: user_books_aggregate(where: {status_id: {_eq: $statusId}}) { + aggregate { + count(columns: [book_id], distinct: true) + } + } + } +} +""" + +BOOK_TARGET_MEMBERSHIP_QUERY = """ +query GetBookTargetMembership($bookId: Int!) { + me { + user_books(where: {book_id: {_eq: $bookId}}, limit: 1, order_by: [{created_at: desc}]) { + id + status_id + } + lists { + id + list_books(where: {book_id: {_eq: $bookId}}, limit: 1) { + id + } + } + } +} +""" + +BOOK_TARGET_MEMBERSHIP_BATCH_QUERY = """ +query GetBookTargetMembershipBatch($bookIds: [Int!]!) { + me { + user_books(where: {book_id: {_in: $bookIds}}, order_by: [{created_at: desc}]) { + id + book_id + status_id + } + lists { + id + list_books(where: {book_id: {_in: $bookIds}}) { + id + book_id + } + } + } +} +""" + +INSERT_USER_BOOK_MUTATION = """ +mutation AddBookToStatus($bookId: Int!, $statusId: Int!) { + insert_user_book(object: {book_id: $bookId, status_id: $statusId}) { + id + error + user_book { + id + book_id + status_id + } + } +} +""" + +UPDATE_USER_BOOK_MUTATION = """ +mutation UpdateBookStatus($userBookId: Int!, $statusId: Int!) { + update_user_book(id: $userBookId, object: {status_id: $statusId}) { + id + error + user_book { + id + book_id + status_id + } + } +} +""" + +DELETE_USER_BOOK_MUTATION = """ +mutation RemoveBookStatus($userBookId: Int!) { + delete_user_book(id: $userBookId) { + id + book_id + user_id + } +} +""" + +INSERT_LIST_BOOK_MUTATION = """ +mutation AddBookToList($bookId: Int!, $listId: Int!) { + insert_list_book(object: {book_id: $bookId, list_id: $listId}) { + id + list_book { + id + book_id + list_id + } + } +} +""" + +DELETE_LIST_BOOK_MUTATION = """ +mutation RemoveBookFromList($listBookId: Int!) { + delete_list_book(id: $listBookId) { + id + list_id + } +} +""" + +SEARCH_FIELD_OPTIONS_QUERY = """ +query SearchFieldOptions( + $query: String!, + $queryType: String!, + $limit: Int!, + $page: Int!, + $sort: String, + $fields: String, + $weights: String +) { + search( + query: $query, + query_type: $queryType, + per_page: $limit, + page: $page, + sort: $sort, + fields: $fields, + weights: $weights + ) { + results + } +} +""" + +SERIES_BY_AUTHOR_IDS_QUERY = """ +query SeriesByAuthorIds($authorIds: [Int!], $limit: Int!) { + series( + where: { + author_id: {_in: $authorIds}, + canonical_id: {_is_null: true}, + state: {_eq: "active"} + }, + limit: $limit, + order_by: [{primary_books_count: desc_nulls_last}, {books_count: desc}, {name: asc}] + ) { + id + name + primary_books_count + books_count + author { + name + } + } +} +""" + +SERIES_BOOKS_BY_ID_QUERY = """ +query GetSeriesBooks($seriesId: Int!) { + series(where: {id: {_eq: $seriesId}}, limit: 1) { + id + name + primary_books_count + book_series( + where: { + book: { + canonical_id: {_is_null: true}, + state: {_in: ["normalized", "normalizing"]} + } + } + order_by: [{position: asc_nulls_last}, {book_id: asc}] + ) { + position + book { + id + title + subtitle + slug + release_date + headline + description + pages + rating + ratings_count + users_count + compilation + editions_count + cached_image + cached_contributors + contributions(where: {contribution: {_eq: "Author"}}) { + author { + name + } + } + featured_book_series { + position + series { + id + name + primary_books_count + } + } + } + } + } +} +""" + +AUTHOR_BOOKS_BY_ID_QUERY = """ +query GetAuthorBooks($authorId: Int!, $limit: Int!, $offset: Int!) { + authors(where: {id: {_eq: $authorId}}, limit: 1) { + name + contributions( + where: { + contributable_type: {_eq: "Book"}, + book: { + canonical_id: {_is_null: true}, + state: {_in: ["normalized", "normalizing"]} + } + }, + order_by: [ + {book: {users_count: desc_nulls_last}}, + {book: {ratings_count: desc_nulls_last}}, + {book: {release_date: asc_nulls_last}}, + {book: {id: asc}} + ], + limit: $limit, + offset: $offset + ) { + contribution + book { + id + title + subtitle + slug + release_date + headline + description + pages + rating + ratings_count + users_count + compilation + editions_count + cached_image + cached_contributors + contributions(where: {contribution: {_eq: "Author"}}) { + author { + name + } + } + featured_book_series { + position + series { + id + name + primary_books_count + } + } + } + } + contributions_aggregate( + where: { + contributable_type: {_eq: "Book"}, + book: { + canonical_id: {_is_null: true}, + state: {_in: ["normalized", "normalizing"]} + } + } + ) { + aggregate { + count + } + } + } +} +""" + +SEARCH_BOOKS_WITH_FIELDS_QUERY = """ +query SearchBooks( + $query: String!, + $limit: Int!, + $page: Int!, + $sort: String, + $fields: String, + $weights: String +) { + search( + query: $query, + query_type: "Book", + per_page: $limit, + page: $page, + sort: $sort, + fields: $fields, + weights: $weights + ) { + results + } +} +""" + +SEARCH_BOOKS_QUERY = """ +query SearchBooks($query: String!, $limit: Int!, $page: Int!, $sort: String) { + search(query: $query, query_type: "Book", per_page: $limit, page: $page, sort: $sort) { + results + } +} +""" + +GET_BOOK_QUERY = """ +query GetBook($id: Int!) { + books(where: {id: {_eq: $id}}, limit: 1) { + id + title + subtitle + slug + release_date + headline + description + pages + cached_image + cached_tags + cached_contributors + contributions(where: {contribution: {_eq: "Author"}}) { + author { + name + } + } + default_physical_edition { + isbn_10 + isbn_13 + } + featured_book_series { + position + series { + id + name + primary_books_count + } + } + editions( + distinct_on: language_id + order_by: [{language_id: asc}, {users_count: desc}] + limit: 200 + ) { + title + language { + language + code2 + code3 + } + } + } +} +""" + +SEARCH_BY_ISBN_QUERY = """ +query SearchByISBN($isbn: String!) { + editions( + where: { + _or: [ + {isbn_10: {_eq: $isbn}}, + {isbn_13: {_eq: $isbn}} + ] + }, + limit: 1 + ) { + isbn_10 + isbn_13 + book { + id + title + subtitle + slug + release_date + headline + description + pages + cached_image + cached_tags + contributions(where: {contribution: {_eq: "Author"}}) { + author { + name + } + } + } + } +} +""" diff --git a/shelfmark/metadata_providers/hardcover/search.py b/shelfmark/metadata_providers/hardcover/search.py new file mode 100644 index 0000000..bd81447 --- /dev/null +++ b/shelfmark/metadata_providers/hardcover/search.py @@ -0,0 +1,844 @@ +"""Search, typeahead, series, and book lookup workflows for Hardcover.""" + +from datetime import UTC, datetime +from typing import TYPE_CHECKING, Any + +from shelfmark.core.cache import cacheable +from shelfmark.core.config import config as app_config +from shelfmark.core.logger import setup_logger +from shelfmark.core.request_helpers import coerce_bool, coerce_int +from shelfmark.metadata_providers import ( + BookMetadata, + MetadataSearchOptions, + SearchResult, + SearchType, + SortOrder, +) + +from .constants import ( + AUTHOR_SUGGESTION_FIELDS, + AUTHOR_SUGGESTION_SORT, + AUTHOR_SUGGESTION_WEIGHTS, + HARDCOVER_LIST_ID_PREFIX, + HARDCOVER_MAX_SERIES_OPTIONS, + HARDCOVER_MIN_TYPEAHEAD_QUERY_LENGTH, + HARDCOVER_PAGE_SIZE, + HARDCOVER_STATUS_PREFIX, + SERIES_SEARCH_FIELDS, + SERIES_SEARCH_SORT, + SERIES_SEARCH_WEIGHTS, + SORT_MAPPING, + TITLE_SUGGESTION_FIELDS, + TITLE_SUGGESTION_SORT, + TITLE_SUGGESTION_WEIGHTS, +) +from .parsing import ( + _extract_typesense_hits, + _normalize_search_text, + _normalize_series_position, + _parse_release_date, + _query_matches_author_name, + _series_allows_split_parts, + _split_part_base_title, + _unwrap_hit_document, +) +from .queries import ( + AUTHOR_BOOKS_BY_ID_QUERY, + GET_BOOK_QUERY, + SEARCH_BOOKS_QUERY, + SEARCH_BOOKS_WITH_FIELDS_QUERY, + SEARCH_BY_ISBN_QUERY, + SEARCH_FIELD_OPTIONS_QUERY, + SERIES_BOOKS_BY_ID_QUERY, + SERIES_BY_AUTHOR_IDS_QUERY, +) + +logger = setup_logger(__name__) + + +class HardcoverSearchMixin: + if TYPE_CHECKING: + api_key: str + + def _detect_list_url(self, query: str) -> tuple[str | None, str] | None: ... + + def _execute_query( + self, + query: str, + variables: dict[str, Any], + *, + raise_on_error: bool = False, + ) -> dict[str, Any] | None: ... + + def _fetch_current_user_books_by_status( + self, status_id: int, page: int, limit: int + ) -> SearchResult: ... + + def _fetch_list_books( + self, slug: str, owner_username: str | None, page: int, limit: int + ) -> SearchResult: ... + + def _fetch_list_books_by_id(self, list_id: int, page: int, limit: int) -> SearchResult: ... + + def _parse_book(self, book: dict[str, Any]) -> BookMetadata: ... + + @staticmethod + def _parse_prefixed_int(value: str, label: str = "target") -> int: ... + + def _parse_search_result(self, item: dict[str, Any]) -> BookMetadata | None: ... + + def get_user_lists(self) -> list[dict[str, str]]: ... + + def _build_search_params( + self, default_query: str, author: str, title: str, series: str + ) -> tuple[str, str | None, str | None]: + """Build search query, fields, and weights based on provided values. + + Returns (query, fields, weights) tuple. Fields/weights are None for general search. + """ + if author and not title and not series: + return author, None, None + if title and not author and not series: + return title, "title,alternative_titles", "5,1" + if author and title and not series: + return f"{title} {author}", "title,alternative_titles,author_names", "5,1,3" + return default_query, None, None + + def get_search_field_options( + self, + field_key: str, + query: str | None = None, + ) -> list[dict[str, str]]: + """Provide dynamic options for Hardcover-specific advanced fields.""" + if field_key == "author": + return self._search_author_options(query or "") + if field_key == "title": + return self._search_title_options(query or "") + if field_key == "series": + return self._search_series_options(query or "") + if field_key == "hardcover_list": + return self.get_user_lists() + return [] + + def _search_field_hits( + self, + *, + query: str, + query_type: str, + limit: int, + sort: str | None, + fields: str | None, + weights: str | None, + ) -> list[dict[str, Any]]: + """Run a Hardcover search request for field-level typeahead options.""" + normalized_query = _normalize_search_text(query) + if not self.api_key or len(normalized_query) < HARDCOVER_MIN_TYPEAHEAD_QUERY_LENGTH: + return [] + + result = self._execute_query( + SEARCH_FIELD_OPTIONS_QUERY, + { + "query": normalized_query, + "queryType": query_type, + "limit": limit, + "page": 1, + "sort": sort, + "fields": fields, + "weights": weights, + }, + ) + if not result: + return [] + + hits, _found_count = _extract_typesense_hits(result) + return hits + + def _search_series_by_matching_author(self, query: str) -> list[dict[str, Any]]: + """Return direct series rows when the query clearly matches an author.""" + author_hits = self._search_field_hits( + query=query, + query_type="Author", + limit=2, + sort=AUTHOR_SUGGESTION_SORT, + fields=AUTHOR_SUGGESTION_FIELDS, + weights=AUTHOR_SUGGESTION_WEIGHTS, + ) + + author_ids: list[int] = [] + for hit in author_hits: + item = _unwrap_hit_document(hit) + if item is None: + continue + + author_name = str(item.get("name") or "").strip() + if not _query_matches_author_name(query, author_name): + continue + + author_id = coerce_int(item.get("id"), 0) + if author_id < 1: + continue + + if author_id not in author_ids: + author_ids.append(author_id) + + if not author_ids: + return [] + + result = self._execute_query( + SERIES_BY_AUTHOR_IDS_QUERY, + { + "authorIds": author_ids, + "limit": 7, + }, + ) + if not result: + return [] + + series_rows = result.get("series", []) + return [row for row in series_rows if isinstance(row, dict)] + + @cacheable(ttl=120, key_prefix="hardcover:author:options") + def _search_author_options(self, query: str) -> list[dict[str, str]]: + """Return typeahead options for Hardcover author search.""" + hits = self._search_field_hits( + query=query, + query_type="Author", + limit=7, + sort=AUTHOR_SUGGESTION_SORT, + fields=AUTHOR_SUGGESTION_FIELDS, + weights=AUTHOR_SUGGESTION_WEIGHTS, + ) + options: list[dict[str, str]] = [] + seen_labels: set[str] = set() + + for hit in hits: + item = _unwrap_hit_document(hit) + if item is None: + continue + + author_id = coerce_int(item.get("id"), 0) + label = str(item.get("name") or "").strip() + normalized_label = label.casefold() + if author_id < 1 or not label or normalized_label in seen_labels: + continue + + seen_labels.add(normalized_label) + options.append({"value": f"id:{author_id}", "label": label}) + + return options + + @cacheable(ttl=120, key_prefix="hardcover:title:options") + def _search_title_options(self, query: str) -> list[dict[str, str]]: + """Return typeahead options for Hardcover title search.""" + hits = self._search_field_hits( + query=query, + query_type="Book", + limit=7, + sort=TITLE_SUGGESTION_SORT, + fields=TITLE_SUGGESTION_FIELDS, + weights=TITLE_SUGGESTION_WEIGHTS, + ) + + exclude_compilations = coerce_bool( + app_config.get("HARDCOVER_EXCLUDE_COMPILATIONS", False), + default=False, + ) + exclude_unreleased = coerce_bool( + app_config.get("HARDCOVER_EXCLUDE_UNRELEASED", False), + default=False, + ) + current_year = datetime.now(UTC).year + + options: list[dict[str, str]] = [] + seen_labels: set[str] = set() + + for hit in hits: + item = _unwrap_hit_document(hit) + if item is None: + continue + + if exclude_compilations and item.get("compilation"): + continue + + if exclude_unreleased: + release_year = item.get("release_year") + try: + if release_year is not None and int(release_year) > current_year: + continue + except TypeError, ValueError: + pass + + label = str(item.get("title") or "").strip() + normalized_label = label.casefold() + if not label or normalized_label in seen_labels: + continue + + seen_labels.add(normalized_label) + options.append({"value": label, "label": label}) + + return options + + def _format_series_option_description(self, item: dict[str, Any]) -> str | None: + """Build a short description for a series suggestion option.""" + author_name = item.get("author_name") + if not author_name: + author_data = item.get("author") + if isinstance(author_data, dict): + author_name = author_data.get("name") + + parts: list[str] = [] + if author_name: + parts.append(f"by {author_name}") + + books_count = item.get("primary_books_count") + if books_count is None: + books_count = item.get("books_count") + + try: + if books_count is not None: + books_count_int = int(books_count) + parts.append(f"{books_count_int} book{'s' if books_count_int != 1 else ''}") + except TypeError, ValueError: + pass + + return " • ".join(parts) if parts else None + + @cacheable(ttl=120, key_prefix="hardcover:series:options") + def _search_series_options(self, query: str) -> list[dict[str, str]]: + """Return typeahead options for Hardcover series search.""" + from concurrent.futures import ThreadPoolExecutor + + with ThreadPoolExecutor(max_workers=2) as executor: + author_future = executor.submit(self._search_series_by_matching_author, query) + series_future = executor.submit( + self._search_field_hits, + query=query, + query_type="Series", + limit=7, + sort=SERIES_SEARCH_SORT, + fields=SERIES_SEARCH_FIELDS, + weights=SERIES_SEARCH_WEIGHTS, + ) + + author_series = author_future.result() + hits = series_future.result() + options: list[dict[str, str]] = [] + seen_values: set[str] = set() + + series_items: list[dict[str, Any]] = [] + series_items.extend(author_series) + series_items.extend(doc for hit in hits if (doc := _unwrap_hit_document(hit)) is not None) + + for item in series_items: + series_id = item.get("id") + name = str(item.get("name") or "").strip() + if series_id is None or not name: + continue + + value = f"id:{series_id}" + if value in seen_values: + continue + seen_values.add(value) + + option: dict[str, str] = { + "value": value, + "label": name, + } + description = self._format_series_option_description(item) + if description: + option["description"] = description + options.append(option) + if len(options) >= HARDCOVER_MAX_SERIES_OPTIONS: + break + + return options + + def _resolve_series_search_value(self, series_value: str) -> dict[str, Any] | None: + """Resolve a series field value to a canonical Hardcover series.""" + normalized_value = _normalize_search_text(series_value) + if not normalized_value: + return None + + if normalized_value.startswith(HARDCOVER_LIST_ID_PREFIX): + try: + return {"id": self._parse_prefixed_int(normalized_value, "series id")} + except ValueError: + logger.debug("Invalid Hardcover series id field value: %s", normalized_value) + return None + + result = self._execute_query( + SEARCH_FIELD_OPTIONS_QUERY, + { + "query": normalized_value, + "queryType": "Series", + "limit": 10, + "page": 1, + "sort": SERIES_SEARCH_SORT, + "fields": SERIES_SEARCH_FIELDS, + "weights": SERIES_SEARCH_WEIGHTS, + }, + ) + if not result: + return None + + hits, _found_count = _extract_typesense_hits(result) + if not hits: + return None + + normalized_lookup = normalized_value.lower() + candidates: list[dict[str, Any]] = [] + for hit in hits: + item = _unwrap_hit_document(hit) + if item is None: + continue + series_id = coerce_int(item.get("id"), 0) + if series_id < 1: + continue + name = str(item.get("name") or "").strip() + if not name: + continue + candidates.append({"id": series_id, "name": name}) + + if not candidates: + return None + + exact_match = next( + ( + candidate + for candidate in candidates + if candidate["name"].lower() == normalized_lookup + ), + None, + ) + return exact_match or candidates[0] + + @cacheable( + ttl_key="METADATA_CACHE_SEARCH_TTL", ttl_default=300, key_prefix="hardcover:series:rows:v4" + ) + def _fetch_series_ordered_rows( + self, + series_id: int, + *, + exclude_compilations: bool, + exclude_unreleased: bool, + ) -> dict[str, Any]: + """Fetch and process all books for a series (cached independently of page).""" + empty: dict[str, Any] = {"rows": [], "series_name": "", "total": 0} + if not self.api_key: + return empty + + result = self._execute_query( + SERIES_BOOKS_BY_ID_QUERY, + {"seriesId": series_id}, + ) + if not result: + return empty + + series_items = result.get("series", []) + if not isinstance(series_items, list) or not series_items: + return empty + + series_data = series_items[0] if isinstance(series_items[0], dict) else {} + series_name = ( + str(series_data.get("name") or "").strip() if isinstance(series_data, dict) else "" + ) + allow_split_parts = _series_allows_split_parts(series_name) + today = datetime.now(UTC).date() + + book_series_rows = ( + series_data.get("book_series", []) if isinstance(series_data, dict) else [] + ) + rows_by_position: dict[float, dict[str, Any]] = {} + for row in book_series_rows: + if not isinstance(row, dict): + continue + book_data = row.get("book", {}) + if not isinstance(book_data, dict) or not book_data: + continue + if exclude_compilations and book_data.get("compilation"): + continue + if not allow_split_parts and _split_part_base_title(str(book_data.get("title") or "")): + continue + + position = _normalize_series_position(row.get("position")) + if position is None: + continue + + release_date = _parse_release_date(book_data.get("release_date")) + if exclude_unreleased and (release_date is None or release_date.date() > today): + continue + + sort_key = ( + 1 if release_date and release_date.date() <= today else 0, + 0 if book_data.get("compilation") else 1, + coerce_int(book_data.get("users_count"), 0), + coerce_int(book_data.get("ratings_count"), 0), + coerce_int(book_data.get("editions_count"), 0), + -coerce_int(book_data.get("id"), 0), + ) + existing_row = rows_by_position.get(position) + if existing_row is None: + rows_by_position[position] = {"row": row, "sort_key": sort_key} + continue + if sort_key > existing_row["sort_key"]: + rows_by_position[position] = {"row": row, "sort_key": sort_key} + + ordered_rows = [ + entry["row"] + for _position, entry in sorted(rows_by_position.items(), key=lambda item: item[0]) + ] + return {"rows": ordered_rows, "series_name": series_name, "total": len(ordered_rows)} + + def _fetch_series_books_by_id( + self, + series_id: int, + page: int, + limit: int, + *, + exclude_compilations: bool, + exclude_unreleased: bool, + ) -> SearchResult: + """Fetch books for a Hardcover series in canonical series order.""" + cached = self._fetch_series_ordered_rows( + series_id, + exclude_compilations=exclude_compilations, + exclude_unreleased=exclude_unreleased, + ) + ordered_rows = cached["rows"] + series_name = cached["series_name"] + total_found = cached["total"] + + offset = (page - 1) * limit + page_rows = ordered_rows[offset : offset + limit] + + books: list[BookMetadata] = [] + for row in page_rows: + book_data = row.get("book", {}) + if not isinstance(book_data, dict) or not book_data: + continue + try: + parsed_book = self._parse_book(book_data) + if not parsed_book: + continue + parsed_book.series_id = str(series_id) + if series_name: + parsed_book.series_name = series_name + parsed_book.series_position = row.get("position") + parsed_book.series_count = total_found + books.append(parsed_book) + except (AttributeError, IndexError, KeyError, TypeError, ValueError) as exc: + logger.debug( + "Failed to parse Hardcover series book for series_id=%s: %s", series_id, exc + ) + + has_more = offset + len(page_rows) < total_found + return SearchResult(books=books, page=page, total_found=total_found, has_more=has_more) + + def _fetch_author_books_by_id( + self, + author_id: int, + page: int, + limit: int, + *, + exclude_compilations: bool, + exclude_unreleased: bool, + ) -> SearchResult: + """Fetch books for a selected Hardcover author.""" + if not self.api_key: + return SearchResult(books=[], page=page, total_found=0, has_more=False) + + offset = (page - 1) * limit + result = self._execute_query( + AUTHOR_BOOKS_BY_ID_QUERY, + {"authorId": author_id, "limit": limit, "offset": offset}, + ) + if not result: + return SearchResult(books=[], page=page, total_found=0, has_more=False) + + author_items = result.get("authors", []) + if not isinstance(author_items, list) or not author_items: + return SearchResult(books=[], page=page, total_found=0, has_more=False) + + author_data = author_items[0] if isinstance(author_items[0], dict) else {} + contributions = ( + author_data.get("contributions", []) if isinstance(author_data, dict) else [] + ) + aggregate = ( + author_data.get("contributions_aggregate", {}) if isinstance(author_data, dict) else {} + ) + total_found = coerce_int( + aggregate.get("aggregate", {}).get("count") if isinstance(aggregate, dict) else 0, + 0, + ) + today = datetime.now(UTC).date() + + books: list[BookMetadata] = [] + for row in contributions: + if not isinstance(row, dict): + continue + contribution = str(row.get("contribution") or "").strip() + if contribution and "author" not in contribution.casefold(): + continue + book_data = row.get("book", {}) + if not isinstance(book_data, dict) or not book_data: + continue + if exclude_compilations and book_data.get("compilation"): + continue + release_date = _parse_release_date(book_data.get("release_date")) + if exclude_unreleased and (release_date is None or release_date.date() > today): + continue + try: + parsed_book = self._parse_book(book_data) + books.append(parsed_book) + except (AttributeError, IndexError, KeyError, TypeError, ValueError) as exc: + logger.debug( + "Failed to parse Hardcover author book for author_id=%s: %s", + author_id, + exc, + ) + + has_more = offset + len(contributions) < total_found + return SearchResult(books=books, page=page, total_found=total_found, has_more=has_more) + + def search(self, options: MetadataSearchOptions) -> list[BookMetadata]: + """Search for books using Hardcover's search API.""" + return self.search_paginated(options).books + + def search_paginated(self, options: MetadataSearchOptions) -> SearchResult: + """Search for books with pagination info.""" + if not self.api_key: + logger.warning("Hardcover API key not configured") + return SearchResult(books=[], page=options.page, total_found=0, has_more=False) + + # Allow pasting a Hardcover list URL directly in the search input + list_url_parts = self._detect_list_url(options.query) + if list_url_parts: + owner_username, list_slug = list_url_parts + return self._fetch_list_books(list_slug, owner_username, options.page, options.limit) + + # Advanced filter list selector (shared fetch path with URL detection) + list_value_from_field = str(options.fields.get("hardcover_list", "")).strip() + if list_value_from_field: + if list_value_from_field.startswith(HARDCOVER_STATUS_PREFIX): + try: + status_id = self._parse_prefixed_int(list_value_from_field, "status") + return self._fetch_current_user_books_by_status( + status_id, options.page, options.limit + ) + except ValueError: + logger.debug("Invalid Hardcover status field value: %s", list_value_from_field) + return SearchResult(books=[], page=options.page, total_found=0, has_more=False) + if list_value_from_field.startswith(HARDCOVER_LIST_ID_PREFIX): + try: + list_id = self._parse_prefixed_int(list_value_from_field, "list") + return self._fetch_list_books_by_id(list_id, options.page, options.limit) + except ValueError: + logger.debug("Invalid hardcover_list field value: %s", list_value_from_field) + return SearchResult(books=[], page=options.page, total_found=0, has_more=False) + return self._fetch_list_books(list_value_from_field, None, options.page, options.limit) + + series_value_from_field = str(options.fields.get("series", "")).strip() + if series_value_from_field: + resolved_series = self._resolve_series_search_value(series_value_from_field) + if not resolved_series: + return SearchResult(books=[], page=options.page, total_found=0, has_more=False) + exclude_compilations = coerce_bool( + app_config.get("HARDCOVER_EXCLUDE_COMPILATIONS", False), + default=False, + ) + exclude_unreleased = coerce_bool( + app_config.get("HARDCOVER_EXCLUDE_UNRELEASED", False), + default=False, + ) + return self._fetch_series_books_by_id( + int(resolved_series["id"]), + options.page, + options.limit, + exclude_compilations=exclude_compilations, + exclude_unreleased=exclude_unreleased, + ) + + author_value_from_field = str(options.fields.get("author", "")).strip() + if author_value_from_field.startswith(HARDCOVER_LIST_ID_PREFIX): + try: + author_id = self._parse_prefixed_int(author_value_from_field, "author id") + except ValueError: + logger.debug("Invalid Hardcover author id field value: %s", author_value_from_field) + return SearchResult(books=[], page=options.page, total_found=0, has_more=False) + exclude_compilations = coerce_bool( + app_config.get("HARDCOVER_EXCLUDE_COMPILATIONS", False), + default=False, + ) + exclude_unreleased = coerce_bool( + app_config.get("HARDCOVER_EXCLUDE_UNRELEASED", False), + default=False, + ) + return self._fetch_author_books_by_id( + author_id, + options.page, + options.limit, + exclude_compilations=exclude_compilations, + exclude_unreleased=exclude_unreleased, + ) + + # Handle ISBN search separately + if options.search_type == SearchType.ISBN: + result = self.search_by_isbn(options.query) + books = [result] if result else [] + return SearchResult(books=books, page=1, total_found=len(books), has_more=False) + + # Build cache key from options (include fields and settings for cache differentiation) + fields_key = ":".join(f"{k}={v}" for k, v in sorted(options.fields.items())) + exclude_compilations = coerce_bool( + app_config.get("HARDCOVER_EXCLUDE_COMPILATIONS", False), + default=False, + ) + exclude_unreleased = coerce_bool( + app_config.get("HARDCOVER_EXCLUDE_UNRELEASED", False), + default=False, + ) + cache_key = f"{options.query}:{options.search_type.value}:{options.sort.value}:{options.limit}:{options.page}:{fields_key}:excl_comp={exclude_compilations}:excl_unrel={exclude_unreleased}" + return self._search_cached(cache_key, options) + + @cacheable(ttl_key="METADATA_CACHE_SEARCH_TTL", ttl_default=300, key_prefix="hardcover:search") + def _search_cached(self, cache_key: str, options: MetadataSearchOptions) -> SearchResult: + """Return cached Hardcover search results.""" + # Determine query and fields based on custom search fields + # Note: Hardcover API requires 'weights' when using 'fields' parameter + author_value = options.fields.get("author", "").strip() + title_value = options.fields.get("title", "").strip() + + # Build query and field configuration based on which fields are provided + query, search_fields, search_weights = self._build_search_params( + options.query, author_value, title_value, "" + ) + + graphql_query = SEARCH_BOOKS_WITH_FIELDS_QUERY if search_fields else SEARCH_BOOKS_QUERY + + # Map abstract sort order to Hardcover's sort parameter + sort_param = SORT_MAPPING.get(options.sort, SORT_MAPPING[SortOrder.RELEVANCE]) + + variables = { + "query": query, + "limit": options.limit, + "page": options.page, + "sort": sort_param, + } + + if search_fields: + variables["fields"] = search_fields + variables["weights"] = search_weights + + try: + result = self._execute_query(graphql_query, variables) + if not result: + logger.debug("Hardcover search: No result from API") + return SearchResult(books=[], page=options.page, total_found=0, has_more=False) + + # Extract hits from Typesense response + hits, found_count = _extract_typesense_hits(result) + + # Parse hits, filtering compilations and unreleased books if enabled + exclude_compilations = coerce_bool( + app_config.get("HARDCOVER_EXCLUDE_COMPILATIONS", False), + default=False, + ) + exclude_unreleased = coerce_bool( + app_config.get("HARDCOVER_EXCLUDE_UNRELEASED", False), + default=False, + ) + current_year = datetime.now(UTC).year + books = [] + for hit in hits: + item = _unwrap_hit_document(hit) + if item is None: + continue + if exclude_compilations and item.get("compilation"): + continue + if exclude_unreleased: + release_year = item.get("release_year") + if release_year is not None and release_year > current_year: + continue + book = self._parse_search_result(item) + if book: + books.append(book) + + logger.info( + "Hardcover search '%s' (fields=%s) returned %s results", + query, + search_fields, + len(books), + ) + + # Calculate if there are more results + results_so_far = (options.page - 1) * HARDCOVER_PAGE_SIZE + len(hits) + has_more = results_so_far < found_count + + return SearchResult( + books=books, page=options.page, total_found=found_count, has_more=has_more + ) + + except AttributeError, KeyError, TypeError, ValueError: + logger.exception("Hardcover search error") + return SearchResult(books=[], page=options.page, total_found=0, has_more=False) + + @cacheable(ttl_key="METADATA_CACHE_BOOK_TTL", ttl_default=600, key_prefix="hardcover:book") + def get_book(self, book_id: str) -> BookMetadata | None: + """Get book details by Hardcover ID.""" + if not self.api_key: + logger.warning("Hardcover API key not configured") + return None + + try: + book_id_int = int(book_id) + result = self._execute_query(GET_BOOK_QUERY, {"id": book_id_int}) + if not result: + return None + + books = result.get("books", []) + if not books: + return None + + return self._parse_book(books[0]) + + except ValueError: + logger.exception("Invalid book ID: %s", book_id) + return None + except AttributeError, KeyError, TypeError: + logger.exception("Hardcover get_book error") + return None + + @cacheable(ttl_key="METADATA_CACHE_BOOK_TTL", ttl_default=600, key_prefix="hardcover:isbn") + def search_by_isbn(self, isbn: str) -> BookMetadata | None: + """Search for a book by ISBN-10 or ISBN-13.""" + if not self.api_key: + logger.warning("Hardcover API key not configured") + return None + + # Clean ISBN (remove hyphens) + clean_isbn = isbn.replace("-", "").strip() + + try: + result = self._execute_query(SEARCH_BY_ISBN_QUERY, {"isbn": clean_isbn}) + if not result: + return None + + editions = result.get("editions", []) + if not editions: + logger.debug("No Hardcover book found for ISBN: %s", isbn) + return None + + edition = editions[0] + book_data = edition.get("book", {}) + if not book_data: + return None + + # Add ISBN data from edition to book data + book_data["isbn_10"] = edition.get("isbn_10") + book_data["isbn_13"] = edition.get("isbn_13") + + return self._parse_book(book_data) + + except AttributeError, IndexError, KeyError, TypeError, ValueError: + logger.exception("Hardcover ISBN search error") + return None diff --git a/shelfmark/metadata_providers/hardcover/settings.py b/shelfmark/metadata_providers/hardcover/settings.py new file mode 100644 index 0000000..5230e8e --- /dev/null +++ b/shelfmark/metadata_providers/hardcover/settings.py @@ -0,0 +1,154 @@ +"""Settings registration for the Hardcover metadata provider.""" + +from typing import Any + +import requests + +from shelfmark.core.config import config as app_config +from shelfmark.core.logger import setup_logger +from shelfmark.core.settings_registry import ( + ActionButton, + CheckboxField, + HeadingField, + PasswordField, + SelectField, + SettingsField, + register_settings, +) + +from .auth import _get_connected_username, _save_connected_user +from .constants import HARDCOVER_API_KEY_MIN_LENGTH +from .parsing import _normalize_hardcover_api_key +from .provider import HardcoverProvider + +logger = setup_logger(__name__) + + +def _test_hardcover_connection(current_values: dict[str, Any] | None = None) -> dict[str, Any]: + """Test the Hardcover API connection using current form values.""" + current_values = current_values or {} + + # Use current form values first, fall back to saved config + raw_key = current_values.get("HARDCOVER_API_KEY") or app_config.get("HARDCOVER_API_KEY", "") + api_key = _normalize_hardcover_api_key(raw_key) + + key_len = len(api_key) if api_key else 0 + logger.debug("Hardcover test: key length=%s", key_len) + + if not api_key: + # Clear any stored connection metadata since there's no key + _save_connected_user(None, None) + return {"success": False, "message": "API key is required"} + + if key_len < HARDCOVER_API_KEY_MIN_LENGTH: + return { + "success": False, + "message": ( + f"API key seems too short ({key_len} chars). " + f"Expected {HARDCOVER_API_KEY_MIN_LENGTH}+ chars." + ), + } + + connection_result = {"success": False, "message": "API request failed - check your API key"} + try: + provider = HardcoverProvider(api_key=api_key) + # Use the 'me' query to test connection (recommended by API docs) + result = provider._execute_query("query { me { id, username } }", {}) + if result is not None: + # Handle both single object and array response formats + me_data = result.get("me", {}) + if isinstance(me_data, list) and me_data: + me_data = me_data[0] + user_id = ( + str(me_data.get("id")) + if isinstance(me_data, dict) and me_data.get("id") is not None + else None + ) + username = ( + me_data.get("username", "Unknown") if isinstance(me_data, dict) else "Unknown" + ) + + # Save connected user metadata for persistent display + per-user list caching + _save_connected_user(user_id, username) + connection_result = {"success": True, "message": f"Connected as: {username}"} + else: + _save_connected_user(None, None) + except (AttributeError, KeyError, requests.RequestException, TypeError, ValueError) as e: + logger.exception("Hardcover connection test failed") + _save_connected_user(None, None) + return {"success": False, "message": f"Connection failed: {e!s}"} + + return connection_result + + +_HARDCOVER_SORT_OPTIONS = [ + {"value": "relevance", "label": "Most relevant"}, + {"value": "popularity", "label": "Most popular"}, + {"value": "rating", "label": "Highest rated"}, + {"value": "newest", "label": "Newest"}, + {"value": "oldest", "label": "Oldest"}, +] + + +@register_settings("hardcover", "Hardcover", icon="book", order=51, group="metadata_providers") +def hardcover_settings() -> list[SettingsField]: + """Hardcover metadata provider settings.""" + # Check for connected username to show status + connected_user = _get_connected_username() + test_button_description = ( + f"Connected as: {connected_user}" if connected_user else "Verify your API key works" + ) + + return [ + HeadingField( + key="hardcover_heading", + title="Hardcover", + description="A modern book tracking and discovery platform with a comprehensive API.", + link_url="https://hardcover.app", + link_text="hardcover.app", + ), + CheckboxField( + key="HARDCOVER_ENABLED", + label="Enable Hardcover", + description="Enable Hardcover as a metadata provider for book searches", + default=False, + ), + PasswordField( + key="HARDCOVER_API_KEY", + label="API Key", + description="Get your API key from hardcover.app/account/api", + required=True, + ), + ActionButton( + key="test_connection", + label="Test Connection", + description=test_button_description, + style="primary", + callback=_test_hardcover_connection, + ), + SelectField( + key="HARDCOVER_DEFAULT_SORT", + label="Default Sort Order", + description="Default sort order for Hardcover search results.", + options=_HARDCOVER_SORT_OPTIONS, + default="relevance", + ), + CheckboxField( + key="HARDCOVER_EXCLUDE_COMPILATIONS", + label="Exclude Compilations", + description="Filter out compilations, anthologies, and omnibus editions from search results", + default=False, + ), + CheckboxField( + key="HARDCOVER_EXCLUDE_UNRELEASED", + label="Exclude Unreleased Books", + description="Filter out books with a release year in the future", + default=False, + ), + CheckboxField( + key="HARDCOVER_AUTO_REMOVE_ON_DOWNLOAD", + label="Auto-Remove from List on Download", + description="Automatically remove a book from the active Hardcover list when you download it", + default=True, + ), + ] diff --git a/shelfmark/metadata_providers/hardcover/targets.py b/shelfmark/metadata_providers/hardcover/targets.py new file mode 100644 index 0000000..9a90532 --- /dev/null +++ b/shelfmark/metadata_providers/hardcover/targets.py @@ -0,0 +1,438 @@ +"""Hardcover list/status target read and mutation workflows.""" + +from typing import TYPE_CHECKING, Any + +from shelfmark.core.cache import cache_key +from shelfmark.core.logger import setup_logger +from shelfmark.core.request_helpers import coerce_int + +from .constants import ( + HARDCOVER_LIST_ID_PREFIX, + HARDCOVER_STATUS_PREFIX, + HARDCOVER_WRITABLE_TARGET_GROUPS, +) +from .models import HardcoverBookTargetState, HardcoverTargetPayloadError +from .queries import ( + BOOK_TARGET_MEMBERSHIP_BATCH_QUERY, + BOOK_TARGET_MEMBERSHIP_QUERY, + DELETE_LIST_BOOK_MUTATION, + DELETE_USER_BOOK_MUTATION, + INSERT_LIST_BOOK_MUTATION, + INSERT_USER_BOOK_MUTATION, + UPDATE_USER_BOOK_MUTATION, +) + +logger = setup_logger(__name__) + + +def _metadata_cache() -> Any: + from shelfmark.metadata_providers import hardcover + + return hardcover.get_metadata_cache() + + +class HardcoverTargetsMixin: + if TYPE_CHECKING: + api_key: str + + def _execute_query( + self, + query: str, + variables: dict[str, Any], + *, + raise_on_error: bool = False, + ) -> dict[str, Any] | None: ... + + def _resolve_current_user_id(self) -> str | None: ... + + def get_user_lists(self) -> list[dict[str, str]]: ... + + def get_book_targets(self, book_id: str) -> list[dict[str, Any]]: + """Get writable Hardcover list/status targets for a specific book.""" + if not self.api_key: + return [] + + book_id_int = coerce_int(book_id, 0) + if book_id_int < 1: + msg = "book_id must be a valid Hardcover book id" + raise ValueError(msg) + + state = self._fetch_book_target_state(book_id_int) + options: list[dict[str, Any]] = [ + dict(option) + for option in self.get_user_lists() + if option.get("group") in HARDCOVER_WRITABLE_TARGET_GROUPS + ] + + for option in options: + value = str(option.get("value") or "").strip() + option["checked"] = self._is_target_checked(value, state) + option["writable"] = True + + return options + + def set_book_target_state( + self, + book_id: str, + target: str, + *, + selected: bool, + ) -> dict[str, Any]: + """Set whether a Hardcover book belongs to a status shelf or user list.""" + if not self.api_key: + msg = "Hardcover is not configured" + raise ValueError(msg) + + book_id_int = coerce_int(book_id, 0) + if book_id_int < 1: + msg = "book_id must be a valid Hardcover book id" + raise ValueError(msg) + + selected_target = str(target or "").strip() + if not selected_target: + msg = "target is required" + raise ValueError(msg) + + if selected_target not in self._get_writable_targets(): + msg = "Unsupported Hardcover target" + raise ValueError(msg) + + state = self._fetch_book_target_state(book_id_int) + status_ids_to_invalidate: set[int] = set() + list_ids_to_invalidate: set[int] = set() + deselected_target: str | None = None + + if selected_target.startswith(HARDCOVER_STATUS_PREFIX): + status_id = self._parse_prefixed_int(selected_target, "status target") + previous_status_id = state.status_id + changed = self._set_status_target_state( + book_id_int, + status_id, + selected=selected, + state=state, + ) + if changed: + if previous_status_id is not None: + status_ids_to_invalidate.add(previous_status_id) + if selected and previous_status_id != status_id: + deselected_target = f"{HARDCOVER_STATUS_PREFIX}{previous_status_id}" + status_ids_to_invalidate.add(status_id) + elif selected_target.startswith(HARDCOVER_LIST_ID_PREFIX): + list_id = self._parse_prefixed_int(selected_target, "list target") + changed = self._set_list_target_state( + book_id_int, + list_id, + selected=selected, + state=state, + ) + if changed: + list_ids_to_invalidate.add(list_id) + else: + msg = "Unsupported Hardcover target" + raise ValueError(msg) + + if changed: + self._invalidate_book_target_caches( + connected_user_id=self._resolve_current_user_id(), + status_ids=status_ids_to_invalidate, + list_ids=list_ids_to_invalidate, + ) + + result_data: dict[str, Any] = {"changed": changed} + if deselected_target: + result_data["deselected_target"] = deselected_target + return result_data + + @staticmethod + def _unwrap_me_data(result: dict | None) -> dict: + """Extract and validate the ``me`` payload from a GraphQL result.""" + if not isinstance(result, dict): + msg = "Hardcover could not load book targets" + raise HardcoverTargetPayloadError(msg) + + me_data = result.get("me", {}) + if isinstance(me_data, list) and me_data: + me_data = me_data[0] + if not isinstance(me_data, dict): + msg = "Hardcover returned an invalid target payload" + raise HardcoverTargetPayloadError(msg) + return me_data + + def _fetch_book_target_state(self, book_id: int) -> HardcoverBookTargetState: + """Load current Hardcover membership state for a specific book.""" + result = self._execute_query( + BOOK_TARGET_MEMBERSHIP_QUERY, + {"bookId": book_id}, + raise_on_error=True, + ) + me_data = self._unwrap_me_data(result) + + user_book_id: int | None = None + status_id: int | None = None + user_books = me_data.get("user_books", []) + if isinstance(user_books, list) and user_books: + latest_user_book = user_books[0] if isinstance(user_books[0], dict) else {} + user_book_id = coerce_int(latest_user_book.get("id"), 0) or None + status_id = coerce_int(latest_user_book.get("status_id"), 0) or None + + list_book_ids: dict[int, int] = {} + for user_list in me_data.get("lists", []): + if not isinstance(user_list, dict): + continue + list_id = coerce_int(user_list.get("id"), 0) + if list_id < 1: + continue + + list_books = user_list.get("list_books", []) + if not isinstance(list_books, list) or not list_books: + continue + + list_book = list_books[0] if isinstance(list_books[0], dict) else {} + list_book_id = coerce_int(list_book.get("id"), 0) + if list_book_id > 0: + list_book_ids[list_id] = list_book_id + + return HardcoverBookTargetState( + user_book_id=user_book_id, + status_id=status_id, + list_book_ids=list_book_ids, + ) + + def _fetch_book_target_states_batch( + self, + book_ids: list[int], + ) -> dict[int, HardcoverBookTargetState]: + """Load Hardcover membership state for multiple books in one query.""" + result = self._execute_query( + BOOK_TARGET_MEMBERSHIP_BATCH_QUERY, + {"bookIds": book_ids}, + raise_on_error=True, + ) + me_data = self._unwrap_me_data(result) + + # Group user_books by book_id (keep only the latest per book) + user_book_by_book: dict[int, dict] = {} + for ub in me_data.get("user_books", []): + if not isinstance(ub, dict): + continue + bid = coerce_int(ub.get("book_id"), 0) + if bid > 0 and bid not in user_book_by_book: + user_book_by_book[bid] = ub + + # Group list_book memberships by book_id + list_book_ids_by_book: dict[int, dict[int, int]] = {} + for user_list in me_data.get("lists", []): + if not isinstance(user_list, dict): + continue + list_id = coerce_int(user_list.get("id"), 0) + if list_id < 1: + continue + for lb in user_list.get("list_books", []): + if not isinstance(lb, dict): + continue + bid = coerce_int(lb.get("book_id"), 0) + lb_id = coerce_int(lb.get("id"), 0) + if bid > 0 and lb_id > 0: + list_book_ids_by_book.setdefault(bid, {})[list_id] = lb_id + + states: dict[int, HardcoverBookTargetState] = {} + for bid in book_ids: + ub = user_book_by_book.get(bid) + states[bid] = HardcoverBookTargetState( + user_book_id=coerce_int(ub.get("id"), 0) or None if ub else None, + status_id=coerce_int(ub.get("status_id"), 0) or None if ub else None, + list_book_ids=list_book_ids_by_book.get(bid, {}), + ) + return states + + def get_book_targets_batch(self, book_ids: list[str]) -> dict[str, list[dict[str, Any]]]: + """Get writable Hardcover list/status targets for multiple books.""" + if not self.api_key or not book_ids: + return {bid: [] for bid in book_ids} + + int_ids = [] + id_map: dict[int, str] = {} + for bid in book_ids: + int_id = coerce_int(bid, 0) + if int_id > 0: + int_ids.append(int_id) + id_map[int_id] = bid + + if not int_ids: + return {bid: [] for bid in book_ids} + + states = self._fetch_book_target_states_batch(int_ids) + writable_options: list[dict[str, Any]] = [ + dict(option) + for option in self.get_user_lists() + if option.get("group") in HARDCOVER_WRITABLE_TARGET_GROUPS + ] + + results: dict[str, list[dict[str, Any]]] = {} + for int_id, str_id in id_map.items(): + state = states.get( + int_id, + HardcoverBookTargetState( + user_book_id=None, + status_id=None, + list_book_ids={}, + ), + ) + options = [dict(opt) for opt in writable_options] + for option in options: + value = str(option.get("value") or "").strip() + option["checked"] = self._is_target_checked(value, state) + option["writable"] = True + results[str_id] = options + + # Fill in any book_ids that didn't parse as valid ints + for bid in book_ids: + if bid not in results: + results[bid] = [] + + return results + + def _get_writable_targets(self) -> set[str]: + """Return the set of writable Hardcover targets for the current user.""" + writable_targets: set[str] = set() + for option in self.get_user_lists(): + value = str(option.get("value") or "").strip() + if ( + option.get("group") in HARDCOVER_WRITABLE_TARGET_GROUPS + and value + and value.startswith((HARDCOVER_STATUS_PREFIX, HARDCOVER_LIST_ID_PREFIX)) + ): + writable_targets.add(value) + return writable_targets + + def _is_target_checked(self, target: str, state: HardcoverBookTargetState) -> bool: + """Return whether a target is currently selected for the book.""" + if target.startswith(HARDCOVER_STATUS_PREFIX): + return state.status_id == self._parse_prefixed_int(target) + if target.startswith(HARDCOVER_LIST_ID_PREFIX): + return self._parse_prefixed_int(target) in state.list_book_ids + return False + + def _set_status_target_state( + self, + book_id: int, + status_id: int, + *, + selected: bool, + state: HardcoverBookTargetState, + ) -> bool: + """Set whether the book belongs to a Hardcover status shelf.""" + if selected: + if state.user_book_id is None: + result = self._execute_query( + INSERT_USER_BOOK_MUTATION, + {"bookId": book_id, "statusId": status_id}, + raise_on_error=True, + ) + self._check_mutation_result(result, "insert_user_book") + return True + + if state.status_id == status_id: + return False + + result = self._execute_query( + UPDATE_USER_BOOK_MUTATION, + {"userBookId": state.user_book_id, "statusId": status_id}, + raise_on_error=True, + ) + self._check_mutation_result(result, "update_user_book") + return True + + if state.user_book_id is None or state.status_id != status_id: + return False + + result = self._execute_query( + DELETE_USER_BOOK_MUTATION, + {"userBookId": state.user_book_id}, + raise_on_error=True, + ) + self._check_mutation_result(result, "delete_user_book", check_error=False) + return True + + def _set_list_target_state( + self, + book_id: int, + list_id: int, + *, + selected: bool, + state: HardcoverBookTargetState, + ) -> bool: + """Set whether the book belongs to a Hardcover list.""" + list_book_id = state.list_book_ids.get(list_id) + + if selected: + if list_book_id is not None: + return False + + result = self._execute_query( + INSERT_LIST_BOOK_MUTATION, + {"bookId": book_id, "listId": list_id}, + raise_on_error=True, + ) + self._check_mutation_result(result, "insert_list_book") + return True + + if list_book_id is None: + return False + + result = self._execute_query( + DELETE_LIST_BOOK_MUTATION, + {"listBookId": list_book_id}, + raise_on_error=True, + ) + self._check_mutation_result(result, "delete_list_book", check_error=False) + return True + + def _invalidate_book_target_caches( + self, + *, + connected_user_id: str | None, + status_ids: set[int], + list_ids: set[int], + ) -> None: + """Invalidate caches affected by a target membership change.""" + metadata_cache = _metadata_cache() + + if connected_user_id: + metadata_cache.invalidate(cache_key("hardcover:user_lists", connected_user_id)) + for status_id in status_ids: + metadata_cache.invalidate_prefix( + cache_key("hardcover:user_books:status", connected_user_id, status_id) + ) + + for list_id in list_ids: + metadata_cache.invalidate_prefix(cache_key("hardcover:list:id", list_id)) + + @staticmethod + def _parse_prefixed_int(value: str, label: str = "target") -> int: + """Parse an integer from a colon-prefixed value like 'status:1' or 'id:42'.""" + try: + return int(value.split(":", 1)[1]) + except (IndexError, ValueError) as exc: + msg = f"Invalid Hardcover {label}" + raise ValueError(msg) from exc + + @staticmethod + def _check_mutation_result(result: Any, key: str, *, check_error: bool = True) -> None: + """Raise if a Hardcover mutation failed. + + When *check_error* is True (the default) the ``error`` field inside + the payload is inspected and surfaced as a ``ValueError``. Pass + ``check_error=False`` for delete mutations that don't return an + error field. + """ + payload = result.get(key, {}) if isinstance(result, dict) else {} + if isinstance(payload, dict): + if check_error: + error_text = str(payload.get("error") or "").strip() + if error_text: + raise ValueError(error_text) + if payload.get("id") is not None: + return + msg = "Hardcover could not complete this action" + raise RuntimeError(msg) diff --git a/tests/metadata/test_hardcover_field_options.py b/tests/metadata/test_hardcover_field_options.py index 59e40c3..1f93821 100644 --- a/tests/metadata/test_hardcover_field_options.py +++ b/tests/metadata/test_hardcover_field_options.py @@ -2,11 +2,13 @@ from shelfmark.metadata_providers.hardcover import HardcoverProvider class TestHardcoverFieldOptions: - def test_search_fields_enable_typeahead_for_series_only(self): + def test_search_fields_enable_typeahead_for_author_and_series(self): provider = HardcoverProvider(api_key="test-token") fields_by_key = {field.key: field for field in provider.search_fields} - assert fields_by_key["author"].suggestions_endpoint is None + assert fields_by_key["author"].suggestions_endpoint == ( + "/api/metadata/field-options?provider=hardcover&field=author" + ) assert fields_by_key["title"].suggestions_endpoint is None assert fields_by_key["series"].suggestions_endpoint == ( "/api/metadata/field-options?provider=hardcover&field=series" @@ -25,9 +27,9 @@ class TestHardcoverFieldOptions: "search": { "results": { "hits": [ - {"document": {"name": "Brandon Sanderson"}}, - {"document": {"name": "Brandon Sanderson"}}, - {"document": {"name": "Brian Sanderson"}}, + {"document": {"id": 1, "name": "Brandon Sanderson"}}, + {"document": {"id": 1, "name": "Brandon Sanderson"}}, + {"document": {"id": 2, "name": "Brian Sanderson"}}, ], "found": 3, } @@ -39,8 +41,8 @@ class TestHardcoverFieldOptions: options = provider.get_search_field_options("author", query="sand") assert options == [ - {"value": "Brandon Sanderson", "label": "Brandon Sanderson"}, - {"value": "Brian Sanderson", "label": "Brian Sanderson"}, + {"value": "id:1", "label": "Brandon Sanderson"}, + {"value": "id:2", "label": "Brian Sanderson"}, ] assert captured["variables"] == { "query": "sand", diff --git a/tests/metadata/test_hardcover_search_author.py b/tests/metadata/test_hardcover_search_author.py index 36c724c..3780a0b 100644 --- a/tests/metadata/test_hardcover_search_author.py +++ b/tests/metadata/test_hardcover_search_author.py @@ -1,4 +1,5 @@ -from shelfmark.metadata_providers.hardcover import _simplify_author_for_search +from shelfmark.metadata_providers import MetadataSearchOptions, SearchResult +from shelfmark.metadata_providers.hardcover import HardcoverProvider, _simplify_author_for_search class TestHardcoverSimplifyAuthorForSearch: @@ -13,3 +14,126 @@ class TestHardcoverSimplifyAuthorForSearch: def test_returns_none_when_no_change(self): assert _simplify_author_for_search("Frank Herbert") is None + + +class TestHardcoverAuthorSearch: + def test_author_text_search_uses_default_book_search_fields(self): + provider = HardcoverProvider(api_key="test-token") + + assert provider._build_search_params("", "Stephen King", "", "") == ( + "Stephen King", + None, + None, + ) + + def test_search_paginated_uses_selected_author_id(self, monkeypatch): + provider = HardcoverProvider(api_key="test-token") + expected = SearchResult(books=[], page=2, total_found=14, has_more=True) + captured: dict[str, int] = {} + + monkeypatch.setattr( + "shelfmark.metadata_providers.hardcover.app_config.get", + lambda key, default=None: { + "HARDCOVER_EXCLUDE_COMPILATIONS": True, + "HARDCOVER_EXCLUDE_UNRELEASED": False, + }.get(key, default), + ) + + def fake_fetch( + author_id: int, + page: int, + limit: int, + *, + exclude_compilations: bool, + exclude_unreleased: bool, + ) -> SearchResult: + captured["author_id"] = author_id + captured["page"] = page + captured["limit"] = limit + captured["exclude_compilations"] = int(exclude_compilations) + captured["exclude_unreleased"] = int(exclude_unreleased) + return expected + + monkeypatch.setattr(provider, "_fetch_author_books_by_id", fake_fetch) + + result = provider.search_paginated( + MetadataSearchOptions( + query="", + page=2, + limit=20, + fields={"author": "id:42"}, + ) + ) + + assert result == expected + assert captured == { + "author_id": 42, + "page": 2, + "limit": 20, + "exclude_compilations": 1, + "exclude_unreleased": 0, + } + + def test_fetch_author_books_by_id_returns_books(self, monkeypatch): + provider = HardcoverProvider(api_key="test-token") + captured: dict[str, object] = {} + + monkeypatch.setattr( + provider, + "_execute_query", + lambda query, variables: ( + captured.update({"query": query, "variables": variables}) + or { + "authors": [ + { + "name": "Stephen King", + "contributions": [ + { + "contribution": "Author, Narrator", + "book": { + "id": 1, + "title": "The Shining", + "subtitle": None, + "slug": "the-shining", + "release_date": "1977-01-28", + "headline": None, + "description": None, + "pages": 447, + "rating": 4.3, + "ratings_count": 1000, + "users_count": 2000, + "compilation": False, + "editions_count": 20, + "cached_image": {}, + "cached_contributors": [{"name": "Stephen King"}], + "contributions": [], + "featured_book_series": None, + }, + } + ], + "contributions_aggregate": {"aggregate": {"count": 1}}, + } + ] + } + ), + ) + + result = provider._fetch_author_books_by_id( + 42, + page=1, + limit=20, + exclude_compilations=True, + exclude_unreleased=True, + ) + + assert "contributions(" in str(captured["query"]) + assert ( + "contribution:" + not in str(captured["query"]).split("contributions(", 1)[1].split(") {", 1)[0] + ) + assert "canonical_id: {_is_null: true}" in str(captured["query"]) + assert captured["variables"] == {"authorId": 42, "limit": 20, "offset": 0} + assert result.total_found == 1 + assert result.has_more is False + assert [book.title for book in result.books] == ["The Shining"] + assert result.books[0].authors == ["Stephen King"]