From bd1ad3495c5e383cdc61e4c23c0b0baf89f5e174 Mon Sep 17 00:00:00 2001 From: Alex Date: Tue, 13 Jan 2026 14:18:23 +0000 Subject: [PATCH] Proxy, prowlarr, ingest dir fixes (#437) --- shelfmark/bypass/internal_bypasser.py | 4 +- shelfmark/config/settings.py | 8 +++ shelfmark/core/settings_registry.py | 10 ++++ shelfmark/download/http.py | 6 +- shelfmark/download/network.py | 59 +++++++++++++++++-- shelfmark/metadata_providers/__init__.py | 1 + shelfmark/metadata_providers/hardcover.py | 32 +++++++++- shelfmark/release_sources/direct_download.py | 2 +- .../prowlarr/clients/sabnzbd.py | 2 + shelfmark/release_sources/prowlarr/handler.py | 48 +++++++++++++-- shelfmark/release_sources/prowlarr/source.py | 33 +++++++---- tests/prowlarr/test_sabnzbd_client.py | 51 ++++++++++++++++ 12 files changed, 227 insertions(+), 29 deletions(-) diff --git a/shelfmark/bypass/internal_bypasser.py b/shelfmark/bypass/internal_bypasser.py index 36a642f1..445054b8 100644 --- a/shelfmark/bypass/internal_bypasser.py +++ b/shelfmark/bypass/internal_bypasser.py @@ -770,7 +770,7 @@ def get(url: str, retry: Optional[int] = None, cancel_flag: Optional[Event] = No cookies = get_cf_cookies_for_domain(urlparse(url).hostname or "") if cookies: try: - response = requests.get(url, cookies=cookies, proxies=get_proxies(), timeout=(5, 10)) + response = requests.get(url, cookies=cookies, proxies=get_proxies(url), timeout=(5, 10)) if response.status_code == 200: logger.debug("Cookies available after lock wait - skipped Chrome") return response.text @@ -1035,7 +1035,7 @@ def _try_with_cached_cookies(url: str, hostname: str) -> Optional[str]: headers['User-Agent'] = stored_ua logger.debug(f"Trying request with cached cookies: {url}") - response = requests.get(url, cookies=cookies, headers=headers, proxies=get_proxies(), timeout=(5, 10)) + response = requests.get(url, cookies=cookies, headers=headers, proxies=get_proxies(url), timeout=(5, 10)) if response.status_code == 200: logger.debug("Cached cookies worked, skipped Chrome bypass") return response.text diff --git a/shelfmark/config/settings.py b/shelfmark/config/settings.py index 849e5027..99609ee8 100644 --- a/shelfmark/config/settings.py +++ b/shelfmark/config/settings.py @@ -493,6 +493,14 @@ def network_settings(): disabled_reason="Proxy settings are not used when Tor routing is enabled.", show_when={"field": "PROXY_MODE", "value": "socks5"}, ), + TextField( + key="NO_PROXY", + label="No Proxy", + description="Comma-separated hosts to bypass proxy (e.g., localhost,127.0.0.1,10.*,*.local)", + disabled=tor_overrides_network, + disabled_reason="Proxy settings are not used when Tor routing is enabled.", + show_when={"field": "PROXY_MODE", "value": ["http", "socks5"]}, + ), ] diff --git a/shelfmark/core/settings_registry.py b/shelfmark/core/settings_registry.py index fabf4b57..350cb2fa 100644 --- a/shelfmark/core/settings_registry.py +++ b/shelfmark/core/settings_registry.py @@ -333,6 +333,16 @@ def migrate_legacy_settings() -> None: if "FILE_ORGANIZATION" in downloads_config or "DESTINATION" in downloads_config: return + # Skip migration if no legacy settings exist (fresh install) + legacy_keys = { + "PROCESSING_MODE", "INGEST_DIR", "LIBRARY_PATH", "USE_BOOK_TITLE", + "LIBRARY_TEMPLATE", "PROCESSING_MODE_AUDIOBOOK", "INGEST_DIR_AUDIOBOOK", + "LIBRARY_PATH_AUDIOBOOK", "LIBRARY_TEMPLATE_AUDIOBOOK", "TORRENT_HARDLINK", + "USE_CONTENT_TYPE_DIRECTORIES", + } + if not any(key in downloads_config for key in legacy_keys): + return + migrated_downloads = {} migrated_sources = {} diff --git a/shelfmark/download/http.py b/shelfmark/download/http.py index 6e0a82d6..69ebf64d 100644 --- a/shelfmark/download/http.py +++ b/shelfmark/download/http.py @@ -206,7 +206,7 @@ def html_get_page( # Try with CF cookies/UA if available (from previous bypass) headers = {} cookies = _apply_cf_bypass(current_url, headers) - response = requests.get(current_url, proxies=get_proxies(), timeout=REQUEST_TIMEOUT, cookies=cookies, headers=headers) + response = requests.get(current_url, proxies=get_proxies(current_url), timeout=REQUEST_TIMEOUT, cookies=cookies, headers=headers) response.raise_for_status() time.sleep(1) return response.text @@ -291,7 +291,7 @@ def download_url( logger.info(f"Downloading: {current_url} (attempt {attempt + 1}/{MAX_DOWNLOAD_RETRIES})") # Try with CF cookies/UA if available cookies = _apply_cf_bypass(current_url, headers) - response = requests.get(current_url, stream=True, proxies=get_proxies(), timeout=REQUEST_TIMEOUT, cookies=cookies, headers=headers) + response = requests.get(current_url, stream=True, proxies=get_proxies(current_url), timeout=REQUEST_TIMEOUT, cookies=cookies, headers=headers) response.raise_for_status() if status_callback: @@ -401,7 +401,7 @@ def _try_resume( resume_headers = {**(base_headers or DOWNLOAD_HEADERS), 'Range': f'bytes={start_byte}-'} cookies = _apply_cf_bypass(url, resume_headers) response = requests.get( - url, stream=True, proxies=get_proxies(), timeout=REQUEST_TIMEOUT, + url, stream=True, proxies=get_proxies(url), timeout=REQUEST_TIMEOUT, headers=resume_headers, cookies=cookies ) diff --git a/shelfmark/download/network.py b/shelfmark/download/network.py index 38d1be3a..35251374 100644 --- a/shelfmark/download/network.py +++ b/shelfmark/download/network.py @@ -1,5 +1,6 @@ """DNS rotation, mirror selection, and network utilities.""" +import fnmatch import requests import urllib.request from typing import Sequence, Tuple, Any, Union, cast, List, Optional, Callable @@ -14,8 +15,58 @@ from shelfmark.core.config import config as app_config from datetime import datetime, timedelta -def get_proxies() -> dict: - """Get current proxy configuration from config singleton.""" +def _get_no_proxy_patterns() -> List[str]: + """Get list of NO_PROXY patterns from config.""" + no_proxy = app_config.get("NO_PROXY", "") + if not no_proxy: + return [] + return [p.strip().lower() for p in no_proxy.split(",") if p.strip()] + + +def should_bypass_proxy(url: str) -> bool: + """Check if a URL should bypass the proxy based on NO_PROXY patterns. + + Supports: + - Exact hostname match: localhost, myhost.local + - Wildcard prefix: *.local matches foo.local + - Wildcard suffix: 10.* matches 10.1.2.3 + """ + if not url: + return False + + patterns = _get_no_proxy_patterns() + if not patterns: + return False + + # Extract hostname from URL + try: + parsed = urllib.parse.urlparse(url) + hostname = (parsed.hostname or "").lower() + except Exception: + return False + + if not hostname: + return False + + for pattern in patterns: + # Use fnmatch for wildcard matching (supports * and ?) + if fnmatch.fnmatch(hostname, pattern): + return True + + return False + + +def get_proxies(url: str = "") -> dict: + """Get current proxy configuration from config singleton. + + Args: + url: Optional URL to check against NO_PROXY patterns. + If provided and matches a pattern, returns empty dict. + """ + # Check NO_PROXY bypass first + if url and should_bypass_proxy(url): + return {} + proxy_mode = app_config.get("PROXY_MODE", "none") if proxy_mode == "socks5": @@ -340,7 +391,7 @@ class DoHResolver: response = self.session.get( self.base_url, params=params, - proxies=get_proxies(), + proxies=get_proxies(self.base_url), timeout=10 # Increased from 5s to handle slow network conditions ) response.raise_for_status() @@ -834,7 +885,7 @@ def _initialize_aa_state() -> None: logger.debug(f"AA_BASE_URL: auto, checking available urls {_aa_urls}") for i, url in enumerate(_aa_urls): try: - response = requests.get(url, proxies=get_proxies(), timeout=3) + response = requests.get(url, proxies=get_proxies(url), timeout=3) if response.status_code == 200: _current_aa_url_index = i _aa_base_url = url diff --git a/shelfmark/metadata_providers/__init__.py b/shelfmark/metadata_providers/__init__.py index 2a7cc606..67bc23c9 100644 --- a/shelfmark/metadata_providers/__init__.py +++ b/shelfmark/metadata_providers/__init__.py @@ -144,6 +144,7 @@ class BookMetadata: genres: List[str] = field(default_factory=list) source_url: Optional[str] = None # Link to book on provider's site subtitle: Optional[str] = None # Book subtitle, if any + search_title: Optional[str] = None # Cleaner title for search queries (provider-specific) # Provider-specific display fields for cards/lists display_fields: List[DisplayField] = field(default_factory=list) diff --git a/shelfmark/metadata_providers/hardcover.py b/shelfmark/metadata_providers/hardcover.py index 36b2634f..40554318 100644 --- a/shelfmark/metadata_providers/hardcover.py +++ b/shelfmark/metadata_providers/hardcover.py @@ -95,6 +95,29 @@ def _build_source_url(slug: str) -> Optional[str]: return f"https://hardcover.app/books/{slug}" if slug else None +def _compute_search_title(title: str, subtitle: Optional[str]) -> Optional[str]: + """Compute a cleaner search title from title and subtitle. + + When Hardcover uses the "Series: Book Title" format, the subtitle contains + the actual book title which is better for searching. For example: + - title: "Mistborn: The Final Empire" + - subtitle: "The Final Empire" + - search_title: "The Final Empire" (better for Prowlarr/indexer searches) + + Skips subtitles that start with series position indicators like "Book One", + "Part 1", "Volume 2" as these are descriptors, not the actual title. + """ + if not subtitle or subtitle not in title: + return None + + # Skip if subtitle starts with series position indicators + skip_prefixes = ('book ', 'part ', 'volume ') + if subtitle.lower().startswith(skip_prefixes): + return None + + return subtitle + + @register_provider_kwargs("hardcover") def _hardcover_kwargs() -> Dict[str, Any]: """Provide Hardcover-specific constructor kwargs.""" @@ -564,6 +587,7 @@ class HardcoverProvider(MetadataProvider): provider_id=str(book_id), title=title, subtitle=subtitle, + search_title=_compute_search_title(title, subtitle), provider_display_name="Hardcover", authors=authors, cover_url=cover_url, @@ -681,11 +705,15 @@ class HardcoverProvider(MetadataProvider): if code3 and code3 not in titles_by_language: titles_by_language[code3] = edition_title + title = book["title"] + subtitle = book.get("subtitle") + return BookMetadata( provider="hardcover", provider_id=str(book["id"]), - title=book["title"], - subtitle=book.get("subtitle"), + title=title, + subtitle=subtitle, + search_title=_compute_search_title(title, subtitle), provider_display_name="Hardcover", authors=authors, isbn_10=isbn_10, diff --git a/shelfmark/release_sources/direct_download.py b/shelfmark/release_sources/direct_download.py index c88f12c8..08fd54a1 100644 --- a/shelfmark/release_sources/direct_download.py +++ b/shelfmark/release_sources/direct_download.py @@ -696,7 +696,7 @@ def _extract_libgen_download_url(link: str, cancel_flag: Optional[Event] = None) headers=downloader.DOWNLOAD_HEADERS, timeout=(5, 10), allow_redirects=True, - proxies=network.get_proxies(), + proxies=network.get_proxies(link), ) if response.status_code != 200: diff --git a/shelfmark/release_sources/prowlarr/clients/sabnzbd.py b/shelfmark/release_sources/prowlarr/clients/sabnzbd.py index 4cdd023c..d0917aee 100644 --- a/shelfmark/release_sources/prowlarr/clients/sabnzbd.py +++ b/shelfmark/release_sources/prowlarr/clients/sabnzbd.py @@ -241,6 +241,7 @@ class SABnzbdClient(DownloadClient): if slot.get("nzo_id") == download_id: status_text = slot.get("status", "").upper() storage = slot.get("storage", "") + logger.debug(f"SABnzbd history: {download_id} status={status_text} storage='{storage}'") if status_text == "COMPLETED": return DownloadStatus( @@ -262,6 +263,7 @@ class SABnzbdClient(DownloadClient): ) # Not found + logger.warning(f"SABnzbd: download {download_id} not found in queue or history") return DownloadStatus.error("Download not found") except Exception as e: error_type = type(e).__name__ diff --git a/shelfmark/release_sources/prowlarr/handler.py b/shelfmark/release_sources/prowlarr/handler.py index 7eec461f..f0ef0655 100644 --- a/shelfmark/release_sources/prowlarr/handler.py +++ b/shelfmark/release_sources/prowlarr/handler.py @@ -178,6 +178,7 @@ class ProwlarrHandler(DownloadHandler): ) -> Optional[str]: """Poll the download client for progress and handle completion.""" try: + logger.debug(f"Starting poll for {download_id} (content_type={task.content_type})") while not cancel_flag.is_set(): status = client.get_status(download_id) progress_callback(status.progress) @@ -185,13 +186,16 @@ class ProwlarrHandler(DownloadHandler): # Check for completion if status.complete: if status.state == DownloadState.ERROR: + logger.error(f"Download {download_id} completed with error: {status.message}") status_callback("error", status.message or "Download failed") return None # Download complete - break to handle file + logger.debug(f"Download {download_id} complete, file_path={status.file_path}") break # Check for error state if status.state == DownloadState.ERROR: + logger.error(f"Download {download_id} error state: {status.message}") status_callback("error", status.message or "Download failed") client.remove(download_id, delete_files=True) return None @@ -214,11 +218,35 @@ class ProwlarrHandler(DownloadHandler): # Handle completed file source_path = client.get_download_path(download_id) if not source_path: - status_callback("error", "Could not locate downloaded file") + logger.error( + f"Download client returned empty path for completed download. " + f"Client: {client.name}, ID: {download_id}. " + f"Check that the download client's completion folder is accessible to Shelfmark." + ) + status_callback( + "error", + f"Download completed in {client.name} but path not returned. " + f"Check volume mappings and category settings." + ) + return None + + # Verify the path actually exists in our filesystem + source_path_obj = Path(source_path) + if not source_path_obj.exists(): + logger.error( + f"Download path does not exist: {source_path}. " + f"Client: {client.name}, ID: {download_id}. " + f"The download client's path may not be mounted in Shelfmark's container. " + f"Ensure both containers use identical volume mappings for the download folder." + ) + status_callback( + "error", + f"Path not accessible: {source_path}. Check volume mappings between {client.name} and Shelfmark." + ) return None result = self._handle_completed_file( - source_path=Path(source_path), + source_path=source_path_obj, protocol=protocol, task=task, status_callback=status_callback, @@ -279,12 +307,22 @@ class ProwlarrHandler(DownloadHandler): return str(staged_path) + except FileNotFoundError as e: + logger.error( + f"Source file not found during staging: {source_path}. " + f"The file may have been moved or deleted by the download client. Error: {e}" + ) + status_callback("error", f"File not found: {source_path}. It may have been moved or deleted.") + return None except PermissionError as e: - logger.error(f"Permission denied staging file: {e}") - status_callback("error", f"Permission denied: {e}") + logger.error( + f"Permission denied staging file from {source_path}. " + f"Check that Shelfmark has read access to the download folder. Error: {e}" + ) + status_callback("error", f"Permission denied accessing {source_path}. Check folder permissions.") return None except Exception as e: - logger.error(f"Staging failed: {e}") + logger.error(f"Staging failed for {source_path}: {e}") status_callback("error", f"Failed to stage file: {e}") return None diff --git a/shelfmark/release_sources/prowlarr/source.py b/shelfmark/release_sources/prowlarr/source.py index 4a31b87f..8e172e77 100644 --- a/shelfmark/release_sources/prowlarr/source.py +++ b/shelfmark/release_sources/prowlarr/source.py @@ -306,8 +306,10 @@ class ProwlarrSource(ReleaseSource): # Build search query query_parts = [] - if book.title: - query_parts.append(book.title) + # Prefer search_title if available (cleaner title for searches) + search_title = book.search_title or book.title + if search_title: + query_parts.append(search_title) if book.authors: # Use first author only - authors may be a list or a single string # that contains multiple comma-separated names (from frontend) @@ -326,31 +328,38 @@ class ProwlarrSource(ReleaseSource): logger.warning("No search query available for book") return [] - # Get selected indexer IDs from config + # Get selected indexer IDs from config (None means search all) indexer_ids = self._get_selected_indexer_ids() - if not indexer_ids: - logger.warning("No indexers selected - configure indexers in Prowlarr settings") - return [] - # Get search categories based on content type # Audiobooks use 3030 (Audio/Audiobook), ebooks use 7000 (Books) search_categories = [3030] if content_type == "audiobook" else [7000] categories = None if expand_search else search_categories self.last_search_type = "expanded" if expand_search else "categories" - logger.debug(f"Searching Prowlarr: query='{query}', indexers={indexer_ids}, categories={categories}") + indexer_desc = f"indexers={indexer_ids}" if indexer_ids else "all enabled indexers" + logger.debug(f"Searching Prowlarr: query='{query}', {indexer_desc}, categories={categories}") def search_indexers(cats: Optional[List[int]]) -> List[dict]: - """Search all indexers with given categories, collecting results.""" + """Search indexers with given categories, collecting results.""" results = [] - for indexer_id in indexer_ids: + if indexer_ids: + # Search specific indexers one at a time + for indexer_id in indexer_ids: + try: + raw = client.search(query=query, indexer_ids=[indexer_id], categories=cats) + if raw: + results.extend(raw) + except Exception as e: + logger.warning(f"Search failed for indexer {indexer_id}: {e}") + else: + # Search all enabled indexers at once try: - raw = client.search(query=query, indexer_ids=[indexer_id], categories=cats) + raw = client.search(query=query, indexer_ids=None, categories=cats) if raw: results.extend(raw) except Exception as e: - logger.warning(f"Search failed for indexer {indexer_id}: {e}") + logger.warning(f"Search failed for all indexers: {e}") return results all_results = [] diff --git a/tests/prowlarr/test_sabnzbd_client.py b/tests/prowlarr/test_sabnzbd_client.py index 13104946..72ed0c4c 100644 --- a/tests/prowlarr/test_sabnzbd_client.py +++ b/tests/prowlarr/test_sabnzbd_client.py @@ -247,6 +247,57 @@ class TestSABnzbdClientGetStatus: assert status.complete is True assert status.file_path == "/downloads/complete/book" + def test_get_status_complete_empty_storage(self, monkeypatch): + """Test status for completed NZB with empty storage path. + + This can happen if SABnzbd category is misconfigured or files are + deleted after completion. The file_path should be empty string. + """ + config_values = { + "SABNZBD_URL": "http://localhost:8080", + "SABNZBD_API_KEY": "abc123", + "SABNZBD_CATEGORY": "cwabd", + } + monkeypatch.setattr( + "shelfmark.release_sources.prowlarr.clients.sabnzbd.config.get", + lambda key, default="": config_values.get(key, default), + ) + + def mock_api_call(mode, params=None): + if mode == "queue": + return {"queue": {"slots": []}} + if mode == "history": + return { + "history": { + "slots": [ + { + "nzo_id": "SABnzbd_nzo_abc123", + "status": "Completed", + "storage": "", # Empty storage path + } + ] + } + } + return {} + + from shelfmark.release_sources.prowlarr.clients.sabnzbd import ( + SABnzbdClient, + ) + + with patch.object(SABnzbdClient, "__init__", lambda x: None): + client = SABnzbdClient() + client.url = "http://localhost:8080" + client.api_key = "abc123" + client._category = "cwabd" + client._api_call = mock_api_call + + status = client.get_status("SABnzbd_nzo_abc123") + + assert status.progress == 100.0 + assert status.state_value == "complete" + assert status.complete is True + assert status.file_path == "" # Empty, not None + def test_get_status_failed(self, monkeypatch): """Test status for failed NZB.""" config_values = {