Proxy, prowlarr, ingest dir fixes (#437)

This commit is contained in:
Alex
2026-01-13 14:18:23 +00:00
committed by GitHub
parent a0079c5a7f
commit bd1ad3495c
12 changed files with 227 additions and 29 deletions
+2 -2
View File
@@ -770,7 +770,7 @@ def get(url: str, retry: Optional[int] = None, cancel_flag: Optional[Event] = No
cookies = get_cf_cookies_for_domain(urlparse(url).hostname or "")
if cookies:
try:
response = requests.get(url, cookies=cookies, proxies=get_proxies(), timeout=(5, 10))
response = requests.get(url, cookies=cookies, proxies=get_proxies(url), timeout=(5, 10))
if response.status_code == 200:
logger.debug("Cookies available after lock wait - skipped Chrome")
return response.text
@@ -1035,7 +1035,7 @@ def _try_with_cached_cookies(url: str, hostname: str) -> Optional[str]:
headers['User-Agent'] = stored_ua
logger.debug(f"Trying request with cached cookies: {url}")
response = requests.get(url, cookies=cookies, headers=headers, proxies=get_proxies(), timeout=(5, 10))
response = requests.get(url, cookies=cookies, headers=headers, proxies=get_proxies(url), timeout=(5, 10))
if response.status_code == 200:
logger.debug("Cached cookies worked, skipped Chrome bypass")
return response.text
+8
View File
@@ -493,6 +493,14 @@ def network_settings():
disabled_reason="Proxy settings are not used when Tor routing is enabled.",
show_when={"field": "PROXY_MODE", "value": "socks5"},
),
TextField(
key="NO_PROXY",
label="No Proxy",
description="Comma-separated hosts to bypass proxy (e.g., localhost,127.0.0.1,10.*,*.local)",
disabled=tor_overrides_network,
disabled_reason="Proxy settings are not used when Tor routing is enabled.",
show_when={"field": "PROXY_MODE", "value": ["http", "socks5"]},
),
]
+10
View File
@@ -333,6 +333,16 @@ def migrate_legacy_settings() -> None:
if "FILE_ORGANIZATION" in downloads_config or "DESTINATION" in downloads_config:
return
# Skip migration if no legacy settings exist (fresh install)
legacy_keys = {
"PROCESSING_MODE", "INGEST_DIR", "LIBRARY_PATH", "USE_BOOK_TITLE",
"LIBRARY_TEMPLATE", "PROCESSING_MODE_AUDIOBOOK", "INGEST_DIR_AUDIOBOOK",
"LIBRARY_PATH_AUDIOBOOK", "LIBRARY_TEMPLATE_AUDIOBOOK", "TORRENT_HARDLINK",
"USE_CONTENT_TYPE_DIRECTORIES",
}
if not any(key in downloads_config for key in legacy_keys):
return
migrated_downloads = {}
migrated_sources = {}
+3 -3
View File
@@ -206,7 +206,7 @@ def html_get_page(
# Try with CF cookies/UA if available (from previous bypass)
headers = {}
cookies = _apply_cf_bypass(current_url, headers)
response = requests.get(current_url, proxies=get_proxies(), timeout=REQUEST_TIMEOUT, cookies=cookies, headers=headers)
response = requests.get(current_url, proxies=get_proxies(current_url), timeout=REQUEST_TIMEOUT, cookies=cookies, headers=headers)
response.raise_for_status()
time.sleep(1)
return response.text
@@ -291,7 +291,7 @@ def download_url(
logger.info(f"Downloading: {current_url} (attempt {attempt + 1}/{MAX_DOWNLOAD_RETRIES})")
# Try with CF cookies/UA if available
cookies = _apply_cf_bypass(current_url, headers)
response = requests.get(current_url, stream=True, proxies=get_proxies(), timeout=REQUEST_TIMEOUT, cookies=cookies, headers=headers)
response = requests.get(current_url, stream=True, proxies=get_proxies(current_url), timeout=REQUEST_TIMEOUT, cookies=cookies, headers=headers)
response.raise_for_status()
if status_callback:
@@ -401,7 +401,7 @@ def _try_resume(
resume_headers = {**(base_headers or DOWNLOAD_HEADERS), 'Range': f'bytes={start_byte}-'}
cookies = _apply_cf_bypass(url, resume_headers)
response = requests.get(
url, stream=True, proxies=get_proxies(), timeout=REQUEST_TIMEOUT,
url, stream=True, proxies=get_proxies(url), timeout=REQUEST_TIMEOUT,
headers=resume_headers, cookies=cookies
)
+55 -4
View File
@@ -1,5 +1,6 @@
"""DNS rotation, mirror selection, and network utilities."""
import fnmatch
import requests
import urllib.request
from typing import Sequence, Tuple, Any, Union, cast, List, Optional, Callable
@@ -14,8 +15,58 @@ from shelfmark.core.config import config as app_config
from datetime import datetime, timedelta
def get_proxies() -> dict:
"""Get current proxy configuration from config singleton."""
def _get_no_proxy_patterns() -> List[str]:
"""Get list of NO_PROXY patterns from config."""
no_proxy = app_config.get("NO_PROXY", "")
if not no_proxy:
return []
return [p.strip().lower() for p in no_proxy.split(",") if p.strip()]
def should_bypass_proxy(url: str) -> bool:
"""Check if a URL should bypass the proxy based on NO_PROXY patterns.
Supports:
- Exact hostname match: localhost, myhost.local
- Wildcard prefix: *.local matches foo.local
- Wildcard suffix: 10.* matches 10.1.2.3
"""
if not url:
return False
patterns = _get_no_proxy_patterns()
if not patterns:
return False
# Extract hostname from URL
try:
parsed = urllib.parse.urlparse(url)
hostname = (parsed.hostname or "").lower()
except Exception:
return False
if not hostname:
return False
for pattern in patterns:
# Use fnmatch for wildcard matching (supports * and ?)
if fnmatch.fnmatch(hostname, pattern):
return True
return False
def get_proxies(url: str = "") -> dict:
"""Get current proxy configuration from config singleton.
Args:
url: Optional URL to check against NO_PROXY patterns.
If provided and matches a pattern, returns empty dict.
"""
# Check NO_PROXY bypass first
if url and should_bypass_proxy(url):
return {}
proxy_mode = app_config.get("PROXY_MODE", "none")
if proxy_mode == "socks5":
@@ -340,7 +391,7 @@ class DoHResolver:
response = self.session.get(
self.base_url,
params=params,
proxies=get_proxies(),
proxies=get_proxies(self.base_url),
timeout=10 # Increased from 5s to handle slow network conditions
)
response.raise_for_status()
@@ -834,7 +885,7 @@ def _initialize_aa_state() -> None:
logger.debug(f"AA_BASE_URL: auto, checking available urls {_aa_urls}")
for i, url in enumerate(_aa_urls):
try:
response = requests.get(url, proxies=get_proxies(), timeout=3)
response = requests.get(url, proxies=get_proxies(url), timeout=3)
if response.status_code == 200:
_current_aa_url_index = i
_aa_base_url = url
+1
View File
@@ -144,6 +144,7 @@ class BookMetadata:
genres: List[str] = field(default_factory=list)
source_url: Optional[str] = None # Link to book on provider's site
subtitle: Optional[str] = None # Book subtitle, if any
search_title: Optional[str] = None # Cleaner title for search queries (provider-specific)
# Provider-specific display fields for cards/lists
display_fields: List[DisplayField] = field(default_factory=list)
+30 -2
View File
@@ -95,6 +95,29 @@ def _build_source_url(slug: str) -> Optional[str]:
return f"https://hardcover.app/books/{slug}" if slug else None
def _compute_search_title(title: str, subtitle: Optional[str]) -> Optional[str]:
"""Compute a cleaner search title from title and subtitle.
When Hardcover uses the "Series: Book Title" format, the subtitle contains
the actual book title which is better for searching. For example:
- title: "Mistborn: The Final Empire"
- subtitle: "The Final Empire"
- search_title: "The Final Empire" (better for Prowlarr/indexer searches)
Skips subtitles that start with series position indicators like "Book One",
"Part 1", "Volume 2" as these are descriptors, not the actual title.
"""
if not subtitle or subtitle not in title:
return None
# Skip if subtitle starts with series position indicators
skip_prefixes = ('book ', 'part ', 'volume ')
if subtitle.lower().startswith(skip_prefixes):
return None
return subtitle
@register_provider_kwargs("hardcover")
def _hardcover_kwargs() -> Dict[str, Any]:
"""Provide Hardcover-specific constructor kwargs."""
@@ -564,6 +587,7 @@ class HardcoverProvider(MetadataProvider):
provider_id=str(book_id),
title=title,
subtitle=subtitle,
search_title=_compute_search_title(title, subtitle),
provider_display_name="Hardcover",
authors=authors,
cover_url=cover_url,
@@ -681,11 +705,15 @@ class HardcoverProvider(MetadataProvider):
if code3 and code3 not in titles_by_language:
titles_by_language[code3] = edition_title
title = book["title"]
subtitle = book.get("subtitle")
return BookMetadata(
provider="hardcover",
provider_id=str(book["id"]),
title=book["title"],
subtitle=book.get("subtitle"),
title=title,
subtitle=subtitle,
search_title=_compute_search_title(title, subtitle),
provider_display_name="Hardcover",
authors=authors,
isbn_10=isbn_10,
+1 -1
View File
@@ -696,7 +696,7 @@ def _extract_libgen_download_url(link: str, cancel_flag: Optional[Event] = None)
headers=downloader.DOWNLOAD_HEADERS,
timeout=(5, 10),
allow_redirects=True,
proxies=network.get_proxies(),
proxies=network.get_proxies(link),
)
if response.status_code != 200:
@@ -241,6 +241,7 @@ class SABnzbdClient(DownloadClient):
if slot.get("nzo_id") == download_id:
status_text = slot.get("status", "").upper()
storage = slot.get("storage", "")
logger.debug(f"SABnzbd history: {download_id} status={status_text} storage='{storage}'")
if status_text == "COMPLETED":
return DownloadStatus(
@@ -262,6 +263,7 @@ class SABnzbdClient(DownloadClient):
)
# Not found
logger.warning(f"SABnzbd: download {download_id} not found in queue or history")
return DownloadStatus.error("Download not found")
except Exception as e:
error_type = type(e).__name__
+43 -5
View File
@@ -178,6 +178,7 @@ class ProwlarrHandler(DownloadHandler):
) -> Optional[str]:
"""Poll the download client for progress and handle completion."""
try:
logger.debug(f"Starting poll for {download_id} (content_type={task.content_type})")
while not cancel_flag.is_set():
status = client.get_status(download_id)
progress_callback(status.progress)
@@ -185,13 +186,16 @@ class ProwlarrHandler(DownloadHandler):
# Check for completion
if status.complete:
if status.state == DownloadState.ERROR:
logger.error(f"Download {download_id} completed with error: {status.message}")
status_callback("error", status.message or "Download failed")
return None
# Download complete - break to handle file
logger.debug(f"Download {download_id} complete, file_path={status.file_path}")
break
# Check for error state
if status.state == DownloadState.ERROR:
logger.error(f"Download {download_id} error state: {status.message}")
status_callback("error", status.message or "Download failed")
client.remove(download_id, delete_files=True)
return None
@@ -214,11 +218,35 @@ class ProwlarrHandler(DownloadHandler):
# Handle completed file
source_path = client.get_download_path(download_id)
if not source_path:
status_callback("error", "Could not locate downloaded file")
logger.error(
f"Download client returned empty path for completed download. "
f"Client: {client.name}, ID: {download_id}. "
f"Check that the download client's completion folder is accessible to Shelfmark."
)
status_callback(
"error",
f"Download completed in {client.name} but path not returned. "
f"Check volume mappings and category settings."
)
return None
# Verify the path actually exists in our filesystem
source_path_obj = Path(source_path)
if not source_path_obj.exists():
logger.error(
f"Download path does not exist: {source_path}. "
f"Client: {client.name}, ID: {download_id}. "
f"The download client's path may not be mounted in Shelfmark's container. "
f"Ensure both containers use identical volume mappings for the download folder."
)
status_callback(
"error",
f"Path not accessible: {source_path}. Check volume mappings between {client.name} and Shelfmark."
)
return None
result = self._handle_completed_file(
source_path=Path(source_path),
source_path=source_path_obj,
protocol=protocol,
task=task,
status_callback=status_callback,
@@ -279,12 +307,22 @@ class ProwlarrHandler(DownloadHandler):
return str(staged_path)
except FileNotFoundError as e:
logger.error(
f"Source file not found during staging: {source_path}. "
f"The file may have been moved or deleted by the download client. Error: {e}"
)
status_callback("error", f"File not found: {source_path}. It may have been moved or deleted.")
return None
except PermissionError as e:
logger.error(f"Permission denied staging file: {e}")
status_callback("error", f"Permission denied: {e}")
logger.error(
f"Permission denied staging file from {source_path}. "
f"Check that Shelfmark has read access to the download folder. Error: {e}"
)
status_callback("error", f"Permission denied accessing {source_path}. Check folder permissions.")
return None
except Exception as e:
logger.error(f"Staging failed: {e}")
logger.error(f"Staging failed for {source_path}: {e}")
status_callback("error", f"Failed to stage file: {e}")
return None
+21 -12
View File
@@ -306,8 +306,10 @@ class ProwlarrSource(ReleaseSource):
# Build search query
query_parts = []
if book.title:
query_parts.append(book.title)
# Prefer search_title if available (cleaner title for searches)
search_title = book.search_title or book.title
if search_title:
query_parts.append(search_title)
if book.authors:
# Use first author only - authors may be a list or a single string
# that contains multiple comma-separated names (from frontend)
@@ -326,31 +328,38 @@ class ProwlarrSource(ReleaseSource):
logger.warning("No search query available for book")
return []
# Get selected indexer IDs from config
# Get selected indexer IDs from config (None means search all)
indexer_ids = self._get_selected_indexer_ids()
if not indexer_ids:
logger.warning("No indexers selected - configure indexers in Prowlarr settings")
return []
# Get search categories based on content type
# Audiobooks use 3030 (Audio/Audiobook), ebooks use 7000 (Books)
search_categories = [3030] if content_type == "audiobook" else [7000]
categories = None if expand_search else search_categories
self.last_search_type = "expanded" if expand_search else "categories"
logger.debug(f"Searching Prowlarr: query='{query}', indexers={indexer_ids}, categories={categories}")
indexer_desc = f"indexers={indexer_ids}" if indexer_ids else "all enabled indexers"
logger.debug(f"Searching Prowlarr: query='{query}', {indexer_desc}, categories={categories}")
def search_indexers(cats: Optional[List[int]]) -> List[dict]:
"""Search all indexers with given categories, collecting results."""
"""Search indexers with given categories, collecting results."""
results = []
for indexer_id in indexer_ids:
if indexer_ids:
# Search specific indexers one at a time
for indexer_id in indexer_ids:
try:
raw = client.search(query=query, indexer_ids=[indexer_id], categories=cats)
if raw:
results.extend(raw)
except Exception as e:
logger.warning(f"Search failed for indexer {indexer_id}: {e}")
else:
# Search all enabled indexers at once
try:
raw = client.search(query=query, indexer_ids=[indexer_id], categories=cats)
raw = client.search(query=query, indexer_ids=None, categories=cats)
if raw:
results.extend(raw)
except Exception as e:
logger.warning(f"Search failed for indexer {indexer_id}: {e}")
logger.warning(f"Search failed for all indexers: {e}")
return results
all_results = []
+51
View File
@@ -247,6 +247,57 @@ class TestSABnzbdClientGetStatus:
assert status.complete is True
assert status.file_path == "/downloads/complete/book"
def test_get_status_complete_empty_storage(self, monkeypatch):
"""Test status for completed NZB with empty storage path.
This can happen if SABnzbd category is misconfigured or files are
deleted after completion. The file_path should be empty string.
"""
config_values = {
"SABNZBD_URL": "http://localhost:8080",
"SABNZBD_API_KEY": "abc123",
"SABNZBD_CATEGORY": "cwabd",
}
monkeypatch.setattr(
"shelfmark.release_sources.prowlarr.clients.sabnzbd.config.get",
lambda key, default="": config_values.get(key, default),
)
def mock_api_call(mode, params=None):
if mode == "queue":
return {"queue": {"slots": []}}
if mode == "history":
return {
"history": {
"slots": [
{
"nzo_id": "SABnzbd_nzo_abc123",
"status": "Completed",
"storage": "", # Empty storage path
}
]
}
}
return {}
from shelfmark.release_sources.prowlarr.clients.sabnzbd import (
SABnzbdClient,
)
with patch.object(SABnzbdClient, "__init__", lambda x: None):
client = SABnzbdClient()
client.url = "http://localhost:8080"
client.api_key = "abc123"
client._category = "cwabd"
client._api_call = mock_api_call
status = client.get_status("SABnzbd_nzo_abc123")
assert status.progress == 100.0
assert status.state_value == "complete"
assert status.complete is True
assert status.file_path == "" # Empty, not None
def test_get_status_failed(self, monkeypatch):
"""Test status for failed NZB."""
config_values = {