From b97e48235b6fd7456148b1516b393aa8eeb3ebde Mon Sep 17 00:00:00 2001 From: Alex Date: Wed, 7 Jan 2026 19:41:46 +0000 Subject: [PATCH] Direct download tweaks (#408) - Simplified bypasser process, removed warmup functionality - Added dedicated fast download sources, tried first. --- .../bypass/internal_bypasser.py | 602 +++++++----------- cwa_book_downloader/config/settings.py | 289 ++++++--- cwa_book_downloader/core/mirrors.py | 213 +++++++ cwa_book_downloader/core/settings_registry.py | 37 +- cwa_book_downloader/download/http.py | 5 + cwa_book_downloader/download/network.py | 9 +- cwa_book_downloader/main.py | 11 +- .../release_sources/direct_download.py | 85 ++- docker-compose.dev.yml | 2 + .../settings/fields/OrderableListField.tsx | 270 ++++---- src/frontend/src/types/settings.ts | 1 + 11 files changed, 858 insertions(+), 666 deletions(-) create mode 100644 cwa_book_downloader/core/mirrors.py diff --git a/cwa_book_downloader/bypass/internal_bypasser.py b/cwa_book_downloader/bypass/internal_bypasser.py index f341fc17..904ec83e 100644 --- a/cwa_book_downloader/bypass/internal_bypasser.py +++ b/cwa_book_downloader/bypass/internal_bypasser.py @@ -1,6 +1,5 @@ import os import random -import signal import socket import subprocess import threading @@ -15,7 +14,7 @@ import requests from seleniumbase import Driver from cwa_book_downloader.bypass import BypassCancelledException -from cwa_book_downloader.bypass.fingerprint import clear_screen_size, get_screen_size +from cwa_book_downloader.bypass.fingerprint import get_screen_size from cwa_book_downloader.config import env from cwa_book_downloader.config.env import LOG_DIR from cwa_book_downloader.config.settings import RECORDING_DIR @@ -42,19 +41,12 @@ DDOS_GUARD_INDICATORS = [ "could not verify your browser automatically", ] -DRIVER = None DISPLAY = { "xvfb": None, "ffmpeg": None, } -LAST_USED = None LOCKED = threading.Lock() -# Flag to track if DNS rotated while a bypass was in progress -# Chrome will be restarted after the current operation completes -_dns_rotation_pending = False -_dns_rotation_lock = threading.Lock() - # Cookie storage - shared with requests library for Cloudflare bypass # Structure: {domain: {cookie_name: {value, expiry, ...}}} _cf_cookies: dict[str, dict] = {} @@ -730,70 +722,48 @@ def _build_host_resolver_rules() -> list[str]: DRIVER_RESET_ERRORS = {"WebDriverException", "SessionNotCreatedException", "TimeoutException", "MaxRetryError"} -def _get(url: str, retry: Optional[int] = None, cancel_flag: Optional[Event] = None) -> str: - """Fetch URL with Cloudflare bypass. Retries on failure.""" - retry = retry if retry is not None else app_config.MAX_RETRY +def _get(url: str, driver: Driver, cancel_flag: Optional[Event] = None) -> str: + """Fetch URL with Cloudflare bypass using provided driver.""" _check_cancellation(cancel_flag, "Bypass cancelled before starting") + logger.debug(f"SB_GET: {url}") + + hostname = urlparse(url).hostname or "" + if has_valid_cf_cookies(hostname): + reconnect_time = 1.0 + logger.debug(f"Using fast reconnect ({reconnect_time}s) - valid cookies exist") + else: + reconnect_time = app_config.DEFAULT_SLEEP + logger.debug(f"Using standard reconnect ({reconnect_time}s) - no cached cookies") + + logger.debug("Opening URL with SeleniumBase...") + driver.uc_open_with_reconnect(url, reconnect_time) + + _check_cancellation(cancel_flag, "Bypass cancelled after page load") + try: - logger.debug(f"SB_GET: {url}") - sb = _get_driver() - - hostname = urlparse(url).hostname or "" - if has_valid_cf_cookies(hostname): - reconnect_time = 1.0 - logger.debug(f"Using fast reconnect ({reconnect_time}s) - valid cookies exist") - else: - reconnect_time = app_config.DEFAULT_SLEEP - logger.debug(f"Using standard reconnect ({reconnect_time}s) - no cached cookies") - - logger.debug("Opening URL with SeleniumBase...") - sb.uc_open_with_reconnect(url, reconnect_time) - - _check_cancellation(cancel_flag, "Bypass cancelled after page load") - - try: - logger.debug(f"Page loaded - URL: {sb.get_current_url()}, Title: {sb.get_title()}") - except Exception as e: - logger.debug(f"Could not get page info: {e}") - - logger.debug("Starting bypass process...") - if _bypass(sb, cancel_flag=cancel_flag): - _extract_cookies_from_driver(sb, url) - return sb.page_source - - logger.warning("Bypass completed but page still shows protection") - try: - body = sb.get_text("body") - logger.debug(f"Page content: {body[:500]}..." if len(body) > 500 else body) - except Exception: - pass - - except BypassCancelledException: - raise + logger.debug(f"Page loaded - URL: {driver.get_current_url()}, Title: {driver.get_title()}") except Exception as e: - error_details = f"{type(e).__name__}: {e}" + logger.debug(f"Could not get page info: {e}") - if retry == 0: - logger.error(f"Failed after all retries: {error_details}") - logger.debug(f"Stack trace: {traceback.format_exc()}") - _reset_driver() - raise + logger.debug("Starting bypass process...") + if _bypass(driver, cancel_flag=cancel_flag): + _extract_cookies_from_driver(driver, url) + return driver.page_source - logger.warning(f"Bypass failed (retry {app_config.MAX_RETRY - retry + 1}/{app_config.MAX_RETRY}): {error_details}") - logger.debug(f"Stack trace: {traceback.format_exc()}") + logger.warning("Bypass completed but page still shows protection") + try: + body = driver.get_text("body") + logger.debug(f"Page content: {body[:500]}..." if len(body) > 500 else body) + except Exception: + pass - if type(e).__name__ in DRIVER_RESET_ERRORS: - logger.info("Restarting bypasser due to browser error...") - _reset_driver() + return "" - _check_cancellation(cancel_flag, "Bypass cancelled before retry") - return _get(url, retry - 1, cancel_flag) def get(url: str, retry: Optional[int] = None, cancel_flag: Optional[Event] = None) -> str: - """Fetch a URL with protection bypass.""" + """Fetch a URL with protection bypass. Creates fresh Chrome instance for each bypass.""" retry = retry if retry is not None else app_config.MAX_RETRY - global LAST_USED with LOCKED: # Try cookies first - another request may have completed bypass while waiting @@ -803,27 +773,55 @@ def get(url: str, retry: Optional[int] = None, cancel_flag: Optional[Event] = No response = requests.get(url, cookies=cookies, proxies=get_proxies(), timeout=(5, 10)) if response.status_code == 200: logger.debug("Cookies available after lock wait - skipped Chrome") - LAST_USED = time.time() return response.text except Exception: pass - result = _get(url, retry, cancel_flag) - LAST_USED = time.time() - return result + # Fresh Chrome for each bypass attempt + driver = None + try: + _ensure_display_initialized() + driver = _create_driver() -def _init_driver() -> Driver: - """Initialize the Chrome driver with undetected-chromedriver settings.""" - global DRIVER - if DRIVER: - _reset_driver() + for attempt in range(retry): + _check_cancellation(cancel_flag, "Bypass cancelled before attempt") + try: + result = _get(url, driver, cancel_flag) + if result: + return result + except BypassCancelledException: + raise + except Exception as e: + error_details = f"{type(e).__name__}: {e}" + logger.warning(f"Bypass failed (attempt {attempt + 1}/{retry}): {error_details}") + logger.debug(f"Stack trace: {traceback.format_exc()}") + + # On driver errors, quit and create fresh driver + if type(e).__name__ in DRIVER_RESET_ERRORS: + logger.info("Restarting Chrome due to browser error...") + _quit_driver(driver) + driver = _create_driver() + + logger.error(f"Bypass failed after {retry} attempts") + return "" + finally: + # Always quit Chrome when done + if driver: + _quit_driver(driver) + +def _create_driver() -> Driver: + """Create a fresh Chrome driver instance.""" chromium_args = _get_chromium_args() screen_width, screen_height = get_screen_size() - logger.debug(f"Initializing Chrome driver with args: {chromium_args}") + logger.debug(f"Creating Chrome driver with args: {chromium_args}") logger.debug(f"Browser screen size: {screen_width}x{screen_height}") + # Start FFmpeg recording if debug mode (record each bypass session) + if app_config.get("DEBUG", False) and DISPLAY["xvfb"] and not DISPLAY["ffmpeg"]: + _start_ffmpeg_recording() + driver = Driver( uc=True, headless=False, @@ -834,10 +832,174 @@ def _init_driver() -> Driver: chromium_arg=chromium_args, ) driver.set_page_load_timeout(60) - DRIVER = driver time.sleep(app_config.DEFAULT_SLEEP) + logger.info("Chrome browser ready") + logger.log_resource_usage() return driver + +def _start_ffmpeg_recording() -> None: + """Start FFmpeg screen recording for debug mode.""" + global DISPLAY + RECORDING_DIR.mkdir(parents=True, exist_ok=True) + display = DISPLAY["xvfb"] + timestamp = datetime.now().strftime("%y%m%d-%H%M%S") + output_file = RECORDING_DIR / f"screen_recording_{timestamp}.mp4" + + screen_width, screen_height = get_screen_size() + display_width = screen_width + 100 + display_height = screen_height + 150 + + ffmpeg_cmd = [ + "ffmpeg", "-y", "-f", "x11grab", + "-video_size", f"{display_width}x{display_height}", + "-i", f":{display.display}", + "-c:v", "libx264", "-preset", "ultrafast", + "-maxrate", "700k", "-bufsize", "1400k", "-crf", "36", + "-pix_fmt", "yuv420p", "-tune", "animation", + "-x264-params", "bframes=0:deblock=-1,-1", + "-r", "15", "-an", + output_file.as_posix(), + "-nostats", "-loglevel", "0" + ] + logger.debug("Starting FFmpeg recording to %s", output_file) + logger.debug_trace(f"FFmpeg command: {' '.join(ffmpeg_cmd)}") + DISPLAY["ffmpeg"] = subprocess.Popen(ffmpeg_cmd) + + +def _close_cdp_sockets() -> int: + """Find and close any sockets connected to CDP port 9222. + + This is a workaround for SeleniumBase not properly closing websocket + connections when using activate_cdp_mode(). Returns count of closed sockets. + """ + import os + closed = 0 + pid = os.getpid() + + try: + fd_path = f'/proc/{pid}/fd' + for fd_name in os.listdir(fd_path): + try: + fd = int(fd_name) + link = os.readlink(f'{fd_path}/{fd_name}') + if 'socket:' not in link: + continue + + # Check if this socket is connected to port 9222 (CDP) + # by reading /proc/net/tcp and matching inode + inode = link.split('[')[1].rstrip(']') + + with open('/proc/net/tcp', 'r') as f: + for line in f: + parts = line.split() + if len(parts) < 10: + continue + # Check if this is our socket and connects to port 9222 (0x2406) + if parts[9] == inode: + remote = parts[2] + remote_port = int(remote.split(':')[1], 16) + if remote_port == 9222: + logger.debug(f"Closing CDP socket fd={fd} inode={inode}") + os.close(fd) + closed += 1 + break + except (ValueError, OSError, IndexError): + continue + except Exception as e: + logger.debug(f"Error scanning for CDP sockets: {e}") + + return closed + + +def _quit_driver(driver: Driver) -> None: + """Quit Chrome driver and clean up resources. + + Proper cleanup sequence for SeleniumBase CDP mode: + 1. Stop CDP browser (closes websocket connections) + 2. Reconnect WebDriver + 3. Close window + 4. Quit driver + 5. Force-kill any lingering processes + + The CDP websocket connection must be explicitly closed before Chrome is killed, + otherwise the sockets end up in CLOSE_WAIT state causing gevent to busy-loop. + + References: + - https://github.com/seleniumbase/SeleniumBase/discussions/3768 + - https://www.selenium.dev/selenium/docs/api/py/selenium_webdriver_common_bidi/selenium.webdriver.common.bidi.cdp.html + """ + if driver is None: + return + + logger.debug("Quitting Chrome driver...") + + # Strategy 1: Stop CDP browser if in CDP mode (closes websocket connections) + # This is the proper SeleniumBase way to close CDP connections + try: + if hasattr(driver, 'cdp') and driver.cdp and hasattr(driver.cdp, 'driver'): + driver.cdp.driver.stop() + logger.debug("Stopped CDP browser (closed websocket)") + time.sleep(0.3) + except Exception as e: + logger.debug(f"CDP stop: {e}") + + # Strategy 2: Reconnect to re-establish WebDriver control before quitting + try: + driver.reconnect() + time.sleep(0.2) + except Exception as e: + logger.debug(f"Reconnect: {e}") + + # Strategy 3: Close the current window/tab + try: + driver.close() + time.sleep(0.2) + except Exception as e: + logger.debug(f"Close window: {e}") + + # Strategy 4: Fallback - explicitly close any remaining CDP sockets + # This catches any sockets that weren't closed by cdp.driver.stop() + closed = _close_cdp_sockets() + if closed: + logger.debug(f"Closed {closed} remaining CDP socket(s)") + + # Strategy 5: Standard quit + try: + driver.quit() + except Exception as e: + logger.debug(f"Quit: {e}") + + # Strategy 6: Force garbage collection + import gc + gc.collect() + + # Strategy 7: Force-kill any lingering Chrome/chromedriver processes + if env.DOCKERMODE: + time.sleep(0.3) + try: + subprocess.run(["pkill", "-9", "-f", "chrom"], capture_output=True, timeout=5) + except Exception as e: + logger.debug(f"pkill chrome: {e}") + + # Strategy 8: Stop ffmpeg recording if running + global DISPLAY + if DISPLAY.get("ffmpeg"): + try: + DISPLAY["ffmpeg"].terminate() + DISPLAY["ffmpeg"].wait(timeout=2) + logger.debug("Stopped ffmpeg recording") + except Exception as e: + logger.debug(f"ffmpeg terminate: {e}") + try: + DISPLAY["ffmpeg"].kill() + except Exception: + pass + DISPLAY["ffmpeg"] = None + + logger.log_resource_usage() + + def _ensure_display_initialized(): """Initialize virtual display if needed. Must be called with LOCKED held.""" global DISPLAY @@ -860,300 +1022,6 @@ def _ensure_display_initialized(): _reset_pyautogui_display_state() -def _get_driver(): - global DRIVER, DISPLAY, LAST_USED - logger.debug("Getting driver...") - LAST_USED = time.time() - - _ensure_display_initialized() - - # Start FFmpeg recording on first actual bypass request (not during warmup) - # This ensures we only record active bypass sessions, not idle time - if app_config.get("DEBUG", False) and DISPLAY["xvfb"] and not DISPLAY["ffmpeg"]: - RECORDING_DIR.mkdir(parents=True, exist_ok=True) - display = DISPLAY["xvfb"] - timestamp = datetime.now().strftime("%y%m%d-%H%M%S") - output_file = RECORDING_DIR / f"screen_recording_{timestamp}.mp4" - - # Get the display size (screen size + padding) - screen_width, screen_height = get_screen_size() - display_width = screen_width + 100 - display_height = screen_height + 150 - - ffmpeg_cmd = [ - "ffmpeg", - "-y", - "-f", "x11grab", - "-video_size", f"{display_width}x{display_height}", - "-i", f":{display.display}", - "-c:v", "libx264", - "-preset", "ultrafast", # or "veryfast" (trade speed for slightly better compression) - "-maxrate", "700k", # Slightly higher bitrate for text clarity - "-bufsize", "1400k", # Buffer size (2x maxrate) - "-crf", "36", # Adjust as needed: higher = smaller, lower = better quality (23 is visually lossless) - "-pix_fmt", "yuv420p", # Crucial for compatibility with most players - "-tune", "animation", # Optimize encoding for screen content - "-x264-params", "bframes=0:deblock=-1,-1", # Optimize for text, disable b-frames and deblocking - "-r", "15", # Reduce frame rate (if content allows) - "-an", # Disable audio recording (if not needed) - output_file.as_posix(), - "-nostats", "-loglevel", "0" - ] - logger.debug("Starting FFmpeg recording to %s", output_file) - logger.debug_trace(f"FFmpeg command: {' '.join(ffmpeg_cmd)}") - DISPLAY["ffmpeg"] = subprocess.Popen(ffmpeg_cmd) - - if not DRIVER: - return _init_driver() - - # Verify the existing driver is actually healthy (browser process still alive) - if not _is_driver_healthy(): - logger.warning("Existing driver is unhealthy (browser may have crashed), reinitializing...") - _reset_driver() - _ensure_display_initialized() # Display was shut down by _reset_driver, reinitialize it - return _init_driver() - - logger.log_resource_usage() - return DRIVER - -def _reset_driver() -> None: - """Reset the browser driver and cleanup all associated processes.""" - logger.log_resource_usage() - logger.info("Shutting down Cloudflare bypasser...") - global DRIVER, DISPLAY - - clear_screen_size() - - if DRIVER: - try: - DRIVER.quit() - except Exception as e: - logger.warning(f"Error quitting driver: {e}") - DRIVER = None - - if DISPLAY["xvfb"]: - try: - DISPLAY["xvfb"].stop() - except Exception as e: - logger.warning(f"Error stopping display: {e}") - DISPLAY["xvfb"] = None - - if DISPLAY["ffmpeg"]: - try: - DISPLAY["ffmpeg"].send_signal(signal.SIGINT) - except Exception as e: - logger.debug(f"Error stopping ffmpeg: {e}") - DISPLAY["ffmpeg"] = None - - time.sleep(0.5) - for process in ["Xvfb", "ffmpeg", "chrom"]: - try: - os.system(f"pkill -f {process}") - except Exception as e: - logger.debug(f"Error killing {process}: {e}") - - time.sleep(0.5) - logger.info("Cloudflare bypasser shut down") - logger.log_resource_usage() - -def _restart_chrome_only() -> None: - """Restart Chrome (not the display) to pick up new DNS settings. - - Called when DNS provider rotates. Display is kept running to avoid slower full restart. - """ - global DRIVER, LAST_USED - - logger.debug("Restarting Chrome to apply new DNS settings...") - - if DRIVER: - try: - DRIVER.quit() - except Exception as e: - logger.debug(f"Error quitting driver during DNS rotation: {e}") - DRIVER = None - - try: - os.system("pkill -f chrom") - except Exception as e: - logger.debug(f"Error killing chrome processes: {e}") - - time.sleep(0.5) - - try: - _init_driver() - LAST_USED = time.time() - logger.debug("Chrome restarted with updated DNS settings") - except Exception as e: - logger.warning(f"Failed to restart Chrome after DNS rotation: {e}") - - -def _on_dns_rotation(provider_name: str, servers: list, doh_url: str) -> None: - """Callback invoked when network.py rotates DNS provider. - - Schedules an async Chrome restart to avoid blocking the current request. - """ - global _dns_rotation_pending - - if DRIVER is None: - return - - with _dns_rotation_lock: - if _dns_rotation_pending: - return - _dns_rotation_pending = True - - def _async_restart(): - global _dns_rotation_pending - logger.debug(f"DNS rotated to {provider_name} - restarting Chrome in background") - with LOCKED: - with _dns_rotation_lock: - _dns_rotation_pending = False - _restart_chrome_only() - - threading.Thread(target=_async_restart, daemon=True).start() - - -def _cleanup_driver() -> None: - """Reset driver after inactivity timeout. - - Uses 4x longer timeout when UI clients are connected. - """ - global LAST_USED - - try: - from cwa_book_downloader.api.websocket import ws_manager - has_active_clients = ws_manager.has_active_connections() - except ImportError: - ws_manager = None - has_active_clients = False - - timeout_minutes = app_config.BYPASS_RELEASE_INACTIVE_MIN - if has_active_clients: - timeout_minutes *= 4 - - with LOCKED: - if not LAST_USED or time.time() - LAST_USED < timeout_minutes * 60: - return - - logger.info(f"Bypasser idle for {timeout_minutes} min - shutting down") - _reset_driver() - LAST_USED = None - logger.log_resource_usage() - - if has_active_clients and ws_manager: - ws_manager.request_warmup_on_next_connect() - logger.debug("Requested warmup on next client connect") - -def _cleanup_loop() -> None: - """Background loop that periodically checks for idle timeout.""" - while True: - _cleanup_driver() - time.sleep(max(app_config.BYPASS_RELEASE_INACTIVE_MIN / 2, 1)) - - -def _init_cleanup_thread() -> None: - """Start the background cleanup thread.""" - threading.Thread(target=_cleanup_loop, daemon=True).start() - -def _should_warmup() -> bool: - """Check if warmup should proceed based on configuration.""" - if not app_config.get("BYPASS_WARMUP_ON_CONNECT", True): - logger.debug("Bypasser warmup disabled via config") - return False - if not env.DOCKERMODE: - logger.debug("Bypasser warmup skipped - not in Docker mode") - return False - if not app_config.get("USE_CF_BYPASS", True): - logger.debug("Bypasser warmup skipped - CF bypass disabled") - return False - if app_config.get("AA_DONATOR_KEY", ""): - logger.debug("Bypasser warmup skipped - AA donator key set") - return False - return True - - -def warmup() -> None: - """Pre-initialize the virtual display and Chrome browser. - - Called when a user connects to the web UI. Skipped if warmup is disabled, - not in Docker mode, CF bypass is disabled, or AA donator key is set. - """ - global LAST_USED - - if not _should_warmup(): - return - - with LOCKED: - if is_warmed_up(): - logger.debug("Bypasser already warmed up") - return - - _cleanup_orphan_processes() - - if DRIVER is not None or DISPLAY["xvfb"] is not None: - logger.info("Resetting stale bypasser state before warmup...") - _reset_driver() - - logger.info("Warming up Cloudflare bypasser...") - - try: - _ensure_display_initialized() - - if DRIVER is None: - logger.info("Pre-initializing Chrome browser...") - _init_driver() - LAST_USED = time.time() - logger.info("Chrome browser ready") - - logger.info("Bypasser warmup complete") - logger.log_resource_usage() - - except Exception as e: - logger.warning(f"Failed to warm up bypasser: {e}") - -def _is_driver_healthy() -> bool: - """Check if the Chrome driver is responsive (not just non-None).""" - if DRIVER is None: - return False - - try: - DRIVER.get_current_url() - return True - except Exception as e: - logger.warning(f"Driver health check failed: {type(e).__name__}: {e}") - return False - - -def is_warmed_up() -> bool: - """Check if the bypasser is fully warmed up (display and browser initialized).""" - if DISPLAY["xvfb"] is None or DRIVER is None: - return False - return _is_driver_healthy() - - -def shutdown_if_idle() -> None: - """Start the inactivity countdown when all WebSocket clients disconnect. - - Sets LAST_USED to start the timer. The cleanup loop shuts down after - BYPASS_RELEASE_INACTIVE_MIN minutes of inactivity. - """ - global LAST_USED - - with LOCKED: - if not is_warmed_up(): - logger.debug("Bypasser already shut down") - return - - LAST_USED = time.time() - logger.info(f"All clients disconnected - shutdown after {app_config.BYPASS_RELEASE_INACTIVE_MIN} min of inactivity") - -_init_cleanup_thread() - -# Register for DNS rotation notifications (Chrome restarts with new DNS settings) -if app_config.get("USE_CF_BYPASS", True) and not app_config.get("USING_EXTERNAL_BYPASSER", False): - network.register_dns_rotation_callback(_on_dns_rotation) - - def _try_with_cached_cookies(url: str, hostname: str) -> Optional[str]: """Attempt request with cached cookies before using Chrome.""" cookies = get_cf_cookies_for_domain(hostname) diff --git a/cwa_book_downloader/config/settings.py b/cwa_book_downloader/config/settings.py index 48b1bb80..3b1508a9 100644 --- a/cwa_book_downloader/config/settings.py +++ b/cwa_book_downloader/config/settings.py @@ -151,34 +151,67 @@ def _get_release_source_options(): _LANGUAGE_OPTIONS = [{"value": lang["code"], "label": lang["language"]} for lang in _SUPPORTED_BOOK_LANGUAGE] -# Default AA mirror URLs (hardcoded base list) -_DEFAULT_AA_URLS = [ - "https://annas-archive.se", - "https://annas-archive.li", - "https://annas-archive.pm", - "https://annas-archive.in", -] - - def _get_aa_base_url_options(): """Build AA URL options dynamically, including additional mirrors from config.""" - from cwa_book_downloader.core.config import config + from cwa_book_downloader.core.mirrors import DEFAULT_AA_MIRRORS, get_aa_mirrors options = [{"value": "auto", "label": "Auto (Recommended)"}] + # Get all mirrors (defaults + custom) + all_mirrors = get_aa_mirrors() + + for url in all_mirrors: + domain = url.replace("https://", "").replace("http://", "") + is_custom = url not in DEFAULT_AA_MIRRORS + label = f"{domain} (custom)" if is_custom else domain + options.append({"value": url, "label": label}) + + return options + + +def _get_zlib_mirror_options(): + """Build Z-Library mirror options for SelectField.""" + from cwa_book_downloader.core.mirrors import DEFAULT_ZLIB_MIRRORS + from cwa_book_downloader.core.config import config + + options = [] + # Add default mirrors - for url in _DEFAULT_AA_URLS: - # Extract domain for label + for url in DEFAULT_ZLIB_MIRRORS: domain = url.replace("https://", "").replace("http://", "") options.append({"value": url, "label": domain}) - # Add any additional mirrors from config - additional = config.get("AA_ADDITIONAL_URLS", "") + # Add custom mirrors + additional = config.get("ZLIB_ADDITIONAL_URLS", "") if additional: for url in additional.split(","): url = url.strip() - if url and url not in _DEFAULT_AA_URLS: - domain = url.replace("https://", "").replace("http://", "") + if url and url not in DEFAULT_ZLIB_MIRRORS: + domain = url.replace("https://", "").replace("http://", "").split("/")[0] + options.append({"value": url, "label": f"{domain} (custom)"}) + + return options + + +def _get_welib_mirror_options(): + """Build Welib mirror options for SelectField.""" + from cwa_book_downloader.core.mirrors import DEFAULT_WELIB_MIRRORS + from cwa_book_downloader.core.config import config + + options = [] + + # Add default mirrors + for url in DEFAULT_WELIB_MIRRORS: + domain = url.replace("https://", "").replace("http://", "") + options.append({"value": url, "label": domain}) + + # Add custom mirrors + additional = config.get("WELIB_ADDITIONAL_URLS", "") + if additional: + for url in additional.split(","): + url = url.strip() + if url and url not in DEFAULT_WELIB_MIRRORS: + domain = url.replace("https://", "").replace("http://", "").split("/")[0] options.append({"value": url, "label": f"{domain} (custom)"}) return options @@ -624,98 +657,118 @@ def download_settings(): ] -def _get_source_priority_options(): - """Build source priority options with dynamic disabled states.""" +def _get_fast_source_options(): + """Fast download sources - display only, not configurable.""" from cwa_book_downloader.core.config import config has_donator_key = bool(config.get("AA_DONATOR_KEY", "")) - use_cf_bypass = config.get("USE_CF_BYPASS", True) - using_external_bypasser = config.get("USING_EXTERNAL_BYPASSER", False) - has_internal_bypasser = use_cf_bypass and not using_external_bypasser return [ { "id": "aa-fast", "label": "Anna's Archive (Fast)", "description": "Fast downloads for donators", + "isPinned": True, "isLocked": not has_donator_key, "disabledReason": "Requires AA Donator Key" if not has_donator_key else None, }, { - "id": "welib", - "label": "Welib", - "description": "Alternative mirror with good availability", - "isLocked": not has_internal_bypasser, - "disabledReason": "Requires internal bypasser" if not has_internal_bypasser else None, + "id": "libgen", + "label": "Library Genesis", + "description": "Instant downloads, no bypass needed", + "isPinned": True, }, + ] + + +def _get_fast_source_defaults(): + """Default values for fast sources display.""" + return [ + {"id": "aa-fast", "enabled": True}, + {"id": "libgen", "enabled": True}, + ] + + +def _get_slow_source_options(): + """Slow download sources - configurable order. All require bypasser.""" + from cwa_book_downloader.core.config import config + + bypass_enabled = config.get("USE_CF_BYPASS", True) + locked = not bypass_enabled + disabled_reason = "Requires Cloudflare bypass" if locked else None + + return [ { "id": "aa-slow-nowait", "label": "Anna's Archive (Slowest, No Waitlist)", - "description": "Partner servers without countdown", + "description": "Partner servers", + "isLocked": locked, + "disabledReason": disabled_reason, }, { "id": "aa-slow-wait", - "label": "Anna's Archive (Slow, Waitlist)", + "label": "Anna's Archive (Slow with Waitlist)", "description": "Partner servers with countdown timer", + "isLocked": locked, + "disabledReason": disabled_reason, }, { - "id": "libgen", - "label": "Libgen (Fast)", - "description": "Library Genesis. Fast downloads, no bypass needed.", + "id": "welib", + "label": "Welib", + "description": "Alternative mirror", + "isLocked": locked, + "disabledReason": disabled_reason, }, { "id": "zlib", "label": "Z-Library", - "description": "Z-Library mirrors (requires Cloudflare bypass)", - "isLocked": not has_internal_bypasser, - "disabledReason": "Requires internal bypasser" if not has_internal_bypasser else None, + "description": "Alternative mirror", + "isLocked": locked, + "disabledReason": disabled_reason, }, ] -def _get_default_source_priority(): - """Default source priority order, respecting legacy env vars. +def _get_slow_source_defaults(): + """Default source priority order for slow sources.""" + from cwa_book_downloader.config.env import _LEGACY_ALLOW_USE_WELIB - ALLOW_USE_WELIB (default true) controls whether welib is enabled. - PRIORITIZE_WELIB (default false) controls whether welib is moved to position 1. - """ - from cwa_book_downloader.config.env import _LEGACY_PRIORITIZE_WELIB, _LEGACY_ALLOW_USE_WELIB - - welib_entry = {"id": "welib", "enabled": _LEGACY_ALLOW_USE_WELIB} - - priority = [ - {"id": "aa-fast", "enabled": True}, - {"id": "libgen", "enabled": True}, + return [ {"id": "aa-slow-nowait", "enabled": True}, {"id": "aa-slow-wait", "enabled": True}, + {"id": "welib", "enabled": _LEGACY_ALLOW_USE_WELIB}, + {"id": "zlib", "enabled": True}, ] - if _LEGACY_PRIORITIZE_WELIB: - priority.insert(1, welib_entry) # After aa-fast - else: - priority.append(welib_entry) # Before zlib - - # Z-Library last - it's quite brittle - priority.append({"id": "zlib", "enabled": True}) - - return priority - @register_settings("download_sources", "Download Sources", icon="download", order=21, group="direct_download") def download_source_settings(): """Settings for download source behavior.""" return [ + PasswordField( + key="AA_DONATOR_KEY", + label="Anna's Archive Donator Key", + description="Enables fast downloads from Anna's Archive.", + ), HeadingField( key="source_priority_heading", title="Source Priority", - description="Configure which download sources to use and in what order.", + description="Sources are tried in order until a download succeeds.", + ), + OrderableListField( + key="FAST_SOURCES_DISPLAY", + label="Fast downloads", + description="Always tried first, no waiting or bypass required.", + options=_get_fast_source_options, + default=_get_fast_source_defaults(), + env_supported=False, ), OrderableListField( key="SOURCE_PRIORITY", - label="Download Source Order", - description="Drag to reorder. Sources are tried from top to bottom until a download succeeds.", - options=_get_source_priority_options, - default=_get_default_source_priority(), + label="Slow downloads", + description="Fallback sources, may have waiting. Requires bypasser. Drag to reorder.", + options=_get_slow_source_options, + default=_get_slow_source_defaults(), ), NumberField( key="MAX_RETRY", @@ -733,29 +786,6 @@ def download_source_settings(): min_value=1, max_value=60, ), - HeadingField( - key="aa_settings_heading", - title="Anna's Archive", - description="Configure Anna's Archive mirror and donator settings.", - ), - SelectField( - key="AA_BASE_URL", - label="Anna's Archive URL", - description="Primary Anna's Archive mirror to use. 'auto' selects automatically. Custom mirrors added below will appear here.", - options=_get_aa_base_url_options, # Callable - includes additional mirrors dynamically - default="auto", - ), - TextField( - key="AA_ADDITIONAL_URLS", - label="Additional AA Mirrors", - description="Comma-separated list of additional Anna's Archive mirror URLs.", - placeholder="https://example.com,https://another.com", - ), - PasswordField( - key="AA_DONATOR_KEY", - label="Anna's Archive Donator Key", - description="Optional donator key for faster downloads from Anna's Archive.", - ), HeadingField( key="content_type_routing_heading", title="Content-Type Routing", @@ -829,20 +859,6 @@ def cloudflare_bypass_settings(): default=True, requires_restart=True, ), - CheckboxField( - key="BYPASS_WARMUP_ON_CONNECT", - label="Warmup on Connect", - description="Pre-warm the bypasser when user connects to Web App UI", - default=True, - ), - NumberField( - key="BYPASS_RELEASE_INACTIVE_MIN", - label="Release Inactive (minutes)", - description="Release bypasser resources after this many minutes of inactivity.", - default=5, - min_value=1, - max_value=60, - ), CheckboxField( key="USING_EXTERNAL_BYPASSER", label="Use External Bypasser", @@ -881,6 +897,83 @@ def cloudflare_bypass_settings(): ] +@register_settings("mirrors", "Mirrors", icon="globe", order=23, group="direct_download") +def mirror_settings(): + """Configure download source mirrors.""" + from cwa_book_downloader.core.mirrors import DEFAULT_ZLIB_MIRRORS, DEFAULT_WELIB_MIRRORS + + return [ + # === ANNA'S ARCHIVE === + HeadingField( + key="aa_mirrors_heading", + title="Anna's Archive", + description="Primary mirror with auto-probe on startup. Additional mirrors used as fallback.", + ), + SelectField( + key="AA_BASE_URL", + label="Primary Mirror", + description="Select 'Auto' to probe mirrors on startup, or choose a specific mirror.", + options=_get_aa_base_url_options, + default="auto", + ), + TextField( + key="AA_ADDITIONAL_URLS", + label="Additional Mirrors", + description="Comma-separated list of custom Anna's Archive mirror URLs.", + ), + + # === LIBGEN === + HeadingField( + key="libgen_mirrors_heading", + title="LibGen", + description="All mirrors are tried during download until one succeeds. Defaults: libgen.gl, libgen.li, libgen.bz, libgen.la, libgen.vg", + ), + TextField( + key="LIBGEN_ADDITIONAL_URLS", + label="Additional Mirrors", + description="Comma-separated list of custom LibGen mirrors to add to the defaults.", + ), + + # === Z-LIBRARY === + HeadingField( + key="zlib_mirrors_heading", + title="Z-Library", + description="Z-Library requires Cloudflare bypass. Only the primary mirror is used.", + ), + SelectField( + key="ZLIB_PRIMARY_URL", + label="Primary Mirror", + description="Z-Library mirror to use for downloads.", + options=_get_zlib_mirror_options, + default=DEFAULT_ZLIB_MIRRORS[0], + ), + TextField( + key="ZLIB_ADDITIONAL_URLS", + label="Additional Mirrors", + description="Comma-separated list of custom Z-Library mirror URLs.", + ), + + # === WELIB === + HeadingField( + key="welib_mirrors_heading", + title="Welib", + description="Welib requires Cloudflare bypass. Only the primary mirror is used.", + ), + SelectField( + key="WELIB_PRIMARY_URL", + label="Primary Mirror", + description="Welib mirror to use for downloads.", + options=_get_welib_mirror_options, + default=DEFAULT_WELIB_MIRRORS[0], + ), + TextField( + key="WELIB_ADDITIONAL_URLS", + label="Additional Mirrors", + description="Comma-separated list of custom Welib mirror URLs.", + ), + ] + + @register_settings("advanced", "Advanced", icon="cog", order=15) def advanced_settings(): """Advanced settings for power users.""" diff --git a/cwa_book_downloader/core/mirrors.py b/cwa_book_downloader/core/mirrors.py new file mode 100644 index 00000000..57cf8577 --- /dev/null +++ b/cwa_book_downloader/core/mirrors.py @@ -0,0 +1,213 @@ +"""Centralized mirror configuration for all download sources.""" + +from typing import List + +# Lazy import to avoid circular imports +_config_module = None + + +def _get_config(): + """Lazy import of config module to avoid circular imports.""" + global _config_module + if _config_module is None: + from cwa_book_downloader.core.config import config + _config_module = config + return _config_module + + +# Default mirror lists (hardcoded fallbacks) +DEFAULT_AA_MIRRORS = [ + "https://annas-archive.se", + "https://annas-archive.li", + "https://annas-archive.pm", + "https://annas-archive.in", +] + +DEFAULT_LIBGEN_MIRRORS = [ + "https://libgen.gl", + "https://libgen.li", + "https://libgen.bz", + "https://libgen.la", + "https://libgen.vg", +] + +DEFAULT_ZLIB_MIRRORS = [ + "https://z-lib.fm", + "https://z-lib.gs", + "https://z-lib.id", + "https://z-library.sk", + "https://zlibrary-global.se", +] + +DEFAULT_WELIB_MIRRORS = [ + "https://welib.org", +] + + +def get_aa_mirrors() -> List[str]: + """ + Get Anna's Archive mirrors from config + defaults. + + Returns: + List of AA mirror URLs, starting with defaults then custom additions. + """ + mirrors = list(DEFAULT_AA_MIRRORS) + config = _get_config() + + additional = config.get("AA_ADDITIONAL_URLS", "") + if additional: + for url in additional.split(","): + url = url.strip() + if url and url not in mirrors: + mirrors.append(url) + + return mirrors + + +def get_libgen_mirrors() -> List[str]: + """ + Get LibGen mirrors: defaults + any additional from config. + + Returns: + List of LibGen mirror URLs (defaults first, then custom additions). + """ + mirrors = list(DEFAULT_LIBGEN_MIRRORS) + config = _get_config() + + additional = config.get("LIBGEN_ADDITIONAL_URLS", "") + if additional: + for url in additional.split(","): + url = url.strip() + if url and url not in mirrors: + mirrors.append(url) + + return mirrors + + +def get_zlib_mirrors() -> List[str]: + """ + Get Z-Library mirrors, with primary first. + + Returns: + List of Z-Library mirror URLs, primary first. + """ + config = _get_config() + + primary = config.get("ZLIB_PRIMARY_URL", DEFAULT_ZLIB_MIRRORS[0]) + mirrors = [primary] + + # Add other defaults (excluding primary) + for url in DEFAULT_ZLIB_MIRRORS: + if url != primary: + mirrors.append(url) + + # Add custom mirrors + additional = config.get("ZLIB_ADDITIONAL_URLS", "") + if additional: + for url in additional.split(","): + url = url.strip() + if url and url not in mirrors: + mirrors.append(url) + + return mirrors + + +def get_zlib_primary_url() -> str: + """ + Get the primary Z-Library mirror URL. + + Returns: + Primary Z-Library mirror URL. + """ + config = _get_config() + return config.get("ZLIB_PRIMARY_URL", DEFAULT_ZLIB_MIRRORS[0]) + + +def get_zlib_url_template() -> str: + """ + Get Z-Library URL template using configured primary mirror. + + Returns: + URL template with {md5} placeholder. + """ + primary = get_zlib_primary_url() + return f"{primary}/md5/{{md5}}" + + +def get_welib_mirrors() -> List[str]: + """ + Get Welib mirrors, with primary first. + + Returns: + List of Welib mirror URLs, primary first. + """ + config = _get_config() + + primary = config.get("WELIB_PRIMARY_URL", DEFAULT_WELIB_MIRRORS[0]) + mirrors = [primary] + + # Add other defaults (excluding primary) + for url in DEFAULT_WELIB_MIRRORS: + if url != primary: + mirrors.append(url) + + # Add custom mirrors + additional = config.get("WELIB_ADDITIONAL_URLS", "") + if additional: + for url in additional.split(","): + url = url.strip() + if url and url not in mirrors: + mirrors.append(url) + + return mirrors + + +def get_welib_primary_url() -> str: + """ + Get the primary Welib mirror URL. + + Returns: + Primary Welib mirror URL. + """ + config = _get_config() + return config.get("WELIB_PRIMARY_URL", DEFAULT_WELIB_MIRRORS[0]) + + +def get_welib_url_template() -> str: + """ + Get Welib URL template using configured primary mirror. + + Returns: + URL template with {md5} placeholder. + """ + primary = get_welib_primary_url() + return f"{primary}/md5/{{md5}}" + + +def get_zlib_cookie_domains() -> set: + """ + Get set of Z-Library domains that need full cookie handling. + + Used by internal_bypasser for CF bypass cookie management. + + Returns: + Set of domain strings (without protocol). + """ + domains = set() + + # Add all default domains + for url in DEFAULT_ZLIB_MIRRORS: + domain = url.replace("https://", "").replace("http://", "").split("/")[0] + domains.add(domain) + + # Add custom domains + config = _get_config() + additional = config.get("ZLIB_ADDITIONAL_URLS", "") + if additional: + for url in additional.split(","): + url = url.strip() + if url: + domain = url.replace("https://", "").replace("http://", "").split("/")[0] + domains.add(domain) + + return domains diff --git a/cwa_book_downloader/core/settings_registry.py b/cwa_book_downloader/core/settings_registry.py index 35fe9b94..a062f91e 100644 --- a/cwa_book_downloader/core/settings_registry.py +++ b/cwa_book_downloader/core/settings_registry.py @@ -85,7 +85,9 @@ class MultiSelectField(FieldBase): @dataclass class OrderableListField(FieldBase): # Options can be a list or a callable that returns a list (for lazy evaluation) - # Each option: {id, label, description?, disabledReason?, isLocked?} + # Each option: {id, label, description?, disabledReason?, isLocked?, section?, isPinned?} + # - isLocked: toggle is disabled (can't enable/disable) + # - isPinned: can't be reordered (but toggle may still work if not also isLocked) options: Any = field(default_factory=list) # Default value: [{id, enabled}, ...] in priority order default: List[Dict[str, Any]] = field(default_factory=list) @@ -310,7 +312,6 @@ def sync_env_to_config() -> None: logger.debug(f"Synced {len(values_to_sync)} ENV values to {tab.name} config: {list(values_to_sync.keys())}") migrate_legacy_settings() - migrate_libgen_priority() def migrate_legacy_settings() -> None: @@ -412,38 +413,6 @@ def migrate_legacy_settings() -> None: logger.info(f"Migrated content-type routing settings: {list(migrated_sources.keys())}") -def migrate_libgen_priority() -> None: - """One-time migration to move libgen to position 1 (after aa-fast).""" - source_config = load_config_file("download_sources") - - if source_config.get("_LIBGEN_PRIORITY_MIGRATED"): - return - - priority = source_config.get("SOURCE_PRIORITY") - if not priority: - save_config_file("download_sources", {"_LIBGEN_PRIORITY_MIGRATED": True}) - return - - libgen_index = None - for i, entry in enumerate(priority): - if entry.get("id") == "libgen": - libgen_index = i - break - - if libgen_index is None or libgen_index == 1: - save_config_file("download_sources", {"_LIBGEN_PRIORITY_MIGRATED": True}) - return - - libgen_entry = priority.pop(libgen_index) - priority.insert(1, libgen_entry) - - save_config_file("download_sources", { - "SOURCE_PRIORITY": priority, - "_LIBGEN_PRIORITY_MIGRATED": True, - }) - logger.info("Migrated SOURCE_PRIORITY: moved libgen to position 1") - - def get_setting_value(field: SettingsField, tab_name: str) -> Any: if isinstance(field, (ActionButton, HeadingField)): return None # Actions and headings don't have values diff --git a/cwa_book_downloader/download/http.py b/cwa_book_downloader/download/http.py index 6d306d52..36a3a40a 100644 --- a/cwa_book_downloader/download/http.py +++ b/cwa_book_downloader/download/http.py @@ -175,6 +175,7 @@ def html_get_page( use_bypasser: bool = False, selector: Optional[network.AAMirrorSelector] = None, cancel_flag: Optional[Event] = None, + status_callback: Optional[Callable[[str, Optional[str]], None]] = None, ) -> str: """Fetch HTML content from a URL with retry mechanism.""" retry = retry if retry is not None else app_config.MAX_RETRY @@ -192,6 +193,8 @@ def html_get_page( try: if use_bypasser_now and _is_cf_bypass_enabled(): logger.debug(f"GET (bypasser): {current_url}") + if status_callback: + status_callback("resolving", "Bypassing protection") try: result = get_bypassed_page(current_url, selector, cancel_flag) return result or "" @@ -223,6 +226,8 @@ def html_get_page( logger.debug(f"403 but cookies now available - retrying with cookies: {current_url}") continue logger.info(f"403 detected; switching to bypasser: {current_url}") + if status_callback: + status_callback("resolving", "Bypassing protection...") use_bypasser_now = True continue logger.warning(f"403 error, giving up: {current_url}") diff --git a/cwa_book_downloader/download/network.py b/cwa_book_downloader/download/network.py index c7cd55ab..e7038b49 100644 --- a/cwa_book_downloader/download/network.py +++ b/cwa_book_downloader/download/network.py @@ -811,12 +811,9 @@ def _looks_like_ip(s: str) -> bool: return s.replace(".", "").replace(":", "").isdigit() def _build_aa_urls() -> List[str]: - """Build list of available AA URLs from config.""" - urls = ["https://annas-archive.se", "https://annas-archive.li", "https://annas-archive.pm", "https://annas-archive.in"] - additional = app_config.get("AA_ADDITIONAL_URLS", "") - if additional: - urls.extend(u.strip() for u in additional.split(",") if u.strip()) - return urls + """Build list of available AA URLs from centralized mirror config.""" + from cwa_book_downloader.core.mirrors import get_aa_mirrors + return get_aa_mirrors() def _initialize_aa_state() -> None: diff --git a/cwa_book_downloader/main.py b/cwa_book_downloader/main.py index 8b71e4a9..28344197 100644 --- a/cwa_book_downloader/main.py +++ b/cwa_book_downloader/main.py @@ -290,22 +290,13 @@ def favicon(_: Any = None) -> Response: """ return send_from_directory(FRONTEND_DIST, 'favicon.ico', mimetype='image/vnd.microsoft.icon') -# Register bypasser warmup callback for when first WebSocket client connects -# and shutdown callback for when all clients disconnect -# Use app_config to read from settings file (not just env var) so UI changes work after restart -if not app_config.get("USING_EXTERNAL_BYPASSER", False): - from cwa_book_downloader.bypass.internal_bypasser import warmup as bypasser_warmup, shutdown_if_idle as bypasser_shutdown - ws_manager.register_on_first_connect(bypasser_warmup) - ws_manager.register_on_all_disconnect(bypasser_shutdown) - logger.info("Registered Cloudflare bypasser warmup/shutdown on WebSocket connect/disconnect") - if DEBUG: import subprocess if app_config.get("USING_EXTERNAL_BYPASSER", False): _stop_gui = lambda: None else: - from cwa_book_downloader.bypass.internal_bypasser import _reset_driver as _stop_gui + from cwa_book_downloader.bypass.internal_bypasser import _cleanup_orphan_processes as _stop_gui @app.route('/api/debug', methods=['GET']) @login_required diff --git a/cwa_book_downloader/release_sources/direct_download.py b/cwa_book_downloader/release_sources/direct_download.py index 8b4b689a..64513535 100644 --- a/cwa_book_downloader/release_sources/direct_download.py +++ b/cwa_book_downloader/release_sources/direct_download.py @@ -62,18 +62,21 @@ _CF_BYPASS_REQUIRED = frozenset({"aa-slow-nowait", "aa-slow-wait", "zlib", "weli # Sources whose URLs come from AA page (multiple mirrors) _AA_PAGE_SOURCES = frozenset({"aa-slow-nowait", "aa-slow-wait"}) -_MD5_URL_TEMPLATES = { - "zlib": "https://z-lib.fm/md5/{md5}", - "welib": "https://welib.org/md5/{md5}", -} +def _get_md5_url_template(source_id: str) -> Optional[str]: + """Get URL template for MD5-based sources from centralized config.""" + from cwa_book_downloader.core import mirrors -_LIBGEN_DOMAINS = [ - "https://libgen.gl", - "https://libgen.li", - "https://libgen.bz", - "https://libgen.la", - "https://libgen.vg", -] + if source_id == "zlib": + return mirrors.get_zlib_url_template() + elif source_id == "welib": + return mirrors.get_welib_url_template() + return None + + +def _get_libgen_domains() -> List[str]: + """Get LibGen domains from centralized config.""" + from cwa_book_downloader.core import mirrors + return mirrors.get_libgen_mirrors() _LIBGEN_GET_PATTERNS = [ re.compile(r']*>\s*]*>GET\s*', re.IGNORECASE), @@ -83,8 +86,28 @@ _LIBGEN_GET_PATTERNS = [ ] def _get_source_priority() -> List[Dict]: - """Get the current source priority configuration.""" - return config.get("SOURCE_PRIORITY") or [] + """Get the full source priority list. + + Fast sources (AA Fast, LibGen) are hardcoded first. + Slow sources come from user config. + """ + # Fast sources - always first, hardcoded + fast_sources = [] + + # AA Fast only if donator key is set + if config.get("AA_DONATOR_KEY"): + fast_sources.append({"id": "aa-fast", "enabled": True}) + + # LibGen always available + fast_sources.append({"id": "libgen", "enabled": True}) + + # User's configured slow sources (config won't contain fast sources) + slow_sources = config.get("SOURCE_PRIORITY") or [] + + # Filter out any legacy fast source entries from old configs + slow_sources = [s for s in slow_sources if s["id"] not in ("aa-fast", "libgen")] + + return fast_sources + slow_sources def _is_source_enabled(source_id: str) -> bool: @@ -344,7 +367,7 @@ def _parse_book_info_page(soup: BeautifulSoup, book_id: str, fetch_download_coun if size == "" and "." in stripped: size = _normalize_size(f) - book_title = _find_in_divs(divs, "🔍")[0].strip("🔍").strip() + book_title = (_find_in_divs(divs, "🔍") or [""])[0].strip("🔍").strip() # Extract basic information description = _extract_book_description(soup) @@ -354,8 +377,8 @@ def _parse_book_info_page(soup: BeautifulSoup, book_id: str, fetch_download_coun preview=preview, title=book_title, content=content, - publisher=_find_in_divs(divs, "icon-[mdi--company]", is_class=True)[0], - author=_find_in_divs(divs, "icon-[mdi--user-edit]", is_class=True)[0], + publisher=(_find_in_divs(divs, "icon-[mdi--company]", is_class=True) or [""])[0], + author=(_find_in_divs(divs, "icon-[mdi--user-edit]", is_class=True) or [""])[0], format=format, size=size, description=description, @@ -543,14 +566,15 @@ def _get_urls_for_source( return [url] # MD5-based sources - generate URL from template - if source_id in _MD5_URL_TEMPLATES: - url = _MD5_URL_TEMPLATES[source_id].format(md5=book_info.id) + template = _get_md5_url_template(source_id) + if template: + url = template.format(md5=book_info.id) _url_source_types[url] = source_id return [url] if source_id == "libgen": urls = [] - for base_url in _LIBGEN_DOMAINS: + for base_url in _get_libgen_domains(): url = f"{base_url}/ads.php?md5={book_info.id}" _url_source_types[url] = "libgen" urls.append(url) @@ -560,7 +584,7 @@ def _get_urls_for_source( if source_id == "welib": if status_callback: status_callback("resolving", "Fetching welib sources") - return _get_download_urls_from_welib(book_info.id, selector=selector, cancel_flag=cancel_flag) + return _get_download_urls_from_welib(book_info.id, selector=selector, cancel_flag=cancel_flag, status_callback=status_callback) # AA page sources - fetch AA page if not already done if source_id in _AA_PAGE_SOURCES: @@ -627,14 +651,21 @@ def _try_download_url( return None -def _get_download_urls_from_welib(book_id: str, selector: Optional[network.AAMirrorSelector] = None, cancel_flag: Optional[Event] = None) -> List[str]: +def _get_download_urls_from_welib( + book_id: str, + selector: Optional[network.AAMirrorSelector] = None, + cancel_flag: Optional[Event] = None, + status_callback: Optional[Callable[[str, Optional[str]], None]] = None +) -> List[str]: """Get download URLs from welib.org (bypasser required).""" + from cwa_book_downloader.core import mirrors + if not _is_source_enabled("welib"): return [] - url = _MD5_URL_TEMPLATES["welib"].format(md5=book_id) + url = mirrors.get_welib_url_template().format(md5=book_id) logger.info(f"Fetching welib download URLs for {book_id}") try: - html = downloader.html_get_page(url, use_bypasser=True, selector=selector or network.AAMirrorSelector(), cancel_flag=cancel_flag) + html = downloader.html_get_page(url, use_bypasser=True, selector=selector or network.AAMirrorSelector(), cancel_flag=cancel_flag, status_callback=status_callback) except Exception as exc: logger.error_trace(f"Welib fetch failed for {book_id}: {exc}") return [] @@ -819,13 +850,13 @@ def _get_download_url( # AA fast download API (JSON response) if link.startswith(f"{network.get_aa_base_url()}/dyn/api/fast_download.json"): - page = downloader.html_get_page(link, selector=sel, cancel_flag=cancel_flag) + page = downloader.html_get_page(link, selector=sel, cancel_flag=cancel_flag, status_callback=status_callback) return downloader.get_absolute_url(link, json.loads(page).get("download_url", "")) - if "/ads.php?md5=" in link and any(domain in link for domain in _LIBGEN_DOMAINS): + if "/ads.php?md5=" in link and any(domain in link for domain in _get_libgen_domains()): return _extract_libgen_download_url(link, cancel_flag) - html = downloader.html_get_page(link, selector=sel, cancel_flag=cancel_flag) + html = downloader.html_get_page(link, selector=sel, cancel_flag=cancel_flag, status_callback=status_callback) if not html: return "" @@ -838,7 +869,7 @@ def _get_download_url( if not dl: # Retry after delay if page not fully loaded time.sleep(2) - html = downloader.html_get_page(link, selector=sel, cancel_flag=cancel_flag) + html = downloader.html_get_page(link, selector=sel, cancel_flag=cancel_flag, status_callback=status_callback) if html: soup = BeautifulSoup(html, "html.parser") dl = soup.find("a", href=True, class_="addDownloadedBook") diff --git a/docker-compose.dev.yml b/docker-compose.dev.yml index b81619d4..246d9f3b 100644 --- a/docker-compose.dev.yml +++ b/docker-compose.dev.yml @@ -8,6 +8,8 @@ services: context: . dockerfile: Dockerfile target: cwa-bd + cap_add: + - SYS_PTRACE environment: DEBUG: true volumes: diff --git a/src/frontend/src/components/settings/fields/OrderableListField.tsx b/src/frontend/src/components/settings/fields/OrderableListField.tsx index 1102584a..48267227 100644 --- a/src/frontend/src/components/settings/fields/OrderableListField.tsx +++ b/src/frontend/src/components/settings/fields/OrderableListField.tsx @@ -15,6 +15,9 @@ interface OrderableListFieldProps { // Represents where the drop indicator should appear type DropPosition = { index: number; position: 'before' | 'after' } | null; +// Merged item type with all properties +type MergedItem = OrderableListItem & OrderableListOption; + /** * Merge current value with options to get full item info. * Items in value take precedence; any options not in value are appended. @@ -22,9 +25,9 @@ type DropPosition = { index: number; position: 'before' | 'after' } | null; const mergeValueWithOptions = ( value: OrderableListItem[], options: OrderableListOption[] -): Array => { +): MergedItem[] => { const optionsMap = new Map(options.map((opt) => [opt.id, opt])); - const result: Array = []; + const result: MergedItem[] = []; // Add items from value (preserves order) for (const item of value) { @@ -35,7 +38,7 @@ const mergeValueWithOptions = ( } } - // Add any remaining options not in value (shouldn't happen normally) + // Add any remaining options not in value for (const option of optionsMap.values()) { result.push({ ...option, id: option.id, enabled: false }); } @@ -56,12 +59,24 @@ export const OrderableListField = ({ const items = mergeValueWithOptions(value ?? [], field.options); + // Check if a move is valid (not crossing pinned items) + const isValidMove = (fromIndex: number): boolean => { + const fromItem = items[fromIndex]; + if (fromItem?.isPinned) return false; + return true; + }; + const handleDragStart = (e: React.DragEvent, index: number) => { + const item = items[index]; + if (item?.isPinned) { + e.preventDefault(); + return; + } + setDraggedIndex(index); dragNodeRef.current = e.currentTarget as HTMLDivElement; e.dataTransfer.effectAllowed = 'move'; e.dataTransfer.setData('text/plain', String(index)); - // Add a slight delay before adding the dragging class for better visual feedback requestAnimationFrame(() => { if (dragNodeRef.current) { dragNodeRef.current.classList.add('opacity-50'); @@ -85,7 +100,12 @@ export const OrderableListField = ({ return; } - // Determine if we're in the top or bottom half of the target + // Can't drop on pinned items + if (items[index]?.isPinned) { + setDropPosition(null); + return; + } + const rect = e.currentTarget.getBoundingClientRect(); const midpoint = rect.top + rect.height / 2; const position = e.clientY < midpoint ? 'before' : 'after'; @@ -94,7 +114,6 @@ export const OrderableListField = ({ }; const handleDragLeave = (e: React.DragEvent) => { - // Only clear if we're leaving the item entirely (not entering a child) const relatedTarget = e.relatedTarget as Node | null; if (!e.currentTarget.contains(relatedTarget)) { setDropPosition(null); @@ -108,27 +127,23 @@ export const OrderableListField = ({ return; } - // Calculate the actual target index based on drop position let targetIndex = dropPosition.index; if (dropPosition.position === 'after') { targetIndex += 1; } - // Adjust if dragging from before the target if (draggedIndex < targetIndex) { targetIndex -= 1; } - if (draggedIndex === targetIndex) { + if (draggedIndex === targetIndex || !isValidMove(draggedIndex)) { handleDragEnd(); return; } - // Reorder the items const newItems = [...items]; const [removed] = newItems.splice(draggedIndex, 1); newItems.splice(targetIndex, 0, removed); - // Convert back to value format const newValue: OrderableListItem[] = newItems.map((item) => ({ id: item.id, enabled: item.enabled, @@ -154,6 +169,7 @@ export const OrderableListField = ({ const moveItem = (fromIndex: number, direction: 'up' | 'down') => { const toIndex = direction === 'up' ? fromIndex - 1 : fromIndex + 1; if (toIndex < 0 || toIndex >= items.length) return; + if (!isValidMove(fromIndex)) return; const newItems = [...items]; [newItems[fromIndex], newItems[toIndex]] = [newItems[toIndex], newItems[fromIndex]]; @@ -166,14 +182,26 @@ export const OrderableListField = ({ onChange(newValue); }; - // Calculate which gap index to show the indicator at (0 = before first item, N = after last item) + const canMoveUp = (index: number): boolean => { + const item = items[index]; + if (item?.isPinned) return false; + if (index === 0) return false; + if (items[index - 1]?.isPinned) return false; + return true; + }; + + const canMoveDown = (index: number): boolean => { + const item = items[index]; + if (item?.isPinned) return false; + if (index === items.length - 1) return false; + return true; + }; + const getDropGapIndex = (): number | null => { if (!dropPosition) return null; - if (dropPosition.position === 'before') { - return dropPosition.index; - } else { - return dropPosition.index + 1; - } + return dropPosition.position === 'before' + ? dropPosition.index + : dropPosition.index + 1; }; const dropGapIndex = getDropGapIndex(); @@ -183,137 +211,131 @@ export const OrderableListField = ({ {items.map((item, index) => { const isDragging = draggedIndex === index; const isItemDisabled = isDisabled || item.isLocked; - // Show indicator before this item if the gap index matches + const isPinned = item.isPinned ?? false; const showIndicatorBefore = dropGapIndex === index; return (
- {/* Drop indicator - absolutely positioned so it doesn't affect layout */} + {/* Drop indicator */} {showIndicatorBefore && (
)}
handleDragStart(e, index)} onDragEnd={handleDragEnd} onDragOver={(e) => handleDragOver(e, index)} onDragLeave={handleDragLeave} onDrop={handleDrop} className={` - flex items-center gap-3 p-3 rounded-lg border + flex items-center gap-3 p-3 rounded-lg transition-all duration-150 - ${isDragging ? 'opacity-50 cursor-grabbing' : 'cursor-grab'} - border-[var(--border-muted)] - ${isDisabled ? 'opacity-60' : 'hover:bg-[var(--hover-surface)]'} + ${isDragging ? 'opacity-50 cursor-grabbing' : isPinned ? 'cursor-default' : 'cursor-grab'} + border border-[var(--border-muted)] + ${isDisabled ? 'opacity-60' : !isPinned ? 'hover:bg-[var(--hover-surface)]' : ''} `} > - {/* Reorder Controls */} -
- - -
- - {/* Label and Description */} -
-
{item.label}
- {item.description && ( -
- {item.description} -
- )} - {item.isLocked && item.disabledReason && ( -
- - - - {item.disabledReason} -
- )} -
- - {/* Toggle Switch */} - {(() => { - // Locked items always show as "off" regardless of enabled state - const showAsEnabled = item.enabled && !item.isLocked; - return ( - - ); - })()} + aria-label="Move up" + > + + + + + +
+ ) : ( +
+ )} + + {/* Label and Description */} +
+
{item.label}
+ {item.description && ( +
+ {item.description} +
+ )} + {item.isLocked && item.disabledReason && ( +
+ + + + {item.disabledReason} +
+ )} +
+ + {/* Toggle Switch */} +
); })} - {/* Drop indicator after last item - use relative container with absolute indicator */} + {/* Drop indicator after last item */} {dropGapIndex === items.length && (
diff --git a/src/frontend/src/types/settings.ts b/src/frontend/src/types/settings.ts index 4c0f1607..c189e413 100644 --- a/src/frontend/src/types/settings.ts +++ b/src/frontend/src/types/settings.ts @@ -101,6 +101,7 @@ export interface OrderableListOption { description?: string; disabledReason?: string; // Explanation when item cannot be enabled isLocked?: boolean; // Item cannot be toggled (e.g., missing dependency) + isPinned?: boolean; // Item cannot be reordered (but can still be toggled if not locked) } export interface OrderableListFieldConfig extends BaseField {