Direct download tweaks (#408)

- Simplified bypasser process, removed warmup functionality
- Added dedicated fast download sources, tried first.
This commit is contained in:
Alex
2026-01-07 19:41:46 +00:00
committed by GitHub
parent 7954ae9138
commit b97e48235b
11 changed files with 858 additions and 666 deletions
+235 -367
View File
@@ -1,6 +1,5 @@
import os
import random
import signal
import socket
import subprocess
import threading
@@ -15,7 +14,7 @@ import requests
from seleniumbase import Driver
from cwa_book_downloader.bypass import BypassCancelledException
from cwa_book_downloader.bypass.fingerprint import clear_screen_size, get_screen_size
from cwa_book_downloader.bypass.fingerprint import get_screen_size
from cwa_book_downloader.config import env
from cwa_book_downloader.config.env import LOG_DIR
from cwa_book_downloader.config.settings import RECORDING_DIR
@@ -42,19 +41,12 @@ DDOS_GUARD_INDICATORS = [
"could not verify your browser automatically",
]
DRIVER = None
DISPLAY = {
"xvfb": None,
"ffmpeg": None,
}
LAST_USED = None
LOCKED = threading.Lock()
# Flag to track if DNS rotated while a bypass was in progress
# Chrome will be restarted after the current operation completes
_dns_rotation_pending = False
_dns_rotation_lock = threading.Lock()
# Cookie storage - shared with requests library for Cloudflare bypass
# Structure: {domain: {cookie_name: {value, expiry, ...}}}
_cf_cookies: dict[str, dict] = {}
@@ -730,70 +722,48 @@ def _build_host_resolver_rules() -> list[str]:
DRIVER_RESET_ERRORS = {"WebDriverException", "SessionNotCreatedException", "TimeoutException", "MaxRetryError"}
def _get(url: str, retry: Optional[int] = None, cancel_flag: Optional[Event] = None) -> str:
"""Fetch URL with Cloudflare bypass. Retries on failure."""
retry = retry if retry is not None else app_config.MAX_RETRY
def _get(url: str, driver: Driver, cancel_flag: Optional[Event] = None) -> str:
"""Fetch URL with Cloudflare bypass using provided driver."""
_check_cancellation(cancel_flag, "Bypass cancelled before starting")
logger.debug(f"SB_GET: {url}")
hostname = urlparse(url).hostname or ""
if has_valid_cf_cookies(hostname):
reconnect_time = 1.0
logger.debug(f"Using fast reconnect ({reconnect_time}s) - valid cookies exist")
else:
reconnect_time = app_config.DEFAULT_SLEEP
logger.debug(f"Using standard reconnect ({reconnect_time}s) - no cached cookies")
logger.debug("Opening URL with SeleniumBase...")
driver.uc_open_with_reconnect(url, reconnect_time)
_check_cancellation(cancel_flag, "Bypass cancelled after page load")
try:
logger.debug(f"SB_GET: {url}")
sb = _get_driver()
hostname = urlparse(url).hostname or ""
if has_valid_cf_cookies(hostname):
reconnect_time = 1.0
logger.debug(f"Using fast reconnect ({reconnect_time}s) - valid cookies exist")
else:
reconnect_time = app_config.DEFAULT_SLEEP
logger.debug(f"Using standard reconnect ({reconnect_time}s) - no cached cookies")
logger.debug("Opening URL with SeleniumBase...")
sb.uc_open_with_reconnect(url, reconnect_time)
_check_cancellation(cancel_flag, "Bypass cancelled after page load")
try:
logger.debug(f"Page loaded - URL: {sb.get_current_url()}, Title: {sb.get_title()}")
except Exception as e:
logger.debug(f"Could not get page info: {e}")
logger.debug("Starting bypass process...")
if _bypass(sb, cancel_flag=cancel_flag):
_extract_cookies_from_driver(sb, url)
return sb.page_source
logger.warning("Bypass completed but page still shows protection")
try:
body = sb.get_text("body")
logger.debug(f"Page content: {body[:500]}..." if len(body) > 500 else body)
except Exception:
pass
except BypassCancelledException:
raise
logger.debug(f"Page loaded - URL: {driver.get_current_url()}, Title: {driver.get_title()}")
except Exception as e:
error_details = f"{type(e).__name__}: {e}"
logger.debug(f"Could not get page info: {e}")
if retry == 0:
logger.error(f"Failed after all retries: {error_details}")
logger.debug(f"Stack trace: {traceback.format_exc()}")
_reset_driver()
raise
logger.debug("Starting bypass process...")
if _bypass(driver, cancel_flag=cancel_flag):
_extract_cookies_from_driver(driver, url)
return driver.page_source
logger.warning(f"Bypass failed (retry {app_config.MAX_RETRY - retry + 1}/{app_config.MAX_RETRY}): {error_details}")
logger.debug(f"Stack trace: {traceback.format_exc()}")
logger.warning("Bypass completed but page still shows protection")
try:
body = driver.get_text("body")
logger.debug(f"Page content: {body[:500]}..." if len(body) > 500 else body)
except Exception:
pass
if type(e).__name__ in DRIVER_RESET_ERRORS:
logger.info("Restarting bypasser due to browser error...")
_reset_driver()
return ""
_check_cancellation(cancel_flag, "Bypass cancelled before retry")
return _get(url, retry - 1, cancel_flag)
def get(url: str, retry: Optional[int] = None, cancel_flag: Optional[Event] = None) -> str:
"""Fetch a URL with protection bypass."""
"""Fetch a URL with protection bypass. Creates fresh Chrome instance for each bypass."""
retry = retry if retry is not None else app_config.MAX_RETRY
global LAST_USED
with LOCKED:
# Try cookies first - another request may have completed bypass while waiting
@@ -803,27 +773,55 @@ def get(url: str, retry: Optional[int] = None, cancel_flag: Optional[Event] = No
response = requests.get(url, cookies=cookies, proxies=get_proxies(), timeout=(5, 10))
if response.status_code == 200:
logger.debug("Cookies available after lock wait - skipped Chrome")
LAST_USED = time.time()
return response.text
except Exception:
pass
result = _get(url, retry, cancel_flag)
LAST_USED = time.time()
return result
# Fresh Chrome for each bypass attempt
driver = None
try:
_ensure_display_initialized()
driver = _create_driver()
def _init_driver() -> Driver:
"""Initialize the Chrome driver with undetected-chromedriver settings."""
global DRIVER
if DRIVER:
_reset_driver()
for attempt in range(retry):
_check_cancellation(cancel_flag, "Bypass cancelled before attempt")
try:
result = _get(url, driver, cancel_flag)
if result:
return result
except BypassCancelledException:
raise
except Exception as e:
error_details = f"{type(e).__name__}: {e}"
logger.warning(f"Bypass failed (attempt {attempt + 1}/{retry}): {error_details}")
logger.debug(f"Stack trace: {traceback.format_exc()}")
# On driver errors, quit and create fresh driver
if type(e).__name__ in DRIVER_RESET_ERRORS:
logger.info("Restarting Chrome due to browser error...")
_quit_driver(driver)
driver = _create_driver()
logger.error(f"Bypass failed after {retry} attempts")
return ""
finally:
# Always quit Chrome when done
if driver:
_quit_driver(driver)
def _create_driver() -> Driver:
"""Create a fresh Chrome driver instance."""
chromium_args = _get_chromium_args()
screen_width, screen_height = get_screen_size()
logger.debug(f"Initializing Chrome driver with args: {chromium_args}")
logger.debug(f"Creating Chrome driver with args: {chromium_args}")
logger.debug(f"Browser screen size: {screen_width}x{screen_height}")
# Start FFmpeg recording if debug mode (record each bypass session)
if app_config.get("DEBUG", False) and DISPLAY["xvfb"] and not DISPLAY["ffmpeg"]:
_start_ffmpeg_recording()
driver = Driver(
uc=True,
headless=False,
@@ -834,10 +832,174 @@ def _init_driver() -> Driver:
chromium_arg=chromium_args,
)
driver.set_page_load_timeout(60)
DRIVER = driver
time.sleep(app_config.DEFAULT_SLEEP)
logger.info("Chrome browser ready")
logger.log_resource_usage()
return driver
def _start_ffmpeg_recording() -> None:
"""Start FFmpeg screen recording for debug mode."""
global DISPLAY
RECORDING_DIR.mkdir(parents=True, exist_ok=True)
display = DISPLAY["xvfb"]
timestamp = datetime.now().strftime("%y%m%d-%H%M%S")
output_file = RECORDING_DIR / f"screen_recording_{timestamp}.mp4"
screen_width, screen_height = get_screen_size()
display_width = screen_width + 100
display_height = screen_height + 150
ffmpeg_cmd = [
"ffmpeg", "-y", "-f", "x11grab",
"-video_size", f"{display_width}x{display_height}",
"-i", f":{display.display}",
"-c:v", "libx264", "-preset", "ultrafast",
"-maxrate", "700k", "-bufsize", "1400k", "-crf", "36",
"-pix_fmt", "yuv420p", "-tune", "animation",
"-x264-params", "bframes=0:deblock=-1,-1",
"-r", "15", "-an",
output_file.as_posix(),
"-nostats", "-loglevel", "0"
]
logger.debug("Starting FFmpeg recording to %s", output_file)
logger.debug_trace(f"FFmpeg command: {' '.join(ffmpeg_cmd)}")
DISPLAY["ffmpeg"] = subprocess.Popen(ffmpeg_cmd)
def _close_cdp_sockets() -> int:
"""Find and close any sockets connected to CDP port 9222.
This is a workaround for SeleniumBase not properly closing websocket
connections when using activate_cdp_mode(). Returns count of closed sockets.
"""
import os
closed = 0
pid = os.getpid()
try:
fd_path = f'/proc/{pid}/fd'
for fd_name in os.listdir(fd_path):
try:
fd = int(fd_name)
link = os.readlink(f'{fd_path}/{fd_name}')
if 'socket:' not in link:
continue
# Check if this socket is connected to port 9222 (CDP)
# by reading /proc/net/tcp and matching inode
inode = link.split('[')[1].rstrip(']')
with open('/proc/net/tcp', 'r') as f:
for line in f:
parts = line.split()
if len(parts) < 10:
continue
# Check if this is our socket and connects to port 9222 (0x2406)
if parts[9] == inode:
remote = parts[2]
remote_port = int(remote.split(':')[1], 16)
if remote_port == 9222:
logger.debug(f"Closing CDP socket fd={fd} inode={inode}")
os.close(fd)
closed += 1
break
except (ValueError, OSError, IndexError):
continue
except Exception as e:
logger.debug(f"Error scanning for CDP sockets: {e}")
return closed
def _quit_driver(driver: Driver) -> None:
"""Quit Chrome driver and clean up resources.
Proper cleanup sequence for SeleniumBase CDP mode:
1. Stop CDP browser (closes websocket connections)
2. Reconnect WebDriver
3. Close window
4. Quit driver
5. Force-kill any lingering processes
The CDP websocket connection must be explicitly closed before Chrome is killed,
otherwise the sockets end up in CLOSE_WAIT state causing gevent to busy-loop.
References:
- https://github.com/seleniumbase/SeleniumBase/discussions/3768
- https://www.selenium.dev/selenium/docs/api/py/selenium_webdriver_common_bidi/selenium.webdriver.common.bidi.cdp.html
"""
if driver is None:
return
logger.debug("Quitting Chrome driver...")
# Strategy 1: Stop CDP browser if in CDP mode (closes websocket connections)
# This is the proper SeleniumBase way to close CDP connections
try:
if hasattr(driver, 'cdp') and driver.cdp and hasattr(driver.cdp, 'driver'):
driver.cdp.driver.stop()
logger.debug("Stopped CDP browser (closed websocket)")
time.sleep(0.3)
except Exception as e:
logger.debug(f"CDP stop: {e}")
# Strategy 2: Reconnect to re-establish WebDriver control before quitting
try:
driver.reconnect()
time.sleep(0.2)
except Exception as e:
logger.debug(f"Reconnect: {e}")
# Strategy 3: Close the current window/tab
try:
driver.close()
time.sleep(0.2)
except Exception as e:
logger.debug(f"Close window: {e}")
# Strategy 4: Fallback - explicitly close any remaining CDP sockets
# This catches any sockets that weren't closed by cdp.driver.stop()
closed = _close_cdp_sockets()
if closed:
logger.debug(f"Closed {closed} remaining CDP socket(s)")
# Strategy 5: Standard quit
try:
driver.quit()
except Exception as e:
logger.debug(f"Quit: {e}")
# Strategy 6: Force garbage collection
import gc
gc.collect()
# Strategy 7: Force-kill any lingering Chrome/chromedriver processes
if env.DOCKERMODE:
time.sleep(0.3)
try:
subprocess.run(["pkill", "-9", "-f", "chrom"], capture_output=True, timeout=5)
except Exception as e:
logger.debug(f"pkill chrome: {e}")
# Strategy 8: Stop ffmpeg recording if running
global DISPLAY
if DISPLAY.get("ffmpeg"):
try:
DISPLAY["ffmpeg"].terminate()
DISPLAY["ffmpeg"].wait(timeout=2)
logger.debug("Stopped ffmpeg recording")
except Exception as e:
logger.debug(f"ffmpeg terminate: {e}")
try:
DISPLAY["ffmpeg"].kill()
except Exception:
pass
DISPLAY["ffmpeg"] = None
logger.log_resource_usage()
def _ensure_display_initialized():
"""Initialize virtual display if needed. Must be called with LOCKED held."""
global DISPLAY
@@ -860,300 +1022,6 @@ def _ensure_display_initialized():
_reset_pyautogui_display_state()
def _get_driver():
global DRIVER, DISPLAY, LAST_USED
logger.debug("Getting driver...")
LAST_USED = time.time()
_ensure_display_initialized()
# Start FFmpeg recording on first actual bypass request (not during warmup)
# This ensures we only record active bypass sessions, not idle time
if app_config.get("DEBUG", False) and DISPLAY["xvfb"] and not DISPLAY["ffmpeg"]:
RECORDING_DIR.mkdir(parents=True, exist_ok=True)
display = DISPLAY["xvfb"]
timestamp = datetime.now().strftime("%y%m%d-%H%M%S")
output_file = RECORDING_DIR / f"screen_recording_{timestamp}.mp4"
# Get the display size (screen size + padding)
screen_width, screen_height = get_screen_size()
display_width = screen_width + 100
display_height = screen_height + 150
ffmpeg_cmd = [
"ffmpeg",
"-y",
"-f", "x11grab",
"-video_size", f"{display_width}x{display_height}",
"-i", f":{display.display}",
"-c:v", "libx264",
"-preset", "ultrafast", # or "veryfast" (trade speed for slightly better compression)
"-maxrate", "700k", # Slightly higher bitrate for text clarity
"-bufsize", "1400k", # Buffer size (2x maxrate)
"-crf", "36", # Adjust as needed: higher = smaller, lower = better quality (23 is visually lossless)
"-pix_fmt", "yuv420p", # Crucial for compatibility with most players
"-tune", "animation", # Optimize encoding for screen content
"-x264-params", "bframes=0:deblock=-1,-1", # Optimize for text, disable b-frames and deblocking
"-r", "15", # Reduce frame rate (if content allows)
"-an", # Disable audio recording (if not needed)
output_file.as_posix(),
"-nostats", "-loglevel", "0"
]
logger.debug("Starting FFmpeg recording to %s", output_file)
logger.debug_trace(f"FFmpeg command: {' '.join(ffmpeg_cmd)}")
DISPLAY["ffmpeg"] = subprocess.Popen(ffmpeg_cmd)
if not DRIVER:
return _init_driver()
# Verify the existing driver is actually healthy (browser process still alive)
if not _is_driver_healthy():
logger.warning("Existing driver is unhealthy (browser may have crashed), reinitializing...")
_reset_driver()
_ensure_display_initialized() # Display was shut down by _reset_driver, reinitialize it
return _init_driver()
logger.log_resource_usage()
return DRIVER
def _reset_driver() -> None:
"""Reset the browser driver and cleanup all associated processes."""
logger.log_resource_usage()
logger.info("Shutting down Cloudflare bypasser...")
global DRIVER, DISPLAY
clear_screen_size()
if DRIVER:
try:
DRIVER.quit()
except Exception as e:
logger.warning(f"Error quitting driver: {e}")
DRIVER = None
if DISPLAY["xvfb"]:
try:
DISPLAY["xvfb"].stop()
except Exception as e:
logger.warning(f"Error stopping display: {e}")
DISPLAY["xvfb"] = None
if DISPLAY["ffmpeg"]:
try:
DISPLAY["ffmpeg"].send_signal(signal.SIGINT)
except Exception as e:
logger.debug(f"Error stopping ffmpeg: {e}")
DISPLAY["ffmpeg"] = None
time.sleep(0.5)
for process in ["Xvfb", "ffmpeg", "chrom"]:
try:
os.system(f"pkill -f {process}")
except Exception as e:
logger.debug(f"Error killing {process}: {e}")
time.sleep(0.5)
logger.info("Cloudflare bypasser shut down")
logger.log_resource_usage()
def _restart_chrome_only() -> None:
"""Restart Chrome (not the display) to pick up new DNS settings.
Called when DNS provider rotates. Display is kept running to avoid slower full restart.
"""
global DRIVER, LAST_USED
logger.debug("Restarting Chrome to apply new DNS settings...")
if DRIVER:
try:
DRIVER.quit()
except Exception as e:
logger.debug(f"Error quitting driver during DNS rotation: {e}")
DRIVER = None
try:
os.system("pkill -f chrom")
except Exception as e:
logger.debug(f"Error killing chrome processes: {e}")
time.sleep(0.5)
try:
_init_driver()
LAST_USED = time.time()
logger.debug("Chrome restarted with updated DNS settings")
except Exception as e:
logger.warning(f"Failed to restart Chrome after DNS rotation: {e}")
def _on_dns_rotation(provider_name: str, servers: list, doh_url: str) -> None:
"""Callback invoked when network.py rotates DNS provider.
Schedules an async Chrome restart to avoid blocking the current request.
"""
global _dns_rotation_pending
if DRIVER is None:
return
with _dns_rotation_lock:
if _dns_rotation_pending:
return
_dns_rotation_pending = True
def _async_restart():
global _dns_rotation_pending
logger.debug(f"DNS rotated to {provider_name} - restarting Chrome in background")
with LOCKED:
with _dns_rotation_lock:
_dns_rotation_pending = False
_restart_chrome_only()
threading.Thread(target=_async_restart, daemon=True).start()
def _cleanup_driver() -> None:
"""Reset driver after inactivity timeout.
Uses 4x longer timeout when UI clients are connected.
"""
global LAST_USED
try:
from cwa_book_downloader.api.websocket import ws_manager
has_active_clients = ws_manager.has_active_connections()
except ImportError:
ws_manager = None
has_active_clients = False
timeout_minutes = app_config.BYPASS_RELEASE_INACTIVE_MIN
if has_active_clients:
timeout_minutes *= 4
with LOCKED:
if not LAST_USED or time.time() - LAST_USED < timeout_minutes * 60:
return
logger.info(f"Bypasser idle for {timeout_minutes} min - shutting down")
_reset_driver()
LAST_USED = None
logger.log_resource_usage()
if has_active_clients and ws_manager:
ws_manager.request_warmup_on_next_connect()
logger.debug("Requested warmup on next client connect")
def _cleanup_loop() -> None:
"""Background loop that periodically checks for idle timeout."""
while True:
_cleanup_driver()
time.sleep(max(app_config.BYPASS_RELEASE_INACTIVE_MIN / 2, 1))
def _init_cleanup_thread() -> None:
"""Start the background cleanup thread."""
threading.Thread(target=_cleanup_loop, daemon=True).start()
def _should_warmup() -> bool:
"""Check if warmup should proceed based on configuration."""
if not app_config.get("BYPASS_WARMUP_ON_CONNECT", True):
logger.debug("Bypasser warmup disabled via config")
return False
if not env.DOCKERMODE:
logger.debug("Bypasser warmup skipped - not in Docker mode")
return False
if not app_config.get("USE_CF_BYPASS", True):
logger.debug("Bypasser warmup skipped - CF bypass disabled")
return False
if app_config.get("AA_DONATOR_KEY", ""):
logger.debug("Bypasser warmup skipped - AA donator key set")
return False
return True
def warmup() -> None:
"""Pre-initialize the virtual display and Chrome browser.
Called when a user connects to the web UI. Skipped if warmup is disabled,
not in Docker mode, CF bypass is disabled, or AA donator key is set.
"""
global LAST_USED
if not _should_warmup():
return
with LOCKED:
if is_warmed_up():
logger.debug("Bypasser already warmed up")
return
_cleanup_orphan_processes()
if DRIVER is not None or DISPLAY["xvfb"] is not None:
logger.info("Resetting stale bypasser state before warmup...")
_reset_driver()
logger.info("Warming up Cloudflare bypasser...")
try:
_ensure_display_initialized()
if DRIVER is None:
logger.info("Pre-initializing Chrome browser...")
_init_driver()
LAST_USED = time.time()
logger.info("Chrome browser ready")
logger.info("Bypasser warmup complete")
logger.log_resource_usage()
except Exception as e:
logger.warning(f"Failed to warm up bypasser: {e}")
def _is_driver_healthy() -> bool:
"""Check if the Chrome driver is responsive (not just non-None)."""
if DRIVER is None:
return False
try:
DRIVER.get_current_url()
return True
except Exception as e:
logger.warning(f"Driver health check failed: {type(e).__name__}: {e}")
return False
def is_warmed_up() -> bool:
"""Check if the bypasser is fully warmed up (display and browser initialized)."""
if DISPLAY["xvfb"] is None or DRIVER is None:
return False
return _is_driver_healthy()
def shutdown_if_idle() -> None:
"""Start the inactivity countdown when all WebSocket clients disconnect.
Sets LAST_USED to start the timer. The cleanup loop shuts down after
BYPASS_RELEASE_INACTIVE_MIN minutes of inactivity.
"""
global LAST_USED
with LOCKED:
if not is_warmed_up():
logger.debug("Bypasser already shut down")
return
LAST_USED = time.time()
logger.info(f"All clients disconnected - shutdown after {app_config.BYPASS_RELEASE_INACTIVE_MIN} min of inactivity")
_init_cleanup_thread()
# Register for DNS rotation notifications (Chrome restarts with new DNS settings)
if app_config.get("USE_CF_BYPASS", True) and not app_config.get("USING_EXTERNAL_BYPASSER", False):
network.register_dns_rotation_callback(_on_dns_rotation)
def _try_with_cached_cookies(url: str, hostname: str) -> Optional[str]:
"""Attempt request with cached cookies before using Chrome."""
cookies = get_cf_cookies_for_domain(hostname)
+191 -98
View File
@@ -151,34 +151,67 @@ def _get_release_source_options():
_LANGUAGE_OPTIONS = [{"value": lang["code"], "label": lang["language"]} for lang in _SUPPORTED_BOOK_LANGUAGE]
# Default AA mirror URLs (hardcoded base list)
_DEFAULT_AA_URLS = [
"https://annas-archive.se",
"https://annas-archive.li",
"https://annas-archive.pm",
"https://annas-archive.in",
]
def _get_aa_base_url_options():
"""Build AA URL options dynamically, including additional mirrors from config."""
from cwa_book_downloader.core.config import config
from cwa_book_downloader.core.mirrors import DEFAULT_AA_MIRRORS, get_aa_mirrors
options = [{"value": "auto", "label": "Auto (Recommended)"}]
# Get all mirrors (defaults + custom)
all_mirrors = get_aa_mirrors()
for url in all_mirrors:
domain = url.replace("https://", "").replace("http://", "")
is_custom = url not in DEFAULT_AA_MIRRORS
label = f"{domain} (custom)" if is_custom else domain
options.append({"value": url, "label": label})
return options
def _get_zlib_mirror_options():
"""Build Z-Library mirror options for SelectField."""
from cwa_book_downloader.core.mirrors import DEFAULT_ZLIB_MIRRORS
from cwa_book_downloader.core.config import config
options = []
# Add default mirrors
for url in _DEFAULT_AA_URLS:
# Extract domain for label
for url in DEFAULT_ZLIB_MIRRORS:
domain = url.replace("https://", "").replace("http://", "")
options.append({"value": url, "label": domain})
# Add any additional mirrors from config
additional = config.get("AA_ADDITIONAL_URLS", "")
# Add custom mirrors
additional = config.get("ZLIB_ADDITIONAL_URLS", "")
if additional:
for url in additional.split(","):
url = url.strip()
if url and url not in _DEFAULT_AA_URLS:
domain = url.replace("https://", "").replace("http://", "")
if url and url not in DEFAULT_ZLIB_MIRRORS:
domain = url.replace("https://", "").replace("http://", "").split("/")[0]
options.append({"value": url, "label": f"{domain} (custom)"})
return options
def _get_welib_mirror_options():
"""Build Welib mirror options for SelectField."""
from cwa_book_downloader.core.mirrors import DEFAULT_WELIB_MIRRORS
from cwa_book_downloader.core.config import config
options = []
# Add default mirrors
for url in DEFAULT_WELIB_MIRRORS:
domain = url.replace("https://", "").replace("http://", "")
options.append({"value": url, "label": domain})
# Add custom mirrors
additional = config.get("WELIB_ADDITIONAL_URLS", "")
if additional:
for url in additional.split(","):
url = url.strip()
if url and url not in DEFAULT_WELIB_MIRRORS:
domain = url.replace("https://", "").replace("http://", "").split("/")[0]
options.append({"value": url, "label": f"{domain} (custom)"})
return options
@@ -624,98 +657,118 @@ def download_settings():
]
def _get_source_priority_options():
"""Build source priority options with dynamic disabled states."""
def _get_fast_source_options():
"""Fast download sources - display only, not configurable."""
from cwa_book_downloader.core.config import config
has_donator_key = bool(config.get("AA_DONATOR_KEY", ""))
use_cf_bypass = config.get("USE_CF_BYPASS", True)
using_external_bypasser = config.get("USING_EXTERNAL_BYPASSER", False)
has_internal_bypasser = use_cf_bypass and not using_external_bypasser
return [
{
"id": "aa-fast",
"label": "Anna's Archive (Fast)",
"description": "Fast downloads for donators",
"isPinned": True,
"isLocked": not has_donator_key,
"disabledReason": "Requires AA Donator Key" if not has_donator_key else None,
},
{
"id": "welib",
"label": "Welib",
"description": "Alternative mirror with good availability",
"isLocked": not has_internal_bypasser,
"disabledReason": "Requires internal bypasser" if not has_internal_bypasser else None,
"id": "libgen",
"label": "Library Genesis",
"description": "Instant downloads, no bypass needed",
"isPinned": True,
},
]
def _get_fast_source_defaults():
"""Default values for fast sources display."""
return [
{"id": "aa-fast", "enabled": True},
{"id": "libgen", "enabled": True},
]
def _get_slow_source_options():
"""Slow download sources - configurable order. All require bypasser."""
from cwa_book_downloader.core.config import config
bypass_enabled = config.get("USE_CF_BYPASS", True)
locked = not bypass_enabled
disabled_reason = "Requires Cloudflare bypass" if locked else None
return [
{
"id": "aa-slow-nowait",
"label": "Anna's Archive (Slowest, No Waitlist)",
"description": "Partner servers without countdown",
"description": "Partner servers",
"isLocked": locked,
"disabledReason": disabled_reason,
},
{
"id": "aa-slow-wait",
"label": "Anna's Archive (Slow, Waitlist)",
"label": "Anna's Archive (Slow with Waitlist)",
"description": "Partner servers with countdown timer",
"isLocked": locked,
"disabledReason": disabled_reason,
},
{
"id": "libgen",
"label": "Libgen (Fast)",
"description": "Library Genesis. Fast downloads, no bypass needed.",
"id": "welib",
"label": "Welib",
"description": "Alternative mirror",
"isLocked": locked,
"disabledReason": disabled_reason,
},
{
"id": "zlib",
"label": "Z-Library",
"description": "Z-Library mirrors (requires Cloudflare bypass)",
"isLocked": not has_internal_bypasser,
"disabledReason": "Requires internal bypasser" if not has_internal_bypasser else None,
"description": "Alternative mirror",
"isLocked": locked,
"disabledReason": disabled_reason,
},
]
def _get_default_source_priority():
"""Default source priority order, respecting legacy env vars.
def _get_slow_source_defaults():
"""Default source priority order for slow sources."""
from cwa_book_downloader.config.env import _LEGACY_ALLOW_USE_WELIB
ALLOW_USE_WELIB (default true) controls whether welib is enabled.
PRIORITIZE_WELIB (default false) controls whether welib is moved to position 1.
"""
from cwa_book_downloader.config.env import _LEGACY_PRIORITIZE_WELIB, _LEGACY_ALLOW_USE_WELIB
welib_entry = {"id": "welib", "enabled": _LEGACY_ALLOW_USE_WELIB}
priority = [
{"id": "aa-fast", "enabled": True},
{"id": "libgen", "enabled": True},
return [
{"id": "aa-slow-nowait", "enabled": True},
{"id": "aa-slow-wait", "enabled": True},
{"id": "welib", "enabled": _LEGACY_ALLOW_USE_WELIB},
{"id": "zlib", "enabled": True},
]
if _LEGACY_PRIORITIZE_WELIB:
priority.insert(1, welib_entry) # After aa-fast
else:
priority.append(welib_entry) # Before zlib
# Z-Library last - it's quite brittle
priority.append({"id": "zlib", "enabled": True})
return priority
@register_settings("download_sources", "Download Sources", icon="download", order=21, group="direct_download")
def download_source_settings():
"""Settings for download source behavior."""
return [
PasswordField(
key="AA_DONATOR_KEY",
label="Anna's Archive Donator Key",
description="Enables fast downloads from Anna's Archive.",
),
HeadingField(
key="source_priority_heading",
title="Source Priority",
description="Configure which download sources to use and in what order.",
description="Sources are tried in order until a download succeeds.",
),
OrderableListField(
key="FAST_SOURCES_DISPLAY",
label="Fast downloads",
description="Always tried first, no waiting or bypass required.",
options=_get_fast_source_options,
default=_get_fast_source_defaults(),
env_supported=False,
),
OrderableListField(
key="SOURCE_PRIORITY",
label="Download Source Order",
description="Drag to reorder. Sources are tried from top to bottom until a download succeeds.",
options=_get_source_priority_options,
default=_get_default_source_priority(),
label="Slow downloads",
description="Fallback sources, may have waiting. Requires bypasser. Drag to reorder.",
options=_get_slow_source_options,
default=_get_slow_source_defaults(),
),
NumberField(
key="MAX_RETRY",
@@ -733,29 +786,6 @@ def download_source_settings():
min_value=1,
max_value=60,
),
HeadingField(
key="aa_settings_heading",
title="Anna's Archive",
description="Configure Anna's Archive mirror and donator settings.",
),
SelectField(
key="AA_BASE_URL",
label="Anna's Archive URL",
description="Primary Anna's Archive mirror to use. 'auto' selects automatically. Custom mirrors added below will appear here.",
options=_get_aa_base_url_options, # Callable - includes additional mirrors dynamically
default="auto",
),
TextField(
key="AA_ADDITIONAL_URLS",
label="Additional AA Mirrors",
description="Comma-separated list of additional Anna's Archive mirror URLs.",
placeholder="https://example.com,https://another.com",
),
PasswordField(
key="AA_DONATOR_KEY",
label="Anna's Archive Donator Key",
description="Optional donator key for faster downloads from Anna's Archive.",
),
HeadingField(
key="content_type_routing_heading",
title="Content-Type Routing",
@@ -829,20 +859,6 @@ def cloudflare_bypass_settings():
default=True,
requires_restart=True,
),
CheckboxField(
key="BYPASS_WARMUP_ON_CONNECT",
label="Warmup on Connect",
description="Pre-warm the bypasser when user connects to Web App UI",
default=True,
),
NumberField(
key="BYPASS_RELEASE_INACTIVE_MIN",
label="Release Inactive (minutes)",
description="Release bypasser resources after this many minutes of inactivity.",
default=5,
min_value=1,
max_value=60,
),
CheckboxField(
key="USING_EXTERNAL_BYPASSER",
label="Use External Bypasser",
@@ -881,6 +897,83 @@ def cloudflare_bypass_settings():
]
@register_settings("mirrors", "Mirrors", icon="globe", order=23, group="direct_download")
def mirror_settings():
"""Configure download source mirrors."""
from cwa_book_downloader.core.mirrors import DEFAULT_ZLIB_MIRRORS, DEFAULT_WELIB_MIRRORS
return [
# === ANNA'S ARCHIVE ===
HeadingField(
key="aa_mirrors_heading",
title="Anna's Archive",
description="Primary mirror with auto-probe on startup. Additional mirrors used as fallback.",
),
SelectField(
key="AA_BASE_URL",
label="Primary Mirror",
description="Select 'Auto' to probe mirrors on startup, or choose a specific mirror.",
options=_get_aa_base_url_options,
default="auto",
),
TextField(
key="AA_ADDITIONAL_URLS",
label="Additional Mirrors",
description="Comma-separated list of custom Anna's Archive mirror URLs.",
),
# === LIBGEN ===
HeadingField(
key="libgen_mirrors_heading",
title="LibGen",
description="All mirrors are tried during download until one succeeds. Defaults: libgen.gl, libgen.li, libgen.bz, libgen.la, libgen.vg",
),
TextField(
key="LIBGEN_ADDITIONAL_URLS",
label="Additional Mirrors",
description="Comma-separated list of custom LibGen mirrors to add to the defaults.",
),
# === Z-LIBRARY ===
HeadingField(
key="zlib_mirrors_heading",
title="Z-Library",
description="Z-Library requires Cloudflare bypass. Only the primary mirror is used.",
),
SelectField(
key="ZLIB_PRIMARY_URL",
label="Primary Mirror",
description="Z-Library mirror to use for downloads.",
options=_get_zlib_mirror_options,
default=DEFAULT_ZLIB_MIRRORS[0],
),
TextField(
key="ZLIB_ADDITIONAL_URLS",
label="Additional Mirrors",
description="Comma-separated list of custom Z-Library mirror URLs.",
),
# === WELIB ===
HeadingField(
key="welib_mirrors_heading",
title="Welib",
description="Welib requires Cloudflare bypass. Only the primary mirror is used.",
),
SelectField(
key="WELIB_PRIMARY_URL",
label="Primary Mirror",
description="Welib mirror to use for downloads.",
options=_get_welib_mirror_options,
default=DEFAULT_WELIB_MIRRORS[0],
),
TextField(
key="WELIB_ADDITIONAL_URLS",
label="Additional Mirrors",
description="Comma-separated list of custom Welib mirror URLs.",
),
]
@register_settings("advanced", "Advanced", icon="cog", order=15)
def advanced_settings():
"""Advanced settings for power users."""
+213
View File
@@ -0,0 +1,213 @@
"""Centralized mirror configuration for all download sources."""
from typing import List
# Lazy import to avoid circular imports
_config_module = None
def _get_config():
"""Lazy import of config module to avoid circular imports."""
global _config_module
if _config_module is None:
from cwa_book_downloader.core.config import config
_config_module = config
return _config_module
# Default mirror lists (hardcoded fallbacks)
DEFAULT_AA_MIRRORS = [
"https://annas-archive.se",
"https://annas-archive.li",
"https://annas-archive.pm",
"https://annas-archive.in",
]
DEFAULT_LIBGEN_MIRRORS = [
"https://libgen.gl",
"https://libgen.li",
"https://libgen.bz",
"https://libgen.la",
"https://libgen.vg",
]
DEFAULT_ZLIB_MIRRORS = [
"https://z-lib.fm",
"https://z-lib.gs",
"https://z-lib.id",
"https://z-library.sk",
"https://zlibrary-global.se",
]
DEFAULT_WELIB_MIRRORS = [
"https://welib.org",
]
def get_aa_mirrors() -> List[str]:
"""
Get Anna's Archive mirrors from config + defaults.
Returns:
List of AA mirror URLs, starting with defaults then custom additions.
"""
mirrors = list(DEFAULT_AA_MIRRORS)
config = _get_config()
additional = config.get("AA_ADDITIONAL_URLS", "")
if additional:
for url in additional.split(","):
url = url.strip()
if url and url not in mirrors:
mirrors.append(url)
return mirrors
def get_libgen_mirrors() -> List[str]:
"""
Get LibGen mirrors: defaults + any additional from config.
Returns:
List of LibGen mirror URLs (defaults first, then custom additions).
"""
mirrors = list(DEFAULT_LIBGEN_MIRRORS)
config = _get_config()
additional = config.get("LIBGEN_ADDITIONAL_URLS", "")
if additional:
for url in additional.split(","):
url = url.strip()
if url and url not in mirrors:
mirrors.append(url)
return mirrors
def get_zlib_mirrors() -> List[str]:
"""
Get Z-Library mirrors, with primary first.
Returns:
List of Z-Library mirror URLs, primary first.
"""
config = _get_config()
primary = config.get("ZLIB_PRIMARY_URL", DEFAULT_ZLIB_MIRRORS[0])
mirrors = [primary]
# Add other defaults (excluding primary)
for url in DEFAULT_ZLIB_MIRRORS:
if url != primary:
mirrors.append(url)
# Add custom mirrors
additional = config.get("ZLIB_ADDITIONAL_URLS", "")
if additional:
for url in additional.split(","):
url = url.strip()
if url and url not in mirrors:
mirrors.append(url)
return mirrors
def get_zlib_primary_url() -> str:
"""
Get the primary Z-Library mirror URL.
Returns:
Primary Z-Library mirror URL.
"""
config = _get_config()
return config.get("ZLIB_PRIMARY_URL", DEFAULT_ZLIB_MIRRORS[0])
def get_zlib_url_template() -> str:
"""
Get Z-Library URL template using configured primary mirror.
Returns:
URL template with {md5} placeholder.
"""
primary = get_zlib_primary_url()
return f"{primary}/md5/{{md5}}"
def get_welib_mirrors() -> List[str]:
"""
Get Welib mirrors, with primary first.
Returns:
List of Welib mirror URLs, primary first.
"""
config = _get_config()
primary = config.get("WELIB_PRIMARY_URL", DEFAULT_WELIB_MIRRORS[0])
mirrors = [primary]
# Add other defaults (excluding primary)
for url in DEFAULT_WELIB_MIRRORS:
if url != primary:
mirrors.append(url)
# Add custom mirrors
additional = config.get("WELIB_ADDITIONAL_URLS", "")
if additional:
for url in additional.split(","):
url = url.strip()
if url and url not in mirrors:
mirrors.append(url)
return mirrors
def get_welib_primary_url() -> str:
"""
Get the primary Welib mirror URL.
Returns:
Primary Welib mirror URL.
"""
config = _get_config()
return config.get("WELIB_PRIMARY_URL", DEFAULT_WELIB_MIRRORS[0])
def get_welib_url_template() -> str:
"""
Get Welib URL template using configured primary mirror.
Returns:
URL template with {md5} placeholder.
"""
primary = get_welib_primary_url()
return f"{primary}/md5/{{md5}}"
def get_zlib_cookie_domains() -> set:
"""
Get set of Z-Library domains that need full cookie handling.
Used by internal_bypasser for CF bypass cookie management.
Returns:
Set of domain strings (without protocol).
"""
domains = set()
# Add all default domains
for url in DEFAULT_ZLIB_MIRRORS:
domain = url.replace("https://", "").replace("http://", "").split("/")[0]
domains.add(domain)
# Add custom domains
config = _get_config()
additional = config.get("ZLIB_ADDITIONAL_URLS", "")
if additional:
for url in additional.split(","):
url = url.strip()
if url:
domain = url.replace("https://", "").replace("http://", "").split("/")[0]
domains.add(domain)
return domains
+3 -34
View File
@@ -85,7 +85,9 @@ class MultiSelectField(FieldBase):
@dataclass
class OrderableListField(FieldBase):
# Options can be a list or a callable that returns a list (for lazy evaluation)
# Each option: {id, label, description?, disabledReason?, isLocked?}
# Each option: {id, label, description?, disabledReason?, isLocked?, section?, isPinned?}
# - isLocked: toggle is disabled (can't enable/disable)
# - isPinned: can't be reordered (but toggle may still work if not also isLocked)
options: Any = field(default_factory=list)
# Default value: [{id, enabled}, ...] in priority order
default: List[Dict[str, Any]] = field(default_factory=list)
@@ -310,7 +312,6 @@ def sync_env_to_config() -> None:
logger.debug(f"Synced {len(values_to_sync)} ENV values to {tab.name} config: {list(values_to_sync.keys())}")
migrate_legacy_settings()
migrate_libgen_priority()
def migrate_legacy_settings() -> None:
@@ -412,38 +413,6 @@ def migrate_legacy_settings() -> None:
logger.info(f"Migrated content-type routing settings: {list(migrated_sources.keys())}")
def migrate_libgen_priority() -> None:
"""One-time migration to move libgen to position 1 (after aa-fast)."""
source_config = load_config_file("download_sources")
if source_config.get("_LIBGEN_PRIORITY_MIGRATED"):
return
priority = source_config.get("SOURCE_PRIORITY")
if not priority:
save_config_file("download_sources", {"_LIBGEN_PRIORITY_MIGRATED": True})
return
libgen_index = None
for i, entry in enumerate(priority):
if entry.get("id") == "libgen":
libgen_index = i
break
if libgen_index is None or libgen_index == 1:
save_config_file("download_sources", {"_LIBGEN_PRIORITY_MIGRATED": True})
return
libgen_entry = priority.pop(libgen_index)
priority.insert(1, libgen_entry)
save_config_file("download_sources", {
"SOURCE_PRIORITY": priority,
"_LIBGEN_PRIORITY_MIGRATED": True,
})
logger.info("Migrated SOURCE_PRIORITY: moved libgen to position 1")
def get_setting_value(field: SettingsField, tab_name: str) -> Any:
if isinstance(field, (ActionButton, HeadingField)):
return None # Actions and headings don't have values
+5
View File
@@ -175,6 +175,7 @@ def html_get_page(
use_bypasser: bool = False,
selector: Optional[network.AAMirrorSelector] = None,
cancel_flag: Optional[Event] = None,
status_callback: Optional[Callable[[str, Optional[str]], None]] = None,
) -> str:
"""Fetch HTML content from a URL with retry mechanism."""
retry = retry if retry is not None else app_config.MAX_RETRY
@@ -192,6 +193,8 @@ def html_get_page(
try:
if use_bypasser_now and _is_cf_bypass_enabled():
logger.debug(f"GET (bypasser): {current_url}")
if status_callback:
status_callback("resolving", "Bypassing protection")
try:
result = get_bypassed_page(current_url, selector, cancel_flag)
return result or ""
@@ -223,6 +226,8 @@ def html_get_page(
logger.debug(f"403 but cookies now available - retrying with cookies: {current_url}")
continue
logger.info(f"403 detected; switching to bypasser: {current_url}")
if status_callback:
status_callback("resolving", "Bypassing protection...")
use_bypasser_now = True
continue
logger.warning(f"403 error, giving up: {current_url}")
+3 -6
View File
@@ -811,12 +811,9 @@ def _looks_like_ip(s: str) -> bool:
return s.replace(".", "").replace(":", "").isdigit()
def _build_aa_urls() -> List[str]:
"""Build list of available AA URLs from config."""
urls = ["https://annas-archive.se", "https://annas-archive.li", "https://annas-archive.pm", "https://annas-archive.in"]
additional = app_config.get("AA_ADDITIONAL_URLS", "")
if additional:
urls.extend(u.strip() for u in additional.split(",") if u.strip())
return urls
"""Build list of available AA URLs from centralized mirror config."""
from cwa_book_downloader.core.mirrors import get_aa_mirrors
return get_aa_mirrors()
def _initialize_aa_state() -> None:
+1 -10
View File
@@ -290,22 +290,13 @@ def favicon(_: Any = None) -> Response:
"""
return send_from_directory(FRONTEND_DIST, 'favicon.ico', mimetype='image/vnd.microsoft.icon')
# Register bypasser warmup callback for when first WebSocket client connects
# and shutdown callback for when all clients disconnect
# Use app_config to read from settings file (not just env var) so UI changes work after restart
if not app_config.get("USING_EXTERNAL_BYPASSER", False):
from cwa_book_downloader.bypass.internal_bypasser import warmup as bypasser_warmup, shutdown_if_idle as bypasser_shutdown
ws_manager.register_on_first_connect(bypasser_warmup)
ws_manager.register_on_all_disconnect(bypasser_shutdown)
logger.info("Registered Cloudflare bypasser warmup/shutdown on WebSocket connect/disconnect")
if DEBUG:
import subprocess
if app_config.get("USING_EXTERNAL_BYPASSER", False):
_stop_gui = lambda: None
else:
from cwa_book_downloader.bypass.internal_bypasser import _reset_driver as _stop_gui
from cwa_book_downloader.bypass.internal_bypasser import _cleanup_orphan_processes as _stop_gui
@app.route('/api/debug', methods=['GET'])
@login_required
@@ -62,18 +62,21 @@ _CF_BYPASS_REQUIRED = frozenset({"aa-slow-nowait", "aa-slow-wait", "zlib", "weli
# Sources whose URLs come from AA page (multiple mirrors)
_AA_PAGE_SOURCES = frozenset({"aa-slow-nowait", "aa-slow-wait"})
_MD5_URL_TEMPLATES = {
"zlib": "https://z-lib.fm/md5/{md5}",
"welib": "https://welib.org/md5/{md5}",
}
def _get_md5_url_template(source_id: str) -> Optional[str]:
"""Get URL template for MD5-based sources from centralized config."""
from cwa_book_downloader.core import mirrors
_LIBGEN_DOMAINS = [
"https://libgen.gl",
"https://libgen.li",
"https://libgen.bz",
"https://libgen.la",
"https://libgen.vg",
]
if source_id == "zlib":
return mirrors.get_zlib_url_template()
elif source_id == "welib":
return mirrors.get_welib_url_template()
return None
def _get_libgen_domains() -> List[str]:
"""Get LibGen domains from centralized config."""
from cwa_book_downloader.core import mirrors
return mirrors.get_libgen_mirrors()
_LIBGEN_GET_PATTERNS = [
re.compile(r'<a\s+href=["\']([^"\']*get\.php\?md5=[^"\']+&key=[^"\']+)["\'][^>]*>\s*<h2[^>]*>GET</h2>\s*</a>', re.IGNORECASE),
@@ -83,8 +86,28 @@ _LIBGEN_GET_PATTERNS = [
]
def _get_source_priority() -> List[Dict]:
"""Get the current source priority configuration."""
return config.get("SOURCE_PRIORITY") or []
"""Get the full source priority list.
Fast sources (AA Fast, LibGen) are hardcoded first.
Slow sources come from user config.
"""
# Fast sources - always first, hardcoded
fast_sources = []
# AA Fast only if donator key is set
if config.get("AA_DONATOR_KEY"):
fast_sources.append({"id": "aa-fast", "enabled": True})
# LibGen always available
fast_sources.append({"id": "libgen", "enabled": True})
# User's configured slow sources (config won't contain fast sources)
slow_sources = config.get("SOURCE_PRIORITY") or []
# Filter out any legacy fast source entries from old configs
slow_sources = [s for s in slow_sources if s["id"] not in ("aa-fast", "libgen")]
return fast_sources + slow_sources
def _is_source_enabled(source_id: str) -> bool:
@@ -344,7 +367,7 @@ def _parse_book_info_page(soup: BeautifulSoup, book_id: str, fetch_download_coun
if size == "" and "." in stripped:
size = _normalize_size(f)
book_title = _find_in_divs(divs, "🔍")[0].strip("🔍").strip()
book_title = (_find_in_divs(divs, "🔍") or [""])[0].strip("🔍").strip()
# Extract basic information
description = _extract_book_description(soup)
@@ -354,8 +377,8 @@ def _parse_book_info_page(soup: BeautifulSoup, book_id: str, fetch_download_coun
preview=preview,
title=book_title,
content=content,
publisher=_find_in_divs(divs, "icon-[mdi--company]", is_class=True)[0],
author=_find_in_divs(divs, "icon-[mdi--user-edit]", is_class=True)[0],
publisher=(_find_in_divs(divs, "icon-[mdi--company]", is_class=True) or [""])[0],
author=(_find_in_divs(divs, "icon-[mdi--user-edit]", is_class=True) or [""])[0],
format=format,
size=size,
description=description,
@@ -543,14 +566,15 @@ def _get_urls_for_source(
return [url]
# MD5-based sources - generate URL from template
if source_id in _MD5_URL_TEMPLATES:
url = _MD5_URL_TEMPLATES[source_id].format(md5=book_info.id)
template = _get_md5_url_template(source_id)
if template:
url = template.format(md5=book_info.id)
_url_source_types[url] = source_id
return [url]
if source_id == "libgen":
urls = []
for base_url in _LIBGEN_DOMAINS:
for base_url in _get_libgen_domains():
url = f"{base_url}/ads.php?md5={book_info.id}"
_url_source_types[url] = "libgen"
urls.append(url)
@@ -560,7 +584,7 @@ def _get_urls_for_source(
if source_id == "welib":
if status_callback:
status_callback("resolving", "Fetching welib sources")
return _get_download_urls_from_welib(book_info.id, selector=selector, cancel_flag=cancel_flag)
return _get_download_urls_from_welib(book_info.id, selector=selector, cancel_flag=cancel_flag, status_callback=status_callback)
# AA page sources - fetch AA page if not already done
if source_id in _AA_PAGE_SOURCES:
@@ -627,14 +651,21 @@ def _try_download_url(
return None
def _get_download_urls_from_welib(book_id: str, selector: Optional[network.AAMirrorSelector] = None, cancel_flag: Optional[Event] = None) -> List[str]:
def _get_download_urls_from_welib(
book_id: str,
selector: Optional[network.AAMirrorSelector] = None,
cancel_flag: Optional[Event] = None,
status_callback: Optional[Callable[[str, Optional[str]], None]] = None
) -> List[str]:
"""Get download URLs from welib.org (bypasser required)."""
from cwa_book_downloader.core import mirrors
if not _is_source_enabled("welib"):
return []
url = _MD5_URL_TEMPLATES["welib"].format(md5=book_id)
url = mirrors.get_welib_url_template().format(md5=book_id)
logger.info(f"Fetching welib download URLs for {book_id}")
try:
html = downloader.html_get_page(url, use_bypasser=True, selector=selector or network.AAMirrorSelector(), cancel_flag=cancel_flag)
html = downloader.html_get_page(url, use_bypasser=True, selector=selector or network.AAMirrorSelector(), cancel_flag=cancel_flag, status_callback=status_callback)
except Exception as exc:
logger.error_trace(f"Welib fetch failed for {book_id}: {exc}")
return []
@@ -819,13 +850,13 @@ def _get_download_url(
# AA fast download API (JSON response)
if link.startswith(f"{network.get_aa_base_url()}/dyn/api/fast_download.json"):
page = downloader.html_get_page(link, selector=sel, cancel_flag=cancel_flag)
page = downloader.html_get_page(link, selector=sel, cancel_flag=cancel_flag, status_callback=status_callback)
return downloader.get_absolute_url(link, json.loads(page).get("download_url", ""))
if "/ads.php?md5=" in link and any(domain in link for domain in _LIBGEN_DOMAINS):
if "/ads.php?md5=" in link and any(domain in link for domain in _get_libgen_domains()):
return _extract_libgen_download_url(link, cancel_flag)
html = downloader.html_get_page(link, selector=sel, cancel_flag=cancel_flag)
html = downloader.html_get_page(link, selector=sel, cancel_flag=cancel_flag, status_callback=status_callback)
if not html:
return ""
@@ -838,7 +869,7 @@ def _get_download_url(
if not dl:
# Retry after delay if page not fully loaded
time.sleep(2)
html = downloader.html_get_page(link, selector=sel, cancel_flag=cancel_flag)
html = downloader.html_get_page(link, selector=sel, cancel_flag=cancel_flag, status_callback=status_callback)
if html:
soup = BeautifulSoup(html, "html.parser")
dl = soup.find("a", href=True, class_="addDownloadedBook")
+2
View File
@@ -8,6 +8,8 @@ services:
context: .
dockerfile: Dockerfile
target: cwa-bd
cap_add:
- SYS_PTRACE
environment:
DEBUG: true
volumes:
@@ -15,6 +15,9 @@ interface OrderableListFieldProps {
// Represents where the drop indicator should appear
type DropPosition = { index: number; position: 'before' | 'after' } | null;
// Merged item type with all properties
type MergedItem = OrderableListItem & OrderableListOption;
/**
* Merge current value with options to get full item info.
* Items in value take precedence; any options not in value are appended.
@@ -22,9 +25,9 @@ type DropPosition = { index: number; position: 'before' | 'after' } | null;
const mergeValueWithOptions = (
value: OrderableListItem[],
options: OrderableListOption[]
): Array<OrderableListItem & OrderableListOption> => {
): MergedItem[] => {
const optionsMap = new Map(options.map((opt) => [opt.id, opt]));
const result: Array<OrderableListItem & OrderableListOption> = [];
const result: MergedItem[] = [];
// Add items from value (preserves order)
for (const item of value) {
@@ -35,7 +38,7 @@ const mergeValueWithOptions = (
}
}
// Add any remaining options not in value (shouldn't happen normally)
// Add any remaining options not in value
for (const option of optionsMap.values()) {
result.push({ ...option, id: option.id, enabled: false });
}
@@ -56,12 +59,24 @@ export const OrderableListField = ({
const items = mergeValueWithOptions(value ?? [], field.options);
// Check if a move is valid (not crossing pinned items)
const isValidMove = (fromIndex: number): boolean => {
const fromItem = items[fromIndex];
if (fromItem?.isPinned) return false;
return true;
};
const handleDragStart = (e: React.DragEvent, index: number) => {
const item = items[index];
if (item?.isPinned) {
e.preventDefault();
return;
}
setDraggedIndex(index);
dragNodeRef.current = e.currentTarget as HTMLDivElement;
e.dataTransfer.effectAllowed = 'move';
e.dataTransfer.setData('text/plain', String(index));
// Add a slight delay before adding the dragging class for better visual feedback
requestAnimationFrame(() => {
if (dragNodeRef.current) {
dragNodeRef.current.classList.add('opacity-50');
@@ -85,7 +100,12 @@ export const OrderableListField = ({
return;
}
// Determine if we're in the top or bottom half of the target
// Can't drop on pinned items
if (items[index]?.isPinned) {
setDropPosition(null);
return;
}
const rect = e.currentTarget.getBoundingClientRect();
const midpoint = rect.top + rect.height / 2;
const position = e.clientY < midpoint ? 'before' : 'after';
@@ -94,7 +114,6 @@ export const OrderableListField = ({
};
const handleDragLeave = (e: React.DragEvent) => {
// Only clear if we're leaving the item entirely (not entering a child)
const relatedTarget = e.relatedTarget as Node | null;
if (!e.currentTarget.contains(relatedTarget)) {
setDropPosition(null);
@@ -108,27 +127,23 @@ export const OrderableListField = ({
return;
}
// Calculate the actual target index based on drop position
let targetIndex = dropPosition.index;
if (dropPosition.position === 'after') {
targetIndex += 1;
}
// Adjust if dragging from before the target
if (draggedIndex < targetIndex) {
targetIndex -= 1;
}
if (draggedIndex === targetIndex) {
if (draggedIndex === targetIndex || !isValidMove(draggedIndex)) {
handleDragEnd();
return;
}
// Reorder the items
const newItems = [...items];
const [removed] = newItems.splice(draggedIndex, 1);
newItems.splice(targetIndex, 0, removed);
// Convert back to value format
const newValue: OrderableListItem[] = newItems.map((item) => ({
id: item.id,
enabled: item.enabled,
@@ -154,6 +169,7 @@ export const OrderableListField = ({
const moveItem = (fromIndex: number, direction: 'up' | 'down') => {
const toIndex = direction === 'up' ? fromIndex - 1 : fromIndex + 1;
if (toIndex < 0 || toIndex >= items.length) return;
if (!isValidMove(fromIndex)) return;
const newItems = [...items];
[newItems[fromIndex], newItems[toIndex]] = [newItems[toIndex], newItems[fromIndex]];
@@ -166,14 +182,26 @@ export const OrderableListField = ({
onChange(newValue);
};
// Calculate which gap index to show the indicator at (0 = before first item, N = after last item)
const canMoveUp = (index: number): boolean => {
const item = items[index];
if (item?.isPinned) return false;
if (index === 0) return false;
if (items[index - 1]?.isPinned) return false;
return true;
};
const canMoveDown = (index: number): boolean => {
const item = items[index];
if (item?.isPinned) return false;
if (index === items.length - 1) return false;
return true;
};
const getDropGapIndex = (): number | null => {
if (!dropPosition) return null;
if (dropPosition.position === 'before') {
return dropPosition.index;
} else {
return dropPosition.index + 1;
}
return dropPosition.position === 'before'
? dropPosition.index
: dropPosition.index + 1;
};
const dropGapIndex = getDropGapIndex();
@@ -183,137 +211,131 @@ export const OrderableListField = ({
{items.map((item, index) => {
const isDragging = draggedIndex === index;
const isItemDisabled = isDisabled || item.isLocked;
// Show indicator before this item if the gap index matches
const isPinned = item.isPinned ?? false;
const showIndicatorBefore = dropGapIndex === index;
return (
<div key={item.id} className="relative">
{/* Drop indicator - absolutely positioned so it doesn't affect layout */}
{/* Drop indicator */}
{showIndicatorBefore && (
<div className="absolute left-1 right-1 h-1 bg-sky-500 rounded-full z-10 -top-1 -translate-y-1/2" />
)}
<div
draggable
draggable={!isPinned}
onDragStart={(e) => handleDragStart(e, index)}
onDragEnd={handleDragEnd}
onDragOver={(e) => handleDragOver(e, index)}
onDragLeave={handleDragLeave}
onDrop={handleDrop}
className={`
flex items-center gap-3 p-3 rounded-lg border
flex items-center gap-3 p-3 rounded-lg
transition-all duration-150
${isDragging ? 'opacity-50 cursor-grabbing' : 'cursor-grab'}
border-[var(--border-muted)]
${isDisabled ? 'opacity-60' : 'hover:bg-[var(--hover-surface)]'}
${isDragging ? 'opacity-50 cursor-grabbing' : isPinned ? 'cursor-default' : 'cursor-grab'}
border border-[var(--border-muted)]
${isDisabled ? 'opacity-60' : !isPinned ? 'hover:bg-[var(--hover-surface)]' : ''}
`}
>
{/* Reorder Controls */}
<div className="flex flex-col flex-shrink-0 -my-1">
<button
type="button"
onClick={(e) => {
e.stopPropagation();
moveItem(index, 'up');
}}
disabled={index === 0}
className={`
p-1.5 sm:p-0.5 rounded transition-colors
${index === 0
? 'text-gray-300 dark:text-gray-600 cursor-not-allowed'
: 'text-gray-400 hover:text-gray-600 dark:hover:text-gray-300 sm:hover:bg-gray-100 sm:dark:hover:bg-gray-700'
}
`}
aria-label="Move up"
>
<svg className="w-5 h-5 sm:w-4 sm:h-4" fill="none" viewBox="0 0 24 24" stroke="currentColor" strokeWidth={2}>
<path strokeLinecap="round" strokeLinejoin="round" d="M5 15l7-7 7 7" />
</svg>
</button>
<button
type="button"
onClick={(e) => {
e.stopPropagation();
moveItem(index, 'down');
}}
disabled={index === items.length - 1}
className={`
p-1.5 sm:p-0.5 rounded transition-colors
${index === items.length - 1
? 'text-gray-300 dark:text-gray-600 cursor-not-allowed'
: 'text-gray-400 hover:text-gray-600 dark:hover:text-gray-300 sm:hover:bg-gray-100 sm:dark:hover:bg-gray-700'
}
`}
aria-label="Move down"
>
<svg className="w-5 h-5 sm:w-4 sm:h-4" fill="none" viewBox="0 0 24 24" stroke="currentColor" strokeWidth={2}>
<path strokeLinecap="round" strokeLinejoin="round" d="M19 9l-7 7-7-7" />
</svg>
</button>
</div>
{/* Label and Description */}
<div className="flex-1 min-w-0">
<div className="font-medium text-sm">{item.label}</div>
{item.description && (
<div className="text-xs text-[var(--text-muted)] mt-0.5">
{item.description}
</div>
)}
{item.isLocked && item.disabledReason && (
<div className="text-xs text-amber-500 mt-0.5 flex items-center gap-1">
<svg
className="w-3 h-3"
fill="currentColor"
viewBox="0 0 20 20"
>
<path
fillRule="evenodd"
d="M8.257 3.099c.765-1.36 2.722-1.36 3.486 0l5.58 9.92c.75 1.334-.213 2.98-1.742 2.98H4.42c-1.53 0-2.493-1.646-1.743-2.98l5.58-9.92zM11 13a1 1 0 11-2 0 1 1 0 012 0zm-1-8a1 1 0 00-1 1v3a1 1 0 002 0V6a1 1 0 00-1-1z"
clipRule="evenodd"
/>
</svg>
{item.disabledReason}
</div>
)}
</div>
{/* Toggle Switch */}
{(() => {
// Locked items always show as "off" regardless of enabled state
const showAsEnabled = item.enabled && !item.isLocked;
return (
<button
type="button"
role="switch"
aria-checked={showAsEnabled}
onClick={(e) => {
e.stopPropagation();
toggleItem(index);
}}
disabled={isItemDisabled}
className={`
relative inline-flex h-6 w-11 flex-shrink-0 items-center rounded-full
transition-colors duration-200 focus:outline-none focus:ring-2
focus:ring-sky-500/50 disabled:opacity-60 disabled:cursor-not-allowed
${showAsEnabled ? 'bg-sky-600' : 'bg-gray-300 dark:bg-gray-600'}
`}
>
<span
{/* Reorder Controls - hidden for pinned items */}
{!isPinned ? (
<div className="flex flex-col flex-shrink-0 -my-1">
<button
type="button"
onClick={(e) => {
e.stopPropagation();
moveItem(index, 'up');
}}
disabled={!canMoveUp(index)}
className={`
inline-block h-4 w-4 transform rounded-full bg-white
shadow-sm transition-transform duration-200
${showAsEnabled ? 'translate-x-6' : 'translate-x-1'}
p-1.5 sm:p-0.5 rounded transition-colors
${!canMoveUp(index)
? 'text-gray-300 dark:text-gray-600 cursor-not-allowed'
: 'text-gray-400 hover:text-gray-600 dark:hover:text-gray-300 sm:hover:bg-gray-100 sm:dark:hover:bg-gray-700'
}
`}
/>
</button>
);
})()}
aria-label="Move up"
>
<svg className="w-5 h-5 sm:w-4 sm:h-4" fill="none" viewBox="0 0 24 24" stroke="currentColor" strokeWidth={2}>
<path strokeLinecap="round" strokeLinejoin="round" d="M5 15l7-7 7 7" />
</svg>
</button>
<button
type="button"
onClick={(e) => {
e.stopPropagation();
moveItem(index, 'down');
}}
disabled={!canMoveDown(index)}
className={`
p-1.5 sm:p-0.5 rounded transition-colors
${!canMoveDown(index)
? 'text-gray-300 dark:text-gray-600 cursor-not-allowed'
: 'text-gray-400 hover:text-gray-600 dark:hover:text-gray-300 sm:hover:bg-gray-100 sm:dark:hover:bg-gray-700'
}
`}
aria-label="Move down"
>
<svg className="w-5 h-5 sm:w-4 sm:h-4" fill="none" viewBox="0 0 24 24" stroke="currentColor" strokeWidth={2}>
<path strokeLinecap="round" strokeLinejoin="round" d="M19 9l-7 7-7-7" />
</svg>
</button>
</div>
) : (
<div className="w-5 sm:w-4 flex-shrink-0" />
)}
{/* Label and Description */}
<div className="flex-1 min-w-0">
<div className="font-medium text-sm">{item.label}</div>
{item.description && (
<div className="text-xs text-[var(--text-muted)] mt-0.5">
{item.description}
</div>
)}
{item.isLocked && item.disabledReason && (
<div className="text-xs text-amber-500 mt-0.5 flex items-center gap-1">
<svg className="w-3 h-3" fill="currentColor" viewBox="0 0 20 20">
<path
fillRule="evenodd"
d="M8.257 3.099c.765-1.36 2.722-1.36 3.486 0l5.58 9.92c.75 1.334-.213 2.98-1.742 2.98H4.42c-1.53 0-2.493-1.646-1.743-2.98l5.58-9.92zM11 13a1 1 0 11-2 0 1 1 0 012 0zm-1-8a1 1 0 00-1 1v3a1 1 0 002 0V6a1 1 0 00-1-1z"
clipRule="evenodd"
/>
</svg>
{item.disabledReason}
</div>
)}
</div>
{/* Toggle Switch */}
<button
type="button"
role="switch"
aria-checked={item.enabled && !item.isLocked}
onClick={(e) => {
e.stopPropagation();
toggleItem(index);
}}
disabled={isItemDisabled}
className={`
relative inline-flex h-6 w-11 flex-shrink-0 items-center rounded-full
transition-colors duration-200 focus:outline-none focus:ring-2
focus:ring-sky-500/50 disabled:opacity-60 disabled:cursor-not-allowed
${item.enabled && !item.isLocked ? 'bg-sky-600' : 'bg-gray-300 dark:bg-gray-600'}
`}
>
<span
className={`
inline-block h-4 w-4 transform rounded-full bg-white
shadow-sm transition-transform duration-200
${item.enabled && !item.isLocked ? 'translate-x-6' : 'translate-x-1'}
`}
/>
</button>
</div>
</div>
);
})}
{/* Drop indicator after last item - use relative container with absolute indicator */}
{/* Drop indicator after last item */}
{dropGapIndex === items.length && (
<div className="relative h-0">
<div className="absolute left-1 right-1 h-1 bg-sky-500 rounded-full z-10 -top-0.5" />
+1
View File
@@ -101,6 +101,7 @@ export interface OrderableListOption {
description?: string;
disabledReason?: string; // Explanation when item cannot be enabled
isLocked?: boolean; // Item cannot be toggled (e.g., missing dependency)
isPinned?: boolean; // Item cannot be reordered (but can still be toggled if not locked)
}
export interface OrderableListFieldConfig extends BaseField {