From e1d94f9cb5ef5602bf5ef7faf445fd295f5b1fc5 Mon Sep 17 00:00:00 2001 From: CaliBrain Date: Mon, 14 Jul 2025 16:05:09 -0400 Subject: [PATCH] Update download handling and URL selection logic (#203) Changed intermediate download file naming in backend.py to use .crdownload extension. Refactored URL selection logic in book_manager.py to adjust order and inclusion based on Cloudflare bypass and donator key. Added extra Chromium arguments in cloudflare_bypasser.py to ignore certificate and SSL errors for improved bypass reliability. --- backend.py | 3 +-- book_manager.py | 9 ++++----- cloudflare_bypasser.py | 7 +++++++ 3 files changed, 12 insertions(+), 7 deletions(-) diff --git a/backend.py b/backend.py index c0e62e5c..77b067dd 100644 --- a/backend.py +++ b/backend.py @@ -136,8 +136,7 @@ def _download_book(book_id: str) -> Optional[str]: if CUSTOM_SCRIPT: logger.info(f"Running custom script: {CUSTOM_SCRIPT}") subprocess.run([CUSTOM_SCRIPT, book_path]) - - intermediate_path = INGEST_DIR / book_id # Without extension + intermediate_path = INGEST_DIR / f"{book_id}.crdownload" final_path = INGEST_DIR / book_name if os.path.exists(book_path): diff --git a/book_manager.py b/book_manager.py index 5209e82d..f46be857 100644 --- a/book_manager.py +++ b/book_manager.py @@ -219,9 +219,7 @@ def _parse_book_info_page(soup: BeautifulSoup, book_id: str) -> BookInfo: pass if USE_CF_BYPASS: - urls = ( - list(slow_urls_no_waitlist) - + list(external_urls_libgen) + urls = (list(external_urls_libgen) + list(slow_urls_with_waitlist) + list(external_urls_z_lib) ) @@ -229,10 +227,11 @@ def _parse_book_info_page(soup: BeautifulSoup, book_id: str) -> BookInfo: urls = ( list(external_urls_libgen) + list(external_urls_z_lib) - + list(slow_urls_no_waitlist) - + list(slow_urls_with_waitlist) ) + if AA_DONATOR_KEY != "": + urls = list(slow_urls_no_waitlist) + urls + for i in range(len(urls)): urls[i] = downloader.get_absolute_url(AA_BASE_URL, urls[i]) diff --git a/cloudflare_bypasser.py b/cloudflare_bypasser.py index 26269d01..a6ae0de5 100644 --- a/cloudflare_bypasser.py +++ b/cloudflare_bypasser.py @@ -110,6 +110,13 @@ def _bypass(sb, max_retries: int = MAX_RETRY) -> None: def _get_chromium_args(): arguments = [ + # Ignore certificate and SSL errors (similar to curl's --insecure) + "--ignore-certificate-errors", + "--ignore-ssl-errors", + "--allow-running-insecure-content", + "--disable-web-security", + "--ignore-certificate-errors-spki-list", + "--ignore-certificate-errors-skip-list" ] # Conditionally add verbose logging arguments