diff --git a/book_manager.py b/book_manager.py index cc66000f..46d90955 100644 --- a/book_manager.py +++ b/book_manager.py @@ -5,9 +5,10 @@ from urllib.parse import urlparse, quote from typing import List, Optional, Dict from bs4 import BeautifulSoup from io import BytesIO +import json from logger import setup_logger -from config import SUPPORTED_FORMATS, BOOK_LANGUAGE +from config import SUPPORTED_FORMATS, BOOK_LANGUAGE, AA_DONATOR_KEY from models import BookInfo import network @@ -207,6 +208,7 @@ def download_book(book_id: str, title: str) -> Optional[BytesIO]: Returns: Optional[BytesIO]: Book content buffer if successful """ + download_links = [ f"https://annas-archive.org/slow_download/{book_id}/0/2", f"https://libgen.li/ads.php?md5={book_id}", @@ -216,6 +218,12 @@ def download_book(book_id: str, title: str) -> Optional[BytesIO]: f"https://annas-archive.org/slow_download/{book_id}/0/1" ] + """If AA_DONATOR_KEY is set, use the fast download URL. Else try other sources.""" + if AA_DONATOR_KEY is not None: + download_links.insert(0, + f"https://annas-archive.org/dyn/api/fast_download.json?md5={book_id}&key={AA_DONATOR_KEY}" + ) + for link in download_links: try: download_url = _get_download_url(link, title) @@ -230,6 +238,11 @@ def download_book(book_id: str, title: str) -> Optional[BytesIO]: def _get_download_url(link: str, title: str) -> Optional[str]: """Extract actual download URL from various source pages.""" + + if link.startswith("https://annas-archive.org/dyn/api/fast_download.json"): + page = network.html_get_page(link) + return json.loads(page).get("download_url") + html = network.html_get_page_cf(link) if not html: return None diff --git a/config.py b/config.py index 51762776..731508f1 100644 --- a/config.py +++ b/config.py @@ -26,6 +26,7 @@ INGEST_DIR.mkdir(exist_ok=True) MAX_RETRY = int(os.getenv("MAX_RETRY", 3)) DEFAULT_SLEEP = int(os.getenv("DEFAULT_SLEEP", 5)) CLOUDFLARE_PROXY = os.getenv("CLOUDFLARE_PROXY_URL", "http://localhost:8000") +AA_DONATOR_KEY = os.getenv("AA_DONATOR_KEY", None) # File format settings SUPPORTED_FORMATS = os.getenv("SUPPORTED_FORMATS", "epub,mobi,azw3,fb2,djvu,cbz,cbr") diff --git a/network.py b/network.py index 9f3da35b..561597b7 100644 --- a/network.py +++ b/network.py @@ -7,7 +7,7 @@ import urllib.request from typing import Optional from logger import setup_logger -from config import MAX_RETRY, DEFAULT_SLEEP, CLOUDFLARE_PROXY +from config import MAX_RETRY, DEFAULT_SLEEP, CLOUDFLARE_PROXY, AA_DONATOR_KEY logger = setup_logger(__name__) @@ -71,7 +71,7 @@ def html_get_page_cf(url: str, retry: int = MAX_RETRY) -> Optional[str]: try: logger.info(f"GET_CF: {url}") response = requests.get( - f"{CLOUDFLARE_PROXY}//html?url={url}&retries=3" + f"{CLOUDFLARE_PROXY}/html?url={url}&retries=3" ) time.sleep(1) return response.text diff --git a/readme.md b/readme.md index 8d3231b5..413efd0f 100644 --- a/readme.md +++ b/readme.md @@ -1,6 +1,6 @@ # 📚 Calibre-Web-Automated-Book-Downloader -![Calibre-Web Automated Book Downloader](static/media/logo.png "Calibre-Web Automated Book Downloader") +![Calibre-Web Automated Book Downloader](static/media/logo.png 'Calibre-Web Automated Book Downloader') An intuitive web interface for searching and requesting book downloads, designed to work seamlessly with [Calibre-Web-Automated](https://github.com/crocodilestick/Calibre-Web-Automated). This project streamlines the process of downloading books and preparing them for integration into your Calibre library. @@ -15,26 +15,30 @@ An intuitive web interface for searching and requesting book downloads, designed ## 🖼️ Screenshots -![Main search interface Screenshot](README_images/search.png "Main search interface") +![Main search interface Screenshot](README_images/search.png 'Main search interface') -![Details modal Screenshot placeholder](README_images/details.png "Details modal") +![Details modal Screenshot placeholder](README_images/details.png 'Details modal') -![Download queue Screenshot placeholder](README_images/downloading.png "Download queue") +![Download queue Screenshot placeholder](README_images/downloading.png 'Download queue') ## 🚀 Quick Start ### Prerequisites + - Docker - Docker Compose - A running instance of [Calibre-Web-Automated](https://github.com/crocodilestick/Calibre-Web-Automated) (recommended) ### Installation Steps + 1. Get the docker-compose.yml: + ```bash curl -O https://raw.githubusercontent.com/calibrain/calibre-web-automated-book-downloader/refs/heads/main/docker-compose.yml ``` 2. Start the service: + ```bash docker compose up -d ``` @@ -46,48 +50,58 @@ An intuitive web interface for searching and requesting book downloads, designed ### Environment Variables #### Application Settings -| Variable | Description | Default Value | -|----------|-------------|---------------| -| `FLASK_PORT` | Web interface port | `8084` | -| `FLASK_DEBUG` | Debug mode toggle | `false` | -| `FLASK_HOST` | Web interface binding | `0.0.0.0` | -| `INGEST_DIR` | Book download directory | `/cwa-book-ingest` | -| `UID` | Runtime user ID | `1000` | -| `GID` | Runtime group ID | `100` | + +| Variable | Description | Default Value | +| ------------- | ----------------------- | ------------------ | +| `FLASK_PORT` | Web interface port | `8084` | +| `FLASK_DEBUG` | Debug mode toggle | `false` | +| `FLASK_HOST` | Web interface binding | `0.0.0.0` | +| `INGEST_DIR` | Book download directory | `/cwa-book-ingest` | +| `UID` | Runtime user ID | `1000` | +| `GID` | Runtime group ID | `100` | #### Download Settings -| Variable | Description | Default Value | -|----------|-------------|---------------| -| `MAX_RETRY` | Maximum retry attempts | `3` | -| `DEFAULT_SLEEP` | Retry delay (seconds) | `5` | -| `MAIN_LOOP_SLEEP_TIME` | Processing loop delay (seconds) | `5` | -| `SUPPORTED_FORMATS` | Supported book formats | `epub,mobi,azw3,fb2,djvu,cbz,cbr` | -| `BOOK_LANGUAGE` | Preferred language for books | `en` | -Note that PDF are NOT supported at the moment (they do not get ingested by CWA, but if you want to just dowload them loclaly, you can add `pdf` to the `SUPPORTED_FORMATS` env +| Variable | Description | Default Value | +| ---------------------- | --------------------------------------------------------- | --------------------------------- | +| `MAX_RETRY` | Maximum retry attempts | `3` | +| `DEFAULT_SLEEP` | Retry delay (seconds) | `5` | +| `MAIN_LOOP_SLEEP_TIME` | Processing loop delay (seconds) | `5` | +| `SUPPORTED_FORMATS` | Supported book formats | `epub,mobi,azw3,fb2,djvu,cbz,cbr` | +| `BOOK_LANGUAGE` | Preferred language for books | `en` | +| `AA_DONATOR_KEY` | Optional Donator key for Anna's Archive fast download API | `` | + +Note that PDF are NOT supported at the moment (they do not get ingested by CWA, but if you want to just download them locally, you can add `pdf` to the `SUPPORTED_FORMATS` env + +If you are a donator on AA, you can use your Key in `AA_DONATOR_API_KEY` to speed up downloads and bypass the wait times. #### Network Settings -| Variable | Description | Default Value | -|----------|-------------|---------------| + +| Variable | Description | Default Value | +| ---------------------- | ----------------------------- | ----------------------- | | `CLOUDFLARE_PROXY_URL` | Cloudflare bypass service URL | `http://localhost:8000` | -| `PORT` | Container external port | `8084` | +| `PORT` | Container external port | `8084` | ### Volume Configuration + ```yaml volumes: - /your/local/path:/cwa-book-ingest ``` + Mount should align with your Calibre-Web-Automated ingest folder. ## 🏗️ Architecture The application consists of two key services: + 1. **calibre-web-automated-bookdownloader**: Main application providing web interface and download functionality 2. **cloudflarebypassforscraping**: Support service for handling Cloudflare-protected websites ## 🏥 Health Monitoring Built-in health checks monitor: + - Web interface availability - Download service status - Cloudflare bypass service connection @@ -97,6 +111,7 @@ Checks run every 30 seconds with a 30-second timeout and 3 retries. ## 📝 Logging Logs are available in: + - Container: `/var/logs/calibre-web-automated-bookdownloader.log` - Docker logs: Access via `docker logs` @@ -111,13 +126,17 @@ This project is licensed under the MIT License - see the [LICENSE](LICENSE) file ## ⚠️ Important Disclaimers ### Copyright Notice + While this tool can access various sources including those that might contain copyrighted material (e.g., Anna's Archive), it is designed for legitimate use only. Users are responsible for: + - Ensuring they have the right to download requested materials - Respecting copyright laws and intellectual property rights - Using the tool in compliance with their local regulations ### Duplicate Downloads Warning + Please note that the current version: + - Does not check for existing files in the download directory - Does not verify if books already exist in your Calibre database - Exercise caution when requesting multiple books to avoid duplicates