diff --git a/Dockerfile b/Dockerfile index e97e74a5..48a26111 100644 --- a/Dockerfile +++ b/Dockerfile @@ -8,13 +8,7 @@ ENV DEBIAN_FRONTEND=noninteractive \ PIP_DISABLE_PIP_VERSION_CHECK=1 \ PIP_DEFAULT_TIMEOUT=100 \ NAME=Calibre-Web-Automated-Book-Downloader \ - FLASK_HOST=0.0.0.0 \ - FLASK_PORT=8084 \ - FLASK_DEBUG=0 \ - STATUS_TIMEOUT=3600 \ PYTHONPATH=/app \ - USE_CF_BYPASS=true \ - AA_BASE_URL=https://annas-archive.org \ UID=1000 \ GID=100 diff --git a/book_manager.py b/book_manager.py index 6de01df8..03252b77 100644 --- a/book_manager.py +++ b/book_manager.py @@ -10,7 +10,7 @@ from logger import setup_logger from config import SUPPORTED_FORMATS, BOOK_LANGUAGE, AA_BASE_URL from env import AA_DONATOR_KEY, USE_CF_BYPASS from models import BookInfo, SearchFilters -import network +import downloader logger = setup_logger(__name__) @@ -61,7 +61,7 @@ def search_books(query: str, filters: SearchFilters) -> List[BookInfo]: f"{filters_query}" ) - html = network.html_get_page(url) + html = downloader.html_get_page(url) if not html: raise Exception("Failed to fetch search results") @@ -128,7 +128,7 @@ def get_book_info(book_id: str) -> BookInfo: BookInfo: Detailed book information """ url = f"{AA_BASE_URL}/md5/{book_id}" - html = network.html_get_page(url) + html = downloader.html_get_page(url) if not html: raise Exception(f"Failed to fetch book info for ID: {book_id}") @@ -207,7 +207,7 @@ def _parse_book_info_page(soup: BeautifulSoup, book_id: str) -> BookInfo: urls = list(external_urls_libgen) + list(external_urls_z_lib) + list(slow_urls_no_waitlist) + list(slow_urls_with_waitlist) for i in range(len(urls)): - urls[i] = network.get_absolute_url(AA_BASE_URL, urls[i]) + urls[i] = downloader.get_absolute_url(AA_BASE_URL, urls[i]) # Extract basic information book_info = BookInfo( @@ -296,7 +296,7 @@ def download_book(book_info: BookInfo, book_path: Path) -> bool: download_url = _get_download_url(link, book_info.title) if download_url != "": logger.info(f"Downloading `{book_info.title}` from `{download_url}`") - data = network.download_url(download_url, book_info.size or "") + data = downloader.download_url(download_url, book_info.size or "") if not data: raise Exception("No data received") @@ -318,10 +318,10 @@ def _get_download_url(link: str, title: str) -> str: url = "" if link.startswith(f"{AA_BASE_URL}/dyn/api/fast_download.json"): - page = network.html_get_page(link) + page = downloader.html_get_page(link) url = json.loads(page).get("download_url") else: - html = network.html_get_page(link) + html = downloader.html_get_page(link) if html == "": return "" @@ -346,4 +346,4 @@ def _get_download_url(link: str, title: str) -> str: else: url = soup.find_all('a', string="GET")[0]['href'] - return network.get_absolute_url(link, url) + return downloader.get_absolute_url(link, url) diff --git a/config.py b/config.py index 6c112ecc..907b0fda 100644 --- a/config.py +++ b/config.py @@ -30,6 +30,30 @@ logger.info(f"STAT INGEST_DIR: {os.stat(env.INGEST_DIR)}") logger.info(f"CROSS_FILE_SYSTEM: {CROSS_FILE_SYSTEM}") # Network settings +_custom_dns = env._CUSTOM_DNS.lower().strip() +_doh_server = "" +if _custom_dns == "google": + CUSTOM_DNS = ["8.8.8.8", "8.8.4.4", "2001:4860:4860::8888", "2001:4860:4860::8844"] + _doh_server = "https://dns.google/dns-query" +elif _custom_dns == "quad9": + CUSTOM_DNS = ["9.9.9.9", "149.112.112.112", "2620:fe::fe", "26620:fe::9"] + _doh_server = "https://dns.quad9.net/dns-query" +elif _custom_dns == "cloudflare": + CUSTOM_DNS = ["1.1.1.1", "1.0.0.1", "2606:4700:4700::1111", "2606:4700:4700::1001"] + _doh_server = "https://cloudflare-dns.com/dns-query" +elif _custom_dns == "opendns": + CUSTOM_DNS = ["208.67.222.222", "208.67.220.220", "2620:119:35::35", "2620:119:53::53"] + _doh_server = "https://doh.opendns.com/dns-query" +else: + _custom_dns_ip = _custom_dns.split(",") + CUSTOM_DNS = [dns.strip() for dns in _custom_dns_ip if dns.replace(":", "").replace(".", "").strip().isdigit()] +logger.info(f"CUSTOM_DNS: {CUSTOM_DNS}") +DOH_SERVER = _doh_server +if env.USE_DOH: + DOH_SERVER = _doh_server +else: + DOH_SERVER = "" +logger.info(f"DOH_SERVER: {DOH_SERVER}") # Proxy settings PROXIES = {} @@ -40,25 +64,10 @@ if env.HTTPS_PROXY: logger.info(f"PROXIES: {PROXIES}") # Anna's Archive settings -aa_available_urls = ["https://annas-archive.org", "https://annas-archive.se", "https://annas-archive.li"] -aa_additional_urls = env.AA_ADDITIONAL_URLS.split(",") -aa_available_urls.extend(aa_additional_urls) - AA_BASE_URL = env._AA_BASE_URL -if AA_BASE_URL == "auto": - logger.info(f"AA_BASE_URL: auto, checking available urls {aa_available_urls}") - for url in aa_available_urls: - try: - import requests - response = requests.get(url) - if response.status_code == 200: - AA_BASE_URL = url - break - except Exception as e: - logger.error_trace(f"Error checking {url}: {e}") - if AA_BASE_URL == "auto": - AA_BASE_URL = aa_available_urls[0] -logger.info(f"AA_BASE_URL: {AA_BASE_URL}") +AA_AVAILABLE_URLS = ["https://annas-archive.org", "https://annas-archive.se", "https://annas-archive.li"] +AA_AVAILABLE_URLS.extend(env._AA_ADDITIONAL_URLS.split(",")) +AA_AVAILABLE_URLS = [url.strip() for url in AA_AVAILABLE_URLS if url.strip()] # File format settings SUPPORTED_FORMATS = env._SUPPORTED_FORMATS.split(",") diff --git a/docker-compose.dev.yml b/docker-compose.dev.yml index b5dd979e..d84e00d2 100644 --- a/docker-compose.dev.yml +++ b/docker-compose.dev.yml @@ -6,9 +6,11 @@ services: dockerfile: Dockerfile environment: FLASK_PORT: 8084 - FLASK_DEBUG: false + DEBUG: true BOOK_LANGUAGE: en USE_BOOK_TITLE: true + CUSTOM_DNS: cloudflare + USE_DOH: true ports: - 8084:8084 restart: unless-stopped diff --git a/downloader.py b/downloader.py new file mode 100644 index 00000000..34d5411e --- /dev/null +++ b/downloader.py @@ -0,0 +1,129 @@ +"""Network operations manager for the book downloader application.""" + +import network +network.init() +import requests +import time +from io import BytesIO +from typing import Optional +from urllib.parse import urlparse +from tqdm import tqdm + +from logger import setup_logger +from config import PROXIES +from env import MAX_RETRY, DEFAULT_SLEEP, USE_CF_BYPASS +if USE_CF_BYPASS: + import cloudflare_bypasser + +logger = setup_logger(__name__) + + +def html_get_page(url: str, retry: int = MAX_RETRY, use_bypasser: bool = False) -> str: + """Fetch HTML content from a URL with retry mechanism. + + Args: + url: Target URL + retry: Number of retry attempts + skip_404: Whether to skip 404 errors + + Returns: + str: HTML content if successful, None otherwise + """ + response = None + try: + logger.debug(f"html_get_page: {url}, retry: {retry}, use_bypasser: {use_bypasser}") + if use_bypasser and USE_CF_BYPASS: + logger.info(f"GET Using Cloudflare Bypasser for: {url}") + response = cloudflare_bypasser.get(url) + logger.debug(f"Cloudflare Bypasser response: {response}") + if response: + return str(response.html) + else: + raise requests.exceptions.RequestException("Failed to bypass Cloudflare") + + logger.info(f"GET: {url}") + response = requests.get(url, proxies=PROXIES) + response.raise_for_status() + logger.debug(f"Success getting: {url}") + time.sleep(1) + return str(response.text) + + except Exception as e: + if retry == 0: + logger.error_trace(f"Failed to fetch page: {url}, error: {e}") + return "" + + if response is not None and response.status_code == 404: + logger.warning(f"404 error for URL: {url}") + return "" + + if response is not None and response.status_code == 403: + if use_bypasser: + logger.warning(f"403 error while using cloudflare bypass for URL: {url}") + return "" + logger.warning(f"403 detected for URL: {url}. Should retry using cloudflare bypass.") + return html_get_page(url, retry - 1, True) + + sleep_time = DEFAULT_SLEEP * (MAX_RETRY - retry + 1) + logger.warning( + f"Retrying GET {url} in {sleep_time} seconds due to error: {e}" + ) + time.sleep(sleep_time) + return html_get_page(url, retry - 1, use_bypasser) + +def download_url(link: str, size: str = "") -> Optional[BytesIO]: + """Download content from URL into a BytesIO buffer. + + Args: + link: URL to download from + + Returns: + BytesIO: Buffer containing downloaded content if successful + """ + try: + logger.info(f"Downloading from: {link}") + response = requests.get(link, stream=True, proxies=PROXIES) + response.raise_for_status() + + total_size : float = 0.0 + try: + # we assume size is in MB + total_size = float(size.strip().replace(" ", "").replace(",", ".").upper()[:-2].strip()) * 1024 * 1024 + except: + total_size = float(response.headers.get('content-length', 0)) + + buffer = BytesIO() + + # Initialize the progress bar with your guess + pbar = tqdm(total=total_size, unit='B', unit_scale=True, desc='Downloading') + for chunk in response.iter_content(chunk_size=1000): + buffer.write(chunk) + pbar.update(len(chunk)) + + pbar.close() + if buffer.tell() * 0.1 < total_size * 0.9: + # Check the content of the buffer if its HTML or binary + if response.headers.get('content-type', '').startswith('text/html'): + logger.warn(f"Failed to download content for {link}. Found HTML content instead.") + return None + return buffer + except requests.exceptions.RequestException as e: + logger.error_trace(f"Failed to download from {link}: {e}") + return None + +def get_absolute_url(base_url: str, url: str) -> str: + """Get absolute URL from relative URL and base URL. + + Args: + base_url: Base URL + url: Relative URL + """ + if url.strip() == "": + return "" + if url.startswith("http"): + return url + parsed_url = urlparse(url) + parsed_base = urlparse(base_url) + if parsed_url.netloc == "" or parsed_url.scheme == "": + parsed_url = parsed_url._replace(netloc=parsed_base.netloc, scheme=parsed_base.scheme) + return parsed_url.geturl() diff --git a/env.py b/env.py index 73084bac..af4af9ac 100644 --- a/env.py +++ b/env.py @@ -16,16 +16,19 @@ HTTP_PROXY = os.getenv("HTTP_PROXY", "").strip() HTTPS_PROXY = os.getenv("HTTPS_PROXY", "").strip() AA_DONATOR_KEY = os.getenv("AA_DONATOR_KEY", "").strip() _AA_BASE_URL = os.getenv("AA_BASE_URL", "auto").strip() -AA_ADDITIONAL_URLS = os.getenv("AA_ADITIINAL_URLS", "") +_AA_ADDITIONAL_URLS = os.getenv("AA_ADDITIONAL_URLS", "").strip() _SUPPORTED_FORMATS = os.getenv("SUPPORTED_FORMATS", "epub,mobi,azw3,fb2,djvu,cbz,cbr").lower() _BOOK_LANGUAGE = os.getenv("BOOK_LANGUAGE", "en").lower() _CUSTOM_SCRIPT = os.getenv("CUSTOM_SCRIPT", "").strip() FLASK_HOST = os.getenv("FLASK_HOST", "0.0.0.0") -FLASK_PORT = int(os.getenv("FLASK_PORT", "5003")) +FLASK_PORT = int(os.getenv("FLASK_PORT", "8084")) FLASK_DEBUG = string_to_bool(os.getenv("FLASK_DEBUG", "False")) +LOG_LEVEL = os.getenv("LOG_LEVEL", "INFO") ENABLE_LOGGING = string_to_bool(os.getenv("ENABLE_LOGGING", "true")) MAIN_LOOP_SLEEP_TIME = int(os.getenv("MAIN_LOOP_SLEEP_TIME", "5")) DOCKERMODE = string_to_bool(os.getenv("DOCKERMODE", "false")) +_CUSTOM_DNS = os.getenv("CUSTOM_DNS", "").strip() +USE_DOH = string_to_bool(os.getenv("USE_DOH", "false")) # Logging settings LOG_FILE = LOG_DIR / "cwa-bookd-downloader.log" \ No newline at end of file diff --git a/logger.py b/logger.py index ddd03c52..122647c6 100644 --- a/logger.py +++ b/logger.py @@ -4,9 +4,17 @@ import logging import sys from pathlib import Path from logging.handlers import RotatingFileHandler -from env import FLASK_DEBUG, LOG_FILE, ENABLE_LOGGING +from env import FLASK_DEBUG, LOG_FILE, ENABLE_LOGGING, LOG_LEVEL +from typing import Any, Optional -def setup_logger(name: str, log_file: Path = LOG_FILE) -> logging.Logger: +class CustomLogger(logging.Logger): + """Custom logger class with additional error_trace method.""" + + def error_trace(self, msg: Any, *args: Any, **kwargs: Any) -> None: + """Log an error message with full stack trace.""" + self.error(msg, *args, exc_info=True, **kwargs) + +def setup_logger(name: str, log_file: Path = LOG_FILE) -> CustomLogger: """Set up and configure a logger instance. Args: @@ -14,18 +22,25 @@ def setup_logger(name: str, log_file: Path = LOG_FILE) -> logging.Logger: log_file: Optional path to log file. If None, logs only to stdout/stderr Returns: - logging.Logger: Configured logger instance + CustomLogger: Configured logger instance with error_trace method """ - logger = logging.getLogger(name) - logger.setLevel(logging.INFO) + # Register our custom logger class + logging.setLoggerClass(CustomLogger) - # Add helper method for error logging with stack trace - def error_trace(self, msg, *args, **kwargs): - """Log an error message with full stack trace.""" - self.error(msg, *args, exc_info=True, **kwargs) - - # Attach the helper method to the logger - logger.error_trace = error_trace.__get__(logger) + # Create logger as CustomLogger instance + logger = CustomLogger(name) + log_level = logging.INFO + if LOG_LEVEL == "DEBUG": + log_level = logging.DEBUG + elif LOG_LEVEL == "INFO": + log_level = logging.INFO + elif LOG_LEVEL == "WARNING": + log_level = logging.WARNING + elif LOG_LEVEL == "ERROR": + log_level = logging.ERROR + elif LOG_LEVEL == "CRITICAL": + log_level = logging.CRITICAL + logger.setLevel(log_level) formatter = logging.Formatter( '%(asctime)s - %(name)s - %(levelname)s - %(filename)s:%(lineno)d - %(message)s' @@ -34,16 +49,13 @@ def setup_logger(name: str, log_file: Path = LOG_FILE) -> logging.Logger: # Console handler for Docker output console_handler = logging.StreamHandler(sys.stdout) console_handler.setFormatter(formatter) - if FLASK_DEBUG: - console_handler.setLevel(logging.DEBUG) - else: - console_handler.setLevel(logging.INFO) - console_handler.addFilter(lambda record: record.levelno < logging.ERROR) # Only allow logs below ERROR + console_handler.setLevel(log_level) + console_handler.addFilter(lambda record: record.levelno < logging.ERROR) # Only allow logs below ERROR to stdout logger.addHandler(console_handler) # Error handler for stderr error_handler = logging.StreamHandler(sys.stderr) - error_handler.setLevel(logging.ERROR) + error_handler.setLevel(logging.ERROR) # Error and above go to stderr error_handler.setFormatter(formatter) logger.addHandler(error_handler) diff --git a/network.py b/network.py index 59fb43eb..a06395cf 100644 --- a/network.py +++ b/network.py @@ -1,21 +1,283 @@ """Network operations manager for the book downloader application.""" import requests -import time -from io import BytesIO import urllib.request -from typing import Optional -from urllib.parse import urlparse -from tqdm import tqdm +from typing import Sequence, Tuple, Any, Union, cast, List, Optional, Callable +import socket +import dns.resolver +from socket import AddressFamily, SocketKind +import urllib.parse +import ssl from logger import setup_logger -from config import PROXIES -from env import MAX_RETRY, DEFAULT_SLEEP, USE_CF_BYPASS -if USE_CF_BYPASS: - import cloudflare_bypasser +from config import PROXIES, AA_BASE_URL, CUSTOM_DNS, AA_AVAILABLE_URLS, DOH_SERVER logger = setup_logger(__name__) -"""Configure urllib opener with appropriate headers.""" + +# Common helper functions for DNS resolution +def _decode_host(host: Union[str, bytes, None]) -> str: + """Convert host to string, handling bytes and None cases.""" + if host is None: + return "" + if isinstance(host, bytes): + return host.decode('utf-8') + return str(host) + +def _decode_port(port: Union[str, bytes, int, None]) -> int: + """Convert port to integer, handling various input types.""" + if port is None: + return 0 + if isinstance(port, (str, bytes)): + return int(port) + return int(port) + +def _is_local_address(host_str: str) -> bool: + """Check if an address is local and should bypass custom DNS.""" + return (host_str == 'localhost' or + host_str.startswith('127.') or + host_str.startswith('::1') or + host_str.startswith('0.0.0.0')) + +# Store the original getaddrinfo function +original_getaddrinfo = socket.getaddrinfo + +class DoHResolver: + """DNS over HTTPS resolver implementation.""" + def __init__(self, provider_url: str, hostname: str, ip: str): + """Initialize DoH resolver with specified provider.""" + self.base_url = provider_url.lower().strip() + self.hostname = hostname # Store the hostname for hostname-based skipping + self.ip = ip # Store IP for direct connections + self.session = requests.Session() + + # Different headers based on provider + if 'google' in self.base_url: + self.session.headers.update({ + 'Accept': 'application/json', + }) + else: + self.session.headers.update({ + 'Accept': 'application/dns-json', + }) + + def resolve(self, hostname: str, record_type: str) -> List[str]: + """Resolve a hostname using DoH. + + Args: + hostname: The hostname to resolve + record_type: The DNS record type (A or AAAA) + + Returns: + List of resolved IP addresses + """ + # Skip resolution for the DoH server itself to prevent recursion + if hostname == self.hostname: + logger.debug(f"Skipping DoH resolution for DoH server itself: {hostname}") + return [self.ip] + + try: + params = { + 'name': hostname, + 'type': 'AAAA' if record_type == 'AAAA' else 'A' + } + + response = self.session.get( + self.base_url, + params=params, + proxies=PROXIES, + timeout=5 + ) + response.raise_for_status() + + data = response.json() + if 'Answer' not in data: + logger.warning(f"DoH resolution failed for {hostname}: {data}") + return [] + + # Extract IP addresses from the response + answers = [answer['data'] for answer in data['Answer'] + if answer.get('type') == (28 if record_type == 'AAAA' else 1)] + logger.debug(f"Resolved {hostname} to {len(answers)} addresses using DoH: {answers}") + return answers + + except Exception as e: + logger.warning(f"DoH resolution failed for {hostname}: {e}") + return [] + +def create_custom_resolver(): + """Create a custom DNS resolver using the configured DNS servers.""" + custom_resolver = dns.resolver.Resolver() + custom_resolver.nameservers = CUSTOM_DNS + return custom_resolver + +def resolve_with_custom_dns(resolver, hostname: str, record_type: str) -> List[str]: + """Resolve hostname using custom DNS resolver. + + Args: + resolver: The DNS resolver to use + hostname: The hostname to resolve + record_type: The DNS record type (A or AAAA) + + Returns: + List of resolved IP addresses + """ + try: + answers = resolver.resolve(hostname, record_type) + return [str(answer) for answer in answers] + except Exception as e: + logger.debug(f"{record_type} resolution failed for {hostname}: {e}") + return [] + +def create_custom_getaddrinfo( + resolve_ipv4: Callable[[str], List[str]], + resolve_ipv6: Callable[[str], List[str]], + skip_check: Optional[Callable[[str], bool]] = None +): + """Create a custom getaddrinfo function that uses the provided resolvers. + + Args: + resolve_ipv4: Function to resolve IPv4 addresses + resolve_ipv6: Function to resolve IPv6 addresses + skip_check: Optional function to check if custom resolution should be skipped + + Returns: + A custom getaddrinfo function + """ + def custom_getaddrinfo( + host: Union[str, bytes, None], + port: Union[str, bytes, int, None], + family: int = 0, + type: int = 0, + proto: int = 0, + flags: int = 0 + ) -> Sequence[Tuple[AddressFamily, SocketKind, int, str, Tuple[Any, ...]]]: + host_str = _decode_host(host) + port_int = _decode_port(port) + + # Skip custom resolution for local addresses or if skip check passes + if _is_local_address(host_str) or (skip_check and skip_check(host_str)): + return original_getaddrinfo(host, port, family, type, proto, flags) + + results: list[Tuple[AddressFamily, SocketKind, int, str, Tuple[Any, ...]]] = [] + + try: + # Try IPv6 first if family allows it + if family == 0 or family == socket.AF_INET6: + logger.debug(f"Resolving IPv6 address for {host_str}") + ipv6_answers = resolve_ipv6(host_str) + for answer in ipv6_answers: + results.append((socket.AF_INET6, cast(SocketKind, type), proto, '', (answer, port_int, 0, 0))) + if ipv6_answers: + logger.debug(f"Found {len(ipv6_answers)} IPv6 addresses for {host_str}") + + # Then try IPv4 + if family == 0 or family == socket.AF_INET: + logger.debug(f"Resolving IPv4 address for {host_str}") + ipv4_answers = resolve_ipv4(host_str) + for answer in ipv4_answers: + results.append((socket.AF_INET, cast(SocketKind, type), proto, '', (answer, port_int))) + if ipv4_answers: + logger.debug(f"Found {len(ipv4_answers)} IPv4 addresses for {host_str}") + + if results: + logger.debug(f"Resolved {host_str} to {len(results)} addresses") + return results + + except Exception as e: + logger.warning(f"Custom DNS resolution failed for {host_str}: {e}, falling back to system DNS") + + # Fall back to system DNS if custom resolution fails + try: + return original_getaddrinfo(host, port, family, type, proto, flags) + except Exception as e: + logger.error(f"System DNS resolution also failed for {host_str}: {e}") + # Last resort: Try to connect to the hostname directly + if family == 0 or family == socket.AF_INET: + logger.warning(f"Using direct hostname as last resort for {host_str}") + return [(socket.AF_INET, cast(SocketKind, type), proto, '', (host_str, port_int))] + else: + raise # Re-raise the exception if we can't provide a last resort + + return custom_getaddrinfo + +def init_doh_resolver(doh_server: str = DOH_SERVER): + """Initialize DNS over HTTPS resolver. + + Args: + doh_server: The DoH server URL + """ + # Pre-resolve the DoH server hostname to prevent recursion + url = urllib.parse.urlparse(doh_server) + server_hostname = url.hostname if url.hostname else '' + server_ip = socket.gethostbyname(server_hostname) + logger.info(f"DoH server {server_hostname} resolved to IP: {server_ip}") + + # Create DoH resolver + doh_resolver = DoHResolver(doh_server, server_hostname, server_ip) + + # Create resolver functions + def resolve_ipv4(hostname: str) -> List[str]: + return doh_resolver.resolve(hostname, 'A') + + def resolve_ipv6(hostname: str) -> List[str]: + return doh_resolver.resolve(hostname, 'AAAA') + + # Skip DoH resolution for the DoH server itself + def skip_doh(hostname: str) -> bool: + return hostname == server_hostname or hostname == server_ip + + # Replace socket.getaddrinfo with our DoH-enabled version + socket.getaddrinfo = cast(Any, create_custom_getaddrinfo( + resolve_ipv4, resolve_ipv6, skip_doh + )) + + logger.info("DoH resolver successfully configured and activated") + return doh_resolver + +def init_custom_resolver(): + """Initialize custom DNS resolver using configured DNS servers.""" + custom_resolver = create_custom_resolver() + + # Create resolver functions + def resolve_ipv4(hostname: str) -> List[str]: + return resolve_with_custom_dns(custom_resolver, hostname, 'A') + + def resolve_ipv6(hostname: str) -> List[str]: + return resolve_with_custom_dns(custom_resolver, hostname, 'AAAA') + + # Replace socket.getaddrinfo with our custom resolver + socket.getaddrinfo = cast(Any, create_custom_getaddrinfo(resolve_ipv4, resolve_ipv6)) + + logger.info("Custom DNS resolver successfully configured and activated") + return custom_resolver + +# Initialize DNS resolvers based on configuration +def init_dns_resolvers(): + """Initialize DNS resolvers based on configuration.""" + if len(CUSTOM_DNS) > 0: + init_custom_resolver() + if DOH_SERVER: + init_doh_resolver() + +# Initialize DNS resolvers +init_dns_resolvers() + +# Check available AA_BASE_URLs if set to auto +if AA_BASE_URL == "auto": + logger.info(f"AA_BASE_URL: auto, checking available urls {AA_AVAILABLE_URLS}") + for url in AA_AVAILABLE_URLS: + try: + response = requests.get(url, proxies=PROXIES) + if response.status_code == 200: + AA_BASE_URL = url + break + except Exception as e: + logger.error_trace(f"Error checking {url}: {e}") + if AA_BASE_URL == "auto": + AA_BASE_URL = AA_AVAILABLE_URLS[0] +logger.info(f"AA_BASE_URL: {AA_BASE_URL}") + +# Configure urllib opener with appropriate headers opener = urllib.request.build_opener() opener.addheaders = [ ('User-agent', 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) ' @@ -24,113 +286,6 @@ opener.addheaders = [ ] urllib.request.install_opener(opener) - -def html_get_page(url: str, retry: int = MAX_RETRY, use_bypasser: bool = False) -> str: - """Fetch HTML content from a URL with retry mechanism. - - Args: - url: Target URL - retry: Number of retry attempts - skip_404: Whether to skip 404 errors - - Returns: - str: HTML content if successful, None otherwise - """ - response = None - try: - logger.debug(f"html_get_page: {url}, retry: {retry}, use_bypasser: {use_bypasser}") - if use_bypasser and USE_CF_BYPASS: - logger.info(f"GET Using Cloudflare Bypasser for: {url}") - response = cloudflare_bypasser.get(url) - logger.debug(f"Cloudflare Bypasser response: {response}") - if response: - return str(response.html) - else: - raise requests.exceptions.RequestException("Failed to bypass Cloudflare") - - logger.info(f"GET: {url}") - response = requests.get(url, proxies=PROXIES) - response.raise_for_status() - logger.debug(f"Success getting: {url}") - time.sleep(1) - return str(response.text) - - except requests.exceptions.RequestException as e: - if retry == 0: - logger.error_trace(f"Failed to fetch page: {url}, error: {e}") - return "" - - if response is not None and response.status_code == 404: - logger.warning(f"404 error for URL: {url}") - return "" - - if response is not None and response.status_code == 403: - if use_bypasser: - logger.warning(f"403 error while using cloudflare bypass for URL: {url}") - return "" - logger.warning(f"403 detected for URL: {url}. Should retry using cloudflare bypass.") - return html_get_page(url, retry - 1, True) - - sleep_time = DEFAULT_SLEEP * (MAX_RETRY - retry + 1) - logger.warning( - f"Retrying GET {url} in {sleep_time} seconds due to error: {e}" - ) - time.sleep(sleep_time) - return html_get_page(url, retry - 1, use_bypasser) - -def download_url(link: str, size: str = "") -> Optional[BytesIO]: - """Download content from URL into a BytesIO buffer. - - Args: - link: URL to download from - - Returns: - BytesIO: Buffer containing downloaded content if successful - """ - try: - logger.info(f"Downloading from: {link}") - response = requests.get(link, stream=True, proxies=PROXIES) - response.raise_for_status() - - total_size : float = 0.0 - try: - # we assume size is in MB - total_size = float(size.strip().replace(" ", "").replace(",", ".").upper()[:-2].strip()) * 1024 * 1024 - except: - total_size = float(response.headers.get('content-length', 0)) - - buffer = BytesIO() - - # Initialize the progress bar with your guess - pbar = tqdm(total=total_size, unit='B', unit_scale=True, desc='Downloading') - for chunk in response.iter_content(chunk_size=1000): - buffer.write(chunk) - pbar.update(len(chunk)) - - pbar.close() - if buffer.tell() * 0.1 < total_size * 0.9: - # Check the content of the buffer if its HTML or binary - if response.headers.get('content-type', '').startswith('text/html'): - logger.warn(f"Failed to download content for {link}. Found HTML content instead.") - return None - return buffer - except requests.exceptions.RequestException as e: - logger.error_trace(f"Failed to download from {link}: {e}") - return None - -def get_absolute_url(base_url: str, url: str) -> str: - """Get absolute URL from relative URL and base URL. - - Args: - base_url: Base URL - url: Relative URL - """ - if url.strip() == "": - return "" - if url.startswith("http"): - return url - parsed_url = urlparse(url) - parsed_base = urlparse(base_url) - if parsed_url.netloc == "" or parsed_url.scheme == "": - parsed_url = parsed_url._replace(netloc=parsed_base.netloc, scheme=parsed_base.scheme) - return parsed_url.geturl() +# Need an empty function to be called by downloader.py +def init(): + pass diff --git a/readme.md b/readme.md index d3280294..99e9b899 100644 --- a/readme.md +++ b/readme.md @@ -60,8 +60,10 @@ An intuitive web interface for searching and requesting book downloads, designed | `UID` | Runtime user ID | `1000` | | `GID` | Runtime group ID | `100` | | `ENABLE_LOGGING` | Enable log file | `true` | +| `LOG_LEVEL` | Log level to use | `info` | If logging is enabld, log folder default location is `/var/log/cwa-book-downloader` +Available log levels: `DEBUG`, `INFO`, `WARNING`, `ERROR`, `CRITICAL`. Higher levels show fewer messages. #### Download Settings @@ -89,11 +91,13 @@ If disabling the cloudflare bypass, you will be using alternative download hosts #### Network Settings -| Variable | Description | Default Value | -| ---------------------- | ----------------------------- | ----------------------- | -| `PORT` | Container external port | `8084` | -| `HTTP_PROXY` | HTTP proxy URL | `` | -| `HTTPS_PROXY` | HTTPS proxy URL | `` | +| Variable | Description | Default Value | +| ---------------------- | ------------------------------- | ----------------------- | +| `AA_ADDITIONAL_URLS` | Proxy URLs for AA (, separated) | `` | +| `HTTP_PROXY` | HTTP proxy URL | `` | +| `HTTPS_PROXY` | HTTPS proxy URL | `` | +| `CUSTOM_DNS` | Custom DNS IP | `` | +| `USE_DOH` | Use DNS over HTTPS | `false` | For proxy configuration, you can specify URLs in the following format: ```bash @@ -107,6 +111,30 @@ HTTPS_PROXY=http://username:password@proxy.example.com:8080 ``` +The `CUSTOM_DNS` setting supports two formats: + +1. **Custom DNS Servers**: A comma-separated list of DNS server IP addresses + - Example: `127.0.0.53,127.0.1.53` (useful for PiHole) + - Supports both IPv4 and IPv6 addresses in the same string + +2. **Preset DNS Providers**: Use one of these predefined options: + - `google` - Google DNS + - `quad9` - Quad9 DNS + - `cloudflare` - Cloudflare DNS + - `opendns` - OpenDNS + +For users experiencing ISP-level website blocks (such as Virgin Media in the UK), using alternative DNS providers like Cloudflare may help bypass these restrictions + +If a `CUSTOM_DNS` is specified from the preset providers, you can also set a `USE_DOH=true` to force using DNS over HTTPS, +which might also help in certain network situations. Note that only `google`, `quad9`, `cloudflare` and `opendns` are +supported for now, and any other value in `CUSTOM_DNS` will make the `USE_DOH` flag ignored. + +Try something like this : +```bash +CUSTOM_DNS=cloudflare +USE_DOH=true +``` + #### Custom configuration | Variable | Description | Default Value | @@ -191,3 +219,4 @@ Please note that the current version: ## 💬 Support For issues or questions, please file an issue on the GitHub repository. + diff --git a/requirements.txt b/requirements.txt index 42affc12..8f9f14e2 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,5 +1,5 @@ flask -requests +requests[socks] beautifulsoup4 tqdm DrissionPage @@ -7,3 +7,4 @@ pyvirtualdisplay types-requests types-beautifulsoup4 types-tqdm +dnspython