diff --git a/.dockerignore b/.dockerignore index 34229039..dcaf924c 100644 --- a/.dockerignore +++ b/.dockerignore @@ -10,3 +10,30 @@ Dockerfile docker-compose.yml docker-compose.*.yml readme.md + +# Python specific +__pycache__/ +*.pyc +*.pyo +*.pyd + +# Environment files +.env* + +# Test & Coverage artifacts +.pytest_cache/ +.coverage +htmlcov/ + +# Logs +*.log + +# Build artifacts +build/ +dist/ +*.egg-info/ + +# Virtual environments +venv/ +.venv/ +env/ diff --git a/.github/workflows/build-and-publish-docker-image.yml b/.github/workflows/build-and-publish-docker-image.yml index 38322dbd..ea0f4cdc 100644 --- a/.github/workflows/build-and-publish-docker-image.yml +++ b/.github/workflows/build-and-publish-docker-image.yml @@ -1,4 +1,4 @@ -name: Create and publish a Docker image +name: Create and publish Docker images on: push: @@ -10,7 +10,7 @@ env: IMAGE_NAME: ${{ github.repository }} jobs: - build-and-push-image: + build-and-push-images: runs-on: ubuntu-latest permissions: contents: read @@ -26,8 +26,10 @@ jobs: registry: ${{ env.REGISTRY }} username: ${{ github.actor }} password: ${{ secrets.GITHUB_TOKEN }} - - name: Extract metadata (tags, labels) for Docker - id: meta + + # Build and push main image + - name: Extract metadata for main image + id: meta-main uses: docker/metadata-action@v5 with: images: ${{ env.REGISTRY }}/${{ env.IMAGE_NAME }} @@ -36,23 +38,57 @@ jobs: type=raw,value={{commit_date 'YYYYMMDD'}} type=sha type=ref,event=branch - type=ref,event=tag + type=ref,event=tag + - name: Set up Docker Buildx uses: docker/setup-buildx-action@v3 - - name: Build and push Docker image - id: push + + - name: Build and push main Docker image + id: push-main uses: docker/build-push-action@v5 with: platforms: linux/amd64,linux/arm64 context: . + target: cwa-bd push: true - tags: ${{ steps.meta.outputs.tags }} - labels: ${{ steps.meta.outputs.labels }} + tags: ${{ steps.meta-main.outputs.tags }} + labels: ${{ steps.meta-main.outputs.labels }} - - name: Generate artifact attestation + - name: Generate artifact attestation for main image uses: actions/attest-build-provenance@v2 with: - subject-name: ${{ env.REGISTRY }}/${{ env.IMAGE_NAME}} - subject-digest: ${{ steps.push.outputs.digest }} + subject-name: ${{ env.REGISTRY }}/${{ env.IMAGE_NAME }} + subject-digest: ${{ steps.push-main.outputs.digest }} + push-to-registry: true + + # Build and push tor image + - name: Extract metadata for tor image + id: meta-tor + uses: docker/metadata-action@v5 + with: + images: ${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}-tor + tags: | + type=raw,value=latest,enable={{is_default_branch}} + type=raw,value={{commit_date 'YYYYMMDD'}} + type=sha + type=ref,event=branch + type=ref,event=tag + + - name: Build and push tor Docker image + id: push-tor + uses: docker/build-push-action@v5 + with: + platforms: linux/amd64,linux/arm64 + context: . + target: cwa-bd-tor + push: true + tags: ${{ steps.meta-tor.outputs.tags }} + labels: ${{ steps.meta-tor.outputs.labels }} + + - name: Generate artifact attestation for tor image + uses: actions/attest-build-provenance@v2 + with: + subject-name: ${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}-tor + subject-digest: ${{ steps.push-tor.outputs.digest }} push-to-registry: true diff --git a/.gitignore b/.gitignore index c2ee6f74..e9019636 100644 --- a/.gitignore +++ b/.gitignore @@ -225,4 +225,5 @@ pyrightconfig.json .history .ionide -# End of https://www.toptal.com/developers/gitignore/api/macos,visualstudiocode,python \ No newline at end of file +# End of https://www.toptal.com/developers/gitignore/api/macos,visualstudiocode,python +/downloaded_files diff --git a/.vscode/launch.json b/.vscode/launch.json index 04284594..6cb50cab 100644 --- a/.vscode/launch.json +++ b/.vscode/launch.json @@ -1,12 +1,29 @@ { "version": "0.2.0", "configurations": [ + { + "name": "Python Debugger: Current File", + "type": "debugpy", + "request": "launch", + "program": "${file}", + "console": "integratedTerminal", + "justMyCode": false, + "env": { + "INGEST_DIR": "/tmp/cwa-book-downloader", + "TEMP_DIR": "/tmp/cwa-book-downloader", + "LOG_LEVEL": "DEBUG", + "LOG_ROOT": "/tmp/cwa-book-downloader", + "ENABLE_LOGGING": "true", + "DOCKERMODE": "false", + "FLASK_DEBUG": "true" + }, + }, { "name": "Docker-compose Dev", - "type": "debugpy", // or "debugpy", Node, etc. + "type": "debugpy", // or "debugpy", Node, etc. "request": "launch", "program": "${workspaceFolder}/app.py", - "preLaunchTask": "docker-compose up (dev)", // Spin up dev containers + "preLaunchTask": "docker-compose up (dev)", // Spin up dev containers "postDebugTask": "docker-compose down (dev)", // Optional: tear them down "env": { "INGEST_DIR": "/tmp/cwa-book-downloader" @@ -17,25 +34,12 @@ "type": "debugpy", "request": "launch", "program": "${workspaceFolder}/app.py", - "preLaunchTask": "docker-compose up (prod)", + "preLaunchTask": "docker-compose up (prod)", "postDebugTask": "docker-compose down (prod)", "env": { "INGEST_DIR": "/tmp/cwa-book-downloader" }, }, - { - "type": "debugpy", - "request": "launch", - "name": "Launch cwa-bd app.py", - "program": "${workspaceFolder}/app.py", - "env": { - "DOCKERMODE": "false", - "INGEST_DIR": "/tmp/cwa-book-downloader" - }, - "presentation": { - "hidden": true - } - }, { "type": "chrome", "request": "launch", diff --git a/Dockerfile b/Dockerfile index 48a26111..d56952e5 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,43 +1,120 @@ -FROM python:3.12-slim +# Use python-slim as the base image +FROM python:3.10-slim AS base +# Set shell to bash with pipefail option +SHELL ["/bin/bash", "-o", "pipefail", "-c"] + +# Consistent environment variables grouped together ENV DEBIAN_FRONTEND=noninteractive \ DOCKERMODE=true \ PYTHONUNBUFFERED=1 \ PYTHONDONTWRITEBYTECODE=1 \ + PYTHONIOENCODING=UTF-8 \ PIP_NO_CACHE_DIR=1 \ PIP_DISABLE_PIP_VERSION_CHECK=1 \ PIP_DEFAULT_TIMEOUT=100 \ NAME=Calibre-Web-Automated-Book-Downloader \ PYTHONPATH=/app \ - UID=1000 \ - GID=100 + # UID/GID will be handled by entrypoint script, but TZ/Locale are still needed + LANG=en_US.UTF-8 \ + LANGUAGE=en_US:en \ + LC_ALL=en_US.UTF-8 -# Install minimal dependencies +# Set ARG for build-time expansion (FLASK_PORT), ENV for runtime access +ENV FLASK_PORT=8084 + +# Configure locale, timezone, and perform initial cleanup in a single layer +# User/group creation is removed RUN apt-get update && \ - apt-get install -y --no-install-recommends --no-install-suggests \ - curl \ + apt-get install -y --no-install-recommends \ + # For locale + locales tzdata \ + # For entrypoint + dumb-init \ + # For dumb display xvfb \ + # For user switching + sudo \ + # --- Chromium Browser --- chromium-driver \ - dumb-init && \ - rm -rf /var/lib/apt/lists/* + # For tkinter (pyautogui) + python3-tk && \ + # Cleanup APT cache *after* all installs in this layer + apt-get purge -y --auto-remove -o APT::AutoRemove::RecommendsImportant=false && \ + apt-get clean && \ + rm -rf /var/lib/apt/lists/* && \ + # Default to UTC timezone but will be overridden by the entrypoint script + ln -snf /usr/share/zoneinfo/UTC /etc/localtime && echo UTC > /etc/timezone && \ + # Configure locale + sed -i '/en_US.UTF-8/s/^# //g' /etc/locale.gen && \ + locale-gen en_US.UTF-8 && \ + echo "LC_ALL=en_US.UTF-8" >> /etc/environment && \ + echo "LANG=en_US.UTF-8" > /etc/locale.conf +# Set working directory WORKDIR /app -# Install Python dependencies including playwright +# Install Python dependencies using pip +# Upgrade pip first, then copy requirements and install +# Copying requirements.txt separately leverages build cache +# No --chown needed as it's copied as root COPY requirements.txt . RUN pip install --no-cache-dir -r requirements.txt && \ - rm -rf /root/.cache /app/.cache + # Clean root's pip cache + rm -rf /root/.cache +# Our custom wanabe curl + +RUN echo "#!/bin/sh" > /usr/local/bin/pyrequests && \ + echo 'python -c "import sys, requests; url=sys.argv[1]; r=requests.get(url); print(r.text); sys.exit(0) if r.ok else sys.exit(1)" "$@"' \ + >> /usr/local/bin/pyrequests && \ + chmod +x /usr/local/bin/pyrequests + +# Copy application code *after* dependencies are installed +# No --chown needed as it's copied as root, entrypoint will handle permissions COPY . . -RUN chmod +x /app/entrypoint.sh && \ - # Create necessary directories - mkdir -p /var/log/cwa-book-downloader && \ - mkdir -p /cwa-book-ingest +# Final setup: permissions and directories in one layer +# Only creating directories and setting executable bits. +# Ownership will be handled by the entrypoint script. +RUN mkdir -p /var/log/cwa-book-downloader /cwa-book-ingest && \ + chmod +x /app/entrypoint.sh /app/tor.sh + # chown is removed + + +# Expose the application port EXPOSE ${FLASK_PORT} +# Add healthcheck for container status +# This will run as root initially, but check localhost which should work if the app binds correctly. HEALTHCHECK --interval=30s --timeout=30s --start-period=5s --retries=3 \ - CMD curl -f http://localhost:${FLASK_PORT}/request/api/status || exit 1 + CMD pyrequests http://localhost:${FLASK_PORT}/request/api/status || exit 1 +# Use dumb-init as the entrypoint to handle signals properly ENTRYPOINT ["/usr/bin/dumb-init", "--"] -CMD ["/app/entrypoint.sh"] \ No newline at end of file + + +FROM base AS cwa-bd + +# Default command to run the application entrypoint script +CMD ["/app/entrypoint.sh"] + +FROM base AS cwa-bd-tor + +ENV ENABLE_TOR=true + +# Install Tor and dependencies +RUN apt-get update && \ + apt-get install -y --no-install-recommends \ + # --- Tor --- + tor iptables && \ + # Cleanup APT cache *after* all installs in this layer + apt-get purge -y --auto-remove -o APT::AutoRemove::RecommendsImportant=false && \ + apt-get clean && \ + rm -rf /var/lib/apt/lists/* + +# Tor configuration is handled by entrypoint/script now or permissions set earlier +# RUN chmod +x /app/tor.sh # This is removed as it's done in the base stage setup + +# Override the default command to run Tor +CMD ["/app/tor.sh"] \ No newline at end of file diff --git a/cloudflare_bypasser.py b/cloudflare_bypasser.py index 423762d1..a33a2f0f 100644 --- a/cloudflare_bypasser.py +++ b/cloudflare_bypasser.py @@ -1,90 +1,50 @@ -import time, os, socket -import network +import time +import os +import socket from urllib.parse import urlparse -from DrissionPage import ChromiumPage # type: ignore -from DrissionPage import ChromiumOptions -from DrissionPage._functions.elements import ChromiumElementsList # type: ignore -from DrissionPage._pages.chromium_tab import ChromiumTab # type: ignore +import threading +import env + +# --- SeleniumBase Import --- +from seleniumbase import Driver +from selenium.webdriver.common.by import By +from selenium.webdriver.support.ui import WebDriverWait +from selenium.webdriver.support import expected_conditions as EC +from selenium.common.exceptions import TimeoutException + +import network from logger import setup_logger -from env import MAX_RETRY, DOCKERMODE, DEFAULT_SLEEP -from config import PROXIES, CUSTOM_DNS, DOH_SERVER, AA_BASE_URL +from env import MAX_RETRY, DEFAULT_SLEEP +from config import PROXIES, CUSTOM_DNS, DOH_SERVER logger = setup_logger(__name__) - -_defaultTab : ChromiumTab | None = None - network.init() -def _search_recursively_shadow_root_with_iframe(ele : ChromiumElementsList) -> ChromiumElementsList | None: - if ele.shadow_root: - if ele.shadow_root.child().tag == "iframe": - return ele.shadow_root.child() - else: - for child in ele.children(): - result = _search_recursively_shadow_root_with_iframe(child) - if result: - return result - return None +DRIVER = None +LAST_USED = None +LOCKED = threading.Lock() +TENTATIVE_CURRENT_URL = None -def _search_recursively_shadow_root_with_cf_input(ele : ChromiumElementsList) -> ChromiumElementsList | None: - if ele.shadow_root: - if ele.shadow_root.ele("tag:input"): - return ele.shadow_root.ele("tag:input") - else: - for child in ele.children(): - result = _search_recursively_shadow_root_with_cf_input(child) - if result: - return result - return None - -def _locate_cf_button(driver : ChromiumTab) -> ChromiumElementsList | None: - button : ChromiumElementsList = None - eles = driver.eles("tag:input") - for ele in eles: - if "name" in ele.attrs.keys() and "type" in ele.attrs.keys(): - if "turnstile" in ele.attrs["name"] and ele.attrs["type"] == "hidden": - button = ele.parent().shadow_root.child()("tag:body").shadow_root("tag:input") - break - - if button: - return button - else: - # If the button is not found, search it recursively - logger.debug("Basic search failed. Searching for button recursively.") - ele = driver.ele("tag:body") - iframe = _search_recursively_shadow_root_with_iframe(ele) - if iframe: - button = _search_recursively_shadow_root_with_cf_input(iframe("tag:body")) - else: - logger.debug("Iframe not found. Button search failed.") - return button - -def _click_verification_button(driver: ChromiumTab) -> None: +def _is_bypassed(sb) -> bool: try: - button = _locate_cf_button(driver) - if button: - logger.debug("Verification button found. Attempting to click.") - button.wait.displayed(timeout=DEFAULT_SLEEP) - button.click() - else: - logger.debug("Verification button not found.") - - except Exception as e: - logger.debug(f"Error clicking verification button: {e}") - -def _is_bypassed(driver: ChromiumTab) -> bool: - try: - title = driver.title.lower() - body = driver.ele("tag:body").text.lower() + title = sb.get_title().lower() + body = sb.get_text("body").lower() # Check both title and body for verification messages verification_texts = [ "just a moment", "verify you are human", "verifying you are human", - "needs to review the security of your connection before proceeding" + "needs to review the security of your connection before proceeding", + "checking your browser", + "checking connection", + "attention required", + "access denied", + "needs to review the security of your connection", + "checking the site connection security", + "enable javascript and cookies to continue", + "ray id", ] - for text in verification_texts: if text in title.lower() or text in body.lower(): return False @@ -94,147 +54,193 @@ def _is_bypassed(driver: ChromiumTab) -> bool: logger.debug(f"Error checking page title: {e}") return False -def _bypass(driver: ChromiumTab, max_retries: int = MAX_RETRY) -> None: +def _bypass(sb, max_retries: int = MAX_RETRY) -> None: try_count = 0 - while not _is_bypassed(driver): - logger.info(f"Starting Cloudflare bypass... Retry: {try_count + 1} / {max_retries}") + while not _is_bypassed(sb): if try_count >= max_retries: logger.warning("Exceeded maximum retries. Bypass failed.") break - - logger.info(f"Attempt {try_count + 1}: Verification page detected. Trying to bypass...") + logger.info(f"Bypass attempt {try_count + 1} / {max_retries}") try_count += 1 - time.sleep(DEFAULT_SLEEP) - _click_verification_button(driver) + wait_time = DEFAULT_SLEEP * (try_count - 1) + logger.info(f"Waiting {wait_time}s before trying...") + time.sleep(wait_time) - time.sleep(DEFAULT_SLEEP) + try: + sb.uc_gui_click_captcha() + except Exception as e: + time.sleep(5) + sb.wait_for_element_visible('body') + try: + sb.uc_gui_click_captcha() + except Exception as e: + time.sleep(DEFAULT_SLEEP) + sb.reconnect(DEFAULT_SLEEP) + sb.uc_gui_click_captcha() - if _is_bypassed(driver): - logger.info("Bypass successful.") - else: - logger.info("Bypass failed.") + if _is_bypassed(sb): + logger.info("Bypass successful.") + else: + logger.info("Bypass failed.") -def _get_chromium_options(arguments: list[str]) -> ChromiumOptions: - options = ChromiumOptions() - for argument in arguments: - options.set_argument(argument) +def _get_chromium_args(): + arguments = [ + "-no-sandbox", + ] + # Add proxy settings if configured if PROXIES: - if 'http' in PROXIES: - options.set_argument(f'--proxy-server={PROXIES["http"]}') - logger.debug(f"Setting HTTP proxy: {PROXIES['http']}") - elif 'https' in PROXIES: - options.set_argument(f'--proxy-server={PROXIES["https"]}') - logger.debug(f"Setting HTTPS proxy: {PROXIES['https']}") + proxy_url = PROXIES.get('https') or PROXIES.get('http') + if proxy_url: + arguments.append(f'--proxy-server={proxy_url}') # --- Add Custom DNS settings --- try: - if len(CUSTOM_DNS) > 0: + if len(CUSTOM_DNS) > 0: if DOH_SERVER: logger.info(f"Configuring DNS over HTTPS (DoH) with server: {DOH_SERVER}") - # Enable the DoH feature - options.set_argument(f'--enable-features=DnsOverHttps') - # TODO: This is probably broken and a halucination, # but it should still default to google DOH so its fine... - options.set_argument(f'--dns-over-https-mode="secure"') - options.set_argument(f'--dns-over-https-servers="{DOH_SERVER}"') - + arguments.extend(['--enable-features=DnsOverHttps', '--dns-over-https-mode=secure', f'--dns-over-https-servers="{DOH_SERVER}"']) doh_hostname = urlparse(DOH_SERVER).hostname if doh_hostname: - doh_ip = socket.gethostbyname(doh_hostname) - options.set_argument(f'--host-resolver-rules=MAP {doh_hostname} {doh_ip}') - logger.debug(f"Setting Chromium --host-resolver-rules='MAP {doh_hostname} {doh_ip}'") - else: - logger.info(f"Applying custom DNS servers: {CUSTOM_DNS}") - # Format: "MAP * , MAP * , ..." - # We create a separate MAP rule for each DNS server. - # Chromium should try them based on its internal logic (likely order/availability). - resolver_rules = [] - for dns_server in CUSTOM_DNS: - resolver_rules.append(f"MAP * {dns_server}") - + try: + arguments.append(f'--host-resolver-rules=MAP {doh_hostname} {socket.gethostbyname(doh_hostname)}') + except socket.gaierror: + logger.warning(f"Could not resolve DoH hostname: {doh_hostname}") + elif CUSTOM_DNS: + resolver_rules = [f"MAP * {dns_server}" for dns_server in CUSTOM_DNS] if resolver_rules: - # Join the rules with " , " (comma and space is a common separator) - host_resolver_rules_value = " , ".join(resolver_rules) - options.set_argument(f'--host-resolver-rules={host_resolver_rules_value}') - logger.debug(f"Setting Chromium --host-resolver-rules='{host_resolver_rules_value}'") + arguments.append(f'--host-resolver-rules={",".join(resolver_rules)}') except Exception as e: logger.error_trace(f"Error configuring DNS settings: {e}") - return options + return arguments -def _genScraper() -> ChromiumPage: - arguments = [ - "-no-first-run", - "-force-color-profile=srgb", - "-metrics-recording-only", - "-password-store=basic", - "-use-mock-keychain", - "-export-tagged-pdf", - "-no-default-browser-check", - "-disable-background-mode", - "-enable-features=NetworkService,NetworkServiceInProcess,LoadCryptoTokenExtension,PermuteTLSExtensions", - "-disable-features=FlashDeprecationWarning,EnablePasswordsAccountStorage", - "-deny-permission-prompts", - "-disable-gpu", - "-no-sandbox", - "-accept-lang=en-US", - "-remote-debugging-port=9222" - ] +CHROMIUM_ARGS = _get_chromium_args() - options = _get_chromium_options(arguments) - # Initialize the browser - driver = ChromiumPage(addr_or_opts=options) +def _get(url, retry : int = MAX_RETRY): + try: + logger.info(f"SB_GET: {url}") + sb = _get_driver() + sb.uc_open_with_disconnect(url) + time.sleep(1) + _bypass(sb) + return sb.page_source + except Exception as e: + if retry == 0: + logger.error_trace(f"Failed to initialize browser: {e}") + _reset_driver() + raise e + logger.error_trace(f"Failed to bypass Cloudflare: {e}. Will retry...") + return _get(url, retry - 1) + +def get(url, retry : int = MAX_RETRY): + global LOCKED, TENTATIVE_CURRENT_URL, LAST_USED + with LOCKED: + TENTATIVE_CURRENT_URL = url + ret = _get(url, retry) + LAST_USED = time.time() + return ret + +def _init_driver(): + global DRIVER + if DRIVER: + _reset_driver() + driver = Driver(uc=True, headless=False, chromium_arg=CHROMIUM_ARGS) + DRIVER = driver + time.sleep(DEFAULT_SLEEP) return driver +def _get_driver(): + global DRIVER + global LAST_USED + LAST_USED = time.time() + if not DRIVER: + return _init_driver() + return DRIVER -def _reset_browser() -> None: - logger.info("Resetting chromiumbrowser") - if not DOCKERMODE: - return - global _defaultTab - # Kill the browser - if _defaultTab: - _defaultTab.close() - _defaultTab = None - # Force kill the browser - os.system("pkill -f -i 'chromium'") - os.system("pkill -f -i 'chrom'") - os.system("pkill -f -i 'xvfb'") - time.sleep(1) - -def _init_browser(retry : int = MAX_RETRY) -> ChromiumTab: - global _defaultTab - if _defaultTab: - return _defaultTab - else: - try: - driver = _genScraper() - _defaultTab = driver.get_tabs()[0] - return _defaultTab - except Exception as e: - if retry > 0: - _reset_browser() - else: - logger.error_trace(f"Failed to initialize browser: {e}") - raise e - return _init_browser(retry - 1) - -def get(url : str, retry : int = MAX_RETRY) -> ChromiumTab: - defaultTab = _init_browser() - defaultTab.get(url) +def _reset_driver(): + logger.info("Resetting driver...") + global DRIVER + if DRIVER: + DRIVER.quit() try: - _bypass(defaultTab) + os.system("pkill -f xvfb") except Exception as e: - if retry > 0: - return get(url, retry - 1) - logger.error_trace(f"Failed to bypass Cloudflare for {url}: {e}") - raise e - return defaultTab + logger.warning(f"Error killing xvfb: {e}") + try: + os.system("pkill -f chrom") + except Exception as e: + logger.warning(f"Error killing chrom: {e}") + DRIVER = None -get(AA_BASE_URL) \ No newline at end of file +def _cleanup_driver(): + global LOCKED + global LAST_USED + with LOCKED: + if LAST_USED: + if time.time() - LAST_USED >= env.BYPASS_RELEASE_INACTIVE_MIN * 60: + _reset_driver() + LAST_USED = None + logger.info("Driver reset due to inactivity.") + +def _cleanup_loop(): + while True: + _cleanup_driver() + time.sleep(max(env.BYPASS_RELEASE_INACTIVE_MIN / 2, 1)) + +def _debug_loop(): + while True: + if DRIVER: + try: + # Get URL with fallback to tentative URL + try: + url = DRIVER.current_url + except Exception as e: + url = TENTATIVE_CURRENT_URL or "unknown_url" + + # Create timestamp and filename + timestamp = time.strftime("%Y%m%d_%H%M%S") + filename = f"screenshot_{timestamp}_{url}" + + # Sanitize filename + sanitized_filename = "".join(c if c.isalnum() or c in ('-', '_', '.') else '_' for c in filename) + sanitized_filename = sanitized_filename[:100] + ".png" # Limit length + + # Ensure screenshots directory exists + screenshots_dir = env.LOG_DIR / "screenshots" + screenshots_dir.mkdir(parents=True, exist_ok=True) + + # Save screenshot + full_path = screenshots_dir / sanitized_filename + + DRIVER.save_screenshot(str(full_path)) + except Exception as e: + pass + time.sleep(1) + +def _init_cleanup_thread(): + cleanup_thread = threading.Thread(target=_cleanup_loop) + cleanup_thread.daemon = True + cleanup_thread.start() + if env.DEBUG: + path = env.LOG_DIR / "screenshots" + path.mkdir(parents=True, exist_ok=True) + debug_thread = threading.Thread(target=_debug_loop) + debug_thread.daemon = True + debug_thread.start() + +def wait_for_result(func, timeout : int = 10, condition : any = True): + start_time = time.time() + while time.time() - start_time < timeout: + result = func() + if condition(result): + return result + time.sleep(0.5) + return None +_init_cleanup_thread() diff --git a/config.py b/config.py index 907b0fda..c1884c49 100644 --- a/config.py +++ b/config.py @@ -18,7 +18,8 @@ with open("data/book-languages.json") as file: # Directory settings BASE_DIR = Path(__file__).resolve().parent logger.info(f"BASE_DIR: {BASE_DIR}") -env.LOG_DIR.mkdir(exist_ok=True) +if env.ENABLE_LOGGING: + env.LOG_DIR.mkdir(exist_ok=True) # Create necessary directories env.TMP_DIR.mkdir(exist_ok=True) diff --git a/docker-compose.dev.yml b/docker-compose.dev.yml index 8b7c65de..2b8b75bf 100644 --- a/docker-compose.dev.yml +++ b/docker-compose.dev.yml @@ -1,20 +1,15 @@ -# If you change the FLASK_PORT, do not forget to change it in ports and healthcheck as well. services: - calibre-web-automated-book-downloader: - build : + calibre-web-automated-book-downloader-dev: + extends: + file: ./docker-compose.yml + service: calibre-web-automated-book-downloader + build: context: . dockerfile: Dockerfile + target: cwa-bd environment: - FLASK_PORT: 8084 LOG_LEVEL: debug - BOOK_LANGUAGE: en - USE_BOOK_TITLE: true - CUSTOM_DNS: cloudflare + FLASK_DEBUG: true USE_DOH: true - ports: - - 8084:8084 - restart: unless-stopped - volumes: - # This is where the books will be downloaded to, usually it would be - # the same as whatever you gave in "calibre-web-automated" - - /tmp/data/calibre-web/ingest:/cwa-book-ingest + CUSTOM_DNS: cloudflare + diff --git a/docker-compose.tor.dev.yml b/docker-compose.tor.dev.yml new file mode 100644 index 00000000..d126d139 --- /dev/null +++ b/docker-compose.tor.dev.yml @@ -0,0 +1,15 @@ +services: + calibre-web-automated-book-downloader-tor-dev: + extends: + file: ./docker-compose.yml + service: calibre-web-automated-book-downloader-tor + build: + context: . + dockerfile: Dockerfile + target: cwa-bd-tor + cap_add: + - NET_ADMIN + - NET_RAW + environment: + LOG_LEVEL: debug + FLASK_DEBUG: true \ No newline at end of file diff --git a/docker-compose.tor.yml b/docker-compose.tor.yml new file mode 100644 index 00000000..6fbea80e --- /dev/null +++ b/docker-compose.tor.yml @@ -0,0 +1,19 @@ +services: + calibre-web-automated-book-downloader-tor: + image: ghcr.io/calibrain/calibre-web-automated-book-downloader-tor:latest + environment: + FLASK_PORT: 8084 + LOG_LEVEL: info + BOOK_LANGUAGE: en + USE_BOOK_TITLE: true + FLASK_DEBUG: false + ENABLE_TOR: true + TZ: America/New_York + ports: + - 8084:8084 + restart: unless-stopped + volumes: + # This is where the books will be downloaded to, usually it would be + # the same as whatever you gave in "calibre-web-automated" + - /tmp/data/calibre-web/ingest:/cwa-book-ingest + - /tmp/cwa-book-downloader:/tmp/cwa-book-downloader diff --git a/docker-compose.yml b/docker-compose.yml index 55f14932..98221850 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -1,22 +1,18 @@ -# If you change the FLASK_PORT, do not forget to change it in ports and healthcheck as well. services: calibre-web-automated-book-downloader: image: ghcr.io/calibrain/calibre-web-automated-book-downloader:latest environment: FLASK_PORT: 8084 - FLASK_DEBUG: false + LOG_LEVEL: info BOOK_LANGUAGE: en + USE_BOOK_TITLE: true + FLASK_DEBUG: false + TZ: America/New_York ports: - 8084:8084 - # Uncomment the following lines if you want to enable healthcheck - #healthcheck: - # test: ["CMD", "curl", "-f", "http://localhost:8084/request/api/status"] - # interval: 30s - # timeout: 30s - # retries: 3 - # start_period: 5s restart: unless-stopped volumes: # This is where the books will be downloaded to, usually it would be # the same as whatever you gave in "calibre-web-automated" - /tmp/data/calibre-web/ingest:/cwa-book-ingest + - /tmp/cwa-book-downloader:/tmp/cwa-book-downloader diff --git a/downloader.py b/downloader.py index 34d5411e..93906504 100644 --- a/downloader.py +++ b/downloader.py @@ -34,18 +34,18 @@ def html_get_page(url: str, retry: int = MAX_RETRY, use_bypasser: bool = False) logger.debug(f"html_get_page: {url}, retry: {retry}, use_bypasser: {use_bypasser}") if use_bypasser and USE_CF_BYPASS: logger.info(f"GET Using Cloudflare Bypasser for: {url}") - response = cloudflare_bypasser.get(url) - logger.debug(f"Cloudflare Bypasser response: {response}") - if response: - return str(response.html) + response_html = cloudflare_bypasser.get(url) + logger.debug(f"Cloudflare Bypasser response length: {len(response_html)}") + if response_html.strip() != "": + return response_html else: raise requests.exceptions.RequestException("Failed to bypass Cloudflare") - - logger.info(f"GET: {url}") - response = requests.get(url, proxies=PROXIES) - response.raise_for_status() - logger.debug(f"Success getting: {url}") - time.sleep(1) + else: + logger.info(f"GET: {url}") + response = requests.get(url, proxies=PROXIES) + response.raise_for_status() + logger.debug(f"Success getting: {url}") + time.sleep(1) return str(response.text) except Exception as e: @@ -53,14 +53,14 @@ def html_get_page(url: str, retry: int = MAX_RETRY, use_bypasser: bool = False) logger.error_trace(f"Failed to fetch page: {url}, error: {e}") return "" - if response is not None and response.status_code == 404: + if use_bypasser and USE_CF_BYPASS: + logger.warning(f"Exception while using cloudflare bypass for URL: {url}") + logger.warning(f"Exception: {e}") + logger.warning(f"Response: {response}") + elif response is not None and response.status_code == 404: logger.warning(f"404 error for URL: {url}") return "" - - if response is not None and response.status_code == 403: - if use_bypasser: - logger.warning(f"403 error while using cloudflare bypass for URL: {url}") - return "" + elif response is not None and response.status_code == 403: logger.warning(f"403 detected for URL: {url}. Should retry using cloudflare bypass.") return html_get_page(url, retry - 1, True) diff --git a/entrypoint.sh b/entrypoint.sh index a82b75b0..fa12c216 100644 --- a/entrypoint.sh +++ b/entrypoint.sh @@ -1,6 +1,11 @@ #!/bin/bash set -e +# Configure timezone +if [ "$TZ" ]; then + ln -snf /usr/share/zoneinfo/$TZ /etc/localtime && echo $TZ > /etc/timezone +fi + # Set UID if not set if [ -z "$UID" ]; then UID=1000 @@ -34,4 +39,4 @@ change_ownership /var/log/cwa-book-downloader change_ownership /cwa-book-ingest # Switch to the user (either newly created or existing) and execute the main command -exec su -s /bin/bash "$USERNAME" -c "python -m app" +exec sudo -E -u "$USERNAME" python3 -m app diff --git a/env.py b/env.py index 545bebe7..0b057530 100644 --- a/env.py +++ b/env.py @@ -4,12 +4,13 @@ from pathlib import Path def string_to_bool(s: str) -> bool: return s.lower() in ["true", "yes", "1", "y"] -LOG_DIR = Path("/var/log/cwa-book-downloader") +LOG_ROOT = Path(os.getenv("LOG_ROOT", "/var/log/")) +LOG_DIR = LOG_ROOT / "cwa-book-downloader" TMP_DIR = Path(os.getenv("TMP_DIR", "/tmp/cwa-book-downloader")) INGEST_DIR = Path(os.getenv("INGEST_DIR", "/cwa-book-ingest")) STATUS_TIMEOUT = int(os.getenv("STATUS_TIMEOUT", "3600")) USE_BOOK_TITLE = string_to_bool(os.getenv("USE_BOOK_TITLE", "false")) -MAX_RETRY = int(os.getenv("MAX_RETRY", "3")) +MAX_RETRY = int(os.getenv("MAX_RETRY", "10")) DEFAULT_SLEEP = int(os.getenv("DEFAULT_SLEEP", "5")) USE_CF_BYPASS = string_to_bool(os.getenv("USE_CF_BYPASS", "true")) HTTP_PROXY = os.getenv("HTTP_PROXY", "").strip() @@ -23,12 +24,22 @@ _CUSTOM_SCRIPT = os.getenv("CUSTOM_SCRIPT", "").strip() FLASK_HOST = os.getenv("FLASK_HOST", "0.0.0.0") FLASK_PORT = int(os.getenv("FLASK_PORT", "8084")) FLASK_DEBUG = string_to_bool(os.getenv("FLASK_DEBUG", "False")) +DEBUG = FLASK_DEBUG LOG_LEVEL = os.getenv("LOG_LEVEL", "INFO").upper() ENABLE_LOGGING = string_to_bool(os.getenv("ENABLE_LOGGING", "true")) MAIN_LOOP_SLEEP_TIME = int(os.getenv("MAIN_LOOP_SLEEP_TIME", "5")) DOCKERMODE = string_to_bool(os.getenv("DOCKERMODE", "false")) _CUSTOM_DNS = os.getenv("CUSTOM_DNS", "").strip() USE_DOH = string_to_bool(os.getenv("USE_DOH", "false")) - +BYPASS_RELEASE_INACTIVE_MIN = int(os.getenv("BYPASS_RELEASE_INACTIVE_MIN", "5")) # Logging settings -LOG_FILE = LOG_DIR / "cwa-bookd-downloader.log" \ No newline at end of file +LOG_FILE = LOG_DIR / "cwa-bookd-downloader.log" + +USING_TOR = string_to_bool(os.getenv("USING_TOR", "false")) +# If using Tor, we don't need to set custom DNS, use DOH, or proxy +if USING_TOR: + _CUSTOM_DNS = "" + USE_DOH = False + HTTP_PROXY = "" + HTTPS_PROXY = "" + \ No newline at end of file diff --git a/logger.py b/logger.py index c317d736..44dad95b 100644 --- a/logger.py +++ b/logger.py @@ -62,6 +62,9 @@ def setup_logger(name: str, log_file: Path = LOG_FILE) -> CustomLogger: # File handler if log file is specified try: if ENABLE_LOGGING: + # Create log directory if it doesn't exist + log_dir = log_file.parent + log_dir.mkdir(parents=True, exist_ok=True) file_handler = RotatingFileHandler( log_file, maxBytes=10485760, # 10MB diff --git a/readme.md b/readme.md index 99e9b899..069cce03 100644 --- a/readme.md +++ b/readme.md @@ -57,6 +57,7 @@ An intuitive web interface for searching and requesting book downloads, designed | `FLASK_DEBUG` | Debug mode toggle | `false` | | `FLASK_HOST` | Web interface binding | `0.0.0.0` | | `INGEST_DIR` | Book download directory | `/cwa-book-ingest` | +| `TZ` | Container timezone | `UTC` | | `UID` | Runtime user ID | `1000` | | `GID` | Runtime group ID | `100` | | `ENABLE_LOGGING` | Enable log file | `true` | @@ -65,6 +66,8 @@ An intuitive web interface for searching and requesting book downloads, designed If logging is enabld, log folder default location is `/var/log/cwa-book-downloader` Available log levels: `DEBUG`, `INFO`, `WARNING`, `ERROR`, `CRITICAL`. Higher levels show fewer messages. +Note that if using TOR, the TZ will be calculated automatically based on IP. + #### Download Settings | Variable | Description | Default Value | @@ -166,12 +169,32 @@ volumes: Mount should align with your Calibre-Web-Automated ingest folder. +## 🧅 Tor Variant + +This application also offers a variant that routes all its traffic through the Tor network. This can be useful for enhanced privacy or bypassing network restrictions. + +To use the Tor variant: + +1. Get the Tor-specific docker-compose file: + ```bash + curl -O https://raw.githubusercontent.com/calibrain/calibre-web-automated-book-downloader/refs/heads/main/docker-compose.tor.yml + ``` +2. Start the service using this file: + ```bash + docker compose -f docker-compose.tor.yml up -d + ``` + +**Important Considerations for Tor:** + +* **Capabilities:** This variant requires the `NET_ADMIN` and `NET_RAW` Docker capabilities to configure `iptables` for transparent Tor proxying. +* **Timezone:** When running in Tor mode, the container will attempt to determine the timezone based on the Tor exit node's IP address and set it automatically. This will override the `TZ` environment variable if it is set. +* **Network Settings:** Custom DNS, DoH, and HTTP(S) proxy settings (`CUSTOM_DNS`, `USE_DOH`, `HTTP_PROXY`, `HTTPS_PROXY`) are ignored when using the Tor variant, as all traffic goes through Tor. + ## 🏗️ Architecture -The application consists of two key services: +The application consists of a single service: 1. **calibre-web-automated-bookdownloader**: Main application providing web interface and download functionality -2. **cloudflarebypassforscraping**: Support service for handling Cloudflare-protected websites ## 🏥 Health Monitoring @@ -182,6 +205,11 @@ Built-in health checks monitor: - Cloudflare bypass service connection Checks run every 30 seconds with a 30-second timeout and 3 retries. +You can enable by adding this to your compose : +``` +HEALTHCHECK --interval=30s --timeout=30s --start-period=5s --retries=3 \ + CMD pyrequests http://localhost:8084/request/api/status || exit 1 +``` ## 📝 Logging diff --git a/requirements.txt b/requirements.txt index 8f9f14e2..b77399bd 100644 --- a/requirements.txt +++ b/requirements.txt @@ -2,9 +2,7 @@ flask requests[socks] beautifulsoup4 tqdm -DrissionPage pyvirtualdisplay -types-requests -types-beautifulsoup4 -types-tqdm dnspython +pyautogui +seleniumbase diff --git a/tor.sh b/tor.sh new file mode 100644 index 00000000..5b11342a --- /dev/null +++ b/tor.sh @@ -0,0 +1,103 @@ +#!/bin/bash +set -e +echo "[*] Installing Tor and dependencies..." +echo "[*] Writing Tor transparent proxy config..." + +cat < /etc/tor/torrc +VirtualAddrNetworkIPv4 10.192.0.0/10 +AutomapHostsOnResolve 1 +TransPort 9040 +DNSPort 53 +Log notice file /var/log/tor/notices.log +EOF + +echo "[*] Setting up DNS..." +cat < /etc/resolv.conf +127.0.0.1 +EOF + +echo "[*] Starting Tor..." +service tor start + +echo "[*] Setting up iptables rules..." + +iptables -F +iptables -t nat -F + +# Don't redirect Tor's own traffic +iptables -t nat -A OUTPUT -m owner --uid-owner debian-tor -j RETURN + +# Allow loopback +iptables -t nat -A OUTPUT -o lo -j RETURN + +# Redirect all TCP to Tor's TransPort +iptables -t nat -A OUTPUT -p tcp --syn -j REDIRECT --to-ports 9040 + +# For UDP DNS queries +iptables -t nat -A OUTPUT -p udp --dport 53 ! -d 127.0.0.1 -j DNAT --to-destination 127.0.0.1:53 + + +# For TCP DNS queries (some DNS queries may use TCP) +iptables -t nat -A OUTPUT -p tcp --dport 53 ! -d 127.0.0.1 -j DNAT --to-destination 127.0.0.1:53 + +echo "[✓] Transparent Tor routing enabled." + +# Wait a bit to ensure Tor has bootstrapped +echo "[*] Waiting for Tor to finish bootstrapping... (up to 5 minutes)" +timeout 300 bash -c ' + while ! grep -q "Bootstrapped 100%" <(tail -n 20 -F /var/log/tor/notices.log 2>/dev/null); do + printf "\r\033[KCurrent log: %s" "$(tail -n 1 /var/log/tor/notices.log 2>/dev/null)" + sleep 1 + done + # Print a newline when finished. + echo "" +' + +echo "[✓] Tor is ready." + +# Check if outgoing IP is using Tor +echo "[*] Verifying Tor connectivity..." +RESULT=$(pyrequests https://check.torproject.org/api/ip) +echo "RESULT: $RESULT" +IS_TOR=$(echo "$RESULT" | grep -oP '"IsTor":\s*\K(true|false)') +IP=$(echo "$RESULT" | grep -oP '"IP":\s*"\K[^"]+') +if [[ "$IS_TOR" == "true" ]]; then + echo "[✓] Success! Traffic is routed through Tor. Current IP: $IP" +else + echo "[✗] Warning: Traffic is NOT using Tor. Current IP: $IP" + exit 1 +fi + +# Set correct timezone +# First check what is the timezone based on the IP +# Then set the timezone + +# Get timezone from IP +TIMEZONE=$(pyrequests https://ipapi.co/timezone) +# If TIMEZONE is not set, use the default timezone +echo "[*] Current Timezone : $(date +%Z). IP Timezone: $TIMEZONE" + +# Set timezone in Docker-compatible way +if [ -f "/usr/share/zoneinfo/$TIMEZONE" ]; then + # Remove existing symlink if it exists + rm -f /etc/localtime + # Create new symlink + ln -sf /usr/share/zoneinfo/$TIMEZONE /etc/localtime + # Set timezone file + echo "$TIMEZONE" > /etc/timezone + # Set TZ environment variable + export TZ=$TIMEZONE + # Verify the change + echo "[✓] Timezone set to $TIMEZONE" + echo "[*] Current time: $(date)" + echo "[*] Timezone verification: $(date +%Z)" +else + echo "[!] Warning: Timezone file not found: $TIMEZONE" + echo "[*] Available timezones:" + ls -la /usr/share/zoneinfo/ + echo "[*] Falling back to container's default timezone: $TZ" +fi + +# Run the entrypoint script +echo "[*] Running entrypoint script..." +./entrypoint.sh