Tor support, use SeleniumBase instead of DrissionPage (#123)

Refactor: Improve Docker build, add Tor support, use SeleniumBase

- Overhauled Dockerfile:
    - Switched to python:3.10-slim base.
    - Implemented multi-stage builds (base, standard, tor).
    - Consolidated RUN layers for efficiency.
    - Added locale/timezone setup.
- Added Tor setup and iptables configuration in dedicated stage/script.
- Replaced DrissionPage with SeleniumBase for Cloudflare bypassing.
- Added Tor support via `docker-compose.tor.yml` and `tor.sh` script.
- Updated `docker-compose.yml` to build locally and changed default port
to 8083.
- Added `docker-compose.dev.yml` and `docker-compose.tor.dev.yml`.
- Updated `entrypoint.sh` for timezone and sudo usage.
- Added/Updated environment variables (`TZ`, `USING_TOR`, etc.).
- Improved `.dockerignore`.
- Updated `readme.md` to reflect port changes, build process, document
`TZ`, and add details about the new Tor variant.
This commit is contained in:
CaliBrain
2025-04-13 02:34:39 -04:00
committed by GitHub
parent e0e7470ac6
commit 57f15b4d95
18 changed files with 602 additions and 277 deletions
+27
View File
@@ -10,3 +10,30 @@ Dockerfile
docker-compose.yml
docker-compose.*.yml
readme.md
# Python specific
__pycache__/
*.pyc
*.pyo
*.pyd
# Environment files
.env*
# Test & Coverage artifacts
.pytest_cache/
.coverage
htmlcov/
# Logs
*.log
# Build artifacts
build/
dist/
*.egg-info/
# Virtual environments
venv/
.venv/
env/
@@ -1,4 +1,4 @@
name: Create and publish a Docker image
name: Create and publish Docker images
on:
push:
@@ -10,7 +10,7 @@ env:
IMAGE_NAME: ${{ github.repository }}
jobs:
build-and-push-image:
build-and-push-images:
runs-on: ubuntu-latest
permissions:
contents: read
@@ -26,8 +26,10 @@ jobs:
registry: ${{ env.REGISTRY }}
username: ${{ github.actor }}
password: ${{ secrets.GITHUB_TOKEN }}
- name: Extract metadata (tags, labels) for Docker
id: meta
# Build and push main image
- name: Extract metadata for main image
id: meta-main
uses: docker/metadata-action@v5
with:
images: ${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}
@@ -36,23 +38,57 @@ jobs:
type=raw,value={{commit_date 'YYYYMMDD'}}
type=sha
type=ref,event=branch
type=ref,event=tag
type=ref,event=tag
- name: Set up Docker Buildx
uses: docker/setup-buildx-action@v3
- name: Build and push Docker image
id: push
- name: Build and push main Docker image
id: push-main
uses: docker/build-push-action@v5
with:
platforms: linux/amd64,linux/arm64
context: .
target: cwa-bd
push: true
tags: ${{ steps.meta.outputs.tags }}
labels: ${{ steps.meta.outputs.labels }}
tags: ${{ steps.meta-main.outputs.tags }}
labels: ${{ steps.meta-main.outputs.labels }}
- name: Generate artifact attestation
- name: Generate artifact attestation for main image
uses: actions/attest-build-provenance@v2
with:
subject-name: ${{ env.REGISTRY }}/${{ env.IMAGE_NAME}}
subject-digest: ${{ steps.push.outputs.digest }}
subject-name: ${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}
subject-digest: ${{ steps.push-main.outputs.digest }}
push-to-registry: true
# Build and push tor image
- name: Extract metadata for tor image
id: meta-tor
uses: docker/metadata-action@v5
with:
images: ${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}-tor
tags: |
type=raw,value=latest,enable={{is_default_branch}}
type=raw,value={{commit_date 'YYYYMMDD'}}
type=sha
type=ref,event=branch
type=ref,event=tag
- name: Build and push tor Docker image
id: push-tor
uses: docker/build-push-action@v5
with:
platforms: linux/amd64,linux/arm64
context: .
target: cwa-bd-tor
push: true
tags: ${{ steps.meta-tor.outputs.tags }}
labels: ${{ steps.meta-tor.outputs.labels }}
- name: Generate artifact attestation for tor image
uses: actions/attest-build-provenance@v2
with:
subject-name: ${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}-tor
subject-digest: ${{ steps.push-tor.outputs.digest }}
push-to-registry: true
+2 -1
View File
@@ -225,4 +225,5 @@ pyrightconfig.json
.history
.ionide
# End of https://www.toptal.com/developers/gitignore/api/macos,visualstudiocode,python
# End of https://www.toptal.com/developers/gitignore/api/macos,visualstudiocode,python
/downloaded_files
+20 -16
View File
@@ -1,12 +1,29 @@
{
"version": "0.2.0",
"configurations": [
{
"name": "Python Debugger: Current File",
"type": "debugpy",
"request": "launch",
"program": "${file}",
"console": "integratedTerminal",
"justMyCode": false,
"env": {
"INGEST_DIR": "/tmp/cwa-book-downloader",
"TEMP_DIR": "/tmp/cwa-book-downloader",
"LOG_LEVEL": "DEBUG",
"LOG_ROOT": "/tmp/cwa-book-downloader",
"ENABLE_LOGGING": "true",
"DOCKERMODE": "false",
"FLASK_DEBUG": "true"
},
},
{
"name": "Docker-compose Dev",
"type": "debugpy", // or "debugpy", Node, etc.
"type": "debugpy", // or "debugpy", Node, etc.
"request": "launch",
"program": "${workspaceFolder}/app.py",
"preLaunchTask": "docker-compose up (dev)", // Spin up dev containers
"preLaunchTask": "docker-compose up (dev)", // Spin up dev containers
"postDebugTask": "docker-compose down (dev)", // Optional: tear them down
"env": {
"INGEST_DIR": "/tmp/cwa-book-downloader"
@@ -17,25 +34,12 @@
"type": "debugpy",
"request": "launch",
"program": "${workspaceFolder}/app.py",
"preLaunchTask": "docker-compose up (prod)",
"preLaunchTask": "docker-compose up (prod)",
"postDebugTask": "docker-compose down (prod)",
"env": {
"INGEST_DIR": "/tmp/cwa-book-downloader"
},
},
{
"type": "debugpy",
"request": "launch",
"name": "Launch cwa-bd app.py",
"program": "${workspaceFolder}/app.py",
"env": {
"DOCKERMODE": "false",
"INGEST_DIR": "/tmp/cwa-book-downloader"
},
"presentation": {
"hidden": true
}
},
{
"type": "chrome",
"request": "launch",
+93 -16
View File
@@ -1,43 +1,120 @@
FROM python:3.12-slim
# Use python-slim as the base image
FROM python:3.10-slim AS base
# Set shell to bash with pipefail option
SHELL ["/bin/bash", "-o", "pipefail", "-c"]
# Consistent environment variables grouped together
ENV DEBIAN_FRONTEND=noninteractive \
DOCKERMODE=true \
PYTHONUNBUFFERED=1 \
PYTHONDONTWRITEBYTECODE=1 \
PYTHONIOENCODING=UTF-8 \
PIP_NO_CACHE_DIR=1 \
PIP_DISABLE_PIP_VERSION_CHECK=1 \
PIP_DEFAULT_TIMEOUT=100 \
NAME=Calibre-Web-Automated-Book-Downloader \
PYTHONPATH=/app \
UID=1000 \
GID=100
# UID/GID will be handled by entrypoint script, but TZ/Locale are still needed
LANG=en_US.UTF-8 \
LANGUAGE=en_US:en \
LC_ALL=en_US.UTF-8
# Install minimal dependencies
# Set ARG for build-time expansion (FLASK_PORT), ENV for runtime access
ENV FLASK_PORT=8084
# Configure locale, timezone, and perform initial cleanup in a single layer
# User/group creation is removed
RUN apt-get update && \
apt-get install -y --no-install-recommends --no-install-suggests \
curl \
apt-get install -y --no-install-recommends \
# For locale
locales tzdata \
# For entrypoint
dumb-init \
# For dumb display
xvfb \
# For user switching
sudo \
# --- Chromium Browser ---
chromium-driver \
dumb-init && \
rm -rf /var/lib/apt/lists/*
# For tkinter (pyautogui)
python3-tk && \
# Cleanup APT cache *after* all installs in this layer
apt-get purge -y --auto-remove -o APT::AutoRemove::RecommendsImportant=false && \
apt-get clean && \
rm -rf /var/lib/apt/lists/* && \
# Default to UTC timezone but will be overridden by the entrypoint script
ln -snf /usr/share/zoneinfo/UTC /etc/localtime && echo UTC > /etc/timezone && \
# Configure locale
sed -i '/en_US.UTF-8/s/^# //g' /etc/locale.gen && \
locale-gen en_US.UTF-8 && \
echo "LC_ALL=en_US.UTF-8" >> /etc/environment && \
echo "LANG=en_US.UTF-8" > /etc/locale.conf
# Set working directory
WORKDIR /app
# Install Python dependencies including playwright
# Install Python dependencies using pip
# Upgrade pip first, then copy requirements and install
# Copying requirements.txt separately leverages build cache
# No --chown needed as it's copied as root
COPY requirements.txt .
RUN pip install --no-cache-dir -r requirements.txt && \
rm -rf /root/.cache /app/.cache
# Clean root's pip cache
rm -rf /root/.cache
# Our custom wanabe curl
RUN echo "#!/bin/sh" > /usr/local/bin/pyrequests && \
echo 'python -c "import sys, requests; url=sys.argv[1]; r=requests.get(url); print(r.text); sys.exit(0) if r.ok else sys.exit(1)" "$@"' \
>> /usr/local/bin/pyrequests && \
chmod +x /usr/local/bin/pyrequests
# Copy application code *after* dependencies are installed
# No --chown needed as it's copied as root, entrypoint will handle permissions
COPY . .
RUN chmod +x /app/entrypoint.sh && \
# Create necessary directories
mkdir -p /var/log/cwa-book-downloader && \
mkdir -p /cwa-book-ingest
# Final setup: permissions and directories in one layer
# Only creating directories and setting executable bits.
# Ownership will be handled by the entrypoint script.
RUN mkdir -p /var/log/cwa-book-downloader /cwa-book-ingest && \
chmod +x /app/entrypoint.sh /app/tor.sh
# chown is removed
# Expose the application port
EXPOSE ${FLASK_PORT}
# Add healthcheck for container status
# This will run as root initially, but check localhost which should work if the app binds correctly.
HEALTHCHECK --interval=30s --timeout=30s --start-period=5s --retries=3 \
CMD curl -f http://localhost:${FLASK_PORT}/request/api/status || exit 1
CMD pyrequests http://localhost:${FLASK_PORT}/request/api/status || exit 1
# Use dumb-init as the entrypoint to handle signals properly
ENTRYPOINT ["/usr/bin/dumb-init", "--"]
CMD ["/app/entrypoint.sh"]
FROM base AS cwa-bd
# Default command to run the application entrypoint script
CMD ["/app/entrypoint.sh"]
FROM base AS cwa-bd-tor
ENV ENABLE_TOR=true
# Install Tor and dependencies
RUN apt-get update && \
apt-get install -y --no-install-recommends \
# --- Tor ---
tor iptables && \
# Cleanup APT cache *after* all installs in this layer
apt-get purge -y --auto-remove -o APT::AutoRemove::RecommendsImportant=false && \
apt-get clean && \
rm -rf /var/lib/apt/lists/*
# Tor configuration is handled by entrypoint/script now or permissions set earlier
# RUN chmod +x /app/tor.sh # This is removed as it's done in the base stage setup
# Override the default command to run Tor
CMD ["/app/tor.sh"]
+187 -181
View File
@@ -1,90 +1,50 @@
import time, os, socket
import network
import time
import os
import socket
from urllib.parse import urlparse
from DrissionPage import ChromiumPage # type: ignore
from DrissionPage import ChromiumOptions
from DrissionPage._functions.elements import ChromiumElementsList # type: ignore
from DrissionPage._pages.chromium_tab import ChromiumTab # type: ignore
import threading
import env
# --- SeleniumBase Import ---
from seleniumbase import Driver
from selenium.webdriver.common.by import By
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
from selenium.common.exceptions import TimeoutException
import network
from logger import setup_logger
from env import MAX_RETRY, DOCKERMODE, DEFAULT_SLEEP
from config import PROXIES, CUSTOM_DNS, DOH_SERVER, AA_BASE_URL
from env import MAX_RETRY, DEFAULT_SLEEP
from config import PROXIES, CUSTOM_DNS, DOH_SERVER
logger = setup_logger(__name__)
_defaultTab : ChromiumTab | None = None
network.init()
def _search_recursively_shadow_root_with_iframe(ele : ChromiumElementsList) -> ChromiumElementsList | None:
if ele.shadow_root:
if ele.shadow_root.child().tag == "iframe":
return ele.shadow_root.child()
else:
for child in ele.children():
result = _search_recursively_shadow_root_with_iframe(child)
if result:
return result
return None
DRIVER = None
LAST_USED = None
LOCKED = threading.Lock()
TENTATIVE_CURRENT_URL = None
def _search_recursively_shadow_root_with_cf_input(ele : ChromiumElementsList) -> ChromiumElementsList | None:
if ele.shadow_root:
if ele.shadow_root.ele("tag:input"):
return ele.shadow_root.ele("tag:input")
else:
for child in ele.children():
result = _search_recursively_shadow_root_with_cf_input(child)
if result:
return result
return None
def _locate_cf_button(driver : ChromiumTab) -> ChromiumElementsList | None:
button : ChromiumElementsList = None
eles = driver.eles("tag:input")
for ele in eles:
if "name" in ele.attrs.keys() and "type" in ele.attrs.keys():
if "turnstile" in ele.attrs["name"] and ele.attrs["type"] == "hidden":
button = ele.parent().shadow_root.child()("tag:body").shadow_root("tag:input")
break
if button:
return button
else:
# If the button is not found, search it recursively
logger.debug("Basic search failed. Searching for button recursively.")
ele = driver.ele("tag:body")
iframe = _search_recursively_shadow_root_with_iframe(ele)
if iframe:
button = _search_recursively_shadow_root_with_cf_input(iframe("tag:body"))
else:
logger.debug("Iframe not found. Button search failed.")
return button
def _click_verification_button(driver: ChromiumTab) -> None:
def _is_bypassed(sb) -> bool:
try:
button = _locate_cf_button(driver)
if button:
logger.debug("Verification button found. Attempting to click.")
button.wait.displayed(timeout=DEFAULT_SLEEP)
button.click()
else:
logger.debug("Verification button not found.")
except Exception as e:
logger.debug(f"Error clicking verification button: {e}")
def _is_bypassed(driver: ChromiumTab) -> bool:
try:
title = driver.title.lower()
body = driver.ele("tag:body").text.lower()
title = sb.get_title().lower()
body = sb.get_text("body").lower()
# Check both title and body for verification messages
verification_texts = [
"just a moment",
"verify you are human",
"verifying you are human",
"needs to review the security of your connection before proceeding"
"needs to review the security of your connection before proceeding",
"checking your browser",
"checking connection",
"attention required",
"access denied",
"needs to review the security of your connection",
"checking the site connection security",
"enable javascript and cookies to continue",
"ray id",
]
for text in verification_texts:
if text in title.lower() or text in body.lower():
return False
@@ -94,147 +54,193 @@ def _is_bypassed(driver: ChromiumTab) -> bool:
logger.debug(f"Error checking page title: {e}")
return False
def _bypass(driver: ChromiumTab, max_retries: int = MAX_RETRY) -> None:
def _bypass(sb, max_retries: int = MAX_RETRY) -> None:
try_count = 0
while not _is_bypassed(driver):
logger.info(f"Starting Cloudflare bypass... Retry: {try_count + 1} / {max_retries}")
while not _is_bypassed(sb):
if try_count >= max_retries:
logger.warning("Exceeded maximum retries. Bypass failed.")
break
logger.info(f"Attempt {try_count + 1}: Verification page detected. Trying to bypass...")
logger.info(f"Bypass attempt {try_count + 1} / {max_retries}")
try_count += 1
time.sleep(DEFAULT_SLEEP)
_click_verification_button(driver)
wait_time = DEFAULT_SLEEP * (try_count - 1)
logger.info(f"Waiting {wait_time}s before trying...")
time.sleep(wait_time)
time.sleep(DEFAULT_SLEEP)
try:
sb.uc_gui_click_captcha()
except Exception as e:
time.sleep(5)
sb.wait_for_element_visible('body')
try:
sb.uc_gui_click_captcha()
except Exception as e:
time.sleep(DEFAULT_SLEEP)
sb.reconnect(DEFAULT_SLEEP)
sb.uc_gui_click_captcha()
if _is_bypassed(driver):
logger.info("Bypass successful.")
else:
logger.info("Bypass failed.")
if _is_bypassed(sb):
logger.info("Bypass successful.")
else:
logger.info("Bypass failed.")
def _get_chromium_options(arguments: list[str]) -> ChromiumOptions:
options = ChromiumOptions()
for argument in arguments:
options.set_argument(argument)
def _get_chromium_args():
arguments = [
"-no-sandbox",
]
# Add proxy settings if configured
if PROXIES:
if 'http' in PROXIES:
options.set_argument(f'--proxy-server={PROXIES["http"]}')
logger.debug(f"Setting HTTP proxy: {PROXIES['http']}")
elif 'https' in PROXIES:
options.set_argument(f'--proxy-server={PROXIES["https"]}')
logger.debug(f"Setting HTTPS proxy: {PROXIES['https']}")
proxy_url = PROXIES.get('https') or PROXIES.get('http')
if proxy_url:
arguments.append(f'--proxy-server={proxy_url}')
# --- Add Custom DNS settings ---
try:
if len(CUSTOM_DNS) > 0:
if len(CUSTOM_DNS) > 0:
if DOH_SERVER:
logger.info(f"Configuring DNS over HTTPS (DoH) with server: {DOH_SERVER}")
# Enable the DoH feature
options.set_argument(f'--enable-features=DnsOverHttps')
# TODO: This is probably broken and a halucination,
# but it should still default to google DOH so its fine...
options.set_argument(f'--dns-over-https-mode="secure"')
options.set_argument(f'--dns-over-https-servers="{DOH_SERVER}"')
arguments.extend(['--enable-features=DnsOverHttps', '--dns-over-https-mode=secure', f'--dns-over-https-servers="{DOH_SERVER}"'])
doh_hostname = urlparse(DOH_SERVER).hostname
if doh_hostname:
doh_ip = socket.gethostbyname(doh_hostname)
options.set_argument(f'--host-resolver-rules=MAP {doh_hostname} {doh_ip}')
logger.debug(f"Setting Chromium --host-resolver-rules='MAP {doh_hostname} {doh_ip}'")
else:
logger.info(f"Applying custom DNS servers: {CUSTOM_DNS}")
# Format: "MAP * <dns1>, MAP * <dns2>, ..."
# We create a separate MAP rule for each DNS server.
# Chromium should try them based on its internal logic (likely order/availability).
resolver_rules = []
for dns_server in CUSTOM_DNS:
resolver_rules.append(f"MAP * {dns_server}")
try:
arguments.append(f'--host-resolver-rules=MAP {doh_hostname} {socket.gethostbyname(doh_hostname)}')
except socket.gaierror:
logger.warning(f"Could not resolve DoH hostname: {doh_hostname}")
elif CUSTOM_DNS:
resolver_rules = [f"MAP * {dns_server}" for dns_server in CUSTOM_DNS]
if resolver_rules:
# Join the rules with " , " (comma and space is a common separator)
host_resolver_rules_value = " , ".join(resolver_rules)
options.set_argument(f'--host-resolver-rules={host_resolver_rules_value}')
logger.debug(f"Setting Chromium --host-resolver-rules='{host_resolver_rules_value}'")
arguments.append(f'--host-resolver-rules={",".join(resolver_rules)}')
except Exception as e:
logger.error_trace(f"Error configuring DNS settings: {e}")
return options
return arguments
def _genScraper() -> ChromiumPage:
arguments = [
"-no-first-run",
"-force-color-profile=srgb",
"-metrics-recording-only",
"-password-store=basic",
"-use-mock-keychain",
"-export-tagged-pdf",
"-no-default-browser-check",
"-disable-background-mode",
"-enable-features=NetworkService,NetworkServiceInProcess,LoadCryptoTokenExtension,PermuteTLSExtensions",
"-disable-features=FlashDeprecationWarning,EnablePasswordsAccountStorage",
"-deny-permission-prompts",
"-disable-gpu",
"-no-sandbox",
"-accept-lang=en-US",
"-remote-debugging-port=9222"
]
CHROMIUM_ARGS = _get_chromium_args()
options = _get_chromium_options(arguments)
# Initialize the browser
driver = ChromiumPage(addr_or_opts=options)
def _get(url, retry : int = MAX_RETRY):
try:
logger.info(f"SB_GET: {url}")
sb = _get_driver()
sb.uc_open_with_disconnect(url)
time.sleep(1)
_bypass(sb)
return sb.page_source
except Exception as e:
if retry == 0:
logger.error_trace(f"Failed to initialize browser: {e}")
_reset_driver()
raise e
logger.error_trace(f"Failed to bypass Cloudflare: {e}. Will retry...")
return _get(url, retry - 1)
def get(url, retry : int = MAX_RETRY):
global LOCKED, TENTATIVE_CURRENT_URL, LAST_USED
with LOCKED:
TENTATIVE_CURRENT_URL = url
ret = _get(url, retry)
LAST_USED = time.time()
return ret
def _init_driver():
global DRIVER
if DRIVER:
_reset_driver()
driver = Driver(uc=True, headless=False, chromium_arg=CHROMIUM_ARGS)
DRIVER = driver
time.sleep(DEFAULT_SLEEP)
return driver
def _get_driver():
global DRIVER
global LAST_USED
LAST_USED = time.time()
if not DRIVER:
return _init_driver()
return DRIVER
def _reset_browser() -> None:
logger.info("Resetting chromiumbrowser")
if not DOCKERMODE:
return
global _defaultTab
# Kill the browser
if _defaultTab:
_defaultTab.close()
_defaultTab = None
# Force kill the browser
os.system("pkill -f -i 'chromium'")
os.system("pkill -f -i 'chrom'")
os.system("pkill -f -i 'xvfb'")
time.sleep(1)
def _init_browser(retry : int = MAX_RETRY) -> ChromiumTab:
global _defaultTab
if _defaultTab:
return _defaultTab
else:
try:
driver = _genScraper()
_defaultTab = driver.get_tabs()[0]
return _defaultTab
except Exception as e:
if retry > 0:
_reset_browser()
else:
logger.error_trace(f"Failed to initialize browser: {e}")
raise e
return _init_browser(retry - 1)
def get(url : str, retry : int = MAX_RETRY) -> ChromiumTab:
defaultTab = _init_browser()
defaultTab.get(url)
def _reset_driver():
logger.info("Resetting driver...")
global DRIVER
if DRIVER:
DRIVER.quit()
try:
_bypass(defaultTab)
os.system("pkill -f xvfb")
except Exception as e:
if retry > 0:
return get(url, retry - 1)
logger.error_trace(f"Failed to bypass Cloudflare for {url}: {e}")
raise e
return defaultTab
logger.warning(f"Error killing xvfb: {e}")
try:
os.system("pkill -f chrom")
except Exception as e:
logger.warning(f"Error killing chrom: {e}")
DRIVER = None
get(AA_BASE_URL)
def _cleanup_driver():
global LOCKED
global LAST_USED
with LOCKED:
if LAST_USED:
if time.time() - LAST_USED >= env.BYPASS_RELEASE_INACTIVE_MIN * 60:
_reset_driver()
LAST_USED = None
logger.info("Driver reset due to inactivity.")
def _cleanup_loop():
while True:
_cleanup_driver()
time.sleep(max(env.BYPASS_RELEASE_INACTIVE_MIN / 2, 1))
def _debug_loop():
while True:
if DRIVER:
try:
# Get URL with fallback to tentative URL
try:
url = DRIVER.current_url
except Exception as e:
url = TENTATIVE_CURRENT_URL or "unknown_url"
# Create timestamp and filename
timestamp = time.strftime("%Y%m%d_%H%M%S")
filename = f"screenshot_{timestamp}_{url}"
# Sanitize filename
sanitized_filename = "".join(c if c.isalnum() or c in ('-', '_', '.') else '_' for c in filename)
sanitized_filename = sanitized_filename[:100] + ".png" # Limit length
# Ensure screenshots directory exists
screenshots_dir = env.LOG_DIR / "screenshots"
screenshots_dir.mkdir(parents=True, exist_ok=True)
# Save screenshot
full_path = screenshots_dir / sanitized_filename
DRIVER.save_screenshot(str(full_path))
except Exception as e:
pass
time.sleep(1)
def _init_cleanup_thread():
cleanup_thread = threading.Thread(target=_cleanup_loop)
cleanup_thread.daemon = True
cleanup_thread.start()
if env.DEBUG:
path = env.LOG_DIR / "screenshots"
path.mkdir(parents=True, exist_ok=True)
debug_thread = threading.Thread(target=_debug_loop)
debug_thread.daemon = True
debug_thread.start()
def wait_for_result(func, timeout : int = 10, condition : any = True):
start_time = time.time()
while time.time() - start_time < timeout:
result = func()
if condition(result):
return result
time.sleep(0.5)
return None
_init_cleanup_thread()
+2 -1
View File
@@ -18,7 +18,8 @@ with open("data/book-languages.json") as file:
# Directory settings
BASE_DIR = Path(__file__).resolve().parent
logger.info(f"BASE_DIR: {BASE_DIR}")
env.LOG_DIR.mkdir(exist_ok=True)
if env.ENABLE_LOGGING:
env.LOG_DIR.mkdir(exist_ok=True)
# Create necessary directories
env.TMP_DIR.mkdir(exist_ok=True)
+9 -14
View File
@@ -1,20 +1,15 @@
# If you change the FLASK_PORT, do not forget to change it in ports and healthcheck as well.
services:
calibre-web-automated-book-downloader:
build :
calibre-web-automated-book-downloader-dev:
extends:
file: ./docker-compose.yml
service: calibre-web-automated-book-downloader
build:
context: .
dockerfile: Dockerfile
target: cwa-bd
environment:
FLASK_PORT: 8084
LOG_LEVEL: debug
BOOK_LANGUAGE: en
USE_BOOK_TITLE: true
CUSTOM_DNS: cloudflare
FLASK_DEBUG: true
USE_DOH: true
ports:
- 8084:8084
restart: unless-stopped
volumes:
# This is where the books will be downloaded to, usually it would be
# the same as whatever you gave in "calibre-web-automated"
- /tmp/data/calibre-web/ingest:/cwa-book-ingest
CUSTOM_DNS: cloudflare
+15
View File
@@ -0,0 +1,15 @@
services:
calibre-web-automated-book-downloader-tor-dev:
extends:
file: ./docker-compose.yml
service: calibre-web-automated-book-downloader-tor
build:
context: .
dockerfile: Dockerfile
target: cwa-bd-tor
cap_add:
- NET_ADMIN
- NET_RAW
environment:
LOG_LEVEL: debug
FLASK_DEBUG: true
+19
View File
@@ -0,0 +1,19 @@
services:
calibre-web-automated-book-downloader-tor:
image: ghcr.io/calibrain/calibre-web-automated-book-downloader-tor:latest
environment:
FLASK_PORT: 8084
LOG_LEVEL: info
BOOK_LANGUAGE: en
USE_BOOK_TITLE: true
FLASK_DEBUG: false
ENABLE_TOR: true
TZ: America/New_York
ports:
- 8084:8084
restart: unless-stopped
volumes:
# This is where the books will be downloaded to, usually it would be
# the same as whatever you gave in "calibre-web-automated"
- /tmp/data/calibre-web/ingest:/cwa-book-ingest
- /tmp/cwa-book-downloader:/tmp/cwa-book-downloader
+5 -9
View File
@@ -1,22 +1,18 @@
# If you change the FLASK_PORT, do not forget to change it in ports and healthcheck as well.
services:
calibre-web-automated-book-downloader:
image: ghcr.io/calibrain/calibre-web-automated-book-downloader:latest
environment:
FLASK_PORT: 8084
FLASK_DEBUG: false
LOG_LEVEL: info
BOOK_LANGUAGE: en
USE_BOOK_TITLE: true
FLASK_DEBUG: false
TZ: America/New_York
ports:
- 8084:8084
# Uncomment the following lines if you want to enable healthcheck
#healthcheck:
# test: ["CMD", "curl", "-f", "http://localhost:8084/request/api/status"]
# interval: 30s
# timeout: 30s
# retries: 3
# start_period: 5s
restart: unless-stopped
volumes:
# This is where the books will be downloaded to, usually it would be
# the same as whatever you gave in "calibre-web-automated"
- /tmp/data/calibre-web/ingest:/cwa-book-ingest
- /tmp/cwa-book-downloader:/tmp/cwa-book-downloader
+16 -16
View File
@@ -34,18 +34,18 @@ def html_get_page(url: str, retry: int = MAX_RETRY, use_bypasser: bool = False)
logger.debug(f"html_get_page: {url}, retry: {retry}, use_bypasser: {use_bypasser}")
if use_bypasser and USE_CF_BYPASS:
logger.info(f"GET Using Cloudflare Bypasser for: {url}")
response = cloudflare_bypasser.get(url)
logger.debug(f"Cloudflare Bypasser response: {response}")
if response:
return str(response.html)
response_html = cloudflare_bypasser.get(url)
logger.debug(f"Cloudflare Bypasser response length: {len(response_html)}")
if response_html.strip() != "":
return response_html
else:
raise requests.exceptions.RequestException("Failed to bypass Cloudflare")
logger.info(f"GET: {url}")
response = requests.get(url, proxies=PROXIES)
response.raise_for_status()
logger.debug(f"Success getting: {url}")
time.sleep(1)
else:
logger.info(f"GET: {url}")
response = requests.get(url, proxies=PROXIES)
response.raise_for_status()
logger.debug(f"Success getting: {url}")
time.sleep(1)
return str(response.text)
except Exception as e:
@@ -53,14 +53,14 @@ def html_get_page(url: str, retry: int = MAX_RETRY, use_bypasser: bool = False)
logger.error_trace(f"Failed to fetch page: {url}, error: {e}")
return ""
if response is not None and response.status_code == 404:
if use_bypasser and USE_CF_BYPASS:
logger.warning(f"Exception while using cloudflare bypass for URL: {url}")
logger.warning(f"Exception: {e}")
logger.warning(f"Response: {response}")
elif response is not None and response.status_code == 404:
logger.warning(f"404 error for URL: {url}")
return ""
if response is not None and response.status_code == 403:
if use_bypasser:
logger.warning(f"403 error while using cloudflare bypass for URL: {url}")
return ""
elif response is not None and response.status_code == 403:
logger.warning(f"403 detected for URL: {url}. Should retry using cloudflare bypass.")
return html_get_page(url, retry - 1, True)
+6 -1
View File
@@ -1,6 +1,11 @@
#!/bin/bash
set -e
# Configure timezone
if [ "$TZ" ]; then
ln -snf /usr/share/zoneinfo/$TZ /etc/localtime && echo $TZ > /etc/timezone
fi
# Set UID if not set
if [ -z "$UID" ]; then
UID=1000
@@ -34,4 +39,4 @@ change_ownership /var/log/cwa-book-downloader
change_ownership /cwa-book-ingest
# Switch to the user (either newly created or existing) and execute the main command
exec su -s /bin/bash "$USERNAME" -c "python -m app"
exec sudo -E -u "$USERNAME" python3 -m app
+15 -4
View File
@@ -4,12 +4,13 @@ from pathlib import Path
def string_to_bool(s: str) -> bool:
return s.lower() in ["true", "yes", "1", "y"]
LOG_DIR = Path("/var/log/cwa-book-downloader")
LOG_ROOT = Path(os.getenv("LOG_ROOT", "/var/log/"))
LOG_DIR = LOG_ROOT / "cwa-book-downloader"
TMP_DIR = Path(os.getenv("TMP_DIR", "/tmp/cwa-book-downloader"))
INGEST_DIR = Path(os.getenv("INGEST_DIR", "/cwa-book-ingest"))
STATUS_TIMEOUT = int(os.getenv("STATUS_TIMEOUT", "3600"))
USE_BOOK_TITLE = string_to_bool(os.getenv("USE_BOOK_TITLE", "false"))
MAX_RETRY = int(os.getenv("MAX_RETRY", "3"))
MAX_RETRY = int(os.getenv("MAX_RETRY", "10"))
DEFAULT_SLEEP = int(os.getenv("DEFAULT_SLEEP", "5"))
USE_CF_BYPASS = string_to_bool(os.getenv("USE_CF_BYPASS", "true"))
HTTP_PROXY = os.getenv("HTTP_PROXY", "").strip()
@@ -23,12 +24,22 @@ _CUSTOM_SCRIPT = os.getenv("CUSTOM_SCRIPT", "").strip()
FLASK_HOST = os.getenv("FLASK_HOST", "0.0.0.0")
FLASK_PORT = int(os.getenv("FLASK_PORT", "8084"))
FLASK_DEBUG = string_to_bool(os.getenv("FLASK_DEBUG", "False"))
DEBUG = FLASK_DEBUG
LOG_LEVEL = os.getenv("LOG_LEVEL", "INFO").upper()
ENABLE_LOGGING = string_to_bool(os.getenv("ENABLE_LOGGING", "true"))
MAIN_LOOP_SLEEP_TIME = int(os.getenv("MAIN_LOOP_SLEEP_TIME", "5"))
DOCKERMODE = string_to_bool(os.getenv("DOCKERMODE", "false"))
_CUSTOM_DNS = os.getenv("CUSTOM_DNS", "").strip()
USE_DOH = string_to_bool(os.getenv("USE_DOH", "false"))
BYPASS_RELEASE_INACTIVE_MIN = int(os.getenv("BYPASS_RELEASE_INACTIVE_MIN", "5"))
# Logging settings
LOG_FILE = LOG_DIR / "cwa-bookd-downloader.log"
LOG_FILE = LOG_DIR / "cwa-bookd-downloader.log"
USING_TOR = string_to_bool(os.getenv("USING_TOR", "false"))
# If using Tor, we don't need to set custom DNS, use DOH, or proxy
if USING_TOR:
_CUSTOM_DNS = ""
USE_DOH = False
HTTP_PROXY = ""
HTTPS_PROXY = ""
+3
View File
@@ -62,6 +62,9 @@ def setup_logger(name: str, log_file: Path = LOG_FILE) -> CustomLogger:
# File handler if log file is specified
try:
if ENABLE_LOGGING:
# Create log directory if it doesn't exist
log_dir = log_file.parent
log_dir.mkdir(parents=True, exist_ok=True)
file_handler = RotatingFileHandler(
log_file,
maxBytes=10485760, # 10MB
+30 -2
View File
@@ -57,6 +57,7 @@ An intuitive web interface for searching and requesting book downloads, designed
| `FLASK_DEBUG` | Debug mode toggle | `false` |
| `FLASK_HOST` | Web interface binding | `0.0.0.0` |
| `INGEST_DIR` | Book download directory | `/cwa-book-ingest` |
| `TZ` | Container timezone | `UTC` |
| `UID` | Runtime user ID | `1000` |
| `GID` | Runtime group ID | `100` |
| `ENABLE_LOGGING` | Enable log file | `true` |
@@ -65,6 +66,8 @@ An intuitive web interface for searching and requesting book downloads, designed
If logging is enabld, log folder default location is `/var/log/cwa-book-downloader`
Available log levels: `DEBUG`, `INFO`, `WARNING`, `ERROR`, `CRITICAL`. Higher levels show fewer messages.
Note that if using TOR, the TZ will be calculated automatically based on IP.
#### Download Settings
| Variable | Description | Default Value |
@@ -166,12 +169,32 @@ volumes:
Mount should align with your Calibre-Web-Automated ingest folder.
## 🧅 Tor Variant
This application also offers a variant that routes all its traffic through the Tor network. This can be useful for enhanced privacy or bypassing network restrictions.
To use the Tor variant:
1. Get the Tor-specific docker-compose file:
```bash
curl -O https://raw.githubusercontent.com/calibrain/calibre-web-automated-book-downloader/refs/heads/main/docker-compose.tor.yml
```
2. Start the service using this file:
```bash
docker compose -f docker-compose.tor.yml up -d
```
**Important Considerations for Tor:**
* **Capabilities:** This variant requires the `NET_ADMIN` and `NET_RAW` Docker capabilities to configure `iptables` for transparent Tor proxying.
* **Timezone:** When running in Tor mode, the container will attempt to determine the timezone based on the Tor exit node's IP address and set it automatically. This will override the `TZ` environment variable if it is set.
* **Network Settings:** Custom DNS, DoH, and HTTP(S) proxy settings (`CUSTOM_DNS`, `USE_DOH`, `HTTP_PROXY`, `HTTPS_PROXY`) are ignored when using the Tor variant, as all traffic goes through Tor.
## 🏗️ Architecture
The application consists of two key services:
The application consists of a single service:
1. **calibre-web-automated-bookdownloader**: Main application providing web interface and download functionality
2. **cloudflarebypassforscraping**: Support service for handling Cloudflare-protected websites
## 🏥 Health Monitoring
@@ -182,6 +205,11 @@ Built-in health checks monitor:
- Cloudflare bypass service connection
Checks run every 30 seconds with a 30-second timeout and 3 retries.
You can enable by adding this to your compose :
```
HEALTHCHECK --interval=30s --timeout=30s --start-period=5s --retries=3 \
CMD pyrequests http://localhost:8084/request/api/status || exit 1
```
## 📝 Logging
+2 -4
View File
@@ -2,9 +2,7 @@ flask
requests[socks]
beautifulsoup4
tqdm
DrissionPage
pyvirtualdisplay
types-requests
types-beautifulsoup4
types-tqdm
dnspython
pyautogui
seleniumbase
+103
View File
@@ -0,0 +1,103 @@
#!/bin/bash
set -e
echo "[*] Installing Tor and dependencies..."
echo "[*] Writing Tor transparent proxy config..."
cat <<EOF > /etc/tor/torrc
VirtualAddrNetworkIPv4 10.192.0.0/10
AutomapHostsOnResolve 1
TransPort 9040
DNSPort 53
Log notice file /var/log/tor/notices.log
EOF
echo "[*] Setting up DNS..."
cat <<EOF > /etc/resolv.conf
127.0.0.1
EOF
echo "[*] Starting Tor..."
service tor start
echo "[*] Setting up iptables rules..."
iptables -F
iptables -t nat -F
# Don't redirect Tor's own traffic
iptables -t nat -A OUTPUT -m owner --uid-owner debian-tor -j RETURN
# Allow loopback
iptables -t nat -A OUTPUT -o lo -j RETURN
# Redirect all TCP to Tor's TransPort
iptables -t nat -A OUTPUT -p tcp --syn -j REDIRECT --to-ports 9040
# For UDP DNS queries
iptables -t nat -A OUTPUT -p udp --dport 53 ! -d 127.0.0.1 -j DNAT --to-destination 127.0.0.1:53
# For TCP DNS queries (some DNS queries may use TCP)
iptables -t nat -A OUTPUT -p tcp --dport 53 ! -d 127.0.0.1 -j DNAT --to-destination 127.0.0.1:53
echo "[✓] Transparent Tor routing enabled."
# Wait a bit to ensure Tor has bootstrapped
echo "[*] Waiting for Tor to finish bootstrapping... (up to 5 minutes)"
timeout 300 bash -c '
while ! grep -q "Bootstrapped 100%" <(tail -n 20 -F /var/log/tor/notices.log 2>/dev/null); do
printf "\r\033[KCurrent log: %s" "$(tail -n 1 /var/log/tor/notices.log 2>/dev/null)"
sleep 1
done
# Print a newline when finished.
echo ""
'
echo "[✓] Tor is ready."
# Check if outgoing IP is using Tor
echo "[*] Verifying Tor connectivity..."
RESULT=$(pyrequests https://check.torproject.org/api/ip)
echo "RESULT: $RESULT"
IS_TOR=$(echo "$RESULT" | grep -oP '"IsTor":\s*\K(true|false)')
IP=$(echo "$RESULT" | grep -oP '"IP":\s*"\K[^"]+')
if [[ "$IS_TOR" == "true" ]]; then
echo "[✓] Success! Traffic is routed through Tor. Current IP: $IP"
else
echo "[✗] Warning: Traffic is NOT using Tor. Current IP: $IP"
exit 1
fi
# Set correct timezone
# First check what is the timezone based on the IP
# Then set the timezone
# Get timezone from IP
TIMEZONE=$(pyrequests https://ipapi.co/timezone)
# If TIMEZONE is not set, use the default timezone
echo "[*] Current Timezone : $(date +%Z). IP Timezone: $TIMEZONE"
# Set timezone in Docker-compatible way
if [ -f "/usr/share/zoneinfo/$TIMEZONE" ]; then
# Remove existing symlink if it exists
rm -f /etc/localtime
# Create new symlink
ln -sf /usr/share/zoneinfo/$TIMEZONE /etc/localtime
# Set timezone file
echo "$TIMEZONE" > /etc/timezone
# Set TZ environment variable
export TZ=$TIMEZONE
# Verify the change
echo "[✓] Timezone set to $TIMEZONE"
echo "[*] Current time: $(date)"
echo "[*] Timezone verification: $(date +%Z)"
else
echo "[!] Warning: Timezone file not found: $TIMEZONE"
echo "[*] Available timezones:"
ls -la /usr/share/zoneinfo/
echo "[*] Falling back to container's default timezone: $TZ"
fi
# Run the entrypoint script
echo "[*] Running entrypoint script..."
./entrypoint.sh