mirror of
https://github.com/calibrain/shelfmark.git
synced 2026-10-04 22:05:45 +01:00
Tor support, use SeleniumBase instead of DrissionPage (#123)
Refactor: Improve Docker build, add Tor support, use SeleniumBase
- Overhauled Dockerfile:
- Switched to python:3.10-slim base.
- Implemented multi-stage builds (base, standard, tor).
- Consolidated RUN layers for efficiency.
- Added locale/timezone setup.
- Added Tor setup and iptables configuration in dedicated stage/script.
- Replaced DrissionPage with SeleniumBase for Cloudflare bypassing.
- Added Tor support via `docker-compose.tor.yml` and `tor.sh` script.
- Updated `docker-compose.yml` to build locally and changed default port
to 8083.
- Added `docker-compose.dev.yml` and `docker-compose.tor.dev.yml`.
- Updated `entrypoint.sh` for timezone and sudo usage.
- Added/Updated environment variables (`TZ`, `USING_TOR`, etc.).
- Improved `.dockerignore`.
- Updated `readme.md` to reflect port changes, build process, document
`TZ`, and add details about the new Tor variant.
This commit is contained in:
@@ -10,3 +10,30 @@ Dockerfile
|
||||
docker-compose.yml
|
||||
docker-compose.*.yml
|
||||
readme.md
|
||||
|
||||
# Python specific
|
||||
__pycache__/
|
||||
*.pyc
|
||||
*.pyo
|
||||
*.pyd
|
||||
|
||||
# Environment files
|
||||
.env*
|
||||
|
||||
# Test & Coverage artifacts
|
||||
.pytest_cache/
|
||||
.coverage
|
||||
htmlcov/
|
||||
|
||||
# Logs
|
||||
*.log
|
||||
|
||||
# Build artifacts
|
||||
build/
|
||||
dist/
|
||||
*.egg-info/
|
||||
|
||||
# Virtual environments
|
||||
venv/
|
||||
.venv/
|
||||
env/
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
name: Create and publish a Docker image
|
||||
name: Create and publish Docker images
|
||||
|
||||
on:
|
||||
push:
|
||||
@@ -10,7 +10,7 @@ env:
|
||||
IMAGE_NAME: ${{ github.repository }}
|
||||
|
||||
jobs:
|
||||
build-and-push-image:
|
||||
build-and-push-images:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -26,8 +26,10 @@ jobs:
|
||||
registry: ${{ env.REGISTRY }}
|
||||
username: ${{ github.actor }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
- name: Extract metadata (tags, labels) for Docker
|
||||
id: meta
|
||||
|
||||
# Build and push main image
|
||||
- name: Extract metadata for main image
|
||||
id: meta-main
|
||||
uses: docker/metadata-action@v5
|
||||
with:
|
||||
images: ${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}
|
||||
@@ -36,23 +38,57 @@ jobs:
|
||||
type=raw,value={{commit_date 'YYYYMMDD'}}
|
||||
type=sha
|
||||
type=ref,event=branch
|
||||
type=ref,event=tag
|
||||
type=ref,event=tag
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
- name: Build and push Docker image
|
||||
id: push
|
||||
|
||||
- name: Build and push main Docker image
|
||||
id: push-main
|
||||
uses: docker/build-push-action@v5
|
||||
with:
|
||||
platforms: linux/amd64,linux/arm64
|
||||
context: .
|
||||
target: cwa-bd
|
||||
push: true
|
||||
tags: ${{ steps.meta.outputs.tags }}
|
||||
labels: ${{ steps.meta.outputs.labels }}
|
||||
tags: ${{ steps.meta-main.outputs.tags }}
|
||||
labels: ${{ steps.meta-main.outputs.labels }}
|
||||
|
||||
- name: Generate artifact attestation
|
||||
- name: Generate artifact attestation for main image
|
||||
uses: actions/attest-build-provenance@v2
|
||||
with:
|
||||
subject-name: ${{ env.REGISTRY }}/${{ env.IMAGE_NAME}}
|
||||
subject-digest: ${{ steps.push.outputs.digest }}
|
||||
subject-name: ${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}
|
||||
subject-digest: ${{ steps.push-main.outputs.digest }}
|
||||
push-to-registry: true
|
||||
|
||||
# Build and push tor image
|
||||
- name: Extract metadata for tor image
|
||||
id: meta-tor
|
||||
uses: docker/metadata-action@v5
|
||||
with:
|
||||
images: ${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}-tor
|
||||
tags: |
|
||||
type=raw,value=latest,enable={{is_default_branch}}
|
||||
type=raw,value={{commit_date 'YYYYMMDD'}}
|
||||
type=sha
|
||||
type=ref,event=branch
|
||||
type=ref,event=tag
|
||||
|
||||
- name: Build and push tor Docker image
|
||||
id: push-tor
|
||||
uses: docker/build-push-action@v5
|
||||
with:
|
||||
platforms: linux/amd64,linux/arm64
|
||||
context: .
|
||||
target: cwa-bd-tor
|
||||
push: true
|
||||
tags: ${{ steps.meta-tor.outputs.tags }}
|
||||
labels: ${{ steps.meta-tor.outputs.labels }}
|
||||
|
||||
- name: Generate artifact attestation for tor image
|
||||
uses: actions/attest-build-provenance@v2
|
||||
with:
|
||||
subject-name: ${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}-tor
|
||||
subject-digest: ${{ steps.push-tor.outputs.digest }}
|
||||
push-to-registry: true
|
||||
|
||||
|
||||
+2
-1
@@ -225,4 +225,5 @@ pyrightconfig.json
|
||||
.history
|
||||
.ionide
|
||||
|
||||
# End of https://www.toptal.com/developers/gitignore/api/macos,visualstudiocode,python
|
||||
# End of https://www.toptal.com/developers/gitignore/api/macos,visualstudiocode,python
|
||||
/downloaded_files
|
||||
|
||||
Vendored
+20
-16
@@ -1,12 +1,29 @@
|
||||
{
|
||||
"version": "0.2.0",
|
||||
"configurations": [
|
||||
{
|
||||
"name": "Python Debugger: Current File",
|
||||
"type": "debugpy",
|
||||
"request": "launch",
|
||||
"program": "${file}",
|
||||
"console": "integratedTerminal",
|
||||
"justMyCode": false,
|
||||
"env": {
|
||||
"INGEST_DIR": "/tmp/cwa-book-downloader",
|
||||
"TEMP_DIR": "/tmp/cwa-book-downloader",
|
||||
"LOG_LEVEL": "DEBUG",
|
||||
"LOG_ROOT": "/tmp/cwa-book-downloader",
|
||||
"ENABLE_LOGGING": "true",
|
||||
"DOCKERMODE": "false",
|
||||
"FLASK_DEBUG": "true"
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "Docker-compose Dev",
|
||||
"type": "debugpy", // or "debugpy", Node, etc.
|
||||
"type": "debugpy", // or "debugpy", Node, etc.
|
||||
"request": "launch",
|
||||
"program": "${workspaceFolder}/app.py",
|
||||
"preLaunchTask": "docker-compose up (dev)", // Spin up dev containers
|
||||
"preLaunchTask": "docker-compose up (dev)", // Spin up dev containers
|
||||
"postDebugTask": "docker-compose down (dev)", // Optional: tear them down
|
||||
"env": {
|
||||
"INGEST_DIR": "/tmp/cwa-book-downloader"
|
||||
@@ -17,25 +34,12 @@
|
||||
"type": "debugpy",
|
||||
"request": "launch",
|
||||
"program": "${workspaceFolder}/app.py",
|
||||
"preLaunchTask": "docker-compose up (prod)",
|
||||
"preLaunchTask": "docker-compose up (prod)",
|
||||
"postDebugTask": "docker-compose down (prod)",
|
||||
"env": {
|
||||
"INGEST_DIR": "/tmp/cwa-book-downloader"
|
||||
},
|
||||
},
|
||||
{
|
||||
"type": "debugpy",
|
||||
"request": "launch",
|
||||
"name": "Launch cwa-bd app.py",
|
||||
"program": "${workspaceFolder}/app.py",
|
||||
"env": {
|
||||
"DOCKERMODE": "false",
|
||||
"INGEST_DIR": "/tmp/cwa-book-downloader"
|
||||
},
|
||||
"presentation": {
|
||||
"hidden": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "chrome",
|
||||
"request": "launch",
|
||||
|
||||
+93
-16
@@ -1,43 +1,120 @@
|
||||
FROM python:3.12-slim
|
||||
# Use python-slim as the base image
|
||||
FROM python:3.10-slim AS base
|
||||
|
||||
# Set shell to bash with pipefail option
|
||||
SHELL ["/bin/bash", "-o", "pipefail", "-c"]
|
||||
|
||||
# Consistent environment variables grouped together
|
||||
ENV DEBIAN_FRONTEND=noninteractive \
|
||||
DOCKERMODE=true \
|
||||
PYTHONUNBUFFERED=1 \
|
||||
PYTHONDONTWRITEBYTECODE=1 \
|
||||
PYTHONIOENCODING=UTF-8 \
|
||||
PIP_NO_CACHE_DIR=1 \
|
||||
PIP_DISABLE_PIP_VERSION_CHECK=1 \
|
||||
PIP_DEFAULT_TIMEOUT=100 \
|
||||
NAME=Calibre-Web-Automated-Book-Downloader \
|
||||
PYTHONPATH=/app \
|
||||
UID=1000 \
|
||||
GID=100
|
||||
# UID/GID will be handled by entrypoint script, but TZ/Locale are still needed
|
||||
LANG=en_US.UTF-8 \
|
||||
LANGUAGE=en_US:en \
|
||||
LC_ALL=en_US.UTF-8
|
||||
|
||||
# Install minimal dependencies
|
||||
# Set ARG for build-time expansion (FLASK_PORT), ENV for runtime access
|
||||
ENV FLASK_PORT=8084
|
||||
|
||||
# Configure locale, timezone, and perform initial cleanup in a single layer
|
||||
# User/group creation is removed
|
||||
RUN apt-get update && \
|
||||
apt-get install -y --no-install-recommends --no-install-suggests \
|
||||
curl \
|
||||
apt-get install -y --no-install-recommends \
|
||||
# For locale
|
||||
locales tzdata \
|
||||
# For entrypoint
|
||||
dumb-init \
|
||||
# For dumb display
|
||||
xvfb \
|
||||
# For user switching
|
||||
sudo \
|
||||
# --- Chromium Browser ---
|
||||
chromium-driver \
|
||||
dumb-init && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
# For tkinter (pyautogui)
|
||||
python3-tk && \
|
||||
# Cleanup APT cache *after* all installs in this layer
|
||||
apt-get purge -y --auto-remove -o APT::AutoRemove::RecommendsImportant=false && \
|
||||
apt-get clean && \
|
||||
rm -rf /var/lib/apt/lists/* && \
|
||||
# Default to UTC timezone but will be overridden by the entrypoint script
|
||||
ln -snf /usr/share/zoneinfo/UTC /etc/localtime && echo UTC > /etc/timezone && \
|
||||
# Configure locale
|
||||
sed -i '/en_US.UTF-8/s/^# //g' /etc/locale.gen && \
|
||||
locale-gen en_US.UTF-8 && \
|
||||
echo "LC_ALL=en_US.UTF-8" >> /etc/environment && \
|
||||
echo "LANG=en_US.UTF-8" > /etc/locale.conf
|
||||
|
||||
# Set working directory
|
||||
WORKDIR /app
|
||||
|
||||
# Install Python dependencies including playwright
|
||||
# Install Python dependencies using pip
|
||||
# Upgrade pip first, then copy requirements and install
|
||||
# Copying requirements.txt separately leverages build cache
|
||||
# No --chown needed as it's copied as root
|
||||
COPY requirements.txt .
|
||||
RUN pip install --no-cache-dir -r requirements.txt && \
|
||||
rm -rf /root/.cache /app/.cache
|
||||
# Clean root's pip cache
|
||||
rm -rf /root/.cache
|
||||
|
||||
# Our custom wanabe curl
|
||||
|
||||
RUN echo "#!/bin/sh" > /usr/local/bin/pyrequests && \
|
||||
echo 'python -c "import sys, requests; url=sys.argv[1]; r=requests.get(url); print(r.text); sys.exit(0) if r.ok else sys.exit(1)" "$@"' \
|
||||
>> /usr/local/bin/pyrequests && \
|
||||
chmod +x /usr/local/bin/pyrequests
|
||||
|
||||
# Copy application code *after* dependencies are installed
|
||||
# No --chown needed as it's copied as root, entrypoint will handle permissions
|
||||
COPY . .
|
||||
RUN chmod +x /app/entrypoint.sh && \
|
||||
# Create necessary directories
|
||||
mkdir -p /var/log/cwa-book-downloader && \
|
||||
mkdir -p /cwa-book-ingest
|
||||
|
||||
# Final setup: permissions and directories in one layer
|
||||
# Only creating directories and setting executable bits.
|
||||
# Ownership will be handled by the entrypoint script.
|
||||
RUN mkdir -p /var/log/cwa-book-downloader /cwa-book-ingest && \
|
||||
chmod +x /app/entrypoint.sh /app/tor.sh
|
||||
# chown is removed
|
||||
|
||||
|
||||
# Expose the application port
|
||||
EXPOSE ${FLASK_PORT}
|
||||
|
||||
# Add healthcheck for container status
|
||||
# This will run as root initially, but check localhost which should work if the app binds correctly.
|
||||
HEALTHCHECK --interval=30s --timeout=30s --start-period=5s --retries=3 \
|
||||
CMD curl -f http://localhost:${FLASK_PORT}/request/api/status || exit 1
|
||||
CMD pyrequests http://localhost:${FLASK_PORT}/request/api/status || exit 1
|
||||
|
||||
# Use dumb-init as the entrypoint to handle signals properly
|
||||
ENTRYPOINT ["/usr/bin/dumb-init", "--"]
|
||||
CMD ["/app/entrypoint.sh"]
|
||||
|
||||
|
||||
FROM base AS cwa-bd
|
||||
|
||||
# Default command to run the application entrypoint script
|
||||
CMD ["/app/entrypoint.sh"]
|
||||
|
||||
FROM base AS cwa-bd-tor
|
||||
|
||||
ENV ENABLE_TOR=true
|
||||
|
||||
# Install Tor and dependencies
|
||||
RUN apt-get update && \
|
||||
apt-get install -y --no-install-recommends \
|
||||
# --- Tor ---
|
||||
tor iptables && \
|
||||
# Cleanup APT cache *after* all installs in this layer
|
||||
apt-get purge -y --auto-remove -o APT::AutoRemove::RecommendsImportant=false && \
|
||||
apt-get clean && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Tor configuration is handled by entrypoint/script now or permissions set earlier
|
||||
# RUN chmod +x /app/tor.sh # This is removed as it's done in the base stage setup
|
||||
|
||||
# Override the default command to run Tor
|
||||
CMD ["/app/tor.sh"]
|
||||
+187
-181
@@ -1,90 +1,50 @@
|
||||
import time, os, socket
|
||||
import network
|
||||
import time
|
||||
import os
|
||||
import socket
|
||||
from urllib.parse import urlparse
|
||||
from DrissionPage import ChromiumPage # type: ignore
|
||||
from DrissionPage import ChromiumOptions
|
||||
from DrissionPage._functions.elements import ChromiumElementsList # type: ignore
|
||||
from DrissionPage._pages.chromium_tab import ChromiumTab # type: ignore
|
||||
import threading
|
||||
import env
|
||||
|
||||
# --- SeleniumBase Import ---
|
||||
from seleniumbase import Driver
|
||||
from selenium.webdriver.common.by import By
|
||||
from selenium.webdriver.support.ui import WebDriverWait
|
||||
from selenium.webdriver.support import expected_conditions as EC
|
||||
from selenium.common.exceptions import TimeoutException
|
||||
|
||||
import network
|
||||
from logger import setup_logger
|
||||
from env import MAX_RETRY, DOCKERMODE, DEFAULT_SLEEP
|
||||
from config import PROXIES, CUSTOM_DNS, DOH_SERVER, AA_BASE_URL
|
||||
from env import MAX_RETRY, DEFAULT_SLEEP
|
||||
from config import PROXIES, CUSTOM_DNS, DOH_SERVER
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
_defaultTab : ChromiumTab | None = None
|
||||
|
||||
network.init()
|
||||
|
||||
def _search_recursively_shadow_root_with_iframe(ele : ChromiumElementsList) -> ChromiumElementsList | None:
|
||||
if ele.shadow_root:
|
||||
if ele.shadow_root.child().tag == "iframe":
|
||||
return ele.shadow_root.child()
|
||||
else:
|
||||
for child in ele.children():
|
||||
result = _search_recursively_shadow_root_with_iframe(child)
|
||||
if result:
|
||||
return result
|
||||
return None
|
||||
DRIVER = None
|
||||
LAST_USED = None
|
||||
LOCKED = threading.Lock()
|
||||
TENTATIVE_CURRENT_URL = None
|
||||
|
||||
def _search_recursively_shadow_root_with_cf_input(ele : ChromiumElementsList) -> ChromiumElementsList | None:
|
||||
if ele.shadow_root:
|
||||
if ele.shadow_root.ele("tag:input"):
|
||||
return ele.shadow_root.ele("tag:input")
|
||||
else:
|
||||
for child in ele.children():
|
||||
result = _search_recursively_shadow_root_with_cf_input(child)
|
||||
if result:
|
||||
return result
|
||||
return None
|
||||
|
||||
def _locate_cf_button(driver : ChromiumTab) -> ChromiumElementsList | None:
|
||||
button : ChromiumElementsList = None
|
||||
eles = driver.eles("tag:input")
|
||||
for ele in eles:
|
||||
if "name" in ele.attrs.keys() and "type" in ele.attrs.keys():
|
||||
if "turnstile" in ele.attrs["name"] and ele.attrs["type"] == "hidden":
|
||||
button = ele.parent().shadow_root.child()("tag:body").shadow_root("tag:input")
|
||||
break
|
||||
|
||||
if button:
|
||||
return button
|
||||
else:
|
||||
# If the button is not found, search it recursively
|
||||
logger.debug("Basic search failed. Searching for button recursively.")
|
||||
ele = driver.ele("tag:body")
|
||||
iframe = _search_recursively_shadow_root_with_iframe(ele)
|
||||
if iframe:
|
||||
button = _search_recursively_shadow_root_with_cf_input(iframe("tag:body"))
|
||||
else:
|
||||
logger.debug("Iframe not found. Button search failed.")
|
||||
return button
|
||||
|
||||
def _click_verification_button(driver: ChromiumTab) -> None:
|
||||
def _is_bypassed(sb) -> bool:
|
||||
try:
|
||||
button = _locate_cf_button(driver)
|
||||
if button:
|
||||
logger.debug("Verification button found. Attempting to click.")
|
||||
button.wait.displayed(timeout=DEFAULT_SLEEP)
|
||||
button.click()
|
||||
else:
|
||||
logger.debug("Verification button not found.")
|
||||
|
||||
except Exception as e:
|
||||
logger.debug(f"Error clicking verification button: {e}")
|
||||
|
||||
def _is_bypassed(driver: ChromiumTab) -> bool:
|
||||
try:
|
||||
title = driver.title.lower()
|
||||
body = driver.ele("tag:body").text.lower()
|
||||
title = sb.get_title().lower()
|
||||
body = sb.get_text("body").lower()
|
||||
|
||||
# Check both title and body for verification messages
|
||||
verification_texts = [
|
||||
"just a moment",
|
||||
"verify you are human",
|
||||
"verifying you are human",
|
||||
"needs to review the security of your connection before proceeding"
|
||||
"needs to review the security of your connection before proceeding",
|
||||
"checking your browser",
|
||||
"checking connection",
|
||||
"attention required",
|
||||
"access denied",
|
||||
"needs to review the security of your connection",
|
||||
"checking the site connection security",
|
||||
"enable javascript and cookies to continue",
|
||||
"ray id",
|
||||
]
|
||||
|
||||
for text in verification_texts:
|
||||
if text in title.lower() or text in body.lower():
|
||||
return False
|
||||
@@ -94,147 +54,193 @@ def _is_bypassed(driver: ChromiumTab) -> bool:
|
||||
logger.debug(f"Error checking page title: {e}")
|
||||
return False
|
||||
|
||||
def _bypass(driver: ChromiumTab, max_retries: int = MAX_RETRY) -> None:
|
||||
def _bypass(sb, max_retries: int = MAX_RETRY) -> None:
|
||||
try_count = 0
|
||||
|
||||
while not _is_bypassed(driver):
|
||||
logger.info(f"Starting Cloudflare bypass... Retry: {try_count + 1} / {max_retries}")
|
||||
while not _is_bypassed(sb):
|
||||
if try_count >= max_retries:
|
||||
logger.warning("Exceeded maximum retries. Bypass failed.")
|
||||
break
|
||||
|
||||
logger.info(f"Attempt {try_count + 1}: Verification page detected. Trying to bypass...")
|
||||
logger.info(f"Bypass attempt {try_count + 1} / {max_retries}")
|
||||
|
||||
try_count += 1
|
||||
time.sleep(DEFAULT_SLEEP)
|
||||
|
||||
_click_verification_button(driver)
|
||||
wait_time = DEFAULT_SLEEP * (try_count - 1)
|
||||
logger.info(f"Waiting {wait_time}s before trying...")
|
||||
time.sleep(wait_time)
|
||||
|
||||
time.sleep(DEFAULT_SLEEP)
|
||||
try:
|
||||
sb.uc_gui_click_captcha()
|
||||
except Exception as e:
|
||||
time.sleep(5)
|
||||
sb.wait_for_element_visible('body')
|
||||
try:
|
||||
sb.uc_gui_click_captcha()
|
||||
except Exception as e:
|
||||
time.sleep(DEFAULT_SLEEP)
|
||||
sb.reconnect(DEFAULT_SLEEP)
|
||||
sb.uc_gui_click_captcha()
|
||||
|
||||
if _is_bypassed(driver):
|
||||
logger.info("Bypass successful.")
|
||||
else:
|
||||
logger.info("Bypass failed.")
|
||||
if _is_bypassed(sb):
|
||||
logger.info("Bypass successful.")
|
||||
else:
|
||||
logger.info("Bypass failed.")
|
||||
|
||||
def _get_chromium_options(arguments: list[str]) -> ChromiumOptions:
|
||||
options = ChromiumOptions()
|
||||
for argument in arguments:
|
||||
options.set_argument(argument)
|
||||
def _get_chromium_args():
|
||||
|
||||
arguments = [
|
||||
"-no-sandbox",
|
||||
]
|
||||
|
||||
# Add proxy settings if configured
|
||||
if PROXIES:
|
||||
if 'http' in PROXIES:
|
||||
options.set_argument(f'--proxy-server={PROXIES["http"]}')
|
||||
logger.debug(f"Setting HTTP proxy: {PROXIES['http']}")
|
||||
elif 'https' in PROXIES:
|
||||
options.set_argument(f'--proxy-server={PROXIES["https"]}')
|
||||
logger.debug(f"Setting HTTPS proxy: {PROXIES['https']}")
|
||||
proxy_url = PROXIES.get('https') or PROXIES.get('http')
|
||||
if proxy_url:
|
||||
arguments.append(f'--proxy-server={proxy_url}')
|
||||
|
||||
# --- Add Custom DNS settings ---
|
||||
try:
|
||||
if len(CUSTOM_DNS) > 0:
|
||||
if len(CUSTOM_DNS) > 0:
|
||||
if DOH_SERVER:
|
||||
logger.info(f"Configuring DNS over HTTPS (DoH) with server: {DOH_SERVER}")
|
||||
|
||||
# Enable the DoH feature
|
||||
options.set_argument(f'--enable-features=DnsOverHttps')
|
||||
|
||||
# TODO: This is probably broken and a halucination,
|
||||
# but it should still default to google DOH so its fine...
|
||||
options.set_argument(f'--dns-over-https-mode="secure"')
|
||||
options.set_argument(f'--dns-over-https-servers="{DOH_SERVER}"')
|
||||
|
||||
arguments.extend(['--enable-features=DnsOverHttps', '--dns-over-https-mode=secure', f'--dns-over-https-servers="{DOH_SERVER}"'])
|
||||
doh_hostname = urlparse(DOH_SERVER).hostname
|
||||
if doh_hostname:
|
||||
doh_ip = socket.gethostbyname(doh_hostname)
|
||||
options.set_argument(f'--host-resolver-rules=MAP {doh_hostname} {doh_ip}')
|
||||
logger.debug(f"Setting Chromium --host-resolver-rules='MAP {doh_hostname} {doh_ip}'")
|
||||
else:
|
||||
logger.info(f"Applying custom DNS servers: {CUSTOM_DNS}")
|
||||
# Format: "MAP * <dns1>, MAP * <dns2>, ..."
|
||||
# We create a separate MAP rule for each DNS server.
|
||||
# Chromium should try them based on its internal logic (likely order/availability).
|
||||
resolver_rules = []
|
||||
for dns_server in CUSTOM_DNS:
|
||||
resolver_rules.append(f"MAP * {dns_server}")
|
||||
|
||||
try:
|
||||
arguments.append(f'--host-resolver-rules=MAP {doh_hostname} {socket.gethostbyname(doh_hostname)}')
|
||||
except socket.gaierror:
|
||||
logger.warning(f"Could not resolve DoH hostname: {doh_hostname}")
|
||||
elif CUSTOM_DNS:
|
||||
resolver_rules = [f"MAP * {dns_server}" for dns_server in CUSTOM_DNS]
|
||||
if resolver_rules:
|
||||
# Join the rules with " , " (comma and space is a common separator)
|
||||
host_resolver_rules_value = " , ".join(resolver_rules)
|
||||
options.set_argument(f'--host-resolver-rules={host_resolver_rules_value}')
|
||||
logger.debug(f"Setting Chromium --host-resolver-rules='{host_resolver_rules_value}'")
|
||||
arguments.append(f'--host-resolver-rules={",".join(resolver_rules)}')
|
||||
except Exception as e:
|
||||
logger.error_trace(f"Error configuring DNS settings: {e}")
|
||||
return options
|
||||
return arguments
|
||||
|
||||
def _genScraper() -> ChromiumPage:
|
||||
arguments = [
|
||||
"-no-first-run",
|
||||
"-force-color-profile=srgb",
|
||||
"-metrics-recording-only",
|
||||
"-password-store=basic",
|
||||
"-use-mock-keychain",
|
||||
"-export-tagged-pdf",
|
||||
"-no-default-browser-check",
|
||||
"-disable-background-mode",
|
||||
"-enable-features=NetworkService,NetworkServiceInProcess,LoadCryptoTokenExtension,PermuteTLSExtensions",
|
||||
"-disable-features=FlashDeprecationWarning,EnablePasswordsAccountStorage",
|
||||
"-deny-permission-prompts",
|
||||
"-disable-gpu",
|
||||
"-no-sandbox",
|
||||
"-accept-lang=en-US",
|
||||
"-remote-debugging-port=9222"
|
||||
]
|
||||
CHROMIUM_ARGS = _get_chromium_args()
|
||||
|
||||
options = _get_chromium_options(arguments)
|
||||
# Initialize the browser
|
||||
driver = ChromiumPage(addr_or_opts=options)
|
||||
def _get(url, retry : int = MAX_RETRY):
|
||||
try:
|
||||
logger.info(f"SB_GET: {url}")
|
||||
sb = _get_driver()
|
||||
sb.uc_open_with_disconnect(url)
|
||||
time.sleep(1)
|
||||
_bypass(sb)
|
||||
return sb.page_source
|
||||
except Exception as e:
|
||||
if retry == 0:
|
||||
logger.error_trace(f"Failed to initialize browser: {e}")
|
||||
_reset_driver()
|
||||
raise e
|
||||
logger.error_trace(f"Failed to bypass Cloudflare: {e}. Will retry...")
|
||||
return _get(url, retry - 1)
|
||||
|
||||
def get(url, retry : int = MAX_RETRY):
|
||||
global LOCKED, TENTATIVE_CURRENT_URL, LAST_USED
|
||||
with LOCKED:
|
||||
TENTATIVE_CURRENT_URL = url
|
||||
ret = _get(url, retry)
|
||||
LAST_USED = time.time()
|
||||
return ret
|
||||
|
||||
def _init_driver():
|
||||
global DRIVER
|
||||
if DRIVER:
|
||||
_reset_driver()
|
||||
driver = Driver(uc=True, headless=False, chromium_arg=CHROMIUM_ARGS)
|
||||
DRIVER = driver
|
||||
time.sleep(DEFAULT_SLEEP)
|
||||
return driver
|
||||
|
||||
def _get_driver():
|
||||
global DRIVER
|
||||
global LAST_USED
|
||||
LAST_USED = time.time()
|
||||
if not DRIVER:
|
||||
return _init_driver()
|
||||
return DRIVER
|
||||
|
||||
def _reset_browser() -> None:
|
||||
logger.info("Resetting chromiumbrowser")
|
||||
if not DOCKERMODE:
|
||||
return
|
||||
global _defaultTab
|
||||
# Kill the browser
|
||||
if _defaultTab:
|
||||
_defaultTab.close()
|
||||
_defaultTab = None
|
||||
# Force kill the browser
|
||||
os.system("pkill -f -i 'chromium'")
|
||||
os.system("pkill -f -i 'chrom'")
|
||||
os.system("pkill -f -i 'xvfb'")
|
||||
time.sleep(1)
|
||||
|
||||
def _init_browser(retry : int = MAX_RETRY) -> ChromiumTab:
|
||||
global _defaultTab
|
||||
if _defaultTab:
|
||||
return _defaultTab
|
||||
else:
|
||||
try:
|
||||
driver = _genScraper()
|
||||
_defaultTab = driver.get_tabs()[0]
|
||||
return _defaultTab
|
||||
except Exception as e:
|
||||
if retry > 0:
|
||||
_reset_browser()
|
||||
else:
|
||||
logger.error_trace(f"Failed to initialize browser: {e}")
|
||||
raise e
|
||||
return _init_browser(retry - 1)
|
||||
|
||||
def get(url : str, retry : int = MAX_RETRY) -> ChromiumTab:
|
||||
defaultTab = _init_browser()
|
||||
defaultTab.get(url)
|
||||
def _reset_driver():
|
||||
logger.info("Resetting driver...")
|
||||
global DRIVER
|
||||
if DRIVER:
|
||||
DRIVER.quit()
|
||||
try:
|
||||
_bypass(defaultTab)
|
||||
os.system("pkill -f xvfb")
|
||||
except Exception as e:
|
||||
if retry > 0:
|
||||
return get(url, retry - 1)
|
||||
logger.error_trace(f"Failed to bypass Cloudflare for {url}: {e}")
|
||||
raise e
|
||||
return defaultTab
|
||||
logger.warning(f"Error killing xvfb: {e}")
|
||||
try:
|
||||
os.system("pkill -f chrom")
|
||||
except Exception as e:
|
||||
logger.warning(f"Error killing chrom: {e}")
|
||||
DRIVER = None
|
||||
|
||||
get(AA_BASE_URL)
|
||||
def _cleanup_driver():
|
||||
global LOCKED
|
||||
global LAST_USED
|
||||
with LOCKED:
|
||||
if LAST_USED:
|
||||
if time.time() - LAST_USED >= env.BYPASS_RELEASE_INACTIVE_MIN * 60:
|
||||
_reset_driver()
|
||||
LAST_USED = None
|
||||
logger.info("Driver reset due to inactivity.")
|
||||
|
||||
def _cleanup_loop():
|
||||
while True:
|
||||
_cleanup_driver()
|
||||
time.sleep(max(env.BYPASS_RELEASE_INACTIVE_MIN / 2, 1))
|
||||
|
||||
def _debug_loop():
|
||||
while True:
|
||||
if DRIVER:
|
||||
try:
|
||||
# Get URL with fallback to tentative URL
|
||||
try:
|
||||
url = DRIVER.current_url
|
||||
except Exception as e:
|
||||
url = TENTATIVE_CURRENT_URL or "unknown_url"
|
||||
|
||||
# Create timestamp and filename
|
||||
timestamp = time.strftime("%Y%m%d_%H%M%S")
|
||||
filename = f"screenshot_{timestamp}_{url}"
|
||||
|
||||
# Sanitize filename
|
||||
sanitized_filename = "".join(c if c.isalnum() or c in ('-', '_', '.') else '_' for c in filename)
|
||||
sanitized_filename = sanitized_filename[:100] + ".png" # Limit length
|
||||
|
||||
# Ensure screenshots directory exists
|
||||
screenshots_dir = env.LOG_DIR / "screenshots"
|
||||
screenshots_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# Save screenshot
|
||||
full_path = screenshots_dir / sanitized_filename
|
||||
|
||||
DRIVER.save_screenshot(str(full_path))
|
||||
except Exception as e:
|
||||
pass
|
||||
time.sleep(1)
|
||||
|
||||
def _init_cleanup_thread():
|
||||
cleanup_thread = threading.Thread(target=_cleanup_loop)
|
||||
cleanup_thread.daemon = True
|
||||
cleanup_thread.start()
|
||||
if env.DEBUG:
|
||||
path = env.LOG_DIR / "screenshots"
|
||||
path.mkdir(parents=True, exist_ok=True)
|
||||
debug_thread = threading.Thread(target=_debug_loop)
|
||||
debug_thread.daemon = True
|
||||
debug_thread.start()
|
||||
|
||||
def wait_for_result(func, timeout : int = 10, condition : any = True):
|
||||
start_time = time.time()
|
||||
while time.time() - start_time < timeout:
|
||||
result = func()
|
||||
if condition(result):
|
||||
return result
|
||||
time.sleep(0.5)
|
||||
return None
|
||||
_init_cleanup_thread()
|
||||
|
||||
@@ -18,7 +18,8 @@ with open("data/book-languages.json") as file:
|
||||
# Directory settings
|
||||
BASE_DIR = Path(__file__).resolve().parent
|
||||
logger.info(f"BASE_DIR: {BASE_DIR}")
|
||||
env.LOG_DIR.mkdir(exist_ok=True)
|
||||
if env.ENABLE_LOGGING:
|
||||
env.LOG_DIR.mkdir(exist_ok=True)
|
||||
|
||||
# Create necessary directories
|
||||
env.TMP_DIR.mkdir(exist_ok=True)
|
||||
|
||||
+9
-14
@@ -1,20 +1,15 @@
|
||||
# If you change the FLASK_PORT, do not forget to change it in ports and healthcheck as well.
|
||||
services:
|
||||
calibre-web-automated-book-downloader:
|
||||
build :
|
||||
calibre-web-automated-book-downloader-dev:
|
||||
extends:
|
||||
file: ./docker-compose.yml
|
||||
service: calibre-web-automated-book-downloader
|
||||
build:
|
||||
context: .
|
||||
dockerfile: Dockerfile
|
||||
target: cwa-bd
|
||||
environment:
|
||||
FLASK_PORT: 8084
|
||||
LOG_LEVEL: debug
|
||||
BOOK_LANGUAGE: en
|
||||
USE_BOOK_TITLE: true
|
||||
CUSTOM_DNS: cloudflare
|
||||
FLASK_DEBUG: true
|
||||
USE_DOH: true
|
||||
ports:
|
||||
- 8084:8084
|
||||
restart: unless-stopped
|
||||
volumes:
|
||||
# This is where the books will be downloaded to, usually it would be
|
||||
# the same as whatever you gave in "calibre-web-automated"
|
||||
- /tmp/data/calibre-web/ingest:/cwa-book-ingest
|
||||
CUSTOM_DNS: cloudflare
|
||||
|
||||
|
||||
@@ -0,0 +1,15 @@
|
||||
services:
|
||||
calibre-web-automated-book-downloader-tor-dev:
|
||||
extends:
|
||||
file: ./docker-compose.yml
|
||||
service: calibre-web-automated-book-downloader-tor
|
||||
build:
|
||||
context: .
|
||||
dockerfile: Dockerfile
|
||||
target: cwa-bd-tor
|
||||
cap_add:
|
||||
- NET_ADMIN
|
||||
- NET_RAW
|
||||
environment:
|
||||
LOG_LEVEL: debug
|
||||
FLASK_DEBUG: true
|
||||
@@ -0,0 +1,19 @@
|
||||
services:
|
||||
calibre-web-automated-book-downloader-tor:
|
||||
image: ghcr.io/calibrain/calibre-web-automated-book-downloader-tor:latest
|
||||
environment:
|
||||
FLASK_PORT: 8084
|
||||
LOG_LEVEL: info
|
||||
BOOK_LANGUAGE: en
|
||||
USE_BOOK_TITLE: true
|
||||
FLASK_DEBUG: false
|
||||
ENABLE_TOR: true
|
||||
TZ: America/New_York
|
||||
ports:
|
||||
- 8084:8084
|
||||
restart: unless-stopped
|
||||
volumes:
|
||||
# This is where the books will be downloaded to, usually it would be
|
||||
# the same as whatever you gave in "calibre-web-automated"
|
||||
- /tmp/data/calibre-web/ingest:/cwa-book-ingest
|
||||
- /tmp/cwa-book-downloader:/tmp/cwa-book-downloader
|
||||
+5
-9
@@ -1,22 +1,18 @@
|
||||
# If you change the FLASK_PORT, do not forget to change it in ports and healthcheck as well.
|
||||
services:
|
||||
calibre-web-automated-book-downloader:
|
||||
image: ghcr.io/calibrain/calibre-web-automated-book-downloader:latest
|
||||
environment:
|
||||
FLASK_PORT: 8084
|
||||
FLASK_DEBUG: false
|
||||
LOG_LEVEL: info
|
||||
BOOK_LANGUAGE: en
|
||||
USE_BOOK_TITLE: true
|
||||
FLASK_DEBUG: false
|
||||
TZ: America/New_York
|
||||
ports:
|
||||
- 8084:8084
|
||||
# Uncomment the following lines if you want to enable healthcheck
|
||||
#healthcheck:
|
||||
# test: ["CMD", "curl", "-f", "http://localhost:8084/request/api/status"]
|
||||
# interval: 30s
|
||||
# timeout: 30s
|
||||
# retries: 3
|
||||
# start_period: 5s
|
||||
restart: unless-stopped
|
||||
volumes:
|
||||
# This is where the books will be downloaded to, usually it would be
|
||||
# the same as whatever you gave in "calibre-web-automated"
|
||||
- /tmp/data/calibre-web/ingest:/cwa-book-ingest
|
||||
- /tmp/cwa-book-downloader:/tmp/cwa-book-downloader
|
||||
|
||||
+16
-16
@@ -34,18 +34,18 @@ def html_get_page(url: str, retry: int = MAX_RETRY, use_bypasser: bool = False)
|
||||
logger.debug(f"html_get_page: {url}, retry: {retry}, use_bypasser: {use_bypasser}")
|
||||
if use_bypasser and USE_CF_BYPASS:
|
||||
logger.info(f"GET Using Cloudflare Bypasser for: {url}")
|
||||
response = cloudflare_bypasser.get(url)
|
||||
logger.debug(f"Cloudflare Bypasser response: {response}")
|
||||
if response:
|
||||
return str(response.html)
|
||||
response_html = cloudflare_bypasser.get(url)
|
||||
logger.debug(f"Cloudflare Bypasser response length: {len(response_html)}")
|
||||
if response_html.strip() != "":
|
||||
return response_html
|
||||
else:
|
||||
raise requests.exceptions.RequestException("Failed to bypass Cloudflare")
|
||||
|
||||
logger.info(f"GET: {url}")
|
||||
response = requests.get(url, proxies=PROXIES)
|
||||
response.raise_for_status()
|
||||
logger.debug(f"Success getting: {url}")
|
||||
time.sleep(1)
|
||||
else:
|
||||
logger.info(f"GET: {url}")
|
||||
response = requests.get(url, proxies=PROXIES)
|
||||
response.raise_for_status()
|
||||
logger.debug(f"Success getting: {url}")
|
||||
time.sleep(1)
|
||||
return str(response.text)
|
||||
|
||||
except Exception as e:
|
||||
@@ -53,14 +53,14 @@ def html_get_page(url: str, retry: int = MAX_RETRY, use_bypasser: bool = False)
|
||||
logger.error_trace(f"Failed to fetch page: {url}, error: {e}")
|
||||
return ""
|
||||
|
||||
if response is not None and response.status_code == 404:
|
||||
if use_bypasser and USE_CF_BYPASS:
|
||||
logger.warning(f"Exception while using cloudflare bypass for URL: {url}")
|
||||
logger.warning(f"Exception: {e}")
|
||||
logger.warning(f"Response: {response}")
|
||||
elif response is not None and response.status_code == 404:
|
||||
logger.warning(f"404 error for URL: {url}")
|
||||
return ""
|
||||
|
||||
if response is not None and response.status_code == 403:
|
||||
if use_bypasser:
|
||||
logger.warning(f"403 error while using cloudflare bypass for URL: {url}")
|
||||
return ""
|
||||
elif response is not None and response.status_code == 403:
|
||||
logger.warning(f"403 detected for URL: {url}. Should retry using cloudflare bypass.")
|
||||
return html_get_page(url, retry - 1, True)
|
||||
|
||||
|
||||
+6
-1
@@ -1,6 +1,11 @@
|
||||
#!/bin/bash
|
||||
set -e
|
||||
|
||||
# Configure timezone
|
||||
if [ "$TZ" ]; then
|
||||
ln -snf /usr/share/zoneinfo/$TZ /etc/localtime && echo $TZ > /etc/timezone
|
||||
fi
|
||||
|
||||
# Set UID if not set
|
||||
if [ -z "$UID" ]; then
|
||||
UID=1000
|
||||
@@ -34,4 +39,4 @@ change_ownership /var/log/cwa-book-downloader
|
||||
change_ownership /cwa-book-ingest
|
||||
|
||||
# Switch to the user (either newly created or existing) and execute the main command
|
||||
exec su -s /bin/bash "$USERNAME" -c "python -m app"
|
||||
exec sudo -E -u "$USERNAME" python3 -m app
|
||||
|
||||
@@ -4,12 +4,13 @@ from pathlib import Path
|
||||
def string_to_bool(s: str) -> bool:
|
||||
return s.lower() in ["true", "yes", "1", "y"]
|
||||
|
||||
LOG_DIR = Path("/var/log/cwa-book-downloader")
|
||||
LOG_ROOT = Path(os.getenv("LOG_ROOT", "/var/log/"))
|
||||
LOG_DIR = LOG_ROOT / "cwa-book-downloader"
|
||||
TMP_DIR = Path(os.getenv("TMP_DIR", "/tmp/cwa-book-downloader"))
|
||||
INGEST_DIR = Path(os.getenv("INGEST_DIR", "/cwa-book-ingest"))
|
||||
STATUS_TIMEOUT = int(os.getenv("STATUS_TIMEOUT", "3600"))
|
||||
USE_BOOK_TITLE = string_to_bool(os.getenv("USE_BOOK_TITLE", "false"))
|
||||
MAX_RETRY = int(os.getenv("MAX_RETRY", "3"))
|
||||
MAX_RETRY = int(os.getenv("MAX_RETRY", "10"))
|
||||
DEFAULT_SLEEP = int(os.getenv("DEFAULT_SLEEP", "5"))
|
||||
USE_CF_BYPASS = string_to_bool(os.getenv("USE_CF_BYPASS", "true"))
|
||||
HTTP_PROXY = os.getenv("HTTP_PROXY", "").strip()
|
||||
@@ -23,12 +24,22 @@ _CUSTOM_SCRIPT = os.getenv("CUSTOM_SCRIPT", "").strip()
|
||||
FLASK_HOST = os.getenv("FLASK_HOST", "0.0.0.0")
|
||||
FLASK_PORT = int(os.getenv("FLASK_PORT", "8084"))
|
||||
FLASK_DEBUG = string_to_bool(os.getenv("FLASK_DEBUG", "False"))
|
||||
DEBUG = FLASK_DEBUG
|
||||
LOG_LEVEL = os.getenv("LOG_LEVEL", "INFO").upper()
|
||||
ENABLE_LOGGING = string_to_bool(os.getenv("ENABLE_LOGGING", "true"))
|
||||
MAIN_LOOP_SLEEP_TIME = int(os.getenv("MAIN_LOOP_SLEEP_TIME", "5"))
|
||||
DOCKERMODE = string_to_bool(os.getenv("DOCKERMODE", "false"))
|
||||
_CUSTOM_DNS = os.getenv("CUSTOM_DNS", "").strip()
|
||||
USE_DOH = string_to_bool(os.getenv("USE_DOH", "false"))
|
||||
|
||||
BYPASS_RELEASE_INACTIVE_MIN = int(os.getenv("BYPASS_RELEASE_INACTIVE_MIN", "5"))
|
||||
# Logging settings
|
||||
LOG_FILE = LOG_DIR / "cwa-bookd-downloader.log"
|
||||
LOG_FILE = LOG_DIR / "cwa-bookd-downloader.log"
|
||||
|
||||
USING_TOR = string_to_bool(os.getenv("USING_TOR", "false"))
|
||||
# If using Tor, we don't need to set custom DNS, use DOH, or proxy
|
||||
if USING_TOR:
|
||||
_CUSTOM_DNS = ""
|
||||
USE_DOH = False
|
||||
HTTP_PROXY = ""
|
||||
HTTPS_PROXY = ""
|
||||
|
||||
@@ -62,6 +62,9 @@ def setup_logger(name: str, log_file: Path = LOG_FILE) -> CustomLogger:
|
||||
# File handler if log file is specified
|
||||
try:
|
||||
if ENABLE_LOGGING:
|
||||
# Create log directory if it doesn't exist
|
||||
log_dir = log_file.parent
|
||||
log_dir.mkdir(parents=True, exist_ok=True)
|
||||
file_handler = RotatingFileHandler(
|
||||
log_file,
|
||||
maxBytes=10485760, # 10MB
|
||||
|
||||
@@ -57,6 +57,7 @@ An intuitive web interface for searching and requesting book downloads, designed
|
||||
| `FLASK_DEBUG` | Debug mode toggle | `false` |
|
||||
| `FLASK_HOST` | Web interface binding | `0.0.0.0` |
|
||||
| `INGEST_DIR` | Book download directory | `/cwa-book-ingest` |
|
||||
| `TZ` | Container timezone | `UTC` |
|
||||
| `UID` | Runtime user ID | `1000` |
|
||||
| `GID` | Runtime group ID | `100` |
|
||||
| `ENABLE_LOGGING` | Enable log file | `true` |
|
||||
@@ -65,6 +66,8 @@ An intuitive web interface for searching and requesting book downloads, designed
|
||||
If logging is enabld, log folder default location is `/var/log/cwa-book-downloader`
|
||||
Available log levels: `DEBUG`, `INFO`, `WARNING`, `ERROR`, `CRITICAL`. Higher levels show fewer messages.
|
||||
|
||||
Note that if using TOR, the TZ will be calculated automatically based on IP.
|
||||
|
||||
#### Download Settings
|
||||
|
||||
| Variable | Description | Default Value |
|
||||
@@ -166,12 +169,32 @@ volumes:
|
||||
|
||||
Mount should align with your Calibre-Web-Automated ingest folder.
|
||||
|
||||
## 🧅 Tor Variant
|
||||
|
||||
This application also offers a variant that routes all its traffic through the Tor network. This can be useful for enhanced privacy or bypassing network restrictions.
|
||||
|
||||
To use the Tor variant:
|
||||
|
||||
1. Get the Tor-specific docker-compose file:
|
||||
```bash
|
||||
curl -O https://raw.githubusercontent.com/calibrain/calibre-web-automated-book-downloader/refs/heads/main/docker-compose.tor.yml
|
||||
```
|
||||
2. Start the service using this file:
|
||||
```bash
|
||||
docker compose -f docker-compose.tor.yml up -d
|
||||
```
|
||||
|
||||
**Important Considerations for Tor:**
|
||||
|
||||
* **Capabilities:** This variant requires the `NET_ADMIN` and `NET_RAW` Docker capabilities to configure `iptables` for transparent Tor proxying.
|
||||
* **Timezone:** When running in Tor mode, the container will attempt to determine the timezone based on the Tor exit node's IP address and set it automatically. This will override the `TZ` environment variable if it is set.
|
||||
* **Network Settings:** Custom DNS, DoH, and HTTP(S) proxy settings (`CUSTOM_DNS`, `USE_DOH`, `HTTP_PROXY`, `HTTPS_PROXY`) are ignored when using the Tor variant, as all traffic goes through Tor.
|
||||
|
||||
## 🏗️ Architecture
|
||||
|
||||
The application consists of two key services:
|
||||
The application consists of a single service:
|
||||
|
||||
1. **calibre-web-automated-bookdownloader**: Main application providing web interface and download functionality
|
||||
2. **cloudflarebypassforscraping**: Support service for handling Cloudflare-protected websites
|
||||
|
||||
## 🏥 Health Monitoring
|
||||
|
||||
@@ -182,6 +205,11 @@ Built-in health checks monitor:
|
||||
- Cloudflare bypass service connection
|
||||
|
||||
Checks run every 30 seconds with a 30-second timeout and 3 retries.
|
||||
You can enable by adding this to your compose :
|
||||
```
|
||||
HEALTHCHECK --interval=30s --timeout=30s --start-period=5s --retries=3 \
|
||||
CMD pyrequests http://localhost:8084/request/api/status || exit 1
|
||||
```
|
||||
|
||||
## 📝 Logging
|
||||
|
||||
|
||||
+2
-4
@@ -2,9 +2,7 @@ flask
|
||||
requests[socks]
|
||||
beautifulsoup4
|
||||
tqdm
|
||||
DrissionPage
|
||||
pyvirtualdisplay
|
||||
types-requests
|
||||
types-beautifulsoup4
|
||||
types-tqdm
|
||||
dnspython
|
||||
pyautogui
|
||||
seleniumbase
|
||||
|
||||
@@ -0,0 +1,103 @@
|
||||
#!/bin/bash
|
||||
set -e
|
||||
echo "[*] Installing Tor and dependencies..."
|
||||
echo "[*] Writing Tor transparent proxy config..."
|
||||
|
||||
cat <<EOF > /etc/tor/torrc
|
||||
VirtualAddrNetworkIPv4 10.192.0.0/10
|
||||
AutomapHostsOnResolve 1
|
||||
TransPort 9040
|
||||
DNSPort 53
|
||||
Log notice file /var/log/tor/notices.log
|
||||
EOF
|
||||
|
||||
echo "[*] Setting up DNS..."
|
||||
cat <<EOF > /etc/resolv.conf
|
||||
127.0.0.1
|
||||
EOF
|
||||
|
||||
echo "[*] Starting Tor..."
|
||||
service tor start
|
||||
|
||||
echo "[*] Setting up iptables rules..."
|
||||
|
||||
iptables -F
|
||||
iptables -t nat -F
|
||||
|
||||
# Don't redirect Tor's own traffic
|
||||
iptables -t nat -A OUTPUT -m owner --uid-owner debian-tor -j RETURN
|
||||
|
||||
# Allow loopback
|
||||
iptables -t nat -A OUTPUT -o lo -j RETURN
|
||||
|
||||
# Redirect all TCP to Tor's TransPort
|
||||
iptables -t nat -A OUTPUT -p tcp --syn -j REDIRECT --to-ports 9040
|
||||
|
||||
# For UDP DNS queries
|
||||
iptables -t nat -A OUTPUT -p udp --dport 53 ! -d 127.0.0.1 -j DNAT --to-destination 127.0.0.1:53
|
||||
|
||||
|
||||
# For TCP DNS queries (some DNS queries may use TCP)
|
||||
iptables -t nat -A OUTPUT -p tcp --dport 53 ! -d 127.0.0.1 -j DNAT --to-destination 127.0.0.1:53
|
||||
|
||||
echo "[✓] Transparent Tor routing enabled."
|
||||
|
||||
# Wait a bit to ensure Tor has bootstrapped
|
||||
echo "[*] Waiting for Tor to finish bootstrapping... (up to 5 minutes)"
|
||||
timeout 300 bash -c '
|
||||
while ! grep -q "Bootstrapped 100%" <(tail -n 20 -F /var/log/tor/notices.log 2>/dev/null); do
|
||||
printf "\r\033[KCurrent log: %s" "$(tail -n 1 /var/log/tor/notices.log 2>/dev/null)"
|
||||
sleep 1
|
||||
done
|
||||
# Print a newline when finished.
|
||||
echo ""
|
||||
'
|
||||
|
||||
echo "[✓] Tor is ready."
|
||||
|
||||
# Check if outgoing IP is using Tor
|
||||
echo "[*] Verifying Tor connectivity..."
|
||||
RESULT=$(pyrequests https://check.torproject.org/api/ip)
|
||||
echo "RESULT: $RESULT"
|
||||
IS_TOR=$(echo "$RESULT" | grep -oP '"IsTor":\s*\K(true|false)')
|
||||
IP=$(echo "$RESULT" | grep -oP '"IP":\s*"\K[^"]+')
|
||||
if [[ "$IS_TOR" == "true" ]]; then
|
||||
echo "[✓] Success! Traffic is routed through Tor. Current IP: $IP"
|
||||
else
|
||||
echo "[✗] Warning: Traffic is NOT using Tor. Current IP: $IP"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Set correct timezone
|
||||
# First check what is the timezone based on the IP
|
||||
# Then set the timezone
|
||||
|
||||
# Get timezone from IP
|
||||
TIMEZONE=$(pyrequests https://ipapi.co/timezone)
|
||||
# If TIMEZONE is not set, use the default timezone
|
||||
echo "[*] Current Timezone : $(date +%Z). IP Timezone: $TIMEZONE"
|
||||
|
||||
# Set timezone in Docker-compatible way
|
||||
if [ -f "/usr/share/zoneinfo/$TIMEZONE" ]; then
|
||||
# Remove existing symlink if it exists
|
||||
rm -f /etc/localtime
|
||||
# Create new symlink
|
||||
ln -sf /usr/share/zoneinfo/$TIMEZONE /etc/localtime
|
||||
# Set timezone file
|
||||
echo "$TIMEZONE" > /etc/timezone
|
||||
# Set TZ environment variable
|
||||
export TZ=$TIMEZONE
|
||||
# Verify the change
|
||||
echo "[✓] Timezone set to $TIMEZONE"
|
||||
echo "[*] Current time: $(date)"
|
||||
echo "[*] Timezone verification: $(date +%Z)"
|
||||
else
|
||||
echo "[!] Warning: Timezone file not found: $TIMEZONE"
|
||||
echo "[*] Available timezones:"
|
||||
ls -la /usr/share/zoneinfo/
|
||||
echo "[*] Falling back to container's default timezone: $TZ"
|
||||
fi
|
||||
|
||||
# Run the entrypoint script
|
||||
echo "[*] Running entrypoint script..."
|
||||
./entrypoint.sh
|
||||
Reference in New Issue
Block a user