Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
cca2587d8a | ||
|
|
e31e9774a3 | ||
|
|
afeae46821 | ||
|
|
29a8d856a6 | ||
|
|
b97e48235b | ||
|
|
7954ae9138 | ||
|
|
06778184af | ||
|
|
abf7f24178 | ||
|
|
3d84c5b42f | ||
|
|
b0206f76f8 | ||
|
|
8cb5335234 | ||
|
|
b2887eb4b0 | ||
|
|
06e468d043 | ||
|
|
c609c0b2bb | ||
|
|
875b705ed3 | ||
|
|
91dd479edb | ||
|
|
e870ada452 | ||
|
|
98aada2f55 | ||
|
|
dbe46e8e61 | ||
|
|
74e657e955 | ||
|
|
a0f8d14c45 | ||
|
|
a99dc1501d | ||
|
|
2cf336d704 | ||
|
|
f154b6994e | ||
|
|
823ceeef4a | ||
|
|
8ed6b94dfb | ||
|
|
2b5983d201 | ||
|
|
a4173eafcb | ||
|
|
15a61a5191 | ||
|
|
0cac541c0b | ||
|
|
85c8c9151d | ||
|
|
4472fbe8cf | ||
|
|
b293bee5f4 | ||
|
|
122a3633c2 | ||
|
|
0e2580030b | ||
|
|
17057ecfbe | ||
|
|
2b831dcfa5 | ||
|
|
78c61e88b3 | ||
|
|
57d85d0748 | ||
|
|
6492bd6a3c | ||
|
|
ed88aac5d5 | ||
|
|
5751910426 | ||
|
|
b02ad7452c | ||
|
|
289666aeef | ||
|
|
cc30d24144 | ||
|
|
50e53a13b0 | ||
|
|
a46d302ba8 | ||
|
|
c5d22e0f91 | ||
|
|
03321a5435 | ||
|
|
6aed906dfe | ||
|
|
742da1c43a | ||
|
|
8ea2fee0bb | ||
|
|
1c24312eb0 | ||
|
|
98e3a2f114 | ||
|
|
cd16f09f2e | ||
|
|
527c5d495d | ||
|
|
f5de2ab143 | ||
|
|
4e5c9b788f | ||
|
|
199d8453eb | ||
|
|
a9854b1a5c | ||
|
|
e4d3a372c8 | ||
|
|
ff44881415 | ||
|
|
9ffedc1fc0 | ||
|
|
00370818f0 | ||
|
|
7d9a82bfea | ||
|
|
207cff96d3 | ||
|
|
c8f21b8f8d | ||
|
|
5e04b6bfb8 | ||
|
|
09bd5ae9f0 | ||
|
|
5f6a81d97d |
@@ -37,3 +37,13 @@ dist/
|
||||
venv/
|
||||
.venv/
|
||||
env/
|
||||
|
||||
# Frontend build artifacts (built in separate stage)
|
||||
src/frontend/node_modules/
|
||||
src/frontend/dist/
|
||||
src/frontend/.vite/
|
||||
|
||||
# Old frontend code (replaced by src/frontend)
|
||||
templates/
|
||||
static/css/
|
||||
static/js/
|
||||
|
||||
@@ -8,7 +8,7 @@ on:
|
||||
workflow_dispatch:
|
||||
env:
|
||||
REGISTRY: ghcr.io
|
||||
IMAGE_NAME: ${{ github.repository }}
|
||||
IMAGE_NAME: ${{ github.repository_owner }}/shelfmark
|
||||
jobs:
|
||||
build-and-push-images:
|
||||
runs-on: ubuntu-latest
|
||||
@@ -20,12 +20,9 @@ jobs:
|
||||
strategy:
|
||||
matrix:
|
||||
include:
|
||||
- suffix: ""
|
||||
target: cwa-bd
|
||||
image_name_suffix: ""
|
||||
- suffix: "-tor"
|
||||
target: cwa-bd-tor
|
||||
image_name_suffix: "-tor"
|
||||
- target: shelfmark
|
||||
- target: shelfmark-lite
|
||||
image_name_suffix: "-lite"
|
||||
steps:
|
||||
- name: Get current date
|
||||
id: date
|
||||
@@ -67,6 +64,7 @@ jobs:
|
||||
push: ${{ github.event_name != 'pull_request' }}
|
||||
build-args: |
|
||||
BUILD_VERSION=${{ steps.date.outputs.date }}-${{ github.sha }}
|
||||
RELEASE_VERSION=${{ github.ref_name }}
|
||||
tags: ${{ steps.meta.outputs.tags }}
|
||||
labels: ${{ steps.meta.outputs.labels }}
|
||||
|
||||
@@ -76,4 +74,71 @@ jobs:
|
||||
with:
|
||||
subject-name: ${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}${{ matrix.image_name_suffix }}
|
||||
subject-digest: ${{ steps.push.outputs.digest }}
|
||||
push-to-registry: true
|
||||
push-to-registry: true
|
||||
|
||||
# Create aliases for backwards compatibility
|
||||
create-aliases:
|
||||
needs: build-and-push-images
|
||||
runs-on: ubuntu-latest
|
||||
if: github.event_name != 'pull_request'
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
env:
|
||||
# Legacy name for backwards compatibility (hardcoded so it works after rename)
|
||||
LEGACY_NAME: calibre-web-automated-book-downloader
|
||||
steps:
|
||||
- name: Log in to registry
|
||||
uses: docker/login-action@v3
|
||||
with:
|
||||
registry: ${{ env.REGISTRY }}
|
||||
username: ${{ github.actor }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
|
||||
- name: Create legacy aliases
|
||||
run: |
|
||||
# Current image names (follows repo name)
|
||||
STANDARD="${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}"
|
||||
LITE="${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}-lite"
|
||||
|
||||
# Legacy image names (hardcoded for backwards compatibility)
|
||||
LEGACY="${{ env.REGISTRY }}/${{ github.repository_owner }}/${{ env.LEGACY_NAME }}"
|
||||
LEGACY_TOR="${LEGACY}-tor"
|
||||
LEGACY_EXTBP="${LEGACY}-extbp"
|
||||
|
||||
SHA_SHORT=$(echo "${{ github.sha }}" | cut -c1-7)
|
||||
|
||||
# Helper function to create alias with all standard tags
|
||||
create_alias() {
|
||||
local SOURCE=$1
|
||||
local ALIAS=$2
|
||||
|
||||
# Always create SHA tag
|
||||
docker buildx imagetools create -t "${ALIAS}:sha-${SHA_SHORT}" "${SOURCE}:sha-${SHA_SHORT}"
|
||||
|
||||
if [[ "${{ github.ref }}" == refs/tags/v* ]]; then
|
||||
VERSION="${{ github.ref_name }}"
|
||||
VERSION_NUM="${VERSION#v}"
|
||||
MINOR="${VERSION_NUM%.*}"
|
||||
|
||||
docker buildx imagetools create -t "${ALIAS}:latest" "${SOURCE}:latest"
|
||||
docker buildx imagetools create -t "${ALIAS}:${VERSION_NUM}" "${SOURCE}:${VERSION_NUM}"
|
||||
docker buildx imagetools create -t "${ALIAS}:${MINOR}" "${SOURCE}:${MINOR}"
|
||||
docker buildx imagetools create -t "${ALIAS}:${VERSION}" "${SOURCE}:${VERSION}"
|
||||
else
|
||||
docker buildx imagetools create -t "${ALIAS}:dev" "${SOURCE}:dev"
|
||||
fi
|
||||
}
|
||||
|
||||
# Create legacy aliases pointing to current images
|
||||
# calibre-web-automated-book-downloader → standard image
|
||||
create_alias "${STANDARD}" "${LEGACY}"
|
||||
|
||||
# calibre-web-automated-book-downloader-tor → standard image
|
||||
create_alias "${STANDARD}" "${LEGACY_TOR}"
|
||||
|
||||
# calibre-web-automated-book-downloader-extbp → lite image
|
||||
create_alias "${LITE}" "${LEGACY_EXTBP}"
|
||||
|
||||
@@ -227,3 +227,7 @@ pyrightconfig.json
|
||||
|
||||
# End of https://www.toptal.com/developers/gitignore/api/macos,visualstudiocode,python
|
||||
/downloaded_files
|
||||
/.local/
|
||||
*.local.*
|
||||
.claude/
|
||||
.playwright-mcp/
|
||||
|
||||
@@ -2,20 +2,21 @@
|
||||
"version": "0.2.0",
|
||||
"configurations": [
|
||||
{
|
||||
"name": "Python Debugger: Current CWABD File",
|
||||
"name": "Python Debugger: Current Shelfmark File",
|
||||
"type": "debugpy",
|
||||
"request": "launch",
|
||||
"program": "${file}",
|
||||
"console": "integratedTerminal",
|
||||
"justMyCode": false,
|
||||
"env": {
|
||||
"INGEST_DIR": "/tmp/cwa-book-downloader",
|
||||
"TEMP_DIR": "/tmp/cwa-book-downloader",
|
||||
"INGEST_DIR": "/tmp/shelfmark",
|
||||
"TEMP_DIR": "/tmp/shelfmark",
|
||||
"LOG_LEVEL": "DEBUG",
|
||||
"LOG_ROOT": "/tmp/cwa-book-downloader",
|
||||
"LOG_ROOT": "/tmp/shelfmark",
|
||||
"ENABLE_LOGGING": "true",
|
||||
"DOCKERMODE": "false",
|
||||
"DEBUG": "true"
|
||||
"DEBUG": "true",
|
||||
"CUSTOM_DNS": "google",
|
||||
},
|
||||
},
|
||||
{
|
||||
@@ -26,7 +27,7 @@
|
||||
"preLaunchTask": "docker-compose up (dev)", // Spin up dev containers
|
||||
"postDebugTask": "docker-compose down (dev)", // Optional: tear them down
|
||||
"env": {
|
||||
"INGEST_DIR": "/tmp/cwa-book-downloader"
|
||||
"INGEST_DIR": "/tmp/shelfmark"
|
||||
},
|
||||
},
|
||||
{
|
||||
@@ -37,7 +38,7 @@
|
||||
"preLaunchTask": "docker-compose up (prod)",
|
||||
"postDebugTask": "docker-compose down (prod)",
|
||||
"env": {
|
||||
"INGEST_DIR": "/tmp/cwa-book-downloader"
|
||||
"INGEST_DIR": "/tmp/shelfmark"
|
||||
},
|
||||
},
|
||||
{
|
||||
@@ -53,9 +54,9 @@
|
||||
],
|
||||
"compounds": [
|
||||
{
|
||||
"name": "Launch CWA-BD",
|
||||
"name": "Launch Shelfmark",
|
||||
"configurations": [
|
||||
"Launch cwa-bd app.py",
|
||||
"Launch Shelfmark app.py",
|
||||
"Launch Browser"
|
||||
]
|
||||
}
|
||||
|
||||
@@ -1,9 +1,37 @@
|
||||
ARG TARGETPLATFORM
|
||||
ARG TARGETARCH
|
||||
ARG BUILDPLATFORM
|
||||
ARG BUILDARCH
|
||||
|
||||
# Frontend build stage.
|
||||
FROM --platform=$BUILDPLATFORM node:20-alpine AS frontend-builder
|
||||
|
||||
# Helpful debug output to see what platforms BuildKit thinks it's using
|
||||
RUN echo "BUILDPLATFORM=$BUILDPLATFORM BUILDARCH=$BUILDARCH TARGETPLATFORM=$TARGETPLATFORM TARGETARCH=$TARGETARCH"
|
||||
|
||||
WORKDIR /frontend
|
||||
|
||||
# Copy frontend package files
|
||||
COPY src/frontend/package*.json ./
|
||||
|
||||
# Install dependencies (cache mount for faster rebuilds)
|
||||
RUN --mount=type=cache,target=/root/.npm \
|
||||
npm ci
|
||||
|
||||
# Copy frontend source
|
||||
COPY src/frontend/ ./
|
||||
|
||||
# Build the frontend
|
||||
RUN npm run build
|
||||
|
||||
# Use python-slim as the base image
|
||||
FROM python:3.10-slim AS base
|
||||
|
||||
# Add build argument for version
|
||||
ARG BUILD_VERSION
|
||||
ENV BUILD_VERSION=${BUILD_VERSION}
|
||||
ARG RELEASE_VERSION
|
||||
ENV RELEASE_VERSION=${RELEASE_VERSION}
|
||||
|
||||
# Set shell to bash with pipefail option
|
||||
SHELL ["/bin/bash", "-o", "pipefail", "-c"]
|
||||
@@ -17,13 +45,12 @@ ENV DEBIAN_FRONTEND=noninteractive \
|
||||
PIP_NO_CACHE_DIR=1 \
|
||||
PIP_DISABLE_PIP_VERSION_CHECK=1 \
|
||||
PIP_DEFAULT_TIMEOUT=100 \
|
||||
NAME=Calibre-Web-Automated-Book-Downloader \
|
||||
NAME=Shelfmark \
|
||||
PYTHONPATH=/app \
|
||||
# UID/GID will be handled by entrypoint script, but TZ/Locale are still needed
|
||||
# PUID/PGID will be handled by entrypoint script, but TZ/Locale are still needed
|
||||
LANG=en_US.UTF-8 \
|
||||
LANGUAGE=en_US:en \
|
||||
LC_ALL=en_US.UTF-8 \
|
||||
APP_ENV=prod
|
||||
LC_ALL=en_US.UTF-8
|
||||
|
||||
# Set ARG for build-time expansion (FLASK_PORT), ENV for runtime access
|
||||
ENV FLASK_PORT=8084
|
||||
@@ -38,18 +65,17 @@ RUN apt-get update && \
|
||||
curl \
|
||||
# For entrypoint
|
||||
dumb-init \
|
||||
# For dumb display
|
||||
xvfb \
|
||||
# For screen recording
|
||||
ffmpeg \
|
||||
# For debug
|
||||
zip iputils-ping \
|
||||
# For user switching
|
||||
sudo \
|
||||
# --- Chromium Browser ---
|
||||
chromium-driver \
|
||||
# For tkinter (pyautogui)
|
||||
python3-tk && \
|
||||
# --- Tor support (activated via USING_TOR=true) ---
|
||||
tor \
|
||||
supervisor \
|
||||
iptables && \
|
||||
# Configure iptables alternatives for tor.sh compatibility
|
||||
update-alternatives --set iptables /usr/sbin/iptables-legacy && \
|
||||
update-alternatives --set ip6tables /usr/sbin/ip6tables-legacy && \
|
||||
# Cleanup APT cache *after* all installs in this layer
|
||||
apt-get purge -y --auto-remove -o APT::AutoRemove::RecommendsImportant=false && \
|
||||
apt-get clean && \
|
||||
@@ -66,61 +92,73 @@ RUN apt-get update && \
|
||||
WORKDIR /app
|
||||
|
||||
# Install Python dependencies using pip
|
||||
# Upgrade pip first, then copy requirements and install
|
||||
# Copying requirements.txt separately leverages build cache
|
||||
COPY requirements.txt .
|
||||
RUN pip install --no-cache-dir -r requirements.txt && \
|
||||
# Clean root's pip cache
|
||||
rm -rf /root/.cache
|
||||
|
||||
# Add this line to grant read/execute permissions to others
|
||||
RUN chmod -R o+rx /usr/bin/chromium && \
|
||||
chmod -R o+rx /usr/bin/chromedriver && \
|
||||
chmod -R o+w /usr/local/lib/python3.10/site-packages/seleniumbase/drivers/
|
||||
# Copying requirements files separately leverages build cache
|
||||
# Cache mount persists pip cache between builds for faster installs
|
||||
COPY requirements-base.txt requirements-shelfmark.txt ./
|
||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||
pip install -r requirements-base.txt
|
||||
|
||||
# Copy application code *after* dependencies are installed
|
||||
COPY . .
|
||||
|
||||
# Copy built frontend from frontend-builder stage
|
||||
COPY --from=frontend-builder /frontend/dist /app/frontend-dist
|
||||
|
||||
# Final setup: permissions and directories in one layer
|
||||
# Only creating directories and setting executable bits.
|
||||
# Ownership will be handled by the entrypoint script.
|
||||
RUN mkdir -p /var/log/cwa-book-downloader /cwa-book-ingest && \
|
||||
RUN mkdir -p /var/log/shelfmark /books && \
|
||||
chmod +x /app/entrypoint.sh /app/tor.sh /app/genDebug.sh
|
||||
|
||||
# Expose the application port
|
||||
EXPOSE ${FLASK_PORT}
|
||||
|
||||
# Add healthcheck for container status
|
||||
# This will run as root initially, but check localhost which should work if the app binds correctly.
|
||||
# Uses /api/health which doesn't require authentication
|
||||
HEALTHCHECK --interval=60s --timeout=60s --start-period=60s --retries=3 \
|
||||
CMD curl -s http://localhost:${FLASK_PORT}/request/api/status > /dev/null || exit 1
|
||||
CMD curl -s http://localhost:${FLASK_PORT}/api/health > /dev/null || exit 1
|
||||
|
||||
# Use dumb-init as the entrypoint to handle signals properly
|
||||
ENTRYPOINT ["/usr/bin/dumb-init", "--"]
|
||||
|
||||
|
||||
FROM base AS cwa-bd
|
||||
FROM base AS shelfmark
|
||||
|
||||
# Default command to run the application entrypoint script
|
||||
CMD ["/app/entrypoint.sh"]
|
||||
|
||||
FROM base AS cwa-bd-tor
|
||||
|
||||
ENV USING_TOR=true
|
||||
|
||||
# Install Tor and dependencies
|
||||
RUN apt-get update && \
|
||||
apt-get install -y --no-install-recommends \
|
||||
# --- Tor ---
|
||||
tor \
|
||||
# --- iptables ---
|
||||
iptables && \
|
||||
update-alternatives --set iptables /usr/sbin/iptables-legacy && \
|
||||
update-alternatives --set ip6tables /usr/sbin/ip6tables-legacy && \
|
||||
# Cleanup APT cache *after* all installs in this layer
|
||||
# For dumb display
|
||||
xvfb \
|
||||
# For screen recording
|
||||
ffmpeg \
|
||||
# --- Chromium ---
|
||||
chromium \
|
||||
# --- ChromeDriver ---
|
||||
chromium-driver \
|
||||
# For tkinter (pyautogui)
|
||||
python3-tk \
|
||||
# For RAR extraction
|
||||
unrar-free && \
|
||||
# Create symlink so rarfile library can find unrar
|
||||
ln -sf /usr/bin/unrar-free /usr/bin/unrar && \
|
||||
# Cleanup APT cache
|
||||
apt-get purge -y --auto-remove -o APT::AutoRemove::RecommendsImportant=false && \
|
||||
apt-get clean && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Override the default command to run Tor
|
||||
# Install additional dependencies (requirements file already copied in base stage)
|
||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||
pip install -r requirements-shelfmark.txt
|
||||
|
||||
# Grant read/execute permissions to others
|
||||
RUN chmod -R o+rx /usr/bin/chromium && \
|
||||
chmod -R o+rx /usr/bin/chromedriver && \
|
||||
chmod -R o+w /usr/local/lib/python3.10/site-packages/seleniumbase/drivers/
|
||||
|
||||
# Default command to run the application entrypoint script
|
||||
CMD ["/app/entrypoint.sh"]
|
||||
|
||||
FROM base AS shelfmark-lite
|
||||
|
||||
ENV USING_EXTERNAL_BYPASSER=true
|
||||
|
||||
CMD ["/app/entrypoint.sh"]
|
||||
|
||||
@@ -0,0 +1,84 @@
|
||||
.PHONY: help install dev build preview typecheck clean up down docker-build refresh restart
|
||||
|
||||
# Frontend directory
|
||||
FRONTEND_DIR := src/frontend
|
||||
|
||||
# Docker compose file
|
||||
COMPOSE_FILE := docker-compose.dev.yml
|
||||
|
||||
# Default target
|
||||
help:
|
||||
@echo "Available targets:"
|
||||
@echo ""
|
||||
@echo "Frontend:"
|
||||
@echo " install - Install frontend dependencies"
|
||||
@echo " dev - Start development server"
|
||||
@echo " build - Build frontend for production"
|
||||
@echo " preview - Preview production build"
|
||||
@echo " typecheck - Run TypeScript type checking"
|
||||
@echo " clean - Remove node_modules and build artifacts"
|
||||
@echo ""
|
||||
@echo "Backend (Docker):"
|
||||
@echo " up - Start backend services"
|
||||
@echo " down - Stop backend services"
|
||||
@echo " restart - Restart backend services (no rebuild)"
|
||||
@echo " docker-build - Build Docker image"
|
||||
@echo " refresh - Rebuild and restart backend services"
|
||||
|
||||
# Install dependencies
|
||||
install:
|
||||
@echo "Installing frontend dependencies..."
|
||||
cd $(FRONTEND_DIR) && npm install
|
||||
|
||||
# Start development server
|
||||
dev:
|
||||
@echo "Starting development server..."
|
||||
cd $(FRONTEND_DIR) && npm run dev
|
||||
|
||||
# Build for production
|
||||
build:
|
||||
@echo "Building frontend for production..."
|
||||
cd $(FRONTEND_DIR) && npm run build
|
||||
|
||||
# Preview production build
|
||||
preview:
|
||||
@echo "Previewing production build..."
|
||||
cd $(FRONTEND_DIR) && npm run preview
|
||||
|
||||
# Type checking
|
||||
typecheck:
|
||||
@echo "Running TypeScript type checking..."
|
||||
cd $(FRONTEND_DIR) && npm run typecheck
|
||||
|
||||
# Clean build artifacts and dependencies
|
||||
clean:
|
||||
@echo "Cleaning build artifacts and dependencies..."
|
||||
rm -rf $(FRONTEND_DIR)/node_modules
|
||||
rm -rf $(FRONTEND_DIR)/dist
|
||||
|
||||
# Start backend services
|
||||
up:
|
||||
@echo "Starting backend services..."
|
||||
docker compose -f $(COMPOSE_FILE) up -d
|
||||
|
||||
# Stop backend services
|
||||
down:
|
||||
@echo "Stopping backend services..."
|
||||
docker compose -f $(COMPOSE_FILE) down
|
||||
|
||||
# Build Docker image
|
||||
docker-build:
|
||||
@echo "Building Docker image..."
|
||||
docker compose -f $(COMPOSE_FILE) build
|
||||
|
||||
# Restart backend services (no rebuild)
|
||||
restart:
|
||||
@echo "Restarting backend services..."
|
||||
docker compose -f $(COMPOSE_FILE) restart
|
||||
|
||||
# Rebuild and restart backend services
|
||||
refresh:
|
||||
@echo "Rebuilding and restarting backend services..."
|
||||
docker compose -f $(COMPOSE_FILE) down
|
||||
docker compose -f $(COMPOSE_FILE) build
|
||||
docker compose -f $(COMPOSE_FILE) up -d
|
||||
|
Before Width: | Height: | Size: 874 KiB |
|
Before Width: | Height: | Size: 233 KiB |
|
After Width: | Height: | Size: 2.2 MiB |
|
After Width: | Height: | Size: 151 KiB |
|
Before Width: | Height: | Size: 110 KiB |
|
After Width: | Height: | Size: 1.1 MiB |
|
After Width: | Height: | Size: 2.3 MiB |
|
Before Width: | Height: | Size: 419 KiB |
|
Before Width: | Height: | Size: 244 KiB |
@@ -1,389 +0,0 @@
|
||||
"""Flask web application for book download service with URL rewrite support."""
|
||||
|
||||
import logging
|
||||
import io, re, os
|
||||
import sqlite3
|
||||
from functools import wraps
|
||||
from flask import Flask, request, jsonify, render_template, send_file, send_from_directory
|
||||
from werkzeug.middleware.proxy_fix import ProxyFix
|
||||
from werkzeug.security import check_password_hash
|
||||
from werkzeug.wrappers import Response
|
||||
from flask import url_for as flask_url_for
|
||||
import typing
|
||||
|
||||
from logger import setup_logger
|
||||
from config import _SUPPORTED_BOOK_LANGUAGE, BOOK_LANGUAGE
|
||||
from env import FLASK_HOST, FLASK_PORT, APP_ENV, CWA_DB_PATH, DEBUG
|
||||
import backend
|
||||
|
||||
from models import SearchFilters
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
app = Flask(__name__)
|
||||
app.wsgi_app = ProxyFix(app.wsgi_app) # type: ignore
|
||||
app.config['SEND_FILE_MAX_AGE_DEFAULT'] = 0 # Disable caching
|
||||
app.config['APPLICATION_ROOT'] = '/'
|
||||
|
||||
# Flask logger
|
||||
app.logger.handlers = logger.handlers
|
||||
app.logger.setLevel(logger.level)
|
||||
# Also handle Werkzeug's logger
|
||||
werkzeug_logger = logging.getLogger('werkzeug')
|
||||
werkzeug_logger.handlers = logger.handlers
|
||||
werkzeug_logger.setLevel(logger.level)
|
||||
|
||||
# Set up authentication defaults
|
||||
# The secret key will reset every time we restart, which will
|
||||
# require users to authenticate again
|
||||
app.config.update(
|
||||
SECRET_KEY = os.urandom(64)
|
||||
)
|
||||
|
||||
def login_required(f):
|
||||
@wraps(f)
|
||||
def decorated_function(*args, **kwargs):
|
||||
# If the CWA_DB_PATH variable exists, but isn't a valid
|
||||
# path, return a server error
|
||||
if CWA_DB_PATH is not None and not os.path.isfile(CWA_DB_PATH):
|
||||
logger.error(f"CWA_DB_PATH is set to {CWA_DB_PATH} but this is not a valid path")
|
||||
return Response("Internal Server Error", 500)
|
||||
if not authenticate():
|
||||
return Response(
|
||||
response="Unauthorized",
|
||||
status=401,
|
||||
headers={
|
||||
"WWW-Authenticate": 'Basic realm="Calibre-Web-Automated-Book-Downloader"',
|
||||
},
|
||||
)
|
||||
return f(*args, **kwargs)
|
||||
return decorated_function
|
||||
|
||||
def register_dual_routes(app : Flask) -> None:
|
||||
"""
|
||||
Register each route both with and without the /request prefix.
|
||||
This function should be called after all routes are defined.
|
||||
"""
|
||||
# Store original url_map rules
|
||||
rules = list(app.url_map.iter_rules())
|
||||
|
||||
# Add /request prefix to each rule
|
||||
for rule in rules:
|
||||
if rule.rule != '/request/' and rule.rule != '/request': # Skip if it's already a request route
|
||||
# Create new routes with /request prefix, both with and without trailing slash
|
||||
base_rule = rule.rule[:-1] if rule.rule.endswith('/') else rule.rule
|
||||
if base_rule == '': # Special case for root path
|
||||
app.add_url_rule('/request', f"root_request",
|
||||
view_func=app.view_functions[rule.endpoint],
|
||||
methods=rule.methods)
|
||||
app.add_url_rule('/request/', f"root_request_slash",
|
||||
view_func=app.view_functions[rule.endpoint],
|
||||
methods=rule.methods)
|
||||
else:
|
||||
app.add_url_rule(f"/request{base_rule}",
|
||||
f"{rule.endpoint}_request",
|
||||
view_func=app.view_functions[rule.endpoint],
|
||||
methods=rule.methods)
|
||||
app.add_url_rule(f"/request{base_rule}/",
|
||||
f"{rule.endpoint}_request_slash",
|
||||
view_func=app.view_functions[rule.endpoint],
|
||||
methods=rule.methods)
|
||||
app.jinja_env.globals['url_for'] = url_for_with_request
|
||||
|
||||
def url_for_with_request(endpoint : str, **values : typing.Any) -> str:
|
||||
"""Generate URLs with /request prefix by default."""
|
||||
if endpoint == 'static':
|
||||
# For static files, add /request prefix
|
||||
url = flask_url_for(endpoint, **values)
|
||||
return f"/request{url}"
|
||||
return flask_url_for(endpoint, **values)
|
||||
|
||||
@app.route('/')
|
||||
@login_required
|
||||
def index() -> str:
|
||||
"""
|
||||
Render main page with search and status table.
|
||||
"""
|
||||
return render_template('index.html', book_languages=_SUPPORTED_BOOK_LANGUAGE, default_language=BOOK_LANGUAGE, debug=DEBUG)
|
||||
|
||||
@app.route('/favico<path:_>')
|
||||
@app.route('/request/favico<path:_>')
|
||||
@app.route('/request/static/favico<path:_>')
|
||||
def favicon(_ : typing.Any) -> Response:
|
||||
return send_from_directory(os.path.join(app.root_path, 'static', 'media'),
|
||||
'favicon.ico', mimetype='image/vnd.microsoft.icon')
|
||||
|
||||
from typing import Union, Tuple
|
||||
|
||||
if DEBUG:
|
||||
import subprocess
|
||||
import time
|
||||
from cloudflare_bypasser import _reset_driver as STOP_GUI
|
||||
@app.route('/debug', methods=['GET'])
|
||||
@login_required
|
||||
def debug() -> Union[Response, Tuple[Response, int]]:
|
||||
"""
|
||||
This will run the /app/debug.sh script, which will generate a debug zip with all the logs
|
||||
The file will be named /tmp/cwa-book-downloader-debug.zip
|
||||
And then return it to the user
|
||||
"""
|
||||
try:
|
||||
# Run the debug script
|
||||
STOP_GUI()
|
||||
time.sleep(1)
|
||||
result = subprocess.run(['/app/genDebug.sh'], capture_output=True, text=True, check=True)
|
||||
if result.returncode != 0:
|
||||
raise Exception(f"Debug script failed: {result.stderr}")
|
||||
logger.info(f"Debug script executed: {result.stdout}")
|
||||
debug_file_path = result.stdout.strip().split('\n')[-1]
|
||||
if not os.path.exists(debug_file_path):
|
||||
logger.error("Debug zip file not found after running debug script")
|
||||
return jsonify({"error": "Failed to generate debug information"}), 500
|
||||
|
||||
# Return the file to the user
|
||||
return send_file(
|
||||
debug_file_path,
|
||||
mimetype='application/zip',
|
||||
download_name=os.path.basename(debug_file_path),
|
||||
as_attachment=True
|
||||
)
|
||||
except subprocess.CalledProcessError as e:
|
||||
logger.error_trace(f"Debug script error: {e}, stdout: {e.stdout}, stderr: {e.stderr}")
|
||||
return jsonify({"error": f"Debug script failed: {e.stderr}"}), 500
|
||||
except Exception as e:
|
||||
logger.error_trace(f"Debug endpoint error: {e}")
|
||||
return jsonify({"error": str(e)}), 500
|
||||
|
||||
if DEBUG:
|
||||
@app.route('/api/restart', methods=['GET'])
|
||||
@login_required
|
||||
def restart() -> Union[Response, Tuple[Response, int]]:
|
||||
"""
|
||||
Restart the application
|
||||
"""
|
||||
os._exit(0)
|
||||
|
||||
@app.route('/api/search', methods=['GET'])
|
||||
@login_required
|
||||
def api_search() -> Union[Response, Tuple[Response, int]]:
|
||||
"""
|
||||
Search for books matching the provided query.
|
||||
|
||||
Query Parameters:
|
||||
query (str): Search term (ISBN, title, author, etc.)
|
||||
isbn (str): Book ISBN
|
||||
author (str): Book Author
|
||||
title (str): Book Title
|
||||
lang (str): Book Language
|
||||
sort (str): Order to sort results
|
||||
content (str): Content type of book
|
||||
format (str): File format filter (pdf, epub, mobi, azw3, fb2, djvu, cbz, cbr)
|
||||
|
||||
Returns:
|
||||
flask.Response: JSON array of matching books or error response.
|
||||
"""
|
||||
query = request.args.get('query', '')
|
||||
|
||||
filters = SearchFilters(
|
||||
isbn = request.args.getlist('isbn'),
|
||||
author = request.args.getlist('author'),
|
||||
title = request.args.getlist('title'),
|
||||
lang = request.args.getlist('lang'),
|
||||
sort = request.args.get('sort'),
|
||||
content = request.args.getlist('content'),
|
||||
format = request.args.getlist('format'),
|
||||
)
|
||||
|
||||
if not query and not any(vars(filters).values()):
|
||||
return jsonify([])
|
||||
|
||||
try:
|
||||
books = backend.search_books(query, filters)
|
||||
return jsonify(books)
|
||||
except Exception as e:
|
||||
logger.error_trace(f"Search error: {e}")
|
||||
return jsonify({"error": str(e)}), 500
|
||||
|
||||
@app.route('/api/info', methods=['GET'])
|
||||
@login_required
|
||||
def api_info() -> Union[Response, Tuple[Response, int]]:
|
||||
"""
|
||||
Get detailed book information.
|
||||
|
||||
Query Parameters:
|
||||
id (str): Book identifier (MD5 hash)
|
||||
|
||||
Returns:
|
||||
flask.Response: JSON object with book details, or an error message.
|
||||
"""
|
||||
book_id = request.args.get('id', '')
|
||||
if not book_id:
|
||||
return jsonify({"error": "No book ID provided"}), 400
|
||||
|
||||
try:
|
||||
book = backend.get_book_info(book_id)
|
||||
if book:
|
||||
return jsonify(book)
|
||||
return jsonify({"error": "Book not found"}), 404
|
||||
except Exception as e:
|
||||
logger.error_trace(f"Info error: {e}")
|
||||
return jsonify({"error": str(e)}), 500
|
||||
|
||||
@app.route('/api/download', methods=['GET'])
|
||||
@login_required
|
||||
def api_download() -> Union[Response, Tuple[Response, int]]:
|
||||
"""
|
||||
Queue a book for download.
|
||||
|
||||
Query Parameters:
|
||||
id (str): Book identifier (MD5 hash)
|
||||
|
||||
Returns:
|
||||
flask.Response: JSON status object indicating success or failure.
|
||||
"""
|
||||
book_id = request.args.get('id', '')
|
||||
if not book_id:
|
||||
return jsonify({"error": "No book ID provided"}), 400
|
||||
|
||||
try:
|
||||
success = backend.queue_book(book_id)
|
||||
if success:
|
||||
return jsonify({"status": "queued"})
|
||||
return jsonify({"error": "Failed to queue book"}), 500
|
||||
except Exception as e:
|
||||
logger.error_trace(f"Download error: {e}")
|
||||
return jsonify({"error": str(e)}), 500
|
||||
|
||||
@app.route('/api/status', methods=['GET'])
|
||||
@login_required
|
||||
def api_status() -> Union[Response, Tuple[Response, int]]:
|
||||
"""
|
||||
Get current download queue status.
|
||||
|
||||
Returns:
|
||||
flask.Response: JSON object with queue status.
|
||||
"""
|
||||
try:
|
||||
status = backend.queue_status()
|
||||
return jsonify(status)
|
||||
except Exception as e:
|
||||
logger.error_trace(f"Status error: {e}")
|
||||
return jsonify({"error": str(e)}), 500
|
||||
|
||||
@app.route('/api/localdownload', methods=['GET'])
|
||||
@login_required
|
||||
def api_local_download() -> Union[Response, Tuple[Response, int]]:
|
||||
"""
|
||||
Download an EPUB file from local storage if available.
|
||||
|
||||
Query Parameters:
|
||||
id (str): Book identifier (MD5 hash)
|
||||
|
||||
Returns:
|
||||
flask.Response: The EPUB file if found, otherwise an error response.
|
||||
"""
|
||||
book_id = request.args.get('id', '')
|
||||
if not book_id:
|
||||
return jsonify({"error": "No book ID provided"}), 400
|
||||
|
||||
try:
|
||||
file_data, book_info = backend.get_book_data(book_id)
|
||||
if file_data is None:
|
||||
# Book data not found or not available
|
||||
return jsonify({"error": "File not found"}), 404
|
||||
# Santize the file name
|
||||
file_name = book_info.title
|
||||
file_name = re.sub(r'[\\/:*?"<>|]', '_', file_name.strip())[:245]
|
||||
file_extension = book_info.format
|
||||
# Prepare the file for sending to the client
|
||||
data = io.BytesIO(file_data)
|
||||
return send_file(
|
||||
data,
|
||||
download_name=f"{file_name}.{file_extension}",
|
||||
as_attachment=True
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
logger.error_trace(f"Local download error: {e}")
|
||||
return jsonify({"error": str(e)}), 500
|
||||
|
||||
@app.errorhandler(404)
|
||||
def not_found_error(error: Exception) -> Union[Response, Tuple[Response, int]]:
|
||||
"""
|
||||
Handle 404 (Not Found) errors.
|
||||
|
||||
Args:
|
||||
error (HTTPException): The 404 error raised by Flask.
|
||||
|
||||
Returns:
|
||||
flask.Response: JSON error message with 404 status.
|
||||
"""
|
||||
logger.warning(f"404 error: {request.url} : {error}")
|
||||
return jsonify({"error": "Resource not found"}), 404
|
||||
|
||||
@app.errorhandler(500)
|
||||
def internal_error(error: Exception) -> Union[Response, Tuple[Response, int]]:
|
||||
"""
|
||||
Handle 500 (Internal Server) errors.
|
||||
|
||||
Args:
|
||||
error (HTTPException): The 500 error raised by Flask.
|
||||
|
||||
Returns:
|
||||
flask.Response: JSON error message with 500 status.
|
||||
"""
|
||||
logger.error_trace(f"500 error: {error}")
|
||||
return jsonify({"error": "Internal server error"}), 500
|
||||
|
||||
def authenticate() -> bool:
|
||||
"""
|
||||
Helper function that validates Basic credentials
|
||||
against a Calibre-Web app.db SQLite database
|
||||
|
||||
Database structure:
|
||||
- Table 'user' with columns: 'name' (username), 'password'
|
||||
"""
|
||||
|
||||
# If the database doesn't exist, the user is always authenticated
|
||||
if not CWA_DB_PATH:
|
||||
return True
|
||||
|
||||
# If no authorization object exists, return false to prompt
|
||||
# a request to the user
|
||||
if not request.authorization:
|
||||
return False
|
||||
|
||||
username = request.authorization.get("username")
|
||||
password = request.authorization.get("password")
|
||||
|
||||
# Validate credentials against database
|
||||
try:
|
||||
conn = sqlite3.connect(CWA_DB_PATH)
|
||||
cur = conn.cursor()
|
||||
cur.execute("SELECT password FROM user WHERE name = ?", (username,))
|
||||
row = cur.fetchone()
|
||||
conn.close()
|
||||
|
||||
# Check if user exists and password is correct
|
||||
if not row or not row[0] or not check_password_hash(row[0], password):
|
||||
logger.error("User not found or password check failed")
|
||||
return False
|
||||
|
||||
except Exception as e:
|
||||
logger.error_trace(f"CWA DB or authentication send_from_directory: {e}")
|
||||
return False
|
||||
|
||||
logger.info(f"Authentication successful for user {username}")
|
||||
return True
|
||||
|
||||
# Register all routes with /request prefix
|
||||
register_dual_routes(app)
|
||||
|
||||
logger.log_resource_usage()
|
||||
|
||||
if __name__ == '__main__':
|
||||
logger.info(f"Starting Flask application on {FLASK_HOST}:{FLASK_PORT} IN {APP_ENV} mode")
|
||||
app.run(
|
||||
host=FLASK_HOST,
|
||||
port=FLASK_PORT,
|
||||
debug=DEBUG
|
||||
)
|
||||
@@ -1,191 +0,0 @@
|
||||
"""Backend logic for the book download application."""
|
||||
|
||||
import threading, time
|
||||
import shutil
|
||||
from pathlib import Path
|
||||
from typing import Dict, List, Optional, Any, Tuple
|
||||
import subprocess
|
||||
import os
|
||||
|
||||
from logger import setup_logger
|
||||
from config import CUSTOM_SCRIPT
|
||||
from env import INGEST_DIR, TMP_DIR, MAIN_LOOP_SLEEP_TIME, USE_BOOK_TITLE
|
||||
from models import book_queue, BookInfo, QueueStatus, SearchFilters
|
||||
import book_manager
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
def _sanitize_filename(filename: str) -> str:
|
||||
"""Sanitize a filename by replacing spaces with underscores and removing invalid characters."""
|
||||
keepcharacters = (' ','.','_')
|
||||
return "".join(c for c in filename if c.isalnum() or c in keepcharacters).rstrip()
|
||||
|
||||
def search_books(query: str, filters: SearchFilters) -> List[Dict[str, Any]]:
|
||||
"""Search for books matching the query.
|
||||
|
||||
Args:
|
||||
query: Search term
|
||||
filters: Search filters object
|
||||
|
||||
Returns:
|
||||
List[Dict]: List of book information dictionaries
|
||||
"""
|
||||
try:
|
||||
books = book_manager.search_books(query, filters)
|
||||
return [_book_info_to_dict(book) for book in books]
|
||||
except Exception as e:
|
||||
logger.error_trace(f"Error searching books: {e}")
|
||||
return []
|
||||
|
||||
def get_book_info(book_id: str) -> Optional[Dict[str, Any]]:
|
||||
"""Get detailed information for a specific book.
|
||||
|
||||
Args:
|
||||
book_id: Book identifier
|
||||
|
||||
Returns:
|
||||
Optional[Dict]: Book information dictionary if found
|
||||
"""
|
||||
try:
|
||||
book = book_manager.get_book_info(book_id)
|
||||
return _book_info_to_dict(book)
|
||||
except Exception as e:
|
||||
logger.error_trace(f"Error getting book info: {e}")
|
||||
return None
|
||||
|
||||
def queue_book(book_id: str) -> bool:
|
||||
"""Add a book to the download queue.
|
||||
|
||||
Args:
|
||||
book_id: Book identifier
|
||||
|
||||
Returns:
|
||||
bool: True if book was successfully queued
|
||||
"""
|
||||
try:
|
||||
book_info = book_manager.get_book_info(book_id)
|
||||
book_queue.add(book_id, book_info)
|
||||
logger.info(f"Book queued: {book_info.title}")
|
||||
return True
|
||||
except Exception as e:
|
||||
logger.error_trace(f"Error queueing book: {e}")
|
||||
return False
|
||||
|
||||
def queue_status() -> Dict[str, Dict[str, Any]]:
|
||||
"""Get current status of the download queue.
|
||||
|
||||
Returns:
|
||||
Dict: Queue status organized by status type
|
||||
"""
|
||||
status = book_queue.get_status()
|
||||
# Convert Enum keys to strings and properly format the response
|
||||
return {
|
||||
status_type.value: books
|
||||
for status_type, books in status.items()
|
||||
}
|
||||
|
||||
def get_book_data(book_id: str) -> Tuple[Optional[bytes], BookInfo]:
|
||||
"""Get book data for a specific book, including its title.
|
||||
|
||||
Args:
|
||||
book_id: Book identifier
|
||||
|
||||
Returns:
|
||||
Tuple[Optional[bytes], str]: Book data if available, and the book title
|
||||
"""
|
||||
try:
|
||||
book_info = book_queue._book_data[book_id]
|
||||
path = book_info.download_path
|
||||
with open(path, "rb") as f:
|
||||
return f.read(), book_info
|
||||
except Exception as e:
|
||||
logger.error_trace(f"Error getting book data: {e}")
|
||||
book_info.download_path = None
|
||||
return None, ""
|
||||
|
||||
def _book_info_to_dict(book: BookInfo) -> Dict[str, Any]:
|
||||
"""Convert BookInfo object to dictionary representation."""
|
||||
return {
|
||||
key: value for key, value in book.__dict__.items()
|
||||
if value is not None
|
||||
}
|
||||
|
||||
def _download_book(book_id: str) -> Optional[str]:
|
||||
"""Download and process a book.
|
||||
|
||||
Args:
|
||||
book_id: Book identifier
|
||||
|
||||
Returns:
|
||||
str: Path to the downloaded book if successful, None otherwise
|
||||
"""
|
||||
try:
|
||||
book_info = book_queue._book_data[book_id]
|
||||
|
||||
if USE_BOOK_TITLE:
|
||||
book_name = _sanitize_filename(book_info.title)
|
||||
else:
|
||||
book_name = book_id
|
||||
book_name += f".{book_info.format}"
|
||||
book_path = TMP_DIR / book_name
|
||||
|
||||
success = book_manager.download_book(book_info, book_path)
|
||||
if not success:
|
||||
raise Exception("Unkown error downloading book")
|
||||
|
||||
if CUSTOM_SCRIPT:
|
||||
logger.info(f"Running custom script: {CUSTOM_SCRIPT}")
|
||||
subprocess.run([CUSTOM_SCRIPT, book_path])
|
||||
intermediate_path = INGEST_DIR / f"{book_id}.crdownload"
|
||||
final_path = INGEST_DIR / book_name
|
||||
|
||||
if os.path.exists(book_path):
|
||||
logger.info(f"Moving book to ingest directory then renaming: {book_path} -> {intermediate_path} -> {final_path}")
|
||||
try:
|
||||
shutil.move(book_path, intermediate_path)
|
||||
except Exception as e:
|
||||
logger.debug(f"Error moving book: {e}, will try copying instead")
|
||||
shutil.copy(book_path, intermediate_path)
|
||||
os.remove(book_path)
|
||||
logger.info(f"Renaming book: {intermediate_path} -> {final_path}")
|
||||
os.rename(intermediate_path, final_path)
|
||||
return str(final_path)
|
||||
except Exception as e:
|
||||
logger.error_trace(f"Error downloading book: {e}")
|
||||
return None
|
||||
|
||||
def download_loop() -> None:
|
||||
"""Background thread for processing download queue."""
|
||||
logger.info("Starting download loop")
|
||||
|
||||
while True:
|
||||
book_id = book_queue.get_next()
|
||||
if not book_id:
|
||||
time.sleep(MAIN_LOOP_SLEEP_TIME)
|
||||
continue
|
||||
|
||||
try:
|
||||
book_queue.update_status(book_id, QueueStatus.DOWNLOADING)
|
||||
download_path = _download_book(book_id)
|
||||
if download_path:
|
||||
book_queue.update_download_path(book_id, download_path)
|
||||
|
||||
new_status = (
|
||||
QueueStatus.AVAILABLE if download_path else QueueStatus.ERROR
|
||||
)
|
||||
book_queue.update_status(book_id, new_status)
|
||||
|
||||
logger.info(
|
||||
f"Book {book_id} download {'successful' if download_path else 'failed'}"
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
logger.error_trace(f"Error in download loop: {e}")
|
||||
book_queue.update_status(book_id, QueueStatus.ERROR)
|
||||
|
||||
# Start download loop in background thread
|
||||
download_thread = threading.Thread(
|
||||
target=download_loop,
|
||||
daemon=True
|
||||
)
|
||||
download_thread.start()
|
||||
@@ -1,378 +0,0 @@
|
||||
"""Book download manager handling search and retrieval operations."""
|
||||
|
||||
import time, json, re
|
||||
from pathlib import Path
|
||||
from urllib.parse import quote
|
||||
from typing import List, Optional, Dict, Union
|
||||
from bs4 import BeautifulSoup, Tag, NavigableString, ResultSet
|
||||
|
||||
import downloader
|
||||
from logger import setup_logger
|
||||
from config import SUPPORTED_FORMATS, BOOK_LANGUAGE, AA_BASE_URL
|
||||
from env import AA_DONATOR_KEY, USE_CF_BYPASS
|
||||
from models import BookInfo, SearchFilters
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
|
||||
def search_books(query: str, filters: SearchFilters) -> List[BookInfo]:
|
||||
"""Search for books matching the query.
|
||||
|
||||
Args:
|
||||
query: Search term (ISBN, title, author, etc.)
|
||||
|
||||
Returns:
|
||||
List[BookInfo]: List of matching books
|
||||
|
||||
Raises:
|
||||
Exception: If no books found or parsing fails
|
||||
"""
|
||||
query_html = quote(query)
|
||||
|
||||
if filters.isbn:
|
||||
# ISBNs are included in query string
|
||||
isbns = " || ".join(
|
||||
[f"('isbn13:{isbn}' || 'isbn10:{isbn}')" for isbn in filters.isbn]
|
||||
)
|
||||
query_html = quote(f"({isbns}) {query}")
|
||||
|
||||
filters_query = ""
|
||||
|
||||
for value in filters.lang or BOOK_LANGUAGE:
|
||||
if value != "all":
|
||||
filters_query += f"&lang={quote(value)}"
|
||||
|
||||
if filters.sort:
|
||||
filters_query += f"&sort={quote(filters.sort)}"
|
||||
|
||||
if filters.content:
|
||||
for value in filters.content:
|
||||
filters_query += f"&content={quote(value)}"
|
||||
|
||||
# Handle format filter
|
||||
formats_to_use = filters.format if filters.format else SUPPORTED_FORMATS
|
||||
|
||||
index = 1
|
||||
for filter_type, filter_values in vars(filters).items():
|
||||
if filter_type == "author" or filter_type == "title" and filter_values:
|
||||
for value in filter_values:
|
||||
filters_query += (
|
||||
f"&termtype_{index}={filter_type}&termval_{index}={quote(value)}"
|
||||
)
|
||||
index += 1
|
||||
|
||||
url = (
|
||||
f"{AA_BASE_URL}"
|
||||
f"/search?index=&page=1&display=table"
|
||||
f"&acc=aa_download&acc=external_download"
|
||||
f"&ext={'&ext='.join(formats_to_use)}"
|
||||
f"&q={query_html}"
|
||||
f"{filters_query}"
|
||||
)
|
||||
|
||||
html = downloader.html_get_page(url)
|
||||
if not html:
|
||||
raise Exception("Failed to fetch search results")
|
||||
|
||||
if "No files found." in html:
|
||||
logger.info(f"No books found for query: {query}")
|
||||
raise Exception("No books found. Please try another query.")
|
||||
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
tbody: Tag | NavigableString | None = soup.find("table")
|
||||
|
||||
if not tbody:
|
||||
logger.warning(f"No results table found for query: {query}")
|
||||
raise Exception("No books found. Please try another query.")
|
||||
|
||||
books = []
|
||||
if isinstance(tbody, Tag):
|
||||
for line_tr in tbody.find_all("tr"):
|
||||
try:
|
||||
book = _parse_search_result_row(line_tr)
|
||||
if book:
|
||||
books.append(book)
|
||||
except Exception as e:
|
||||
logger.error_trace(f"Failed to parse search result row: {e}")
|
||||
|
||||
books.sort(
|
||||
key=lambda x: (
|
||||
SUPPORTED_FORMATS.index(x.format)
|
||||
if x.format in SUPPORTED_FORMATS
|
||||
else len(SUPPORTED_FORMATS)
|
||||
)
|
||||
)
|
||||
|
||||
return books
|
||||
|
||||
|
||||
def _parse_search_result_row(row: Tag) -> Optional[BookInfo]:
|
||||
"""Parse a single search result row into a BookInfo object."""
|
||||
try:
|
||||
cells = row.find_all("td")
|
||||
preview_img = cells[0].find("img")
|
||||
preview = preview_img["src"] if preview_img else None
|
||||
|
||||
return BookInfo(
|
||||
id=row.find_all("a")[0]["href"].split("/")[-1],
|
||||
preview=preview,
|
||||
title=cells[1].find("span").next,
|
||||
author=cells[2].find("span").next,
|
||||
publisher=cells[3].find("span").next,
|
||||
year=cells[4].find("span").next,
|
||||
language=cells[7].find("span").next,
|
||||
format=cells[9].find("span").next.lower(),
|
||||
size=cells[10].find("span").next,
|
||||
)
|
||||
except Exception as e:
|
||||
logger.error_trace(f"Error parsing search result row: {e}")
|
||||
return None
|
||||
|
||||
|
||||
def get_book_info(book_id: str) -> BookInfo:
|
||||
"""Get detailed information for a specific book.
|
||||
|
||||
Args:
|
||||
book_id: Book identifier (MD5 hash)
|
||||
|
||||
Returns:
|
||||
BookInfo: Detailed book information
|
||||
"""
|
||||
url = f"{AA_BASE_URL}/md5/{book_id}"
|
||||
html = downloader.html_get_page(url)
|
||||
|
||||
if not html:
|
||||
raise Exception(f"Failed to fetch book info for ID: {book_id}")
|
||||
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
|
||||
return _parse_book_info_page(soup, book_id)
|
||||
|
||||
|
||||
def _parse_book_info_page(soup: BeautifulSoup, book_id: str) -> BookInfo:
|
||||
"""Parse the book info page HTML into a BookInfo object."""
|
||||
data = soup.select_one("body > main > div:nth-of-type(1)")
|
||||
|
||||
if not data:
|
||||
raise Exception(f"Failed to parse book info for ID: {book_id}")
|
||||
|
||||
preview: str = ""
|
||||
|
||||
node = data.select_one("div:nth-of-type(1) > img")
|
||||
if node:
|
||||
preview_value = node.get("src", "")
|
||||
if isinstance(preview_value, list):
|
||||
preview = preview_value[0]
|
||||
else:
|
||||
preview = preview_value
|
||||
|
||||
data = soup.find_all("div", {"class": "main-inner"})[0].find_next("div")
|
||||
divs = list(data.children)
|
||||
format = divs[13].text.split(" · ")[1].strip().lower()
|
||||
size = divs[13].text.split(" · ")[2].strip().lower()
|
||||
|
||||
every_url = soup.find_all("a")
|
||||
slow_urls_no_waitlist = set()
|
||||
slow_urls_with_waitlist = set()
|
||||
external_urls_libgen = set()
|
||||
external_urls_z_lib = set()
|
||||
external_urls_welib = set()
|
||||
|
||||
for url in every_url:
|
||||
try:
|
||||
if url.text.strip().lower().startswith("slow partner server"):
|
||||
if (
|
||||
url.next is not None
|
||||
and url.next.next is not None
|
||||
and "waitlist" in url.next.next.strip().lower()
|
||||
):
|
||||
internal_text = url.next.next.strip().lower()
|
||||
if "no waitlist" in internal_text:
|
||||
slow_urls_no_waitlist.add(url["href"])
|
||||
else:
|
||||
slow_urls_with_waitlist.add(url["href"])
|
||||
elif (
|
||||
url.next is not None
|
||||
and url.next.next is not None
|
||||
and "click “GET” at the top" in url.next.next.text.strip()
|
||||
):
|
||||
libgen_url = url["href"]
|
||||
# TODO : Temporary fix ? Maybe get URLs from https://open-slum.org/ ?
|
||||
libgen_url = libgen_url = re.sub(r'libgen\.(\w+)', 'libgen.bz', url["href"])
|
||||
external_urls_libgen.add(libgen_url)
|
||||
elif url.text.strip().lower().startswith("z-lib"):
|
||||
if ".onion/" not in url["href"]:
|
||||
external_urls_z_lib.add(url["href"])
|
||||
except:
|
||||
pass
|
||||
|
||||
|
||||
urls = []
|
||||
urls += list(slow_urls_no_waitlist) if USE_CF_BYPASS else []
|
||||
urls += list(external_urls_libgen)
|
||||
urls += list( _get_download_urls_from_welib(book_id)) if USE_CF_BYPASS else []
|
||||
urls += list(slow_urls_with_waitlist) if USE_CF_BYPASS else []
|
||||
urls += list(external_urls_z_lib)
|
||||
|
||||
for i in range(len(urls)):
|
||||
urls[i] = downloader.get_absolute_url(AA_BASE_URL, urls[i])
|
||||
|
||||
# Remove empty urls
|
||||
urls = [url for url in urls if url != ""]
|
||||
|
||||
# Extract basic information
|
||||
book_info = BookInfo(
|
||||
id=book_id,
|
||||
preview=preview,
|
||||
title=divs[7].next.strip(),
|
||||
publisher=divs[11].text.strip(),
|
||||
author=divs[9].text.strip(),
|
||||
format=format,
|
||||
size=size,
|
||||
download_urls=urls,
|
||||
)
|
||||
|
||||
# Extract additional metadata
|
||||
info = _extract_book_metadata(divs[-6])
|
||||
book_info.info = info
|
||||
|
||||
# Set language and year from metadata if available
|
||||
if info.get("Language"):
|
||||
book_info.language = info["Language"][0]
|
||||
if info.get("Year"):
|
||||
book_info.year = info["Year"][0]
|
||||
|
||||
return book_info
|
||||
|
||||
def _get_download_urls_from_welib(book_id: str) -> List[str]:
|
||||
"""Get download urls from welib.org."""
|
||||
url = f"https://welib.org/md5/{book_id}"
|
||||
html = downloader.html_get_page(url, use_bypasser=True)
|
||||
if not html:
|
||||
return []
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
download_links = soup.find_all("a", href=True)
|
||||
download_links = [link["href"] for link in download_links]
|
||||
download_links = [link for link in download_links if "/slow_download/" in link]
|
||||
download_links = [downloader.get_absolute_url(url, link) for link in download_links]
|
||||
return set(download_links)
|
||||
|
||||
def _extract_book_metadata(
|
||||
metadata_divs
|
||||
) -> Dict[str, List[str]]:
|
||||
"""Extract metadata from book info divs."""
|
||||
info: Dict[str, List[str]] = {}
|
||||
|
||||
# Process the first set of metadata
|
||||
sub_datas = metadata_divs.find_all("div")[0]
|
||||
sub_datas = list(sub_datas.children)
|
||||
for sub_data in sub_datas:
|
||||
if sub_data.text.strip() == "":
|
||||
continue
|
||||
sub_data = list(sub_data.children)
|
||||
key = sub_data[0].text.strip()
|
||||
value = sub_data[1].text.strip()
|
||||
if key not in info:
|
||||
info[key] = set()
|
||||
info[key].add(value)
|
||||
|
||||
# make set into list
|
||||
for key, value in info.items():
|
||||
info[key] = list(value)
|
||||
|
||||
# Filter relevant metadata
|
||||
relevant_prefixes = [
|
||||
"ISBN-",
|
||||
"ALTERNATIVE",
|
||||
"ASIN",
|
||||
"Goodreads",
|
||||
"Language",
|
||||
"Year",
|
||||
]
|
||||
return {
|
||||
k.strip(): v
|
||||
for k, v in info.items()
|
||||
if any(k.lower().startswith(prefix.lower()) for prefix in relevant_prefixes)
|
||||
and "filename" not in k.lower()
|
||||
}
|
||||
|
||||
|
||||
def download_book(book_info: BookInfo, book_path: Path) -> bool:
|
||||
"""Download a book from available sources.
|
||||
|
||||
Args:
|
||||
book_id: Book identifier (MD5 hash)
|
||||
title: Book title for logging
|
||||
|
||||
Returns:
|
||||
Optional[BytesIO]: Book content buffer if successful
|
||||
"""
|
||||
|
||||
if len(book_info.download_urls) == 0:
|
||||
book_info = get_book_info(book_info.id)
|
||||
download_links = book_info.download_urls
|
||||
|
||||
# If AA_DONATOR_KEY is set, use the fast download URL. Else try other sources.
|
||||
if AA_DONATOR_KEY != "":
|
||||
download_links.insert(
|
||||
0,
|
||||
f"{AA_BASE_URL}/dyn/api/fast_download.json?md5={book_info.id}&key={AA_DONATOR_KEY}",
|
||||
)
|
||||
|
||||
for link in download_links:
|
||||
try:
|
||||
download_url = _get_download_url(link, book_info.title)
|
||||
if download_url != "":
|
||||
logger.info(f"Downloading `{book_info.title}` from `{download_url}`")
|
||||
data = downloader.download_url(download_url, book_info.size or "")
|
||||
if not data:
|
||||
raise Exception("No data received")
|
||||
|
||||
logger.info(f"Download finished. Writing to {book_path}")
|
||||
with open(book_path, "wb") as f:
|
||||
f.write(data.getbuffer())
|
||||
logger.info(f"Writing `{book_info.title}` successfully")
|
||||
return True
|
||||
|
||||
except Exception as e:
|
||||
logger.error_trace(f"Failed to download from {link}: {e}")
|
||||
continue
|
||||
|
||||
return False
|
||||
|
||||
|
||||
def _get_download_url(link: str, title: str) -> str:
|
||||
"""Extract actual download URL from various source pages."""
|
||||
|
||||
url = ""
|
||||
|
||||
if link.startswith(f"{AA_BASE_URL}/dyn/api/fast_download.json"):
|
||||
page = downloader.html_get_page(link)
|
||||
url = json.loads(page).get("download_url")
|
||||
else:
|
||||
html = downloader.html_get_page(link)
|
||||
|
||||
if html == "":
|
||||
return ""
|
||||
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
|
||||
if link.startswith("https://z-lib."):
|
||||
download_link = soup.find_all("a", href=True, class_="addDownloadedBook")
|
||||
if download_link:
|
||||
url = download_link[0]["href"]
|
||||
elif "/slow_download/" in link:
|
||||
download_links = soup.find_all("a", href=True, string="📚 Download now")
|
||||
if not download_links:
|
||||
countdown = soup.find_all("span", class_="js-partner-countdown")
|
||||
if countdown:
|
||||
sleep_time = int(countdown[0].text)
|
||||
logger.info(f"Waiting {sleep_time}s for {title}")
|
||||
time.sleep(sleep_time)
|
||||
url = _get_download_url(link, title)
|
||||
else:
|
||||
url = download_links[0]["href"]
|
||||
else:
|
||||
url = soup.find_all("a", string="GET")[0]["href"]
|
||||
|
||||
return downloader.get_absolute_url(link, url)
|
||||
@@ -1,311 +0,0 @@
|
||||
import time
|
||||
import os
|
||||
import socket
|
||||
from urllib.parse import urlparse
|
||||
import threading
|
||||
import env
|
||||
from env import LOG_DIR, DEBUG
|
||||
import signal
|
||||
from datetime import datetime
|
||||
import subprocess
|
||||
|
||||
# --- SeleniumBase Import ---
|
||||
from seleniumbase import Driver
|
||||
from selenium.webdriver.common.by import By
|
||||
from selenium.webdriver.support.ui import WebDriverWait
|
||||
from selenium.webdriver.support import expected_conditions as EC
|
||||
from selenium.common.exceptions import TimeoutException
|
||||
|
||||
import network
|
||||
from logger import setup_logger
|
||||
from env import MAX_RETRY, DEFAULT_SLEEP
|
||||
from config import PROXIES, CUSTOM_DNS, DOH_SERVER, VIRTUAL_SCREEN_SIZE, RECORDING_DIR
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
network.init()
|
||||
|
||||
DRIVER = None
|
||||
DISPLAY = {
|
||||
"xvfb": None,
|
||||
"ffmpeg": None,
|
||||
}
|
||||
LAST_USED = None
|
||||
LOCKED = threading.Lock()
|
||||
TENTATIVE_CURRENT_URL = None
|
||||
|
||||
def _reset_pyautogui_display_state():
|
||||
try:
|
||||
import pyautogui
|
||||
import Xlib.display
|
||||
pyautogui._pyautogui_x11._display = (
|
||||
Xlib.display.Display(os.environ['DISPLAY'])
|
||||
)
|
||||
except Exception as e:
|
||||
logger.warning(f"Error resetting pyautogui display state: {e}")
|
||||
|
||||
def _is_bypassed(sb) -> bool:
|
||||
try:
|
||||
title = sb.get_title().lower()
|
||||
body = sb.get_text("body").lower()
|
||||
|
||||
# Check both title and body for verification messages
|
||||
verification_texts = [
|
||||
"just a moment",
|
||||
"verify you are human",
|
||||
"verifying you are human",
|
||||
"needs to review the security of your connection before proceeding",
|
||||
"checking your browser",
|
||||
"checking connection",
|
||||
"attention required",
|
||||
"access denied",
|
||||
"needs to review the security of your connection",
|
||||
"checking the site connection security",
|
||||
"enable javascript and cookies to continue",
|
||||
"ray id",
|
||||
]
|
||||
for text in verification_texts:
|
||||
if text in title.lower() or text in body.lower():
|
||||
return False
|
||||
|
||||
return True
|
||||
except Exception as e:
|
||||
logger.debug(f"Error checking page title: {e}")
|
||||
return False
|
||||
|
||||
def _bypass_method_1(sb) -> bool:
|
||||
try:
|
||||
sb.uc_gui_click_captcha()
|
||||
except Exception as e:
|
||||
logger.debug_trace(f"Error clicking captcha: {e}")
|
||||
time.sleep(5)
|
||||
sb.wait_for_element_visible('body')
|
||||
try:
|
||||
sb.uc_gui_click_captcha()
|
||||
except Exception as e:
|
||||
logger.debug_trace(f"Error clicking captcha again: {e}")
|
||||
time.sleep(DEFAULT_SLEEP)
|
||||
sb.uc_gui_click_captcha()
|
||||
return _is_bypassed(sb)
|
||||
|
||||
def _bypass(sb, max_retries: int = MAX_RETRY) -> None:
|
||||
try_count = 0
|
||||
|
||||
while not _is_bypassed(sb):
|
||||
if try_count >= max_retries:
|
||||
logger.warning("Exceeded maximum retries. Bypass failed.")
|
||||
break
|
||||
logger.info(f"Bypass attempt {try_count + 1} / {max_retries}")
|
||||
|
||||
try_count += 1
|
||||
|
||||
wait_time = DEFAULT_SLEEP * (try_count - 1)
|
||||
logger.info(f"Waiting {wait_time}s before trying...")
|
||||
time.sleep(wait_time)
|
||||
|
||||
if _bypass_method_1(sb):
|
||||
return
|
||||
|
||||
logger.info("Bypass failed.")
|
||||
|
||||
def _get_chromium_args():
|
||||
|
||||
arguments = [
|
||||
# Ignore certificate and SSL errors (similar to curl's --insecure)
|
||||
"--ignore-certificate-errors",
|
||||
"--ignore-ssl-errors",
|
||||
"--allow-running-insecure-content",
|
||||
"--ignore-certificate-errors-spki-list",
|
||||
"--ignore-certificate-errors-skip-list"
|
||||
]
|
||||
|
||||
# Conditionally add verbose logging arguments
|
||||
if DEBUG:
|
||||
arguments.extend([
|
||||
"--enable-logging", # Enable Chrome browser logging
|
||||
"--v=1", # Set verbosity level for Chrome logs
|
||||
"--log-file=" + str(LOG_DIR / "chrome_browser.log")
|
||||
])
|
||||
|
||||
# Add proxy settings if configured
|
||||
if PROXIES:
|
||||
proxy_url = PROXIES.get('https') or PROXIES.get('http')
|
||||
if proxy_url:
|
||||
arguments.append(f'--proxy-server={proxy_url}')
|
||||
|
||||
# --- Add Custom DNS settings ---
|
||||
try:
|
||||
if len(CUSTOM_DNS) > 0:
|
||||
if DOH_SERVER:
|
||||
logger.info(f"Configuring DNS over HTTPS (DoH) with server: {DOH_SERVER}")
|
||||
|
||||
# TODO: This is probably broken and a halucination,
|
||||
# but it should still default to google DOH so its fine...
|
||||
arguments.extend(['--enable-features=DnsOverHttps', '--dns-over-https-mode=secure', f'--dns-over-https-servers="{DOH_SERVER}"'])
|
||||
doh_hostname = urlparse(DOH_SERVER).hostname
|
||||
if doh_hostname:
|
||||
try:
|
||||
arguments.append(f'--host-resolver-rules=MAP {doh_hostname} {socket.gethostbyname(doh_hostname)}')
|
||||
except socket.gaierror:
|
||||
logger.warning(f"Could not resolve DoH hostname: {doh_hostname}")
|
||||
elif CUSTOM_DNS:
|
||||
resolver_rules = [f"MAP * {dns_server}" for dns_server in CUSTOM_DNS]
|
||||
if resolver_rules:
|
||||
arguments.append(f'--host-resolver-rules={",".join(resolver_rules)}')
|
||||
except Exception as e:
|
||||
logger.error_trace(f"Error configuring DNS settings: {e}")
|
||||
return arguments
|
||||
|
||||
CHROMIUM_ARGS = _get_chromium_args()
|
||||
|
||||
def _get(url, retry : int = MAX_RETRY):
|
||||
try:
|
||||
logger.info(f"SB_GET: {url}")
|
||||
sb = _get_driver()
|
||||
sb.uc_open_with_reconnect(url, DEFAULT_SLEEP)
|
||||
time.sleep(DEFAULT_SLEEP)
|
||||
_bypass(sb)
|
||||
if _is_bypassed(sb):
|
||||
logger.info("Bypass successful.")
|
||||
return sb.page_source
|
||||
except Exception as e:
|
||||
if retry == 0:
|
||||
logger.error_trace(f"Failed to initialize browser: {e}")
|
||||
_reset_driver()
|
||||
raise e
|
||||
logger.error_trace(f"Failed to bypass Cloudflare: {e}. Will retry...")
|
||||
return _get(url, retry - 1)
|
||||
|
||||
def get(url, retry : int = MAX_RETRY):
|
||||
global LOCKED, TENTATIVE_CURRENT_URL, LAST_USED
|
||||
with LOCKED:
|
||||
TENTATIVE_CURRENT_URL = url
|
||||
ret = _get(url, retry)
|
||||
LAST_USED = time.time()
|
||||
return ret
|
||||
|
||||
def _init_driver():
|
||||
global DRIVER
|
||||
if DRIVER:
|
||||
_reset_driver()
|
||||
driver = Driver(uc=True, headless=False, size=f"{VIRTUAL_SCREEN_SIZE[0]},{VIRTUAL_SCREEN_SIZE[1]}", chromium_arg=CHROMIUM_ARGS)
|
||||
DRIVER = driver
|
||||
time.sleep(DEFAULT_SLEEP)
|
||||
return driver
|
||||
|
||||
def _get_driver():
|
||||
global DRIVER, DISPLAY
|
||||
global LAST_USED
|
||||
logger.info("Getting driver...")
|
||||
LAST_USED = time.time()
|
||||
if env.DOCKERMODE and env.USE_CF_BYPASS and not DISPLAY["xvfb"]:
|
||||
from pyvirtualdisplay import Display
|
||||
display = Display(visible=False, size=VIRTUAL_SCREEN_SIZE)
|
||||
display.start()
|
||||
logger.info("Display started")
|
||||
DISPLAY["xvfb"] = display
|
||||
time.sleep(DEFAULT_SLEEP)
|
||||
_reset_pyautogui_display_state()
|
||||
|
||||
if env.DEBUG:
|
||||
timestamp = datetime.now().strftime("%y%m%d-%H%M%S")
|
||||
output_file = RECORDING_DIR / f"screen_recording_{timestamp}.mp4"
|
||||
|
||||
ffmpeg_cmd = [
|
||||
"ffmpeg",
|
||||
"-y",
|
||||
"-f", "x11grab",
|
||||
"-video_size", f"{VIRTUAL_SCREEN_SIZE[0]}x{VIRTUAL_SCREEN_SIZE[1]}",
|
||||
"-i", f":{display.display}",
|
||||
"-c:v", "libx264",
|
||||
"-preset", "ultrafast", # or "veryfast" (trade speed for slightly better compression)
|
||||
"-maxrate", "700k", # Slightly higher bitrate for text clarity
|
||||
"-bufsize", "1400k", # Buffer size (2x maxrate)
|
||||
"-crf", "36", # Adjust as needed: higher = smaller, lower = better quality (23 is visually lossless)
|
||||
"-pix_fmt", "yuv420p", # Crucial for compatibility with most players
|
||||
"-tune", "animation", # Optimize encoding for screen content
|
||||
"-x264-params", "bframes=0:deblock=-1,-1", # Optimize for text, disable b-frames and deblocking
|
||||
"-r", "15", # Reduce frame rate (if content allows)
|
||||
"-an", # Disable audio recording (if not needed)
|
||||
output_file.as_posix(),
|
||||
"-nostats", "-loglevel", "0"
|
||||
]
|
||||
logger.info("Starting FFmpeg recording to %s", output_file)
|
||||
logger.debug_trace(f"FFmpeg command: {' '.join(ffmpeg_cmd)}")
|
||||
DISPLAY["ffmpeg"] = subprocess.Popen(ffmpeg_cmd)
|
||||
if not DRIVER:
|
||||
return _init_driver()
|
||||
logger.log_resource_usage()
|
||||
return DRIVER
|
||||
|
||||
def _reset_driver():
|
||||
logger.log_resource_usage()
|
||||
logger.info("Resetting driver...")
|
||||
global DRIVER, DISPLAY
|
||||
if DRIVER:
|
||||
try:
|
||||
DRIVER.quit()
|
||||
DRIVER = None
|
||||
except Exception as e:
|
||||
logger.warning(f"Error quitting driver: {e}")
|
||||
time.sleep(0.5)
|
||||
if DISPLAY["xvfb"]:
|
||||
try:
|
||||
DISPLAY["xvfb"].stop()
|
||||
DISPLAY["xvfb"] = None
|
||||
except Exception as e:
|
||||
logger.warning(f"Error stopping display: {e}")
|
||||
time.sleep(0.5)
|
||||
try:
|
||||
os.system("pkill -f Xvfb")
|
||||
except Exception as e:
|
||||
logger.debug(f"Error killing Xvfb: {e}")
|
||||
time.sleep(0.5)
|
||||
if DISPLAY["ffmpeg"]:
|
||||
try:
|
||||
DISPLAY["ffmpeg"].send_signal(signal.SIGINT)
|
||||
DISPLAY["ffmpeg"] = None
|
||||
except Exception as e:
|
||||
logger.debug(f"Error stopping ffmpeg: {e}")
|
||||
time.sleep(0.5)
|
||||
try:
|
||||
os.system("pkill -f ffmpeg")
|
||||
except Exception as e:
|
||||
logger.debug(f"Error killing ffmpeg: {e}")
|
||||
time.sleep(0.5)
|
||||
try:
|
||||
os.system("pkill -f chrom")
|
||||
except Exception as e:
|
||||
logger.debug(f"Error killing chrom: {e}")
|
||||
time.sleep(0.5)
|
||||
logger.info("Driver reset.")
|
||||
logger.log_resource_usage()
|
||||
|
||||
def _cleanup_driver():
|
||||
global LOCKED
|
||||
global LAST_USED
|
||||
with LOCKED:
|
||||
if LAST_USED:
|
||||
if time.time() - LAST_USED >= env.BYPASS_RELEASE_INACTIVE_MIN * 60:
|
||||
_reset_driver()
|
||||
LAST_USED = None
|
||||
logger.info("Driver reset due to inactivity.")
|
||||
|
||||
def _cleanup_loop():
|
||||
while True:
|
||||
_cleanup_driver()
|
||||
time.sleep(max(env.BYPASS_RELEASE_INACTIVE_MIN / 2, 1))
|
||||
|
||||
def _init_cleanup_thread():
|
||||
cleanup_thread = threading.Thread(target=_cleanup_loop)
|
||||
cleanup_thread.daemon = True
|
||||
cleanup_thread.start()
|
||||
|
||||
def wait_for_result(func, timeout : int = 10, condition : any = True):
|
||||
start_time = time.time()
|
||||
while time.time() - start_time < timeout:
|
||||
result = func()
|
||||
if condition(result):
|
||||
return result
|
||||
time.sleep(0.5)
|
||||
return None
|
||||
_init_cleanup_thread()
|
||||
@@ -0,0 +1,20 @@
|
||||
# Uses external Cloudflare bypasser (FlareSolverr/ByParr) instead of built-in Selenium
|
||||
services:
|
||||
shelfmark-lite:
|
||||
image: ghcr.io/calibrain/shelfmark-lite:dev
|
||||
environment:
|
||||
# TZ: America/New_York
|
||||
EXT_BYPASSER_URL: http://flaresolverr:8191
|
||||
# PUID: 1000
|
||||
# PGID: 1000
|
||||
ports:
|
||||
- 8084:8084
|
||||
restart: unless-stopped
|
||||
volumes:
|
||||
- /path/to/books:/books # Book destination directory
|
||||
- /path/to/config:/config # App configuration
|
||||
# Download client mount - path must match your torrent/usenet client's volume exactly
|
||||
# - /path/to/downloads:/path/to/downloads
|
||||
|
||||
flaresolverr:
|
||||
image: ghcr.io/flaresolverr/flaresolverr:latest
|
||||
@@ -0,0 +1,21 @@
|
||||
# Routes all traffic through Tor - requires NET_ADMIN capability
|
||||
services:
|
||||
shelfmark-tor:
|
||||
image: ghcr.io/calibrain/shelfmark:dev
|
||||
environment:
|
||||
FLASK_PORT: 8084
|
||||
# TZ: America/New_York
|
||||
USING_TOR: true
|
||||
# PUID: 1000
|
||||
# PGID: 1000
|
||||
cap_add:
|
||||
- NET_ADMIN
|
||||
- NET_RAW
|
||||
ports:
|
||||
- 8084:8084
|
||||
restart: unless-stopped
|
||||
volumes:
|
||||
- /path/to/books:/books # Book destination directory
|
||||
- /path/to/config:/config # App configuration
|
||||
# Download client mount - path must match your torrent/usenet client's volume exactly
|
||||
# - /path/to/downloads:/path/to/downloads
|
||||
@@ -0,0 +1,16 @@
|
||||
services:
|
||||
shelfmark:
|
||||
image: ghcr.io/calibrain/shelfmark:dev
|
||||
container_name: shelfmark
|
||||
environment:
|
||||
# TZ: America/New_York
|
||||
# PUID: 1000
|
||||
# PGID: 1000
|
||||
ports:
|
||||
- 8084:8084
|
||||
restart: unless-stopped
|
||||
volumes:
|
||||
- /path/to/books:/books # Book destination directory
|
||||
- /path/to/config:/config # App configuration
|
||||
# Download client mount - path must match your torrent/usenet client's volume exactly
|
||||
# - /path/to/downloads:/path/to/downloads
|
||||
@@ -0,0 +1,16 @@
|
||||
services:
|
||||
shelfmark-lite:
|
||||
image: ghcr.io/calibrain/shelfmark-lite:latest
|
||||
environment:
|
||||
# TZ: America/New_York
|
||||
# EXT_BYPASSER_URL: http://flaresolverr:8191 #If using Flaresolverr
|
||||
# PUID: 1000
|
||||
# PGID: 1000
|
||||
ports:
|
||||
- 8084:8084
|
||||
restart: unless-stopped
|
||||
volumes:
|
||||
- /path/to/books:/books # Book destination directory
|
||||
- /path/to/config:/config # App configuration
|
||||
# Download client mount - path must match your torrent/usenet client's volume exactly
|
||||
# - /path/to/downloads:/path/to/downloads
|
||||
@@ -0,0 +1,21 @@
|
||||
# Routes all traffic through Tor - requires NET_ADMIN capability
|
||||
services:
|
||||
shelfmark-tor:
|
||||
image: ghcr.io/calibrain/shelfmark:latest
|
||||
environment:
|
||||
FLASK_PORT: 8084
|
||||
# TZ: America/New_York
|
||||
USING_TOR: true
|
||||
# PUID: 1000
|
||||
# PGID: 1000
|
||||
cap_add:
|
||||
- NET_ADMIN
|
||||
- NET_RAW
|
||||
ports:
|
||||
- 8084:8084
|
||||
restart: unless-stopped
|
||||
volumes:
|
||||
- /path/to/books:/books # Book destination directory
|
||||
- /path/to/config:/config # App configuration
|
||||
# Download client mount - path must match your torrent/usenet client's volume exactly
|
||||
# - /path/to/downloads:/path/to/downloads
|
||||
@@ -0,0 +1,16 @@
|
||||
services:
|
||||
shelfmark:
|
||||
image: ghcr.io/calibrain/shelfmark:latest
|
||||
container_name: shelfmark
|
||||
environment:
|
||||
# TZ: America/New_York
|
||||
# PUID: 1000
|
||||
# PGID: 1000
|
||||
ports:
|
||||
- 8084:8084
|
||||
restart: unless-stopped
|
||||
volumes:
|
||||
- /path/to/books:/books # Book destination directory
|
||||
- /path/to/config:/config # App configuration
|
||||
# Download client mount - path must match your torrent/usenet client's volume exactly
|
||||
# - /path/to/downloads:/path/to/downloads
|
||||
@@ -1,99 +0,0 @@
|
||||
"""Configuration settings for the book downloader application."""
|
||||
|
||||
import os
|
||||
from pathlib import Path
|
||||
import json
|
||||
import env
|
||||
from logger import setup_logger
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
for key, value in env.__dict__.items():
|
||||
if not key.startswith('_'):
|
||||
if key == "AA_DONATOR_KEY" and value.strip() != "":
|
||||
value = "REDACTED"
|
||||
logger.info(f"{key}: {value}")
|
||||
|
||||
with open("data/book-languages.json") as file:
|
||||
_SUPPORTED_BOOK_LANGUAGE = json.load(file)
|
||||
|
||||
# Directory settings
|
||||
BASE_DIR = Path(__file__).resolve().parent
|
||||
logger.info(f"BASE_DIR: {BASE_DIR}")
|
||||
if env.ENABLE_LOGGING:
|
||||
env.LOG_DIR.mkdir(exist_ok=True)
|
||||
|
||||
# Create necessary directories
|
||||
env.TMP_DIR.mkdir(exist_ok=True)
|
||||
env.INGEST_DIR.mkdir(exist_ok=True)
|
||||
|
||||
CROSS_FILE_SYSTEM = os.stat(env.TMP_DIR).st_dev != os.stat(env.INGEST_DIR).st_dev
|
||||
logger.info(f"STAT TMP_DIR: {os.stat(env.TMP_DIR)}")
|
||||
logger.info(f"STAT INGEST_DIR: {os.stat(env.INGEST_DIR)}")
|
||||
logger.info(f"CROSS_FILE_SYSTEM: {CROSS_FILE_SYSTEM}")
|
||||
|
||||
# Network settings
|
||||
_custom_dns = env._CUSTOM_DNS.lower().strip()
|
||||
_doh_server = ""
|
||||
if _custom_dns == "google":
|
||||
CUSTOM_DNS = ["8.8.8.8", "8.8.4.4", "2001:4860:4860::8888", "2001:4860:4860::8844"]
|
||||
_doh_server = "https://dns.google/dns-query"
|
||||
elif _custom_dns == "quad9":
|
||||
CUSTOM_DNS = ["9.9.9.9", "149.112.112.112", "2620:fe::fe", "26620:fe::9"]
|
||||
_doh_server = "https://dns.quad9.net/dns-query"
|
||||
elif _custom_dns == "cloudflare":
|
||||
CUSTOM_DNS = ["1.1.1.1", "1.0.0.1", "2606:4700:4700::1111", "2606:4700:4700::1001"]
|
||||
_doh_server = "https://cloudflare-dns.com/dns-query"
|
||||
elif _custom_dns == "opendns":
|
||||
CUSTOM_DNS = ["208.67.222.222", "208.67.220.220", "2620:119:35::35", "2620:119:53::53"]
|
||||
_doh_server = "https://doh.opendns.com/dns-query"
|
||||
else:
|
||||
_custom_dns_ip = _custom_dns.split(",")
|
||||
CUSTOM_DNS = [dns.strip() for dns in _custom_dns_ip if dns.replace(":", "").replace(".", "").strip().isdigit()]
|
||||
logger.info(f"CUSTOM_DNS: {CUSTOM_DNS}")
|
||||
DOH_SERVER = _doh_server
|
||||
if env.USE_DOH:
|
||||
DOH_SERVER = _doh_server
|
||||
else:
|
||||
DOH_SERVER = ""
|
||||
logger.info(f"DOH_SERVER: {DOH_SERVER}")
|
||||
|
||||
# Proxy settings
|
||||
PROXIES = {}
|
||||
if env.HTTP_PROXY:
|
||||
PROXIES["http"] = env.HTTP_PROXY
|
||||
if env.HTTPS_PROXY:
|
||||
PROXIES["https"] = env.HTTPS_PROXY
|
||||
logger.info(f"PROXIES: {PROXIES}")
|
||||
|
||||
# Anna's Archive settings
|
||||
AA_BASE_URL = env._AA_BASE_URL
|
||||
AA_AVAILABLE_URLS = ["https://annas-archive.org", "https://annas-archive.se", "https://annas-archive.li"]
|
||||
AA_AVAILABLE_URLS.extend(env._AA_ADDITIONAL_URLS.split(","))
|
||||
AA_AVAILABLE_URLS = [url.strip() for url in AA_AVAILABLE_URLS if url.strip()]
|
||||
|
||||
# File format settings
|
||||
SUPPORTED_FORMATS = env._SUPPORTED_FORMATS.split(",")
|
||||
logger.info(f"SUPPORTED_FORMATS: {SUPPORTED_FORMATS}")
|
||||
|
||||
# Complex language processing logic kept in config.py
|
||||
BOOK_LANGUAGE = env._BOOK_LANGUAGE.split(',')
|
||||
BOOK_LANGUAGE = [l for l in BOOK_LANGUAGE if l in [lang['code'] for lang in _SUPPORTED_BOOK_LANGUAGE]]
|
||||
if len(BOOK_LANGUAGE) == 0:
|
||||
BOOK_LANGUAGE = ['en']
|
||||
|
||||
# Custom script settings with validation logic
|
||||
CUSTOM_SCRIPT = env._CUSTOM_SCRIPT
|
||||
if CUSTOM_SCRIPT:
|
||||
if not os.path.exists(CUSTOM_SCRIPT):
|
||||
logger.warn(f"CUSTOM_SCRIPT {CUSTOM_SCRIPT} does not exist")
|
||||
CUSTOM_SCRIPT = ""
|
||||
elif not os.access(CUSTOM_SCRIPT, os.X_OK):
|
||||
logger.warn(f"CUSTOM_SCRIPT {CUSTOM_SCRIPT} is not executable")
|
||||
CUSTOM_SCRIPT = ""
|
||||
|
||||
# Debugging settings
|
||||
VIRTUAL_SCREEN_SIZE = (1024, 768)
|
||||
RECORDING_DIR = env.LOG_DIR / "recording"
|
||||
if env.DEBUG:
|
||||
RECORDING_DIR.mkdir(parents=True, exist_ok=True)
|
||||
@@ -0,0 +1,25 @@
|
||||
# Local development - External bypasser variant (lite)
|
||||
services:
|
||||
shelfmark-lite-dev:
|
||||
extends:
|
||||
file: ./compose/edge/docker-compose.extbp.yml
|
||||
service: shelfmark-lite
|
||||
build:
|
||||
context: .
|
||||
dockerfile: Dockerfile
|
||||
target: shelfmark-lite
|
||||
environment:
|
||||
DEBUG: true
|
||||
EXT_BYPASSER_URL: http://flaresolverr:8191
|
||||
EXT_BYPASSER_PATH: /v1
|
||||
EXT_BYPASSER_TIMEOUT: 60000
|
||||
volumes:
|
||||
- ./.local/config:/config
|
||||
- ./.local/books:/books
|
||||
- ./.local/log:/var/log/shelfmark
|
||||
- ./.local/tmp:/tmp/shelfmark
|
||||
# Download client mount - path must match your torrent/usenet client's volume exactly
|
||||
# - /path/to/downloads:/path/to/downloads
|
||||
|
||||
flaresolverr:
|
||||
image: ghcr.io/flaresolverr/flaresolverr:latest
|
||||
@@ -0,0 +1,20 @@
|
||||
# Local development - Tor variant
|
||||
services:
|
||||
shelfmark-tor-dev:
|
||||
extends:
|
||||
file: ./compose/edge/docker-compose.tor.yml
|
||||
service: shelfmark-tor
|
||||
build:
|
||||
context: .
|
||||
dockerfile: Dockerfile
|
||||
target: shelfmark
|
||||
environment:
|
||||
DEBUG: true
|
||||
USING_TOR: true
|
||||
volumes:
|
||||
- ./.local/config:/config
|
||||
- ./.local/books:/books
|
||||
- ./.local/log:/var/log/shelfmark
|
||||
- ./.local/tmp:/tmp/shelfmark
|
||||
# Download client mount - path must match your torrent/usenet client's volume exactly
|
||||
# - /path/to/downloads:/path/to/downloads
|
||||
@@ -1,17 +1,22 @@
|
||||
# Local development - builds from source with debug enabled
|
||||
services:
|
||||
calibre-web-automated-book-downloader-dev:
|
||||
shelfmark-dev:
|
||||
extends:
|
||||
file: ./docker-compose.yml
|
||||
service: calibre-web-automated-book-downloader
|
||||
file: ./compose/edge/docker-compose.yml
|
||||
service: shelfmark
|
||||
build:
|
||||
context: .
|
||||
dockerfile: Dockerfile
|
||||
target: cwa-bd
|
||||
target: shelfmark
|
||||
cap_add:
|
||||
- SYS_PTRACE
|
||||
environment:
|
||||
DEBUG: true
|
||||
APP_ENV: dev
|
||||
USE_DOH: true
|
||||
CUSTOM_DNS: cloudflare
|
||||
volumes:
|
||||
- /tmp/cwa-book-downloader:/tmp/cwa-book-downloader
|
||||
- /tmp/cwa-book-downloader-log:/var/log/cwa-book-downloader
|
||||
- ./.local/config:/config
|
||||
- ./.local/books:/books
|
||||
- ./.local/log:/var/log/shelfmark
|
||||
- ./.local/tmp:/tmp/shelfmark
|
||||
- ./shelfmark:/app/shelfmark:ro
|
||||
# Download client mount - path must match your torrent/usenet client's volume exactly
|
||||
# - /path/to/downloads:/path/to/downloads
|
||||
|
||||
@@ -1,15 +0,0 @@
|
||||
services:
|
||||
calibre-web-automated-book-downloader-tor-dev:
|
||||
extends:
|
||||
file: ./docker-compose.tor.yml
|
||||
service: calibre-web-automated-book-downloader-tor
|
||||
build:
|
||||
context: .
|
||||
dockerfile: Dockerfile
|
||||
target: cwa-bd-tor
|
||||
environment:
|
||||
DEBUG: true
|
||||
APP_ENV: dev
|
||||
volumes:
|
||||
- /tmp/cwa-book-downloader:/tmp/cwa-book-downloader
|
||||
- /tmp/cwa-book-downloader-log:/var/log/cwa-book-downloader
|
||||
@@ -1,21 +0,0 @@
|
||||
services:
|
||||
calibre-web-automated-book-downloader-tor:
|
||||
image: ghcr.io/calibrain/calibre-web-automated-book-downloader-tor:latest
|
||||
environment:
|
||||
FLASK_PORT: 8084
|
||||
LOG_LEVEL: info
|
||||
BOOK_LANGUAGE: en
|
||||
USE_BOOK_TITLE: true
|
||||
TZ: America/New_York
|
||||
USING_TOR: true
|
||||
APP_ENV: prod
|
||||
cap_add:
|
||||
- NET_ADMIN
|
||||
- NET_RAW
|
||||
ports:
|
||||
- 8084:8084
|
||||
restart: unless-stopped
|
||||
volumes:
|
||||
# This is where the books will be downloaded to, usually it would be
|
||||
# the same as whatever you gave in "calibre-web-automated"
|
||||
- /tmp/data/calibre-web/ingest:/cwa-book-ingest
|
||||
@@ -1,24 +0,0 @@
|
||||
services:
|
||||
calibre-web-automated-book-downloader:
|
||||
image: ghcr.io/calibrain/calibre-web-automated-book-downloader:latest
|
||||
container_name: calibre-web-automated-book-downloader
|
||||
environment:
|
||||
FLASK_PORT: 8084
|
||||
LOG_LEVEL: info
|
||||
BOOK_LANGUAGE: en
|
||||
USE_BOOK_TITLE: true
|
||||
TZ: America/New_York
|
||||
APP_ENV: prod
|
||||
UID: 1000
|
||||
GID: 100
|
||||
CWA_DB_PATH: /auth/app.db
|
||||
ports:
|
||||
- 8084:8084
|
||||
restart: unless-stopped
|
||||
volumes:
|
||||
# This is where the books will be downloaded to, usually it would be
|
||||
# the same as whatever you gave in "calibre-web-automated"
|
||||
- /tmp/data/calibre-web/ingest:/cwa-book-ingest
|
||||
# This is the location of CWA's app.db, which contains authentication
|
||||
# details
|
||||
#- /cwa/config/path/app.db:/auth/app.db:ro
|
||||
@@ -0,0 +1,628 @@
|
||||
# Plugin Settings Integration Guide
|
||||
|
||||
This guide explains how to add configuration settings to plugins (Metadata Providers and Release Sources) so they appear in the Settings UI.
|
||||
|
||||
## Overview
|
||||
|
||||
The settings system uses a decorator-based registration pattern. Plugins register their settings when their module is imported, and the frontend dynamically renders the appropriate UI based on the schema provided by the backend.
|
||||
|
||||
**Key features:**
|
||||
- Settings are defined in Python and automatically rendered in the React frontend
|
||||
- Values persist across container restarts via JSON config files
|
||||
- Changes take effect immediately without restart (unless marked otherwise)
|
||||
|
||||
## Quick Start
|
||||
|
||||
Add settings to your plugin in 3 steps:
|
||||
|
||||
```python
|
||||
from shelfmark.core.settings_registry import (
|
||||
register_settings,
|
||||
TextField,
|
||||
PasswordField,
|
||||
ActionButton,
|
||||
)
|
||||
|
||||
@register_settings(
|
||||
name="my_plugin", # Unique identifier
|
||||
display_name="My Plugin", # Shown in sidebar
|
||||
icon="wrench", # Icon name
|
||||
order=100, # Sort order (lower = higher in list)
|
||||
group="metadata_providers" # Optional: group in sidebar
|
||||
)
|
||||
def my_plugin_settings():
|
||||
return [
|
||||
PasswordField(
|
||||
key="MY_PLUGIN_API_KEY",
|
||||
label="API Key",
|
||||
description="Your API key from the provider",
|
||||
required=True,
|
||||
),
|
||||
ActionButton(
|
||||
key="test_connection",
|
||||
label="Test Connection",
|
||||
style="primary",
|
||||
callback=_test_connection,
|
||||
),
|
||||
]
|
||||
|
||||
def _test_connection():
|
||||
# Perform connection test
|
||||
return {"success": True, "message": "Connected successfully!"}
|
||||
```
|
||||
|
||||
## Available Field Types
|
||||
|
||||
### TextField
|
||||
|
||||
Single-line text input for strings.
|
||||
|
||||
```python
|
||||
TextField(
|
||||
key="MY_SETTING", # Config key
|
||||
label="Setting Name", # Display label
|
||||
description="Help text", # Optional description below field
|
||||
default="", # Default value
|
||||
placeholder="Enter value", # Placeholder text
|
||||
max_length=100, # Optional max characters
|
||||
required=False, # Is this field required?
|
||||
requires_restart=False, # Does changing this need a restart?
|
||||
show_when=None, # Conditional visibility (see below)
|
||||
disabled_when=None, # Conditional disable (see below)
|
||||
)
|
||||
```
|
||||
|
||||
### PasswordField
|
||||
|
||||
Masked input for sensitive values (API keys, passwords). Values are never echoed back to the frontend.
|
||||
|
||||
```python
|
||||
PasswordField(
|
||||
key="API_KEY",
|
||||
label="API Key",
|
||||
description="Your secret API key",
|
||||
placeholder="sk-...",
|
||||
required=True,
|
||||
)
|
||||
```
|
||||
|
||||
### NumberField
|
||||
|
||||
Numeric input with optional min/max constraints.
|
||||
|
||||
```python
|
||||
NumberField(
|
||||
key="TIMEOUT",
|
||||
label="Timeout (seconds)",
|
||||
description="Connection timeout in seconds",
|
||||
default=30,
|
||||
min_value=5,
|
||||
max_value=300,
|
||||
step=1, # Increment step
|
||||
required=False,
|
||||
)
|
||||
```
|
||||
|
||||
### CheckboxField
|
||||
|
||||
Toggle switch for boolean values.
|
||||
|
||||
```python
|
||||
CheckboxField(
|
||||
key="ENABLE_FEATURE",
|
||||
label="Enable Feature",
|
||||
description="Turn this feature on or off",
|
||||
default=False,
|
||||
)
|
||||
```
|
||||
|
||||
### SelectField
|
||||
|
||||
Dropdown for single-choice selection.
|
||||
|
||||
```python
|
||||
SelectField(
|
||||
key="LOG_LEVEL",
|
||||
label="Log Level",
|
||||
description="Logging verbosity",
|
||||
default="info",
|
||||
options=[
|
||||
{"value": "debug", "label": "Debug"},
|
||||
{"value": "info", "label": "Info"},
|
||||
{"value": "warning", "label": "Warning"},
|
||||
{"value": "error", "label": "Error"},
|
||||
],
|
||||
)
|
||||
```
|
||||
|
||||
### MultiSelectField
|
||||
|
||||
Multi-choice selection from a list of options.
|
||||
|
||||
```python
|
||||
MultiSelectField(
|
||||
key="SUPPORTED_FORMATS",
|
||||
label="Supported Formats",
|
||||
description="Select which formats to support",
|
||||
default=["epub", "mobi"],
|
||||
options=[
|
||||
{"value": "epub", "label": "EPUB"},
|
||||
{"value": "mobi", "label": "MOBI"},
|
||||
{"value": "pdf", "label": "PDF"},
|
||||
{"value": "azw3", "label": "AZW3"},
|
||||
],
|
||||
)
|
||||
```
|
||||
|
||||
### ActionButton
|
||||
|
||||
Button that executes a callback function. Does not store a value.
|
||||
|
||||
```python
|
||||
ActionButton(
|
||||
key="test_connection", # Unique key for the action
|
||||
label="Test Connection", # Button text
|
||||
description="Test the API connection",
|
||||
style="primary", # "default", "primary", or "danger"
|
||||
callback=my_callback_fn, # Function to execute
|
||||
)
|
||||
|
||||
def my_callback_fn():
|
||||
"""Callback must return dict with 'success' and 'message' keys."""
|
||||
try:
|
||||
# Perform action
|
||||
return {"success": True, "message": "Connection successful!"}
|
||||
except Exception as e:
|
||||
return {"success": False, "message": f"Failed: {str(e)}"}
|
||||
```
|
||||
|
||||
### HeadingField
|
||||
|
||||
Display-only section heading with optional link. Does not store a value.
|
||||
|
||||
```python
|
||||
HeadingField(
|
||||
key="section_heading", # Unique key
|
||||
title="Configuration", # Heading text
|
||||
description="Configure the plugin settings below",
|
||||
link_url="https://example.com/docs", # Optional link
|
||||
link_text="View Documentation", # Link text
|
||||
)
|
||||
```
|
||||
|
||||
## Common Field Properties
|
||||
|
||||
All field types support these common properties:
|
||||
|
||||
| Property | Type | Default | Description |
|
||||
|----------|------|---------|-------------|
|
||||
| `key` | `str` | Required | Unique identifier for this setting |
|
||||
| `label` | `str` | Required | Display label in the UI |
|
||||
| `description` | `str` | `""` | Help text shown below the field |
|
||||
| `default` | `Any` | `None` | Default value if not set |
|
||||
| `required` | `bool` | `False` | Whether the field must have a value |
|
||||
| `disabled` | `bool` | `False` | Disable the field (greyed out) |
|
||||
| `disabled_reason` | `str` | `""` | Explanation shown when disabled |
|
||||
| `requires_restart` | `bool` | `False` | Whether changes require container restart |
|
||||
| `show_when` | `dict` | `None` | Conditional visibility (see below) |
|
||||
| `disabled_when` | `dict` | `None` | Conditional disable (see below) |
|
||||
|
||||
## Conditional Visibility
|
||||
|
||||
Fields can be shown/hidden based on other field values using `show_when`:
|
||||
|
||||
```python
|
||||
# Only show DNS servers field when custom DNS is selected
|
||||
TextField(
|
||||
key="CUSTOM_DNS_SERVERS",
|
||||
label="DNS Servers",
|
||||
description="Comma-separated DNS server IPs",
|
||||
show_when={"field": "DNS_PROVIDER", "value": "manual"},
|
||||
)
|
||||
```
|
||||
|
||||
The field will only be visible when the referenced field has the specified value.
|
||||
|
||||
## Conditional Disable
|
||||
|
||||
Fields can be enabled/disabled based on other field values using `disabled_when`:
|
||||
|
||||
```python
|
||||
# Disable timeout field when feature is disabled
|
||||
NumberField(
|
||||
key="FEATURE_TIMEOUT",
|
||||
label="Timeout (seconds)",
|
||||
description="Request timeout",
|
||||
default=30,
|
||||
disabled_when={
|
||||
"field": "FEATURE_ENABLED",
|
||||
"value": False,
|
||||
"reason": "Enable the feature first"
|
||||
},
|
||||
)
|
||||
```
|
||||
|
||||
The field will be greyed out with the specified reason when the condition is met.
|
||||
|
||||
## Settings Groups
|
||||
|
||||
Register a group to organize related settings tabs in the sidebar:
|
||||
|
||||
```python
|
||||
from shelfmark.core.settings_registry import register_group
|
||||
|
||||
# Register a group (do this once, usually in a central config file)
|
||||
register_group(
|
||||
name="my_group",
|
||||
display_name="My Group",
|
||||
icon="folder",
|
||||
order=50,
|
||||
)
|
||||
|
||||
# Then register settings to the group
|
||||
@register_settings(
|
||||
name="plugin_a",
|
||||
display_name="Plugin A",
|
||||
icon="puzzle",
|
||||
order=51,
|
||||
group="my_group", # Assigns to the group
|
||||
)
|
||||
def plugin_a_settings():
|
||||
return [...]
|
||||
```
|
||||
|
||||
**Existing groups:**
|
||||
- `direct_download` (order=20): For download-related settings
|
||||
- `metadata_providers` (order=50): For metadata provider plugins
|
||||
|
||||
## Value Resolution Priority
|
||||
|
||||
Settings values are resolved in this order (highest priority first):
|
||||
|
||||
1. **Config File** - Stored in `CONFIG_DIR/plugins/<tab_name>.json`
|
||||
2. **Field Default** - Value specified in the field definition
|
||||
|
||||
The `general` tab uses `CONFIG_DIR/settings.json` instead of the plugins subdirectory.
|
||||
|
||||
## Reading Setting Values
|
||||
|
||||
Use the `config` singleton to read setting values in your plugin code:
|
||||
|
||||
```python
|
||||
from shelfmark.core.config import config
|
||||
|
||||
# Get a setting value with default fallback
|
||||
api_key = config.get("MY_PLUGIN_API_KEY", "")
|
||||
timeout = config.get("MY_PLUGIN_TIMEOUT", 30)
|
||||
|
||||
# Or access as attributes (raises AttributeError if not found)
|
||||
api_key = config.MY_PLUGIN_API_KEY
|
||||
|
||||
# Check all cached settings
|
||||
all_settings = config.get_all()
|
||||
```
|
||||
|
||||
The config singleton:
|
||||
- Automatically resolves values from config files with field defaults as fallback
|
||||
- Caches values for performance
|
||||
- Refreshes automatically when settings are updated via the UI
|
||||
|
||||
## Complete Example: Metadata Provider
|
||||
|
||||
Here's a complete example for a metadata provider plugin:
|
||||
|
||||
```python
|
||||
# shelfmark/metadata_providers/my_provider.py
|
||||
|
||||
from shelfmark.metadata_providers.base import (
|
||||
MetadataProvider,
|
||||
register_provider,
|
||||
)
|
||||
from shelfmark.core.settings_registry import (
|
||||
register_settings,
|
||||
HeadingField,
|
||||
TextField,
|
||||
PasswordField,
|
||||
CheckboxField,
|
||||
ActionButton,
|
||||
)
|
||||
from shelfmark.core.config import config
|
||||
|
||||
|
||||
def _test_connection():
|
||||
"""Test API connection callback."""
|
||||
api_key = config.get("MY_PROVIDER_API_KEY", "")
|
||||
if not api_key:
|
||||
return {"success": False, "message": "API key not configured"}
|
||||
|
||||
try:
|
||||
# Perform actual connection test
|
||||
# response = requests.get(...)
|
||||
return {"success": True, "message": "Connected to My Provider API"}
|
||||
except Exception as e:
|
||||
return {"success": False, "message": f"Connection failed: {str(e)}"}
|
||||
|
||||
|
||||
@register_settings(
|
||||
name="my_provider",
|
||||
display_name="My Provider",
|
||||
icon="book",
|
||||
order=53,
|
||||
group="metadata_providers",
|
||||
)
|
||||
def my_provider_settings():
|
||||
"""Define settings for this metadata provider."""
|
||||
return [
|
||||
HeadingField(
|
||||
key="my_provider_heading",
|
||||
title="My Provider",
|
||||
description="A metadata provider for book information",
|
||||
link_url="https://myprovider.com",
|
||||
link_text="Visit My Provider",
|
||||
),
|
||||
PasswordField(
|
||||
key="MY_PROVIDER_API_KEY",
|
||||
label="API Key",
|
||||
description="Your My Provider API key",
|
||||
placeholder="Enter your API key",
|
||||
required=True,
|
||||
),
|
||||
CheckboxField(
|
||||
key="MY_PROVIDER_INCLUDE_COVERS",
|
||||
label="Include Cover Images",
|
||||
description="Fetch cover images when searching",
|
||||
default=True,
|
||||
),
|
||||
TextField(
|
||||
key="MY_PROVIDER_BASE_URL",
|
||||
label="API Base URL",
|
||||
description="Override the default API endpoint",
|
||||
default="https://api.myprovider.com/v1",
|
||||
required=False,
|
||||
),
|
||||
ActionButton(
|
||||
key="test_connection",
|
||||
label="Test Connection",
|
||||
description="Verify your API key works",
|
||||
style="primary",
|
||||
callback=_test_connection,
|
||||
),
|
||||
]
|
||||
|
||||
|
||||
@register_provider("my_provider")
|
||||
class MyProvider(MetadataProvider):
|
||||
"""My Provider metadata implementation."""
|
||||
|
||||
name = "my_provider"
|
||||
display_name = "My Provider"
|
||||
requires_auth = True
|
||||
|
||||
def __init__(self, api_key: str = None):
|
||||
self.api_key = api_key or config.get("MY_PROVIDER_API_KEY", "")
|
||||
self.base_url = config.get(
|
||||
"MY_PROVIDER_BASE_URL",
|
||||
"https://api.myprovider.com/v1"
|
||||
)
|
||||
|
||||
def is_available(self) -> bool:
|
||||
return bool(self.api_key)
|
||||
|
||||
def search(self, query: str):
|
||||
# Implementation...
|
||||
pass
|
||||
|
||||
def get_book(self, book_id: str):
|
||||
# Implementation...
|
||||
pass
|
||||
```
|
||||
|
||||
## Complete Example: Release Source
|
||||
|
||||
Here's a complete example for a release source plugin:
|
||||
|
||||
```python
|
||||
# shelfmark/release_sources/my_source.py
|
||||
|
||||
from shelfmark.release_sources.base import (
|
||||
ReleaseSource,
|
||||
DownloadHandler,
|
||||
register_source,
|
||||
register_handler,
|
||||
)
|
||||
from shelfmark.core.settings_registry import (
|
||||
register_settings,
|
||||
HeadingField,
|
||||
TextField,
|
||||
NumberField,
|
||||
CheckboxField,
|
||||
SelectField,
|
||||
ActionButton,
|
||||
)
|
||||
from shelfmark.core.config import config
|
||||
|
||||
|
||||
def _test_source():
|
||||
"""Test source availability callback."""
|
||||
base_url = config.get("MY_SOURCE_URL", "https://mysource.com")
|
||||
try:
|
||||
# Test connectivity
|
||||
return {"success": True, "message": f"Source available at {base_url}"}
|
||||
except Exception as e:
|
||||
return {"success": False, "message": f"Source unavailable: {str(e)}"}
|
||||
|
||||
|
||||
@register_settings(
|
||||
name="my_source",
|
||||
display_name="My Source",
|
||||
icon="download",
|
||||
order=25,
|
||||
group="direct_download",
|
||||
)
|
||||
def my_source_settings():
|
||||
"""Define settings for this release source."""
|
||||
return [
|
||||
HeadingField(
|
||||
key="my_source_heading",
|
||||
title="My Source Configuration",
|
||||
description="Configure the My Source download provider",
|
||||
),
|
||||
CheckboxField(
|
||||
key="MY_SOURCE_ENABLED",
|
||||
label="Enable My Source",
|
||||
description="Include My Source in download fallback chain",
|
||||
default=True,
|
||||
),
|
||||
TextField(
|
||||
key="MY_SOURCE_URL",
|
||||
label="Source URL",
|
||||
description="Base URL for the source",
|
||||
default="https://mysource.com",
|
||||
show_when={"field": "MY_SOURCE_ENABLED", "value": True},
|
||||
),
|
||||
NumberField(
|
||||
key="MY_SOURCE_TIMEOUT",
|
||||
label="Timeout (seconds)",
|
||||
description="Request timeout",
|
||||
default=30,
|
||||
min_value=10,
|
||||
max_value=120,
|
||||
show_when={"field": "MY_SOURCE_ENABLED", "value": True},
|
||||
),
|
||||
SelectField(
|
||||
key="MY_SOURCE_PRIORITY",
|
||||
label="Priority",
|
||||
description="Where in the fallback chain to try this source",
|
||||
default="normal",
|
||||
options=[
|
||||
{"value": "high", "label": "High (try first)"},
|
||||
{"value": "normal", "label": "Normal"},
|
||||
{"value": "low", "label": "Low (try last)"},
|
||||
],
|
||||
show_when={"field": "MY_SOURCE_ENABLED", "value": True},
|
||||
),
|
||||
ActionButton(
|
||||
key="test_source",
|
||||
label="Test Source",
|
||||
description="Check if the source is accessible",
|
||||
style="primary",
|
||||
callback=_test_source,
|
||||
),
|
||||
]
|
||||
|
||||
|
||||
@register_source("my_source")
|
||||
class MySource(ReleaseSource):
|
||||
"""My Source release source implementation."""
|
||||
|
||||
name = "my_source"
|
||||
display_name = "My Source"
|
||||
|
||||
def __init__(self):
|
||||
self.enabled = config.get("MY_SOURCE_ENABLED", True)
|
||||
self.base_url = config.get("MY_SOURCE_URL", "https://mysource.com")
|
||||
self.timeout = config.get("MY_SOURCE_TIMEOUT", 30)
|
||||
|
||||
def is_available(self) -> bool:
|
||||
return self.enabled
|
||||
|
||||
def search(self, book):
|
||||
# Implementation...
|
||||
pass
|
||||
|
||||
|
||||
@register_handler("my_source")
|
||||
class MySourceHandler(DownloadHandler):
|
||||
"""Handler for downloading from My Source."""
|
||||
|
||||
name = "my_source"
|
||||
|
||||
def download(self, release, output_path):
|
||||
# Implementation...
|
||||
pass
|
||||
```
|
||||
|
||||
## Best Practices
|
||||
|
||||
1. **Use descriptive keys**: Keys should be uppercase and prefixed with your plugin name (e.g., `MY_PLUGIN_API_KEY`)
|
||||
|
||||
2. **Provide helpful descriptions**: Include enough detail in descriptions to help users understand what each setting does
|
||||
|
||||
3. **Set sensible defaults**: Users should be able to get started without configuring everything
|
||||
|
||||
4. **Use conditional visibility**: Hide advanced options behind enabling checkboxes to reduce UI clutter
|
||||
|
||||
5. **Include a test button**: ActionButtons that test connections help users verify their configuration
|
||||
|
||||
6. **Mark restart-required settings**: Use `requires_restart=True` for settings that can't be applied live
|
||||
|
||||
7. **Group related settings**: Use HeadingField to visually separate sections, and put plugins in appropriate groups
|
||||
|
||||
8. **Handle missing values gracefully**: Always provide fallbacks when reading settings in your code
|
||||
|
||||
## API Reference
|
||||
|
||||
### Backend Routes
|
||||
|
||||
| Method | Endpoint | Description |
|
||||
|--------|----------|-------------|
|
||||
| GET | `/api/settings` | Get all settings tabs, groups, and values |
|
||||
| GET | `/api/settings/<tab_name>` | Get a specific settings tab |
|
||||
| PUT | `/api/settings/<tab_name>` | Update settings for a tab |
|
||||
| POST | `/api/settings/<tab_name>/action/<action_key>` | Execute an action button callback |
|
||||
|
||||
### Response Format
|
||||
|
||||
**GET /api/settings**
|
||||
```json
|
||||
{
|
||||
"groups": [
|
||||
{"name": "direct_download", "displayName": "Direct Download", "icon": "download", "order": 20}
|
||||
],
|
||||
"tabs": [
|
||||
{
|
||||
"name": "my_plugin",
|
||||
"displayName": "My Plugin",
|
||||
"icon": "book",
|
||||
"order": 53,
|
||||
"group": "metadata_providers",
|
||||
"fields": [
|
||||
{
|
||||
"type": "password",
|
||||
"key": "MY_PLUGIN_API_KEY",
|
||||
"label": "API Key",
|
||||
"description": "Your API key",
|
||||
"hasValue": true,
|
||||
"value": "",
|
||||
"required": true,
|
||||
"disabled": false,
|
||||
"requiresRestart": false
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
**PUT /api/settings/<tab_name>**
|
||||
```json
|
||||
// Request
|
||||
{"MY_PLUGIN_API_KEY": "new-value", "MY_PLUGIN_TIMEOUT": 60}
|
||||
|
||||
// Response
|
||||
{
|
||||
"success": true,
|
||||
"message": "Settings updated",
|
||||
"updated": ["MY_PLUGIN_API_KEY", "MY_PLUGIN_TIMEOUT"],
|
||||
"requiresRestart": false
|
||||
}
|
||||
```
|
||||
|
||||
**POST /api/settings/<tab_name>/action/<action_key>**
|
||||
```json
|
||||
// Response
|
||||
{
|
||||
"success": true,
|
||||
"message": "Connection successful!"
|
||||
}
|
||||
```
|
||||
@@ -0,0 +1,75 @@
|
||||
# URL Search Parameters
|
||||
|
||||
You can trigger searches directly via URL by adding query parameters. This enables bookmarking searches and sharing links.
|
||||
|
||||
## Basic Usage
|
||||
|
||||
```
|
||||
http://your-server:8084/?q=harry+potter
|
||||
```
|
||||
|
||||
## Supported Parameters
|
||||
|
||||
| Parameter | Description | Example |
|
||||
|-----------|-------------|---------|
|
||||
| `q` or `query` | Main search query | `/?q=dune` |
|
||||
| `author` | Filter by author name | `/?author=frank+herbert` |
|
||||
| `title` | Filter by book title | `/?title=foundation` |
|
||||
| `isbn` | Filter by ISBN | `/?isbn=978-0747532699` |
|
||||
| `lang` | Filter by language (ISO 639-1 code) | `/?lang=en` |
|
||||
| `format` | Filter by file format | `/?format=epub` |
|
||||
| `content` | Filter by content type | `/?content=fiction` |
|
||||
| `sort` | Sort order for results | `/?sort=newest` |
|
||||
|
||||
## Multiple Values
|
||||
|
||||
Some parameters support multiple values by repeating the parameter:
|
||||
|
||||
```
|
||||
/?lang=en&lang=de&lang=fr
|
||||
/?format=epub&format=mobi&format=azw3
|
||||
```
|
||||
|
||||
## Examples
|
||||
|
||||
**Simple search:**
|
||||
```
|
||||
/?q=lord+of+the+rings
|
||||
```
|
||||
|
||||
**Search with author filter:**
|
||||
```
|
||||
/?q=dune&author=frank+herbert
|
||||
```
|
||||
|
||||
**Search with format and language:**
|
||||
```
|
||||
/?q=harry+potter&format=epub&lang=en
|
||||
```
|
||||
|
||||
**Author search with multiple formats:**
|
||||
```
|
||||
/?author=stephen+king&format=epub&format=mobi
|
||||
```
|
||||
|
||||
**Search with sort order:**
|
||||
```
|
||||
/?q=science+fiction&sort=newest
|
||||
```
|
||||
|
||||
## Search Mode Behavior
|
||||
|
||||
### Direct Download Mode (default)
|
||||
|
||||
All parameters are used to filter results from Anna's Archive.
|
||||
|
||||
### Universal Mode
|
||||
|
||||
Only `q` and `sort` are used. Other parameters (author, title, format, etc.) are silently ignored since metadata providers have their own search capabilities.
|
||||
|
||||
## Notes
|
||||
|
||||
- URL parameters are read once on page load
|
||||
- The URL is not updated when you perform searches manually
|
||||
- Spaces should be encoded as `+` or `%20`
|
||||
- Invalid or unknown parameters are silently ignored
|
||||
@@ -1,131 +0,0 @@
|
||||
"""Network operations manager for the book downloader application."""
|
||||
|
||||
import network
|
||||
network.init()
|
||||
import requests
|
||||
import time
|
||||
from io import BytesIO
|
||||
from typing import Optional
|
||||
from urllib.parse import urlparse
|
||||
from tqdm import tqdm
|
||||
|
||||
from logger import setup_logger
|
||||
from config import PROXIES
|
||||
from env import MAX_RETRY, DEFAULT_SLEEP, USE_CF_BYPASS
|
||||
if USE_CF_BYPASS:
|
||||
import cloudflare_bypasser
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
|
||||
def html_get_page(url: str, retry: int = MAX_RETRY, use_bypasser: bool = False) -> str:
|
||||
"""Fetch HTML content from a URL with retry mechanism.
|
||||
|
||||
Args:
|
||||
url: Target URL
|
||||
retry: Number of retry attempts
|
||||
skip_404: Whether to skip 404 errors
|
||||
|
||||
Returns:
|
||||
str: HTML content if successful, None otherwise
|
||||
"""
|
||||
response = None
|
||||
try:
|
||||
logger.debug(f"html_get_page: {url}, retry: {retry}, use_bypasser: {use_bypasser}")
|
||||
if use_bypasser and USE_CF_BYPASS:
|
||||
logger.info(f"GET Using Cloudflare Bypasser for: {url}")
|
||||
response_html = cloudflare_bypasser.get(url)
|
||||
logger.debug(f"Cloudflare Bypasser response length: {len(response_html)}")
|
||||
if response_html.strip() != "":
|
||||
return response_html
|
||||
else:
|
||||
raise requests.exceptions.RequestException("Failed to bypass Cloudflare")
|
||||
else:
|
||||
logger.info(f"GET: {url}")
|
||||
response = requests.get(url, proxies=PROXIES)
|
||||
response.raise_for_status()
|
||||
logger.debug(f"Success getting: {url}")
|
||||
time.sleep(1)
|
||||
return str(response.text)
|
||||
|
||||
except Exception as e:
|
||||
if retry == 0:
|
||||
logger.error_trace(f"Failed to fetch page: {url}, error: {e}")
|
||||
return ""
|
||||
|
||||
if use_bypasser and USE_CF_BYPASS:
|
||||
logger.warning(f"Exception while using cloudflare bypass for URL: {url}")
|
||||
logger.warning(f"Exception: {e}")
|
||||
logger.warning(f"Response: {response}")
|
||||
elif response is not None and response.status_code == 404:
|
||||
logger.warning(f"404 error for URL: {url}")
|
||||
return ""
|
||||
elif response is not None and response.status_code == 403:
|
||||
logger.warning(f"403 detected for URL: {url}. Should retry using cloudflare bypass.")
|
||||
return html_get_page(url, retry - 1, True)
|
||||
|
||||
sleep_time = DEFAULT_SLEEP * (MAX_RETRY - retry + 1)
|
||||
logger.warning(
|
||||
f"Retrying GET {url} in {sleep_time} seconds due to error: {e}"
|
||||
)
|
||||
time.sleep(sleep_time)
|
||||
return html_get_page(url, retry - 1, use_bypasser)
|
||||
|
||||
def download_url(link: str, size: str = "") -> Optional[BytesIO]:
|
||||
"""Download content from URL into a BytesIO buffer.
|
||||
|
||||
Args:
|
||||
link: URL to download from
|
||||
|
||||
Returns:
|
||||
BytesIO: Buffer containing downloaded content if successful
|
||||
"""
|
||||
try:
|
||||
logger.info(f"Downloading from: {link}")
|
||||
response = requests.get(link, stream=True, proxies=PROXIES)
|
||||
response.raise_for_status()
|
||||
|
||||
total_size : float = 0.0
|
||||
try:
|
||||
# we assume size is in MB
|
||||
total_size = float(size.strip().replace(" ", "").replace(",", ".").upper()[:-2].strip()) * 1024 * 1024
|
||||
except:
|
||||
total_size = float(response.headers.get('content-length', 0))
|
||||
|
||||
buffer = BytesIO()
|
||||
|
||||
# Initialize the progress bar with your guess
|
||||
pbar = tqdm(total=total_size, unit='B', unit_scale=True, desc='Downloading')
|
||||
for chunk in response.iter_content(chunk_size=1000):
|
||||
buffer.write(chunk)
|
||||
pbar.update(len(chunk))
|
||||
|
||||
pbar.close()
|
||||
if buffer.tell() * 0.1 < total_size * 0.9:
|
||||
# Check the content of the buffer if its HTML or binary
|
||||
if response.headers.get('content-type', '').startswith('text/html'):
|
||||
logger.warn(f"Failed to download content for {link}. Found HTML content instead.")
|
||||
return None
|
||||
return buffer
|
||||
except requests.exceptions.RequestException as e:
|
||||
logger.error_trace(f"Failed to download from {link}: {e}")
|
||||
return None
|
||||
|
||||
def get_absolute_url(base_url: str, url: str) -> str:
|
||||
"""Get absolute URL from relative URL and base URL.
|
||||
|
||||
Args:
|
||||
base_url: Base URL
|
||||
url: Relative URL
|
||||
"""
|
||||
if url.strip() == "":
|
||||
return ""
|
||||
if url.strip("#") == "":
|
||||
return ""
|
||||
if url.startswith("http"):
|
||||
return url
|
||||
parsed_url = urlparse(url)
|
||||
parsed_base = urlparse(base_url)
|
||||
if parsed_url.netloc == "" or parsed_url.scheme == "":
|
||||
parsed_url = parsed_url._replace(netloc=parsed_base.netloc, scheme=parsed_base.scheme)
|
||||
return parsed_url.geturl()
|
||||
@@ -1,7 +1,7 @@
|
||||
#!/bin/bash
|
||||
LOG_DIR=${LOG_ROOT:-/var/log/}/cwa-book-downloader
|
||||
LOG_DIR=${LOG_ROOT:-/var/log/}/shelfmark
|
||||
mkdir -p $LOG_DIR
|
||||
LOG_FILE=${LOG_DIR}/cwa-bd_entrypoint.log
|
||||
LOG_FILE=${LOG_DIR}/shelfmark_entrypoint.log
|
||||
|
||||
# Cleanup any existing files or folders in the log directory
|
||||
rm -rf $LOG_DIR/*
|
||||
@@ -20,6 +20,7 @@ set -e
|
||||
|
||||
# Print build version
|
||||
echo "Build version: $BUILD_VERSION"
|
||||
echo "Release version: $RELEASE_VERSION"
|
||||
|
||||
# Configure timezone
|
||||
if [ "$TZ" ]; then
|
||||
@@ -27,34 +28,57 @@ if [ "$TZ" ]; then
|
||||
ln -snf /usr/share/zoneinfo/$TZ /etc/localtime && echo $TZ > /etc/timezone
|
||||
fi
|
||||
|
||||
# Set UID if not set
|
||||
if [ -z "$UID" ]; then
|
||||
UID=1000
|
||||
# Determine user ID with proper precedence:
|
||||
# 1. PUID (LinuxServer.io standard - recommended)
|
||||
# 2. UID (legacy, for backward compatibility with existing installs)
|
||||
# 3. Default to 1000
|
||||
#
|
||||
# Note: $UID is a bash builtin that's always set. We use `printenv` to detect
|
||||
# if UID was explicitly set as an environment variable (e.g., via docker-compose).
|
||||
if [ -n "$PUID" ]; then
|
||||
RUN_UID="$PUID"
|
||||
echo "Using PUID=$RUN_UID"
|
||||
elif printenv UID >/dev/null 2>&1; then
|
||||
RUN_UID="$(printenv UID)"
|
||||
echo "Using UID=$RUN_UID (legacy - consider migrating to PUID)"
|
||||
else
|
||||
RUN_UID=1000
|
||||
echo "Using default UID=$RUN_UID"
|
||||
fi
|
||||
|
||||
# Set GID if not set
|
||||
if [ -z "$GID" ]; then
|
||||
GID=100
|
||||
# Determine group ID with proper precedence:
|
||||
# 1. PGID (LinuxServer.io standard - recommended)
|
||||
# 2. GID (legacy, for backward compatibility with existing installs)
|
||||
# 3. Default to 1000
|
||||
if [ -n "$PGID" ]; then
|
||||
RUN_GID="$PGID"
|
||||
echo "Using PGID=$RUN_GID"
|
||||
elif [ -n "$GID" ]; then
|
||||
RUN_GID="$GID"
|
||||
echo "Using GID=$RUN_GID (legacy - consider migrating to PGID)"
|
||||
else
|
||||
RUN_GID=1000
|
||||
echo "Using default GID=$RUN_GID"
|
||||
fi
|
||||
|
||||
if ! getent group "$GID" >/dev/null; then
|
||||
echo "Adding group $GID with name appuser"
|
||||
groupadd -g "$GID" appuser
|
||||
if ! getent group "$RUN_GID" >/dev/null; then
|
||||
echo "Adding group $RUN_GID with name appuser"
|
||||
groupadd -g "$RUN_GID" appuser
|
||||
fi
|
||||
|
||||
# Create user if it doesn't exist
|
||||
if ! id -u "$UID" >/dev/null 2>&1; then
|
||||
echo "Adding user $UID with name appuser"
|
||||
useradd -u "$UID" -g "$GID" -d /app -s /sbin/nologin appuser
|
||||
if ! id -u "$RUN_UID" >/dev/null 2>&1; then
|
||||
echo "Adding user $RUN_UID with name appuser"
|
||||
useradd -u "$RUN_UID" -g "$RUN_GID" -d /app -s /sbin/nologin appuser
|
||||
fi
|
||||
|
||||
# Get username for the UID (whether we just created it or it existed)
|
||||
USERNAME=$(getent passwd "$UID" | cut -d: -f1)
|
||||
echo "Username for UID $UID is $USERNAME"
|
||||
USERNAME=$(getent passwd "$RUN_UID" | cut -d: -f1)
|
||||
echo "Username for UID $RUN_UID is $USERNAME"
|
||||
|
||||
test_write() {
|
||||
folder=$1
|
||||
test_file=$folder/calibre-web-automated-book-downloader_TEST_WRITE
|
||||
test_file=$folder/shelfmark_TEST_WRITE
|
||||
mkdir -p $folder
|
||||
(
|
||||
echo 0123456789_TEST | sudo -E -u "$USERNAME" HOME=/app tee $test_file > /dev/null
|
||||
@@ -83,7 +107,16 @@ make_writable() {
|
||||
else
|
||||
echo "Folder $folder is not writable, changing ownership"
|
||||
change_ownership $folder
|
||||
chmod g+r,g+w $folder || echo "Failed to change group permissions for ${folder}, continuing..."
|
||||
chmod -R g+r,g+w $folder || echo "Failed to change group permissions for ${folder}, continuing..."
|
||||
fi
|
||||
# Fix any misowned subdirectories/files (e.g., from previous runs as root)
|
||||
if [ -d "$folder" ]; then
|
||||
misowned_count=$(find "$folder" -mindepth 1 \( ! -user "$RUN_UID" -o ! -group "$RUN_GID" \) 2>/dev/null | wc -l)
|
||||
if [ "$misowned_count" -gt 0 ]; then
|
||||
echo "Fixing ownership of $misowned_count files/directories in $folder"
|
||||
find "$folder" -mindepth 1 \( ! -user "$RUN_UID" -o ! -group "$RUN_GID" \) \
|
||||
-exec chown "$RUN_UID:$RUN_GID" {} \; 2>/dev/null || true
|
||||
fi
|
||||
fi
|
||||
test_write $folder || echo "Failed to test write to ${folder}, continuing..."
|
||||
}
|
||||
@@ -92,28 +125,73 @@ make_writable() {
|
||||
change_ownership() {
|
||||
folder=$1
|
||||
mkdir -p $folder
|
||||
echo "Changing ownership of $folder to $USERNAME:$GID"
|
||||
chown -R "${UID}" "${folder}" || echo "Failed to change user ownership for ${folder}, continuing..."
|
||||
chown -R ":${GID}" "${folder}" || echo "Failed to change group ownership for ${folder}, continuing..."
|
||||
echo "Changing ownership of $folder to $USERNAME:$RUN_GID"
|
||||
chown -R "${RUN_UID}" "${folder}" || echo "Failed to change user ownership for ${folder}, continuing..."
|
||||
chown -R ":${RUN_GID}" "${folder}" || echo "Failed to change group ownership for ${folder}, continuing..."
|
||||
}
|
||||
|
||||
change_ownership /app
|
||||
change_ownership /var/log/cwa-book-downloader
|
||||
change_ownership /tmp/cwa-book-downloader
|
||||
change_ownership /var/log/shelfmark
|
||||
change_ownership /tmp/shelfmark
|
||||
|
||||
# Test write to all folders
|
||||
make_writable /cwa-book-ingest
|
||||
make_writable ${CONFIG_DIR:-/config}
|
||||
make_writable ${INGEST_DIR:-/books}
|
||||
|
||||
# Set the command to run based on the environment
|
||||
is_prod=$(echo "$APP_ENV" | tr '[:upper:]' '[:lower:]')
|
||||
if [ "$is_prod" = "prod" ]; then
|
||||
command="gunicorn -t 300 -b ${FLASK_HOST:-0.0.0.0}:${FLASK_PORT:-8084} app:app"
|
||||
else
|
||||
command="python3 app.py"
|
||||
# Fix permissions on directories configured in settings
|
||||
echo "Checking for additional configured directories..."
|
||||
if [ -f /app/scripts/fix_permissions.py ]; then
|
||||
configured_dirs=$(python3 /app/scripts/fix_permissions.py 2>/dev/null || echo "")
|
||||
if [ -n "$configured_dirs" ]; then
|
||||
echo "$configured_dirs" | while read -r dir; do
|
||||
if [ -n "$dir" ] && [ -d "$dir" ]; then
|
||||
echo "Checking configured directory: $dir"
|
||||
make_writable "$dir"
|
||||
fi
|
||||
done
|
||||
fi
|
||||
fi
|
||||
|
||||
# IF DEBUG
|
||||
if [ "$DEBUG" = "true" ]; then
|
||||
# Fallback to root if config dir is still not writable (common on NAS/Unraid after upgrade from v0.4.0)
|
||||
CONFIG_PATH=${CONFIG_DIR:-/config}
|
||||
set +e
|
||||
test_write "$CONFIG_PATH" >/dev/null 2>&1
|
||||
config_ok=$?
|
||||
set -e
|
||||
|
||||
if [ $config_ok -ne 0 ] && [ "$RUN_UID" != "0" ]; then
|
||||
config_owner=$(stat -c '%u' "$CONFIG_PATH" 2>/dev/null || echo "unknown")
|
||||
if [ "$config_owner" = "0" ]; then
|
||||
echo ""
|
||||
echo "========================================================"
|
||||
echo "WARNING: Permission issue detected!"
|
||||
echo ""
|
||||
echo "Config directory is owned by root but PUID=$RUN_UID."
|
||||
echo "This typically happens after upgrading from v0.4.0 where"
|
||||
echo "PUID/PGID settings were not respected."
|
||||
echo ""
|
||||
echo "Falling back to running as root to prevent data loss."
|
||||
echo ""
|
||||
echo "To fix this permanently, run on your HOST machine:"
|
||||
echo " chown -R $RUN_UID:$RUN_GID /path/to/config"
|
||||
echo ""
|
||||
echo "Then restart the container."
|
||||
echo "========================================================"
|
||||
echo ""
|
||||
RUN_UID=0
|
||||
RUN_GID=0
|
||||
USERNAME=root
|
||||
fi
|
||||
fi
|
||||
|
||||
# Always run Gunicorn (even when DEBUG=true) to ensure Socket.IO WebSocket
|
||||
# upgrades work reliably on customer machines.
|
||||
# Map app LOG_LEVEL (often DEBUG/INFO/...) to gunicorn's --log-level (lowercase).
|
||||
gunicorn_loglevel=$([ "$DEBUG" = "true" ] && echo debug || echo "${LOG_LEVEL:-info}" | tr '[:upper:]' '[:lower:]')
|
||||
command="gunicorn --log-level ${gunicorn_loglevel} --access-logfile - --error-logfile - --worker-class geventwebsocket.gunicorn.workers.GeventWebSocketWorker --workers 1 -t 300 -b ${FLASK_HOST:-0.0.0.0}:${FLASK_PORT:-8084} shelfmark.main:app"
|
||||
|
||||
# If DEBUG and not using an external bypass
|
||||
if [ "$DEBUG" = "true" ] && [ "$USING_EXTERNAL_BYPASSER" != "true" ]; then
|
||||
set +e
|
||||
set -x
|
||||
echo "vvvvvvvvvvvv DEBUG MODE vvvvvvvvvvvv"
|
||||
@@ -166,15 +244,24 @@ if [ "$DEBUG" = "true" ]; then
|
||||
echo "^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^"
|
||||
fi
|
||||
|
||||
# Hacky way to verify /tmp has at least 1MB of space and is writable/readable
|
||||
# Verify /tmp has at least 1MB of space and is writable/readable
|
||||
echo "Verifying /tmp has enough space"
|
||||
rm -f /tmp/test.cwa-bd
|
||||
for i in {1..150000}; do printf "%04d\n" $i; done > /tmp/test.cwa-bd
|
||||
sum=$(python3 -c "print(sum(int(l.strip()) for l in open('/tmp/test.cwa-bd').readlines()))")
|
||||
[ "$sum" == 11250075000 ] && echo "Success: /tmp is writable" || (echo "Failure: /tmp is not writable" && exit 1)
|
||||
rm /tmp/test.cwa-bd
|
||||
rm -f /tmp/test.shelfmark
|
||||
if dd if=/dev/zero of=/tmp/test.shelfmark bs=1M count=1 2>/dev/null && \
|
||||
[ "$(wc -c < /tmp/test.shelfmark)" -eq 1048576 ]; then
|
||||
rm -f /tmp/test.shelfmark
|
||||
echo "Success: /tmp is writable and readable"
|
||||
else
|
||||
echo "Failure: /tmp is not writable or has insufficient space"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "Running command: '$command' as '$USERNAME' in '$APP_ENV' mode"
|
||||
echo "Running command: '$command' as '$USERNAME' (debug=$is_debug)"
|
||||
|
||||
# Set umask for file permissions (default: 0022 = files 644, dirs 755)
|
||||
UMASK_VALUE=${UMASK:-0022}
|
||||
echo "Setting umask to $UMASK_VALUE"
|
||||
umask $UMASK_VALUE
|
||||
|
||||
# Stop logging
|
||||
exec 1>&3 2>&4
|
||||
|
||||
@@ -1,51 +0,0 @@
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
def string_to_bool(s: str) -> bool:
|
||||
return s.lower() in ["true", "yes", "1", "y"]
|
||||
|
||||
CWA_DB = os.getenv("CWA_DB_PATH")
|
||||
CWA_DB_PATH = Path(CWA_DB) if CWA_DB else None
|
||||
LOG_ROOT = Path(os.getenv("LOG_ROOT", "/var/log/"))
|
||||
LOG_DIR = LOG_ROOT / "cwa-book-downloader"
|
||||
TMP_DIR = Path(os.getenv("TMP_DIR", "/tmp/cwa-book-downloader"))
|
||||
INGEST_DIR = Path(os.getenv("INGEST_DIR", "/cwa-book-ingest"))
|
||||
STATUS_TIMEOUT = int(os.getenv("STATUS_TIMEOUT", "3600"))
|
||||
USE_BOOK_TITLE = string_to_bool(os.getenv("USE_BOOK_TITLE", "false"))
|
||||
MAX_RETRY = int(os.getenv("MAX_RETRY", "10"))
|
||||
DEFAULT_SLEEP = int(os.getenv("DEFAULT_SLEEP", "5"))
|
||||
USE_CF_BYPASS = string_to_bool(os.getenv("USE_CF_BYPASS", "true"))
|
||||
HTTP_PROXY = os.getenv("HTTP_PROXY", "").strip()
|
||||
HTTPS_PROXY = os.getenv("HTTPS_PROXY", "").strip()
|
||||
AA_DONATOR_KEY = os.getenv("AA_DONATOR_KEY", "").strip()
|
||||
_AA_BASE_URL = os.getenv("AA_BASE_URL", "auto").strip()
|
||||
_AA_ADDITIONAL_URLS = os.getenv("AA_ADDITIONAL_URLS", "").strip()
|
||||
_SUPPORTED_FORMATS = os.getenv("SUPPORTED_FORMATS", "epub,mobi,azw3,fb2,djvu,cbz,cbr").lower()
|
||||
_BOOK_LANGUAGE = os.getenv("BOOK_LANGUAGE", "en").lower()
|
||||
_CUSTOM_SCRIPT = os.getenv("CUSTOM_SCRIPT", "").strip()
|
||||
FLASK_HOST = os.getenv("FLASK_HOST", "0.0.0.0")
|
||||
FLASK_PORT = int(os.getenv("FLASK_PORT", "8084"))
|
||||
DEBUG = string_to_bool(os.getenv("DEBUG", "False"))
|
||||
# If debug is true, we want to log everything
|
||||
if DEBUG:
|
||||
LOG_LEVEL = "DEBUG"
|
||||
else:
|
||||
LOG_LEVEL = os.getenv("LOG_LEVEL", "INFO").upper()
|
||||
ENABLE_LOGGING = string_to_bool(os.getenv("ENABLE_LOGGING", "true"))
|
||||
MAIN_LOOP_SLEEP_TIME = int(os.getenv("MAIN_LOOP_SLEEP_TIME", "5"))
|
||||
DOCKERMODE = string_to_bool(os.getenv("DOCKERMODE", "false"))
|
||||
_CUSTOM_DNS = os.getenv("CUSTOM_DNS", "").strip()
|
||||
USE_DOH = string_to_bool(os.getenv("USE_DOH", "false"))
|
||||
BYPASS_RELEASE_INACTIVE_MIN = int(os.getenv("BYPASS_RELEASE_INACTIVE_MIN", "5"))
|
||||
APP_ENV = os.getenv("APP_ENV", "prod").lower()
|
||||
# Logging settings
|
||||
LOG_FILE = LOG_DIR / "cwa-book-downloader.log"
|
||||
|
||||
USING_TOR = string_to_bool(os.getenv("USING_TOR", "false"))
|
||||
# If using Tor, we don't need to set custom DNS, use DOH, or proxy
|
||||
if USING_TOR:
|
||||
_CUSTOM_DNS = ""
|
||||
USE_DOH = False
|
||||
HTTP_PROXY = ""
|
||||
HTTPS_PROXY = ""
|
||||
|
||||
@@ -2,8 +2,8 @@
|
||||
|
||||
# Set up log paths
|
||||
LOG_ROOT=${LOG_ROOT:-"/var/log"}
|
||||
LOG_DIR="$LOG_ROOT/cwa-book-downloader"
|
||||
OUTPUT_FILE_NAME="cwa-book-downloader-debug_BUILD-${BUILD_VERSION:-local}_$(date +%Y%m%d-%H%M%S)"
|
||||
LOG_DIR="$LOG_ROOT/shelfmark"
|
||||
OUTPUT_FILE_NAME="shelfmark-debug_BUILD-${BUILD_VERSION:-local}_RELEASE-${RELEASE_VERSION:-NA}_$(date +%Y%m%d-%H%M%S)"
|
||||
OUTPUT_FILE="/tmp/$OUTPUT_FILE_NAME.zip"
|
||||
|
||||
# Create LOG_DIR if it doesn't exist
|
||||
@@ -18,17 +18,17 @@ echo "" >> "$LOG_DIR/system_info.txt"
|
||||
|
||||
# Add disk usage
|
||||
echo "=== Disk Usage ===" >> "$LOG_DIR/system_info.txt"
|
||||
df -h >> "$LOG_DIR/system_info.txt"
|
||||
df -h >> "$LOG_DIR/system_info.txt" 2>&1
|
||||
echo "" >> "$LOG_DIR/system_info.txt"
|
||||
|
||||
# Add memory info
|
||||
echo "=== Memory Info ===" >> "$LOG_DIR/system_info.txt"
|
||||
free -h >> "$LOG_DIR/system_info.txt"
|
||||
free -h >> "$LOG_DIR/system_info.txt" 2>&1
|
||||
echo "" >> "$LOG_DIR/system_info.txt"
|
||||
|
||||
# Add running processes
|
||||
echo "=== Running Processes ===" >> "$LOG_DIR/system_info.txt"
|
||||
ps aux >> "$LOG_DIR/system_info.txt"
|
||||
ps aux >> "$LOG_DIR/system_info.txt" 2>&1
|
||||
echo "" >> "$LOG_DIR/system_info.txt"
|
||||
|
||||
# Add network information using basic commands
|
||||
@@ -37,17 +37,17 @@ echo "=== Network Information ===" > "$LOG_DIR/network_info.txt"
|
||||
# Try to get basic connectivity information
|
||||
echo "=== Basic Connectivity ===" >> "$LOG_DIR/network_info.txt"
|
||||
echo "Hostname resolution:" >> "$LOG_DIR/network_info.txt"
|
||||
cat /etc/hosts 2>/dev/null >> "$LOG_DIR/network_info.txt" || echo "Unable to read /etc/hosts" >> "$LOG_DIR/network_info.txt"
|
||||
cat /etc/hosts >> "$LOG_DIR/network_info.txt" 2>&1 || echo "Unable to read /etc/hosts" >> "$LOG_DIR/network_info.txt"
|
||||
echo "" >> "$LOG_DIR/network_info.txt"
|
||||
|
||||
echo "DNS configuration:" >> "$LOG_DIR/network_info.txt"
|
||||
cat /etc/resolv.conf 2>/dev/null >> "$LOG_DIR/network_info.txt" || echo "Unable to read /etc/resolv.conf" >> "$LOG_DIR/network_info.txt"
|
||||
cat /etc/resolv.conf >> "$LOG_DIR/network_info.txt" 2>&1 || echo "Unable to read /etc/resolv.conf" >> "$LOG_DIR/network_info.txt"
|
||||
echo "" >> "$LOG_DIR/network_info.txt"
|
||||
|
||||
# Try to get interface information from /proc
|
||||
echo "=== Network Interfaces (/proc) ===" >> "$LOG_DIR/network_info.txt"
|
||||
if [ -f "/proc/net/dev" ]; then
|
||||
cat /proc/net/dev >> "$LOG_DIR/network_info.txt"
|
||||
cat /proc/net/dev >> "$LOG_DIR/network_info.txt" 2>&1
|
||||
else
|
||||
echo "Not available: /proc/net/dev not found" >> "$LOG_DIR/network_info.txt"
|
||||
fi
|
||||
@@ -55,9 +55,9 @@ echo "" >> "$LOG_DIR/network_info.txt"
|
||||
|
||||
# Try connectivity tests
|
||||
echo "=== Internet Connectivity ===" >> "$LOG_DIR/network_info.txt"
|
||||
ping -c 3 1.1.1.1 2>/dev/null >> "$LOG_DIR/network_info.txt" || echo "Ping command failed or not available" >> "$LOG_DIR/network_info.txt"
|
||||
ping -c 3 1.1.1.1 >> "$LOG_DIR/network_info.txt" 2>&1 || echo "Ping command failed or not available" >> "$LOG_DIR/network_info.txt"
|
||||
echo "" >> "$LOG_DIR/network_info.txt"
|
||||
ping -c 3 one.one.one.one 2>/dev/null >> "$LOG_DIR/network_info.txt" || echo "DNS resolution test failed" >> "$LOG_DIR/network_info.txt"
|
||||
ping -c 3 one.one.one.one >> "$LOG_DIR/network_info.txt" 2>&1 || echo "DNS resolution test failed" >> "$LOG_DIR/network_info.txt"
|
||||
echo "" >> "$LOG_DIR/network_info.txt"
|
||||
|
||||
# Test IPv6 connectivity
|
||||
@@ -77,7 +77,7 @@ echo "" >> "$LOG_DIR/network_info.txt"
|
||||
|
||||
# Try IPv6 connectivity test using Cloudflare's IPv6 DNS
|
||||
echo "Testing IPv6 connectivity to Cloudflare DNS:" >> "$LOG_DIR/network_info.txt"
|
||||
ping6 -c 3 2606:4700:4700::1111 2>/dev/null >> "$LOG_DIR/network_info.txt" || echo "IPv6 ping failed or not available" >> "$LOG_DIR/network_info.txt"
|
||||
ping6 -c 3 2606:4700:4700::1111 >> "$LOG_DIR/network_info.txt" 2>&1 || echo "IPv6 ping failed or not available" >> "$LOG_DIR/network_info.txt"
|
||||
echo "" >> "$LOG_DIR/network_info.txt"
|
||||
|
||||
# Test SSL connectivity
|
||||
@@ -92,24 +92,36 @@ echo "" >> "$LOG_DIR/network_info.txt"
|
||||
|
||||
# Add installed packages
|
||||
echo "=== Installed Python Packages ===" > "$LOG_DIR/packages.txt"
|
||||
pip list 2>/dev/null >> "$LOG_DIR/packages.txt" || echo "pip not found" >> "$LOG_DIR/packages.txt"
|
||||
pip list >> "$LOG_DIR/packages.txt" 2>&1 || echo "pip not found" >> "$LOG_DIR/packages.txt"
|
||||
echo "" >> "$LOG_DIR/packages.txt"
|
||||
|
||||
# Check Permissions
|
||||
echo "=== Permissions ===" > "$LOG_DIR/permissions.txt"
|
||||
echo "ls -all /app" >> "$LOG_DIR/permissions.txt"
|
||||
ls -all /app >> "$LOG_DIR/permissions.txt"
|
||||
ls -all /app >> "$LOG_DIR/permissions.txt" 2>&1
|
||||
echo "" >> "$LOG_DIR/permissions.txt"
|
||||
echo "ls -all /cwa-book-ingest" >> "$LOG_DIR/permissions.txt"
|
||||
ls -all /cwa-book-ingest >> "$LOG_DIR/permissions.txt"
|
||||
echo "ls -all ${INGEST_DIR:-/books}" >> "$LOG_DIR/permissions.txt"
|
||||
ls -all ${INGEST_DIR:-/books} >> "$LOG_DIR/permissions.txt" 2>&1
|
||||
echo "" >> "$LOG_DIR/permissions.txt"
|
||||
echo "ls -all /var/log/cwa-book-downloader" >> "$LOG_DIR/permissions.txt"
|
||||
ls -all /var/log/cwa-book-downloader >> "$LOG_DIR/permissions.txt"
|
||||
echo "ls -all /var/log/shelfmark" >> "$LOG_DIR/permissions.txt"
|
||||
ls -all /var/log/shelfmark >> "$LOG_DIR/permissions.txt" 2>&1
|
||||
echo "" >> "$LOG_DIR/permissions.txt"
|
||||
echo "ls -all /tmp/cwa-book-downloader" >> "$LOG_DIR/permissions.txt"
|
||||
ls -all /tmp/cwa-book-downloader >> "$LOG_DIR/permissions.txt"
|
||||
echo "ls -all /tmp/shelfmark" >> "$LOG_DIR/permissions.txt"
|
||||
ls -all /tmp/shelfmark >> "$LOG_DIR/permissions.txt" 2>&1
|
||||
echo "" >> "$LOG_DIR/permissions.txt"
|
||||
|
||||
# Check Iptables (NAT)
|
||||
echo "=== IPtables NAT Rules ===" > "$LOG_DIR/iptables_nat.txt"
|
||||
iptables -t nat -L -v -n >> "$LOG_DIR/iptables_nat.txt" 2>&1
|
||||
|
||||
# Check DNS Resolution details
|
||||
echo "=== DNS Resolution Test ===" > "$LOG_DIR/dns_test.txt"
|
||||
echo "Resolving google.com:" >> "$LOG_DIR/dns_test.txt"
|
||||
nslookup google.com >> "$LOG_DIR/dns_test.txt" 2>&1
|
||||
echo "" >> "$LOG_DIR/dns_test.txt"
|
||||
echo "Resolving check.torproject.org:" >> "$LOG_DIR/dns_test.txt"
|
||||
nslookup check.torproject.org >> "$LOG_DIR/dns_test.txt" 2>&1
|
||||
|
||||
|
||||
# Check if running in Docker
|
||||
echo "=== Container Info ===" > "$LOG_DIR/container_info.txt"
|
||||
@@ -122,19 +134,58 @@ else
|
||||
fi
|
||||
|
||||
# Add environment variables (redacting sensitive info)
|
||||
env | grep -v -E "(AA_DONATOR_KEY)" | sort > "$LOG_DIR/environment.txt"
|
||||
env | grep -v -E "(AA_DONATOR_KEY|HARDCOVER_API_KEY|_KEY=|_SECRET=|_PASSWORD=|_TOKEN=)" | sort > "$LOG_DIR/environment.txt"
|
||||
|
||||
echo "--- HTTPBin ---" > $LOG_DIR/network_info.txt
|
||||
pyrequests https://httpbin.org/get >> $LOG_DIR/network_info.txt
|
||||
ehco ""
|
||||
# Add configuration files (redacting sensitive values)
|
||||
CONFIG_DIR=${CONFIG_DIR:-"/config"}
|
||||
if [ -d "$CONFIG_DIR" ]; then
|
||||
mkdir -p "$LOG_DIR/config"
|
||||
|
||||
# Copy and redact main settings file
|
||||
if [ -f "$CONFIG_DIR/settings.json" ]; then
|
||||
# Redact sensitive fields (API keys, passwords, tokens)
|
||||
sed -E 's/("(AA_DONATOR_KEY|HARDCOVER_API_KEY|[^"]*_KEY|[^"]*_SECRET|[^"]*_PASSWORD|[^"]*_TOKEN)"[[:space:]]*:[[:space:]]*")[^"]+"/\1[REDACTED]"/g' \
|
||||
"$CONFIG_DIR/settings.json" > "$LOG_DIR/config/settings.json" 2>/dev/null
|
||||
fi
|
||||
|
||||
# Copy and redact plugin config files
|
||||
if [ -d "$CONFIG_DIR/plugins" ]; then
|
||||
mkdir -p "$LOG_DIR/config/plugins"
|
||||
for config_file in "$CONFIG_DIR/plugins"/*.json; do
|
||||
if [ -f "$config_file" ]; then
|
||||
filename=$(basename "$config_file")
|
||||
sed -E 's/("(AA_DONATOR_KEY|HARDCOVER_API_KEY|[^"]*_KEY|[^"]*_SECRET|[^"]*_PASSWORD|[^"]*_TOKEN)"[[:space:]]*:[[:space:]]*")[^"]+"/\1[REDACTED]"/g' \
|
||||
"$config_file" > "$LOG_DIR/config/plugins/$filename" 2>/dev/null
|
||||
fi
|
||||
done
|
||||
fi
|
||||
|
||||
echo "Configuration files copied (sensitive values redacted)" >> "$LOG_DIR/container_info.txt"
|
||||
else
|
||||
echo "Config directory not found at $CONFIG_DIR" >> "$LOG_DIR/container_info.txt"
|
||||
fi
|
||||
|
||||
echo "--- HTTPBin ---" >> $LOG_DIR/network_info.txt
|
||||
curl -s https://httpbin.org/get >> $LOG_DIR/network_info.txt 2>&1
|
||||
echo "" >> $LOG_DIR/network_info.txt
|
||||
echo "--- HowsMySSL ---" >> $LOG_DIR/network_info.txt
|
||||
pyrequests https://www.howsmyssl.com/a/check >> $LOG_DIR/network_info.txt
|
||||
ehco ""
|
||||
curl -s https://www.howsmyssl.com/a/check >> $LOG_DIR/network_info.txt 2>&1
|
||||
echo "" >> $LOG_DIR/network_info.txt
|
||||
echo "--- IPInfo ---" >> $LOG_DIR/network_info.txt
|
||||
pyrequests https://ipinfo.io >> $LOG_DIR/network_info.txt
|
||||
ehco ""
|
||||
curl -s https://ipinfo.io >> $LOG_DIR/network_info.txt 2>&1
|
||||
echo "" >> $LOG_DIR/network_info.txt
|
||||
echo "--- Cloudflare Trace ---" >> $LOG_DIR/network_info.txt
|
||||
pyrequests https://1.1.1.1/cdn-cgi/trace >> $LOG_DIR/network_info.txt
|
||||
curl -s https://1.1.1.1/cdn-cgi/trace >> $LOG_DIR/network_info.txt 2>&1
|
||||
|
||||
# Copy Tor logs if they exist
|
||||
if [ -f "/var/log/tor/notices.log" ]; then
|
||||
cp "/var/log/tor/notices.log" "$LOG_DIR/tor_notices.log"
|
||||
fi
|
||||
|
||||
# Copy Supervisor logs if they exist
|
||||
if [ -d "/var/log/supervisor" ]; then
|
||||
cp -rf "/var/log/supervisor/" "$LOG_DIR/supervisor/"
|
||||
fi
|
||||
|
||||
# Create the zip file directly from LOG_DIR
|
||||
ln -s "$LOG_DIR" /tmp/$OUTPUT_FILE_NAME
|
||||
|
||||
@@ -1,131 +0,0 @@
|
||||
"""Data structures and models used across the application."""
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Dict, List, Optional
|
||||
from enum import Enum
|
||||
from datetime import datetime, timedelta
|
||||
from threading import Lock
|
||||
from pathlib import Path
|
||||
from env import INGEST_DIR, STATUS_TIMEOUT
|
||||
|
||||
class QueueStatus(str, Enum):
|
||||
"""Enum for possible book queue statuses."""
|
||||
QUEUED = "queued"
|
||||
DOWNLOADING = "downloading"
|
||||
AVAILABLE = "available"
|
||||
ERROR = "error"
|
||||
DONE = "done"
|
||||
|
||||
@dataclass
|
||||
class BookInfo:
|
||||
"""Data class representing book information."""
|
||||
id: str
|
||||
title: str
|
||||
preview: Optional[str] = None
|
||||
author: Optional[str] = None
|
||||
publisher: Optional[str] = None
|
||||
year: Optional[str] = None
|
||||
language: Optional[str] = None
|
||||
format: Optional[str] = None
|
||||
size: Optional[str] = None
|
||||
info: Optional[Dict[str, List[str]]] = None
|
||||
download_urls: List[str] = field(default_factory=list)
|
||||
download_path: Optional[str] = None
|
||||
|
||||
class BookQueue:
|
||||
"""Thread-safe book queue manager."""
|
||||
def __init__(self) -> None:
|
||||
self._queue: set[str] = set()
|
||||
self._lock = Lock()
|
||||
self._status: dict[str, QueueStatus] = {}
|
||||
self._book_data: dict[str, BookInfo]= {}
|
||||
self._status_timestamps: dict[str, datetime] = {} # Track when each status was last updated
|
||||
self._status_timeout = timedelta(seconds=STATUS_TIMEOUT) # 1 hour timeout
|
||||
|
||||
def add(self, book_id: str, book_data: BookInfo) -> None:
|
||||
"""Add a book to the queue."""
|
||||
with self._lock:
|
||||
self._queue.add(book_id)
|
||||
self._book_data[book_id] = book_data
|
||||
self._update_status(book_id, QueueStatus.QUEUED)
|
||||
|
||||
def get_next(self) -> Optional[str]:
|
||||
"""Get next book ID from queue."""
|
||||
with self._lock:
|
||||
return self._queue.pop() if self._queue else None
|
||||
|
||||
def _update_status(self, book_id: str, status: QueueStatus) -> None:
|
||||
"""Internal method to update status and timestamp."""
|
||||
self._status[book_id] = status
|
||||
self._status_timestamps[book_id] = datetime.now()
|
||||
|
||||
def update_status(self, book_id: str, status: QueueStatus) -> None:
|
||||
"""Update status of a book in the queue."""
|
||||
with self._lock:
|
||||
self._update_status(book_id, status)
|
||||
|
||||
def update_download_path(self, book_id: str, download_path: str) -> None:
|
||||
"""Update the download path of a book in the queue."""
|
||||
with self._lock:
|
||||
self._book_data[book_id].download_path = download_path
|
||||
|
||||
def get_status(self) -> Dict[QueueStatus, Dict[str, BookInfo]]:
|
||||
"""Get current queue status."""
|
||||
self.refresh()
|
||||
with self._lock:
|
||||
result: Dict[QueueStatus, Dict[str, BookInfo]] = {status: {} for status in QueueStatus}
|
||||
for book_id, status in self._status.items():
|
||||
if book_id in self._book_data:
|
||||
result[status][book_id] = self._book_data[book_id]
|
||||
return result
|
||||
|
||||
def refresh(self) -> None:
|
||||
"""Remove any books that are done downloading or have stale status."""
|
||||
with self._lock:
|
||||
current_time = datetime.now()
|
||||
|
||||
# Create a list of items to remove to avoid modifying dict during iteration
|
||||
to_remove = []
|
||||
|
||||
for book_id, status in self._status.items():
|
||||
path = self._book_data[book_id].download_path
|
||||
if path and not Path(path).exists():
|
||||
self._book_data[book_id].download_path = None
|
||||
path = None
|
||||
|
||||
# Check for completed downloads
|
||||
if status == QueueStatus.AVAILABLE:
|
||||
if not path:
|
||||
self._update_status(book_id, QueueStatus.DONE)
|
||||
|
||||
# Check for stale status entries
|
||||
last_update = self._status_timestamps.get(book_id)
|
||||
if last_update and (current_time - last_update) > self._status_timeout:
|
||||
if status == QueueStatus.DONE or status == QueueStatus.ERROR or status == QueueStatus.AVAILABLE:
|
||||
to_remove.append(book_id)
|
||||
|
||||
# Remove stale entries
|
||||
for book_id in to_remove:
|
||||
del self._status[book_id]
|
||||
del self._status_timestamps[book_id]
|
||||
if book_id in self._book_data:
|
||||
del self._book_data[book_id]
|
||||
|
||||
def set_status_timeout(self, hours: int) -> None:
|
||||
"""Set the status timeout duration in hours."""
|
||||
with self._lock:
|
||||
self._status_timeout = timedelta(hours=hours)
|
||||
|
||||
|
||||
# Global instance of BookQueue
|
||||
book_queue = BookQueue()
|
||||
|
||||
@dataclass
|
||||
class SearchFilters:
|
||||
isbn: Optional[List[str]] = None
|
||||
author: Optional[List[str]] = None
|
||||
title: Optional[List[str]] = None
|
||||
lang: Optional[List[str]] = None
|
||||
sort: Optional[str] = None
|
||||
content: Optional[List[str]] = None
|
||||
format: Optional[List[str]] = None
|
||||
@@ -1,350 +0,0 @@
|
||||
"""Network operations manager for the book downloader application."""
|
||||
|
||||
import requests
|
||||
import urllib.request
|
||||
from typing import Sequence, Tuple, Any, Union, cast, List, Optional, Callable
|
||||
import socket
|
||||
import dns.resolver
|
||||
from socket import AddressFamily, SocketKind
|
||||
import urllib.parse
|
||||
import ssl
|
||||
import ipaddress
|
||||
|
||||
from logger import setup_logger
|
||||
from config import PROXIES, AA_BASE_URL, CUSTOM_DNS, AA_AVAILABLE_URLS, DOH_SERVER
|
||||
import config
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
# Common helper functions for DNS resolution
|
||||
def _decode_host(host: Union[str, bytes, None]) -> str:
|
||||
"""Convert host to string, handling bytes and None cases."""
|
||||
if host is None:
|
||||
return ""
|
||||
if isinstance(host, bytes):
|
||||
return host.decode('utf-8')
|
||||
return str(host)
|
||||
|
||||
def _decode_port(port: Union[str, bytes, int, None]) -> int:
|
||||
"""Convert port to integer, handling various input types."""
|
||||
if port is None:
|
||||
return 0
|
||||
if isinstance(port, (str, bytes)):
|
||||
return int(port)
|
||||
return int(port)
|
||||
|
||||
def _is_local_address(host_str: str) -> bool:
|
||||
"""Check if an address is local and should bypass custom DNS."""
|
||||
"""Check if an address is local or private and should bypass custom DNS."""
|
||||
# Localhost checks
|
||||
if (host_str == 'localhost' or
|
||||
host_str.startswith('127.') or
|
||||
host_str == '::1' or
|
||||
host_str == '0.0.0.0'):
|
||||
return True
|
||||
|
||||
# IPv4 private ranges (RFC 1918)
|
||||
if (host_str.startswith('10.') or
|
||||
(host_str.startswith('172.') and
|
||||
len(host_str.split('.')) > 1 and
|
||||
16 <= int(host_str.split('.')[1]) <= 31) or
|
||||
host_str.startswith('192.168.')):
|
||||
return True
|
||||
|
||||
# IPv6 private ranges
|
||||
if (host_str.startswith('fc') or
|
||||
host_str.startswith('fd') or # Unique local addresses (fc00::/7)
|
||||
host_str.startswith('fe80:')): # Link-local addresses (fe80::/10)
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
def _is_ip_address(host_str: str) -> bool:
|
||||
"""Check if a string is a valid IP address (IPv4 or IPv6)."""
|
||||
try:
|
||||
ipaddress.ip_address(host_str)
|
||||
return True
|
||||
except ValueError:
|
||||
return False
|
||||
|
||||
# Store the original getaddrinfo function
|
||||
original_getaddrinfo = socket.getaddrinfo
|
||||
|
||||
class DoHResolver:
|
||||
"""DNS over HTTPS resolver implementation."""
|
||||
def __init__(self, provider_url: str, hostname: str, ip: str):
|
||||
"""Initialize DoH resolver with specified provider."""
|
||||
self.base_url = provider_url.lower().strip()
|
||||
self.hostname = hostname # Store the hostname for hostname-based skipping
|
||||
self.ip = ip # Store IP for direct connections
|
||||
self.session = requests.Session()
|
||||
|
||||
# Different headers based on provider
|
||||
if 'google' in self.base_url:
|
||||
self.session.headers.update({
|
||||
'Accept': 'application/json',
|
||||
})
|
||||
else:
|
||||
self.session.headers.update({
|
||||
'Accept': 'application/dns-json',
|
||||
})
|
||||
|
||||
def resolve(self, hostname: str, record_type: str) -> List[str]:
|
||||
"""Resolve a hostname using DoH.
|
||||
|
||||
Args:
|
||||
hostname: The hostname to resolve
|
||||
record_type: The DNS record type (A or AAAA)
|
||||
|
||||
Returns:
|
||||
List of resolved IP addresses
|
||||
"""
|
||||
# Check if hostname is already an IP address, no need to resolve
|
||||
if _is_ip_address(hostname):
|
||||
logger.debug(f"Skipping DoH resolution for IP address: {hostname}")
|
||||
return [hostname]
|
||||
|
||||
# Check if hostname is a private IP address, and skip DoH if it is
|
||||
if _is_local_address(hostname):
|
||||
logger.debug(f"Skipping DoH resolution for private IP: {hostname}")
|
||||
return [hostname]
|
||||
|
||||
# Skip resolution for the DoH server itself to prevent recursion
|
||||
if hostname == self.hostname:
|
||||
logger.debug(f"Skipping DoH resolution for DoH server itself: {hostname}")
|
||||
return [self.ip]
|
||||
|
||||
try:
|
||||
params = {
|
||||
'name': hostname,
|
||||
'type': 'AAAA' if record_type == 'AAAA' else 'A'
|
||||
}
|
||||
|
||||
response = self.session.get(
|
||||
self.base_url,
|
||||
params=params,
|
||||
proxies=PROXIES,
|
||||
timeout=5
|
||||
)
|
||||
response.raise_for_status()
|
||||
|
||||
data = response.json()
|
||||
if 'Answer' not in data:
|
||||
logger.warning(f"DoH resolution failed for {hostname}: {data}")
|
||||
return []
|
||||
|
||||
# Extract IP addresses from the response
|
||||
answers = [answer['data'] for answer in data['Answer']
|
||||
if answer.get('type') == (28 if record_type == 'AAAA' else 1)]
|
||||
logger.debug(f"Resolved {hostname} to {len(answers)} addresses using DoH: {answers}")
|
||||
return answers
|
||||
|
||||
except Exception as e:
|
||||
logger.warning(f"DoH resolution failed for {hostname}: {e}")
|
||||
return []
|
||||
|
||||
def create_custom_resolver():
|
||||
"""Create a custom DNS resolver using the configured DNS servers."""
|
||||
custom_resolver = dns.resolver.Resolver()
|
||||
custom_resolver.nameservers = CUSTOM_DNS
|
||||
return custom_resolver
|
||||
|
||||
def resolve_with_custom_dns(resolver, hostname: str, record_type: str) -> List[str]:
|
||||
"""Resolve hostname using custom DNS resolver.
|
||||
|
||||
Args:
|
||||
resolver: The DNS resolver to use
|
||||
hostname: The hostname to resolve
|
||||
record_type: The DNS record type (A or AAAA)
|
||||
|
||||
Returns:
|
||||
List of resolved IP addresses
|
||||
"""
|
||||
try:
|
||||
answers = resolver.resolve(hostname, record_type)
|
||||
return [str(answer) for answer in answers]
|
||||
except Exception as e:
|
||||
logger.debug(f"{record_type} resolution failed for {hostname}: {e}")
|
||||
return []
|
||||
|
||||
def create_custom_getaddrinfo(
|
||||
resolve_ipv4: Callable[[str], List[str]],
|
||||
resolve_ipv6: Callable[[str], List[str]],
|
||||
skip_check: Optional[Callable[[str], bool]] = None
|
||||
):
|
||||
"""Create a custom getaddrinfo function that uses the provided resolvers.
|
||||
|
||||
Args:
|
||||
resolve_ipv4: Function to resolve IPv4 addresses
|
||||
resolve_ipv6: Function to resolve IPv6 addresses
|
||||
skip_check: Optional function to check if custom resolution should be skipped
|
||||
|
||||
Returns:
|
||||
A custom getaddrinfo function
|
||||
"""
|
||||
def custom_getaddrinfo(
|
||||
host: Union[str, bytes, None],
|
||||
port: Union[str, bytes, int, None],
|
||||
family: int = 0,
|
||||
type: int = 0,
|
||||
proto: int = 0,
|
||||
flags: int = 0
|
||||
) -> Sequence[Tuple[AddressFamily, SocketKind, int, str, Tuple[Any, ...]]]:
|
||||
host_str = _decode_host(host)
|
||||
port_int = _decode_port(port)
|
||||
|
||||
# Skip custom resolution for IP addresses, local addresses, or if skip check passes
|
||||
if _is_ip_address(host_str) or _is_local_address(host_str) or (skip_check and skip_check(host_str)):
|
||||
logger.debug(f"Using system DNS for IP address or local/private address: {host_str}")
|
||||
return original_getaddrinfo(host, port, family, type, proto, flags)
|
||||
|
||||
results: list[Tuple[AddressFamily, SocketKind, int, str, Tuple[Any, ...]]] = []
|
||||
|
||||
try:
|
||||
# Try IPv6 first if family allows it
|
||||
if family == 0 or family == socket.AF_INET6:
|
||||
logger.debug(f"Resolving IPv6 address for {host_str}")
|
||||
ipv6_answers = resolve_ipv6(host_str)
|
||||
for answer in ipv6_answers:
|
||||
results.append((socket.AF_INET6, cast(SocketKind, type), proto, '', (answer, port_int, 0, 0)))
|
||||
if ipv6_answers:
|
||||
logger.debug(f"Found {len(ipv6_answers)} IPv6 addresses for {host_str}")
|
||||
|
||||
# Then try IPv4
|
||||
if family == 0 or family == socket.AF_INET:
|
||||
logger.debug(f"Resolving IPv4 address for {host_str}")
|
||||
ipv4_answers = resolve_ipv4(host_str)
|
||||
for answer in ipv4_answers:
|
||||
results.append((socket.AF_INET, cast(SocketKind, type), proto, '', (answer, port_int)))
|
||||
if ipv4_answers:
|
||||
logger.debug(f"Found {len(ipv4_answers)} IPv4 addresses for {host_str}")
|
||||
|
||||
if results:
|
||||
logger.debug(f"Resolved {host_str} to {len(results)} addresses")
|
||||
return results
|
||||
|
||||
except Exception as e:
|
||||
logger.warning(f"Custom DNS resolution failed for {host_str}: {e}, falling back to system DNS")
|
||||
|
||||
# Fall back to system DNS if custom resolution fails
|
||||
try:
|
||||
return original_getaddrinfo(host, port, family, type, proto, flags)
|
||||
except Exception as e:
|
||||
logger.error(f"System DNS resolution also failed for {host_str}: {e}")
|
||||
# Last resort: Try to connect to the hostname directly
|
||||
if family == 0 or family == socket.AF_INET:
|
||||
logger.warning(f"Using direct hostname as last resort for {host_str}")
|
||||
return [(socket.AF_INET, cast(SocketKind, type), proto, '', (host_str, port_int))]
|
||||
else:
|
||||
raise # Re-raise the exception if we can't provide a last resort
|
||||
|
||||
return custom_getaddrinfo
|
||||
|
||||
def init_doh_resolver(doh_server: str = DOH_SERVER):
|
||||
"""Initialize DNS over HTTPS resolver.
|
||||
|
||||
Args:
|
||||
doh_server: The DoH server URL
|
||||
"""
|
||||
# Pre-resolve the DoH server hostname to prevent recursion
|
||||
url = urllib.parse.urlparse(doh_server)
|
||||
server_hostname = url.hostname if url.hostname else ''
|
||||
|
||||
# Use system DNS for DoH server to prevent circular dependencies
|
||||
try:
|
||||
# Temporarily restore original getaddrinfo to resolve DoH server
|
||||
temp_getaddrinfo = socket.getaddrinfo
|
||||
socket.getaddrinfo = original_getaddrinfo
|
||||
|
||||
server_ip = socket.gethostbyname(server_hostname)
|
||||
logger.info(f"DoH server {server_hostname} resolved to IP: {server_ip}")
|
||||
|
||||
# Restore custom getaddrinfo if it was previously set
|
||||
socket.getaddrinfo = temp_getaddrinfo
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to resolve DoH server {server_hostname}: {e}")
|
||||
# Fall back to a known public DNS if resolution fails
|
||||
server_ip = "1.1.1.1"
|
||||
logger.info(f"Using fallback IP for DoH server: {server_ip}")
|
||||
|
||||
# Create DoH resolver
|
||||
doh_resolver = DoHResolver(doh_server, server_hostname, server_ip)
|
||||
|
||||
# Create resolver functions
|
||||
def resolve_ipv4(hostname: str) -> List[str]:
|
||||
return doh_resolver.resolve(hostname, 'A')
|
||||
|
||||
def resolve_ipv6(hostname: str) -> List[str]:
|
||||
return doh_resolver.resolve(hostname, 'AAAA')
|
||||
|
||||
# Skip DoH resolution for the DoH server itself, IP addresses, and private addresses
|
||||
def skip_doh(hostname: str) -> bool:
|
||||
return (hostname == server_hostname or
|
||||
hostname == server_ip or
|
||||
_is_ip_address(hostname) or
|
||||
_is_local_address(hostname))
|
||||
|
||||
# Replace socket.getaddrinfo with our DoH-enabled version
|
||||
socket.getaddrinfo = cast(Any, create_custom_getaddrinfo(
|
||||
resolve_ipv4, resolve_ipv6, skip_doh
|
||||
))
|
||||
|
||||
logger.info("DoH resolver successfully configured and activated")
|
||||
return doh_resolver
|
||||
|
||||
def init_custom_resolver():
|
||||
"""Initialize custom DNS resolver using configured DNS servers."""
|
||||
custom_resolver = create_custom_resolver()
|
||||
|
||||
# Create resolver functions
|
||||
def resolve_ipv4(hostname: str) -> List[str]:
|
||||
return resolve_with_custom_dns(custom_resolver, hostname, 'A')
|
||||
|
||||
def resolve_ipv6(hostname: str) -> List[str]:
|
||||
return resolve_with_custom_dns(custom_resolver, hostname, 'AAAA')
|
||||
|
||||
# Replace socket.getaddrinfo with our custom resolver
|
||||
socket.getaddrinfo = cast(Any, create_custom_getaddrinfo(resolve_ipv4, resolve_ipv6))
|
||||
|
||||
logger.info("Custom DNS resolver successfully configured and activated")
|
||||
return custom_resolver
|
||||
|
||||
# Initialize DNS resolvers based on configuration
|
||||
def init_dns_resolvers():
|
||||
"""Initialize DNS resolvers based on configuration."""
|
||||
if len(CUSTOM_DNS) > 0:
|
||||
init_custom_resolver()
|
||||
if DOH_SERVER:
|
||||
init_doh_resolver()
|
||||
|
||||
# Initialize DNS resolvers
|
||||
init_dns_resolvers()
|
||||
|
||||
# Check available AA_BASE_URLs if set to auto
|
||||
if AA_BASE_URL == "auto":
|
||||
logger.info(f"AA_BASE_URL: auto, checking available urls {AA_AVAILABLE_URLS}")
|
||||
for url in AA_AVAILABLE_URLS:
|
||||
try:
|
||||
response = requests.get(url, proxies=PROXIES)
|
||||
if response.status_code == 200:
|
||||
AA_BASE_URL = url
|
||||
break
|
||||
except Exception as e:
|
||||
logger.error_trace(f"Error checking {url}: {e}")
|
||||
if AA_BASE_URL == "auto":
|
||||
AA_BASE_URL = AA_AVAILABLE_URLS[0]
|
||||
config.AA_BASE_URL = AA_BASE_URL
|
||||
logger.info(f"AA_BASE_URL: {AA_BASE_URL}")
|
||||
|
||||
# Configure urllib opener with appropriate headers
|
||||
opener = urllib.request.build_opener()
|
||||
opener.addheaders = [
|
||||
('User-agent', 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) '
|
||||
'AppleWebKit/537.36 (KHTML, like Gecko) '
|
||||
'Chrome/129.0.0.0 Safari/537.3')
|
||||
]
|
||||
urllib.request.install_opener(opener)
|
||||
|
||||
# Need an empty function to be called by downloader.py
|
||||
def init():
|
||||
pass
|
||||
@@ -0,0 +1,26 @@
|
||||
[project]
|
||||
name = "shelfmark"
|
||||
version = "0.1.0"
|
||||
description = "Shelfmark - Book Downloader"
|
||||
requires-python = ">=3.10"
|
||||
|
||||
[tool.pytest.ini_options]
|
||||
testpaths = ["tests"]
|
||||
python_files = ["test_*.py", "*_test.py"]
|
||||
python_classes = ["Test*"]
|
||||
python_functions = ["test_*"]
|
||||
addopts = [
|
||||
"-v",
|
||||
"--tb=short",
|
||||
]
|
||||
markers = [
|
||||
"integration: marks tests that require running services (deselect with '-m \"not integration\"')",
|
||||
"slow: marks tests as slow (deselect with '-m \"not slow\"')",
|
||||
"e2e: marks end-to-end tests that require the full application stack",
|
||||
]
|
||||
|
||||
[tool.mypy]
|
||||
python_version = "3.10"
|
||||
warn_return_any = true
|
||||
warn_unused_ignores = true
|
||||
ignore_missing_imports = true
|
||||
@@ -1,255 +1,250 @@
|
||||
# 📚 Calibre-Web-Automated-Book-Downloader
|
||||
# 📚 Shelfmark: Book Downloader
|
||||
|
||||

|
||||
Formerly *Calibre Web Automated Book Downloader (CWABD)*
|
||||
|
||||
An intuitive web interface for searching and requesting book downloads, designed to work seamlessly with [Calibre-Web-Automated](https://github.com/crocodilestick/Calibre-Web-Automated). This project streamlines the process of downloading books and preparing them for integration into your Calibre library.
|
||||
<img src="src/frontend/public/logo.png" alt="Shelfmark" width="200">
|
||||
|
||||
Shelfmark is a unified web interface for searching and downloading books and audiobooks from multiple sources - all in one place. Works out of the box with popular web sources, no configuration required. Add metadata providers, additional release sources, and download clients to create a single hub for building your digital library.
|
||||
|
||||
**Fully standalone** - no external dependencies required. Works great alongside library tools like [Calibre-Web-Automated](https://github.com/crocodilestick/Calibre-Web-Automated), [Booklore](https://github.com/booklore-app/booklore) or [Audiobookshelf](https://github.com/advplyr/audiobookshelf) for automatic import.
|
||||
|
||||
## ✨ Features
|
||||
|
||||
- 🌐 User-friendly web interface for book search and download
|
||||
- 🔄 Automated download to your specified ingest folder
|
||||
- 🔌 Seamless integration with Calibre-Web-Automated
|
||||
- 📖 Support for multiple book formats (epub, mobi, azw3, fb2, djvu, cbz, cbr)
|
||||
- 🛡️ Cloudflare bypass capability for reliable downloads
|
||||
- 🐳 Docker-based deployment for quick setup
|
||||
- **One-Stop Interface** - A clean, modern UI to search, browse, and download from multiple sources in one place
|
||||
- **Multiple sources** - Popular archive websites, Torrent, Usenet and IRC download support
|
||||
- **Audiobook support** - Full audiobook search and download with dedicated processing
|
||||
- **Real-Time Progress** - Unified download queue with live status updates across all sources
|
||||
- **Two Search Modes**:
|
||||
- **Direct** - Search and download books from popular web sources
|
||||
- **Universal** - Search metadata providers (Hardcover, Open Library) for richer book and audiobook discovery, with multi-source downloads
|
||||
- **Cloudflare Bypass** - Built-in bypasser for reliable access to protected sources
|
||||
|
||||
## 🖼️ Screenshots
|
||||
|
||||

|
||||
**Home screen**
|
||||

|
||||
|
||||

|
||||
**Search results**
|
||||

|
||||
|
||||

|
||||
**Multi-source downloads**
|
||||

|
||||
|
||||
**Download queue**
|
||||

|
||||
|
||||
## 🚀 Quick Start
|
||||
|
||||
### Prerequisites
|
||||
|
||||
- Docker
|
||||
- Docker Compose
|
||||
- A running instance of [Calibre-Web-Automated](https://github.com/crocodilestick/Calibre-Web-Automated) (recommended)
|
||||
- Docker & Docker Compose
|
||||
|
||||
### Installation Steps
|
||||
|
||||
1. Get the docker-compose.yml:
|
||||
### Installation
|
||||
|
||||
1. Download the docker-compose file:
|
||||
```bash
|
||||
curl -O https://raw.githubusercontent.com/calibrain/calibre-web-automated-book-downloader/refs/heads/main/docker-compose.yml
|
||||
curl -O https://raw.githubusercontent.com/calibrain/shelfmark/main/compose/stable/docker-compose.yml
|
||||
```
|
||||
|
||||
2. Start the service:
|
||||
|
||||
```bash
|
||||
docker compose up -d
|
||||
```
|
||||
|
||||
3. Access the web interface at `http://localhost:8084`
|
||||
> **Edge users**: If you're tracking the main branch (`:dev` tag), use compose files from `compose/edge/` instead.
|
||||
|
||||
## ⚙️ Configuration
|
||||
3. Open `http://localhost:8084`
|
||||
|
||||
### Environment Variables
|
||||
That's it! Configure settings through the web interface as needed.
|
||||
|
||||
#### Application Settings
|
||||
|
||||
| Variable | Description | Default Value |
|
||||
| ----------------- | ----------------------- | ------------------ |
|
||||
| `FLASK_PORT` | Web interface port | `8084` |
|
||||
| `FLASK_HOST` | Web interface binding | `0.0.0.0` |
|
||||
| `DEBUG` | Debug mode toggle | `false` |
|
||||
| `INGEST_DIR` | Book download directory | `/cwa-book-ingest` |
|
||||
| `TZ` | Container timezone | `UTC` |
|
||||
| `UID` | Runtime user ID | `1000` |
|
||||
| `GID` | Runtime group ID | `100` |
|
||||
| `CWA_DB_PATH` | Calibre-Web's database | None |
|
||||
| `ENABLE_LOGGING` | Enable log file | `true` |
|
||||
| `LOG_LEVEL` | Log level to use | `info` |
|
||||
|
||||
If you wish to enable authentication, you must set `CWA_DB_PATH` to point to Calibre-Web's `app.db`, in order to match the username and password.
|
||||
|
||||
If logging is enabld, log folder default location is `/var/log/cwa-book-downloader`
|
||||
Available log levels: `DEBUG`, `INFO`, `WARNING`, `ERROR`, `CRITICAL`. Higher levels show fewer messages.
|
||||
|
||||
Note that if using TOR, the TZ will be calculated automatically based on IP.
|
||||
|
||||
#### Download Settings
|
||||
|
||||
| Variable | Description | Default Value |
|
||||
| ---------------------- | --------------------------------------------------------- | --------------------------------- |
|
||||
| `MAX_RETRY` | Maximum retry attempts | `3` |
|
||||
| `DEFAULT_SLEEP` | Retry delay (seconds) | `5` |
|
||||
| `MAIN_LOOP_SLEEP_TIME` | Processing loop delay (seconds) | `5` |
|
||||
| `SUPPORTED_FORMATS` | Supported book formats | `epub,mobi,azw3,fb2,djvu,cbz,cbr` |
|
||||
| `BOOK_LANGUAGE` | Preferred language for books | `en` |
|
||||
| `AA_DONATOR_KEY` | Optional Donator key for Anna's Archive fast download API | `` |
|
||||
| `USE_BOOK_TITLE` | Use book title as filename instead of ID | `false` |
|
||||
|
||||
If you change `BOOK_LANGUAGE`, you can add multiple comma separated languages, such as `en,fr,ru` etc.
|
||||
|
||||
#### AA
|
||||
|
||||
| Variable | Description | Default Value |
|
||||
| ---------------------- | --------------------------------------------------------- | --------------------------------- |
|
||||
| `AA_BASE_URL` | Base URL of Annas-Archive (could be changed for a proxy) | `https://annas-archive.org` |
|
||||
| `USE_CF_BYPASS` | Disable CF bypass and use alternative links instead | `true` |
|
||||
|
||||
If you are a donator on AA, you can use your Key in `AA_DONATOR_KEY` to speed up downloads and bypass the wait times.
|
||||
If disabling the cloudflare bypass, you will be using alternative download hosts, such as libgen or z-lib, but they usually have a delay before getting the more recent books and their collection is not as big as aa's. But this setting should work for the majority of books.
|
||||
|
||||
#### Network Settings
|
||||
|
||||
| Variable | Description | Default Value |
|
||||
| ---------------------- | ------------------------------- | ----------------------- |
|
||||
| `AA_ADDITIONAL_URLS` | Proxy URLs for AA (, separated) | `` |
|
||||
| `HTTP_PROXY` | HTTP proxy URL | `` |
|
||||
| `HTTPS_PROXY` | HTTPS proxy URL | `` |
|
||||
| `CUSTOM_DNS` | Custom DNS IP | `` |
|
||||
| `USE_DOH` | Use DNS over HTTPS | `false` |
|
||||
|
||||
For proxy configuration, you can specify URLs in the following format:
|
||||
```bash
|
||||
# Basic proxy
|
||||
HTTP_PROXY=http://proxy.example.com:8080
|
||||
HTTPS_PROXY=http://proxy.example.com:8080
|
||||
|
||||
# Proxy with authentication
|
||||
HTTP_PROXY=http://username:password@proxy.example.com:8080
|
||||
HTTPS_PROXY=http://username:password@proxy.example.com:8080
|
||||
```
|
||||
|
||||
|
||||
The `CUSTOM_DNS` setting supports two formats:
|
||||
|
||||
1. **Custom DNS Servers**: A comma-separated list of DNS server IP addresses
|
||||
- Example: `127.0.0.53,127.0.1.53` (useful for PiHole)
|
||||
- Supports both IPv4 and IPv6 addresses in the same string
|
||||
|
||||
2. **Preset DNS Providers**: Use one of these predefined options:
|
||||
- `google` - Google DNS
|
||||
- `quad9` - Quad9 DNS
|
||||
- `cloudflare` - Cloudflare DNS
|
||||
- `opendns` - OpenDNS
|
||||
|
||||
For users experiencing ISP-level website blocks (such as Virgin Media in the UK), using alternative DNS providers like Cloudflare may help bypass these restrictions
|
||||
|
||||
If a `CUSTOM_DNS` is specified from the preset providers, you can also set a `USE_DOH=true` to force using DNS over HTTPS,
|
||||
which might also help in certain network situations. Note that only `google`, `quad9`, `cloudflare` and `opendns` are
|
||||
supported for now, and any other value in `CUSTOM_DNS` will make the `USE_DOH` flag ignored.
|
||||
|
||||
Try something like this :
|
||||
```bash
|
||||
CUSTOM_DNS=cloudflare
|
||||
USE_DOH=true
|
||||
```
|
||||
|
||||
#### Custom configuration
|
||||
|
||||
| Variable | Description | Default Value |
|
||||
| ---------------------- | ----------------------------------------------------------- | ----------------------- |
|
||||
| `CUSTOM_SCRIPT` | Path to an executable script that tuns after each download | `` |
|
||||
|
||||
If `CUSTOM_SCRIPT` is set, it will be executed after each successful download but before the file is moved to the ingest directory. This allows for custom processing like format conversion or validation.
|
||||
|
||||
The script is called with the full path of the downloaded file as its argument. Important notes:
|
||||
- The script must preserve the original filename for proper processing
|
||||
- The file can be modified or even deleted if needed
|
||||
- The file will be moved to `/cwa-book-ingest` after the script execution (if not deleted)
|
||||
|
||||
You can specify these configuration in this format :
|
||||
```
|
||||
environment:
|
||||
- CUSTOM_SCRIPT=/scripts/process-book.sh
|
||||
|
||||
volumes:
|
||||
- local/scripts/custom_script.sh:/scripts/process-book.sh
|
||||
```
|
||||
|
||||
### Volume Configuration
|
||||
### Volume Setup
|
||||
|
||||
```yaml
|
||||
volumes:
|
||||
- /your/local/path:/cwa-book-ingest
|
||||
- /cwa/config/path/app.db:/auth/app.db:ro
|
||||
```
|
||||
**Note** - If your library volume is on a cifs share, you will get a "database locked" error until you add **nobrl** to your mount line in your fstab file. e.g. //192.168.1.1/Books /media/books cifs credentials=.smbcredentials,uid=1000,gid=1000,iocharset=utf8,**nobrl** - See https://github.com/crocodilestick/Calibre-Web-Automated/issues/64#issuecomment-2712769777
|
||||
|
||||
Mount should align with your Calibre-Web-Automated ingest folder.
|
||||
|
||||
## 🧅 Tor Variant
|
||||
|
||||
This application also offers a variant that routes all its traffic through the Tor network. This can be useful for enhanced privacy or bypassing network restrictions.
|
||||
|
||||
To use the Tor variant:
|
||||
|
||||
1. Get the Tor-specific docker-compose file:
|
||||
```bash
|
||||
curl -O https://raw.githubusercontent.com/calibrain/calibre-web-automated-book-downloader/refs/heads/main/docker-compose.tor.yml
|
||||
```
|
||||
2. Start the service using this file:
|
||||
```bash
|
||||
docker compose -f docker-compose.tor.yml up -d
|
||||
```
|
||||
|
||||
**Important Considerations for Tor:**
|
||||
|
||||
* **Capabilities:** This variant requires the `NET_ADMIN` and `NET_RAW` Docker capabilities to configure `iptables` for transparent Tor proxying.
|
||||
* **Timezone:** When running in Tor mode, the container will attempt to determine the timezone based on the Tor exit node's IP address and set it automatically. This will override the `TZ` environment variable if it is set.
|
||||
* **Network Settings:** Custom DNS, DoH, and HTTP(S) proxy settings (`CUSTOM_DNS`, `USE_DOH`, `HTTP_PROXY`, `HTTPS_PROXY`) are ignored when using the Tor variant, as all traffic goes through Tor.
|
||||
|
||||
## 🏗️ Architecture
|
||||
|
||||
The application consists of a single service:
|
||||
|
||||
1. **calibre-web-automated-bookdownloader**: Main application providing web interface and download functionality
|
||||
|
||||
## 🏥 Health Monitoring
|
||||
|
||||
Built-in health checks monitor:
|
||||
|
||||
- Web interface availability
|
||||
- Download service status
|
||||
- Cloudflare bypass service connection
|
||||
|
||||
Checks run every 30 seconds with a 30-second timeout and 3 retries.
|
||||
You can enable by adding this to your compose :
|
||||
```
|
||||
HEALTHCHECK --interval=30s --timeout=30s --start-period=5s --retries=3 \
|
||||
CMD pyrequests http://localhost:8084/request/api/status || exit 1
|
||||
- /your/config/path:/config # Config, database, and artwork cache directory
|
||||
- /your/download/path:/books # Downloaded books
|
||||
- /client/path:/client/path # Optional: For Torrent/Usenet downloads, match your client directory exactly.
|
||||
```
|
||||
|
||||
## 📝 Logging
|
||||
> **Tip**: Point the download volume to your CWA or Booklore ingest folder for automatic import.
|
||||
|
||||
Logs are available in:
|
||||
> **Note**: CIFS shares require `nobrl` mount option to avoid database lock errors.
|
||||
|
||||
- Container: `/var/logs/cwa-book-downloader.log`
|
||||
- Docker logs: Access via `docker logs`
|
||||
## ⚙️ Configuration
|
||||
|
||||
## 🤝 Contributing
|
||||
### Search Modes
|
||||
|
||||
Contributions are welcome! Feel free to submit a Pull Request.
|
||||
**Direct** (default)
|
||||
- Works out of the box, no setup required
|
||||
- Searches a huge library of books directly
|
||||
- Returns downloadable releases immediately
|
||||
|
||||
## 📄 License
|
||||
**Universal**
|
||||
- Cleaner search results via metadata providers (Hardcover is recommended)
|
||||
- Aggregates releases from multiple configured sources
|
||||
- Full Audiobook support
|
||||
- Requires manual setup (API keys, additional sources)
|
||||
|
||||
This project is licensed under the MIT License - see the [LICENSE](LICENSE) file for details.
|
||||
### Environment Variables
|
||||
|
||||
## ⚠️ Important Disclaimers
|
||||
Environment variables work for initial setup and Docker deployments. They serve as defaults that can be overridden in the web interface.
|
||||
|
||||
| Variable | Description | Default |
|
||||
|----------|-------------|---------|
|
||||
| `FLASK_PORT` | Web interface port | `8084` |
|
||||
| `INGEST_DIR` | Book download directory | `/books` |
|
||||
| `TZ` | Container timezone | `UTC` |
|
||||
| `PUID` / `PGID` | Runtime user/group ID (also supports legacy `UID`/`GID`) | `1000` / `1000` |
|
||||
| `SEARCH_MODE` | `direct` or `universal` | `direct` |
|
||||
| `USING_TOR` | Enable Tor routing (requires `NET_ADMIN` capability) | `false` |
|
||||
|
||||
Some of the additional options available in Settings:
|
||||
- **AA Donator Key** - Use your paid account to skip Cloudflare challenges entirely and use faster, direct downloads
|
||||
- **Prowlarr** - Configure indexers and download clients to download books and audiobooks
|
||||
- **IRC** - Add details for IRC book sources and download directly from the UI
|
||||
- **Library Link** - Add a link to your Calibre-Web or Booklore instance in the UI header
|
||||
- **File processing** - Customiseable download paths, file renaming and directory creation with template-based renaming
|
||||
- **Network Resilience** - Auto DNS rotation and mirror fallback when sources are unreachable. Custom proxy support (SOCK5 + HTTP/S), Tor routing.
|
||||
- **Format & Language** - Filter downloads by preferred formats, languages and sorting order
|
||||
- **Metadata Providers** - Configure API keys for Hardcover, Open Library, etc.
|
||||
|
||||
## 🐳 Docker Variants
|
||||
|
||||
### Standard
|
||||
```bash
|
||||
docker compose up -d
|
||||
```
|
||||
|
||||
The full-featured image with built-in Cloudflare bypass.
|
||||
|
||||
#### Enable Tor Routing
|
||||
Routes all traffic through Tor for enhanced privacy:
|
||||
```bash
|
||||
curl -O https://raw.githubusercontent.com/calibrain/shelfmark/main/compose/stable/docker-compose.tor.yml
|
||||
docker compose -f docker-compose.tor.yml up -d
|
||||
```
|
||||
|
||||
**Notes:**
|
||||
- Requires `NET_ADMIN` and `NET_RAW` capabilities
|
||||
- Timezone is auto-detected from Tor exit node
|
||||
- Custom DNS/proxy settings are ignored when Tor is active
|
||||
|
||||
### Lite
|
||||
A smaller image without the built-in Cloudflare bypasser. Ideal for:
|
||||
|
||||
- **External bypassers** - Already running FlareSolverr or ByParr for other services
|
||||
- **Fast downloads** - Using fast download sources
|
||||
- **Alternative sources only** - Exclusively using Prowlarr, IRC, or other sources
|
||||
- **Audiobooks** - Using Shelfmark exclusively for audiobooks
|
||||
|
||||
```bash
|
||||
curl -O https://raw.githubusercontent.com/calibrain/shelfmark/main/compose/stable/docker-compose.lite.yml
|
||||
docker compose -f docker-compose.lite.yml up -d
|
||||
```
|
||||
|
||||
If you need Cloudflare bypass with the Lite image, configure an external resolver (FlareSolverr/ByParr) in Settings under the Cloudflare tab.
|
||||
|
||||
## 🔐 Authentication
|
||||
|
||||
Authentication is optional but recommended for shared or exposed instances. Enable in Settings.
|
||||
|
||||
**Alternative**: If you're running Calibre-Web, you can reuse its user database by mounting it:
|
||||
|
||||
```yaml
|
||||
volumes:
|
||||
- /path/to/calibre-web/app.db:/auth/app.db:ro
|
||||
```
|
||||
|
||||
## Health Monitoring
|
||||
|
||||
The application exposes a health endpoint at `/api/health` (no authentication required). Add a health check to your compose:
|
||||
|
||||
```yaml
|
||||
healthcheck:
|
||||
test: ["CMD", "curl", "-sf", "http://localhost:8084/api/health"]
|
||||
interval: 30s
|
||||
timeout: 30s
|
||||
retries: 3
|
||||
```
|
||||
|
||||
## Logging
|
||||
|
||||
Logs are available via:
|
||||
- `docker logs <container-name>`
|
||||
- `/var/log/shelfmark/` inside the container (when `ENABLE_LOGGING=true`)
|
||||
|
||||
Log level is configurable via Settings or `LOG_LEVEL` environment variable.
|
||||
|
||||
## Development
|
||||
|
||||
```bash
|
||||
# Frontend development
|
||||
make install # Install dependencies
|
||||
make dev # Start Vite dev server (localhost:5173)
|
||||
make build # Production build
|
||||
make typecheck # TypeScript checks
|
||||
|
||||
# Backend (Docker)
|
||||
make up # Start backend via docker-compose.dev.yml
|
||||
make down # Stop services
|
||||
make refresh # Rebuild and restart
|
||||
make restart # Restart container
|
||||
```
|
||||
|
||||
The frontend dev server proxies to the backend on port 8084.
|
||||
|
||||
### Architecture
|
||||
|
||||
```
|
||||
┌─────────────────────────────────────────────────────────────┐
|
||||
│ Web Interface │
|
||||
│ (React + TypeScript + Vite) │
|
||||
├─────────────────────────────────────────────────────────────┤
|
||||
│ Flask Backend │
|
||||
│ (REST API + WebSocket) │
|
||||
├───────────────────┬─────────────────────┬───────────────────┤
|
||||
│ Metadata Providers│ Download Queue │ Cloudflare │
|
||||
│ │ & Orchestrator │ Bypass │
|
||||
├───────────────────┼─────────────────────┼───────────────────┤
|
||||
│ • Hardcover │ • Task scheduling │ • Internal │
|
||||
│ • Open Library │ • Progress tracking │ • External │
|
||||
│ │ • Retry logic │ (FlareSolverr) │
|
||||
├───────────────────┴─────────────────────┴───────────────────┤
|
||||
│ Release Sources │
|
||||
├─────────────────────────────────────────────────────────────┤
|
||||
│ • Direct Download (Anna's Archive → Libgen → Welib) │
|
||||
├─────────────────────────────────────────────────────────────┤
|
||||
│ Network Layer │
|
||||
├─────────────────────────────────────────────────────────────┤
|
||||
│ • Auto DNS rotation • Mirror failover • Resume support │
|
||||
└─────────────────────────────────────────────────────────────┘
|
||||
```
|
||||
|
||||
The backend uses a plugin architecture. Metadata providers and release sources register via decorators and are automatically discovered.
|
||||
|
||||
## Contributing
|
||||
|
||||
Contributions are welcome! Please file issues or submit pull requests on GitHub.
|
||||
|
||||
> **Note**: Additional release sources and download clients are under active development. Want to add support for your favorite source? Check out the plugin architecture above and submit a PR!
|
||||
|
||||
## License
|
||||
|
||||
MIT License - see [LICENSE](LICENSE) for details.
|
||||
|
||||
## ⚠️ Disclaimers
|
||||
|
||||
### Copyright Notice
|
||||
|
||||
While this tool can access various sources including those that might contain copyrighted material (e.g., Anna's Archive), it is designed for legitimate use only. Users are responsible for:
|
||||
|
||||
This tool can access various sources including those that might contain copyrighted material. Users are responsible for:
|
||||
- Ensuring they have the right to download requested materials
|
||||
- Respecting copyright laws and intellectual property rights
|
||||
- Using the tool in compliance with their local regulations
|
||||
|
||||
### Duplicate Downloads Warning
|
||||
### Library Integration
|
||||
|
||||
Please note that the current version:
|
||||
Downloads are written atomically (via intermediate `.crdownload` files) to prevent partial files from being ingested. However, if your library tool (CWA, Booklore, Calibre) is actively scanning or importing, there's a small chance of race conditions. If you experience database errors or import failures, try pausing your library's auto-import during bulk downloads.
|
||||
|
||||
- Does not check for existing files in the download directory
|
||||
- Does not verify if books already exist in your Calibre database
|
||||
- Exercise caution when requesting multiple books to avoid duplicates
|
||||
|
||||
## 💬 Support
|
||||
|
||||
For issues or questions, please file an issue on the GitHub repository.
|
||||
## Support
|
||||
|
||||
For issues or questions, please [file an issue](https://github.com/calibrain/shelfmark/issues) on GitHub.
|
||||
|
||||
@@ -0,0 +1,17 @@
|
||||
flask
|
||||
flask-cors
|
||||
flask-socketio
|
||||
python-socketio
|
||||
requests[socks]
|
||||
beautifulsoup4
|
||||
tqdm
|
||||
dnspython
|
||||
gunicorn
|
||||
gevent
|
||||
gevent-websocket
|
||||
psutil
|
||||
emoji
|
||||
rarfile
|
||||
qbittorrent-api
|
||||
transmission-rpc
|
||||
deluge-client
|
||||
@@ -0,0 +1,4 @@
|
||||
pyvirtualdisplay
|
||||
pyautogui
|
||||
seleniumbase>=4.45.6
|
||||
python-xlib
|
||||
@@ -1,11 +0,0 @@
|
||||
flask
|
||||
requests[socks]
|
||||
beautifulsoup4
|
||||
tqdm
|
||||
pyvirtualdisplay
|
||||
dnspython
|
||||
pyautogui
|
||||
seleniumbase==4.41
|
||||
gunicorn
|
||||
python-xlib
|
||||
psutil
|
||||
@@ -0,0 +1,92 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Fix permissions on all configured directories.
|
||||
|
||||
This script is called by the entrypoint to ensure all user-configured
|
||||
directories have correct ownership. It reads directory paths from:
|
||||
- CONFIG_DIR environment variable
|
||||
- Config files in CONFIG_DIR/plugins/
|
||||
|
||||
Outputs directory paths that need permission fixing (one per line).
|
||||
The entrypoint handles the actual chown operations.
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def get_directories_from_config() -> set[str]:
|
||||
"""Extract all directory paths from config files."""
|
||||
directories = set()
|
||||
|
||||
config_dir = Path(os.getenv("CONFIG_DIR", "/config"))
|
||||
plugins_dir = config_dir / "plugins"
|
||||
|
||||
if not plugins_dir.exists():
|
||||
return directories
|
||||
|
||||
# Keys that contain directory paths
|
||||
directory_keys = {
|
||||
# Main destinations
|
||||
"DESTINATION",
|
||||
"DESTINATION_AUDIOBOOK",
|
||||
# Content type routing directories
|
||||
"AA_CONTENT_TYPE_DIR_FICTION",
|
||||
"AA_CONTENT_TYPE_DIR_NON_FICTION",
|
||||
"AA_CONTENT_TYPE_DIR_UNKNOWN",
|
||||
"AA_CONTENT_TYPE_DIR_MAGAZINE",
|
||||
"AA_CONTENT_TYPE_DIR_COMIC",
|
||||
"AA_CONTENT_TYPE_DIR_STANDARDS",
|
||||
"AA_CONTENT_TYPE_DIR_MUSICAL_SCORE",
|
||||
"AA_CONTENT_TYPE_DIR_OTHER",
|
||||
# Legacy keys (in case of old configs)
|
||||
"INGEST_DIR",
|
||||
"INGEST_DIR_AUDIOBOOK",
|
||||
"INGEST_DIR_BOOK_FICTION",
|
||||
"INGEST_DIR_BOOK_NON_FICTION",
|
||||
"INGEST_DIR_BOOK_UNKNOWN",
|
||||
"INGEST_DIR_MAGAZINE",
|
||||
"INGEST_DIR_COMIC_BOOK",
|
||||
"INGEST_DIR_STANDARDS_DOCUMENT",
|
||||
"INGEST_DIR_MUSICAL_SCORE",
|
||||
"INGEST_DIR_OTHER",
|
||||
"LIBRARY_PATH",
|
||||
"LIBRARY_PATH_AUDIOBOOK",
|
||||
}
|
||||
|
||||
# Read all JSON config files
|
||||
for config_file in plugins_dir.glob("*.json"):
|
||||
try:
|
||||
with open(config_file, "r") as f:
|
||||
config = json.load(f)
|
||||
|
||||
for key in directory_keys:
|
||||
if key in config:
|
||||
value = config[key]
|
||||
if value and isinstance(value, str) and value.startswith("/"):
|
||||
directories.add(value)
|
||||
except (json.JSONDecodeError, OSError):
|
||||
continue
|
||||
|
||||
return directories
|
||||
|
||||
|
||||
def main():
|
||||
"""Output all configured directories that exist."""
|
||||
directories = get_directories_from_config()
|
||||
|
||||
# Filter to directories that actually exist
|
||||
existing = []
|
||||
for dir_path in directories:
|
||||
path = Path(dir_path)
|
||||
if path.exists() and path.is_dir():
|
||||
existing.append(dir_path)
|
||||
|
||||
# Output one directory per line
|
||||
for dir_path in sorted(existing):
|
||||
print(dir_path)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,431 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Test script for download client implementations.
|
||||
|
||||
Usage:
|
||||
1. Start the test stack:
|
||||
docker compose -f docker-compose.test-clients.yml up -d
|
||||
|
||||
2. Wait for containers to initialize (first run takes ~30s)
|
||||
|
||||
3. Run this script to verify clients are accessible:
|
||||
python scripts/test_clients.py
|
||||
|
||||
4. Access cwabd at http://localhost:8084
|
||||
- Go to Settings > Prowlarr > Download Clients
|
||||
- Select a client from the dropdown
|
||||
- Click "Test Connection" to verify
|
||||
|
||||
Web UIs:
|
||||
- cwabd: http://localhost:8084
|
||||
- qBittorrent: http://localhost:8080
|
||||
- Transmission: http://localhost:9091
|
||||
- Deluge: http://localhost:8112
|
||||
- NZBGet: http://localhost:6789
|
||||
- SABnzbd: http://localhost:8085
|
||||
|
||||
Prerequisites (for running this script locally):
|
||||
pip install requests transmission-rpc deluge-client qbittorrent-api
|
||||
|
||||
First-Time Setup:
|
||||
qBittorrent:
|
||||
- Check container logs for temporary password: docker logs test-qbittorrent
|
||||
- Login at http://localhost:8080, change password to something known
|
||||
- Default username is 'admin'
|
||||
|
||||
Transmission:
|
||||
- No setup needed, credentials pre-configured (admin/admin)
|
||||
|
||||
Deluge:
|
||||
1. Access Web UI at http://localhost:8112 (default password: deluge)
|
||||
2. Add auth line to .local/test-clients/deluge/config/auth:
|
||||
echo "admin:admin:10" >> .local/test-clients/deluge/config/auth
|
||||
3. Restart: docker restart test-deluge
|
||||
|
||||
NZBGet:
|
||||
- No setup needed, credentials pre-configured (admin/admin)
|
||||
|
||||
SABnzbd:
|
||||
- Complete the setup wizard at http://localhost:8085
|
||||
- API key will be auto-detected by this script
|
||||
- In cwabd, copy API key from SABnzbd Config > General
|
||||
"""
|
||||
|
||||
import sys
|
||||
import time
|
||||
|
||||
# Test configuration - matches docker-compose.test-clients.yml
|
||||
CONFIG = {
|
||||
# Usenet clients
|
||||
"nzbget": {
|
||||
"url": "http://localhost:6789",
|
||||
"username": "admin",
|
||||
"password": "admin",
|
||||
},
|
||||
"sabnzbd": {
|
||||
"url": "http://localhost:8085",
|
||||
"api_key": None, # Will be read from config on first run
|
||||
},
|
||||
# Torrent clients
|
||||
"qbittorrent": {
|
||||
"url": "http://localhost:8080",
|
||||
"username": "admin",
|
||||
"password": "5NCngsHXm", # Temp password from: docker logs test-qbittorrent | grep password
|
||||
},
|
||||
"transmission": {
|
||||
"url": "http://localhost:9091",
|
||||
"username": "admin",
|
||||
"password": "admin",
|
||||
},
|
||||
"deluge": {
|
||||
"host": "localhost",
|
||||
"port": 58846,
|
||||
"username": "admin",
|
||||
"password": "admin",
|
||||
},
|
||||
}
|
||||
|
||||
# Test magnet link (Ubuntu ISO - legal, small metadata)
|
||||
TEST_MAGNET = "magnet:?xt=urn:btih:3b245504cf5f11bbdbe1201cea6a6bf45aee1bc0&dn=ubuntu-22.04.3-live-server-amd64.iso"
|
||||
|
||||
|
||||
def test_nzbget():
|
||||
"""Test NZBGet connection."""
|
||||
import requests
|
||||
|
||||
print("\n" + "=" * 50)
|
||||
print("Testing NZBGet")
|
||||
print("=" * 50)
|
||||
|
||||
url = CONFIG["nzbget"]["url"]
|
||||
username = CONFIG["nzbget"]["username"]
|
||||
password = CONFIG["nzbget"]["password"]
|
||||
|
||||
try:
|
||||
# Test connection via JSON-RPC
|
||||
rpc_url = f"{url}/jsonrpc"
|
||||
response = requests.post(
|
||||
rpc_url,
|
||||
json={"method": "version", "params": []},
|
||||
auth=(username, password),
|
||||
timeout=10,
|
||||
)
|
||||
response.raise_for_status()
|
||||
result = response.json()
|
||||
version = result.get("result", "unknown")
|
||||
print(f" Connected to NZBGet {version}")
|
||||
|
||||
# Test status
|
||||
response = requests.post(
|
||||
rpc_url,
|
||||
json={"method": "status", "params": []},
|
||||
auth=(username, password),
|
||||
timeout=10,
|
||||
)
|
||||
status = response.json().get("result", {})
|
||||
print(f" Server state: {'Paused' if status.get('ServerPaused') else 'Running'}")
|
||||
print(f" Downloads in queue: {status.get('DownloadedSizeMB', 0)} MB downloaded")
|
||||
|
||||
print(" SUCCESS: NZBGet is working!")
|
||||
return True
|
||||
|
||||
except requests.exceptions.ConnectionError:
|
||||
print(" ERROR: Could not connect to NZBGet")
|
||||
print(" Is the container running? docker ps | grep nzbget")
|
||||
return False
|
||||
except Exception as e:
|
||||
print(f" ERROR: {e}")
|
||||
return False
|
||||
|
||||
|
||||
def test_sabnzbd():
|
||||
"""Test SABnzbd connection."""
|
||||
import requests
|
||||
|
||||
print("\n" + "=" * 50)
|
||||
print("Testing SABnzbd")
|
||||
print("=" * 50)
|
||||
|
||||
url = CONFIG["sabnzbd"]["url"]
|
||||
api_key = CONFIG["sabnzbd"]["api_key"]
|
||||
|
||||
# Try to get API key from config if not set
|
||||
if not api_key:
|
||||
try:
|
||||
import os
|
||||
ini_path = ".local/test-clients/sabnzbd/config/sabnzbd.ini"
|
||||
if os.path.exists(ini_path):
|
||||
with open(ini_path) as f:
|
||||
for line in f:
|
||||
if line.startswith("api_key"):
|
||||
api_key = line.split("=")[1].strip()
|
||||
print(f" Found API key in config: {api_key[:8]}...")
|
||||
break
|
||||
except Exception as e:
|
||||
print(f" Could not read API key from config: {e}")
|
||||
|
||||
if not api_key:
|
||||
print(" ERROR: No API key configured")
|
||||
print(" Please access http://localhost:8085 and complete initial setup")
|
||||
print(" Then copy the API key from Config > General")
|
||||
return False
|
||||
|
||||
try:
|
||||
# Test connection
|
||||
response = requests.get(
|
||||
f"{url}/api",
|
||||
params={"apikey": api_key, "mode": "version", "output": "json"},
|
||||
timeout=10,
|
||||
)
|
||||
response.raise_for_status()
|
||||
result = response.json()
|
||||
version = result.get("version", "unknown")
|
||||
print(f" Connected to SABnzbd {version}")
|
||||
|
||||
# Test queue status
|
||||
response = requests.get(
|
||||
f"{url}/api",
|
||||
params={"apikey": api_key, "mode": "queue", "output": "json"},
|
||||
timeout=10,
|
||||
)
|
||||
queue = response.json().get("queue", {})
|
||||
print(f" Queue status: {queue.get('status', 'unknown')}")
|
||||
print(f" Items in queue: {len(queue.get('slots', []))}")
|
||||
|
||||
print(" SUCCESS: SABnzbd is working!")
|
||||
return True
|
||||
|
||||
except requests.exceptions.ConnectionError:
|
||||
print(" ERROR: Could not connect to SABnzbd")
|
||||
print(" Is the container running? docker ps | grep sabnzbd")
|
||||
return False
|
||||
except Exception as e:
|
||||
print(f" ERROR: {e}")
|
||||
return False
|
||||
|
||||
|
||||
def test_qbittorrent():
|
||||
"""Test qBittorrent connection."""
|
||||
print("\n" + "=" * 50)
|
||||
print("Testing qBittorrent")
|
||||
print("=" * 50)
|
||||
|
||||
try:
|
||||
import qbittorrentapi
|
||||
|
||||
url = CONFIG["qbittorrent"]["url"]
|
||||
username = CONFIG["qbittorrent"]["username"]
|
||||
password = CONFIG["qbittorrent"]["password"]
|
||||
|
||||
# Parse URL for host/port
|
||||
from urllib.parse import urlparse
|
||||
parsed = urlparse(url)
|
||||
|
||||
client = qbittorrentapi.Client(
|
||||
host=parsed.hostname,
|
||||
port=parsed.port or 8080,
|
||||
username=username,
|
||||
password=password,
|
||||
)
|
||||
|
||||
# Test connection
|
||||
client.auth_log_in()
|
||||
version = client.app.version
|
||||
print(f" Connected to qBittorrent {version}")
|
||||
|
||||
# Get torrent list
|
||||
torrents = client.torrents_info()
|
||||
print(f" Active torrents: {len(torrents)}")
|
||||
|
||||
# Test adding a torrent (then remove it)
|
||||
print(" Testing add/remove torrent...")
|
||||
result = client.torrents_add(urls=TEST_MAGNET, is_paused=True)
|
||||
if result == "Ok.":
|
||||
# Wait a moment for it to be added
|
||||
time.sleep(1)
|
||||
torrents = client.torrents_info()
|
||||
if torrents:
|
||||
test_torrent = torrents[-1] # Most recently added
|
||||
print(f" Added test torrent: {test_torrent.name[:50]}...")
|
||||
print(f" Status: {test_torrent.state}")
|
||||
|
||||
# Remove it
|
||||
client.torrents_delete(torrent_hashes=test_torrent.hash, delete_files=True)
|
||||
print(" Removed test torrent")
|
||||
else:
|
||||
print(f" Add result: {result}")
|
||||
|
||||
print(" SUCCESS: qBittorrent is working!")
|
||||
return True
|
||||
|
||||
except ImportError:
|
||||
print(" ERROR: qbittorrent-api not installed")
|
||||
print(" Run: pip install qbittorrent-api")
|
||||
return False
|
||||
except Exception as e:
|
||||
print(f" ERROR: {e}")
|
||||
if "Forbidden" in str(e) or "401" in str(e):
|
||||
print("\n Authentication failed. Check password:")
|
||||
print(" 1. docker logs test-qbittorrent | grep password")
|
||||
print(" 2. Login to http://localhost:8080 and set a known password")
|
||||
return False
|
||||
|
||||
|
||||
def test_transmission():
|
||||
"""Test Transmission connection."""
|
||||
print("\n" + "=" * 50)
|
||||
print("Testing Transmission")
|
||||
print("=" * 50)
|
||||
|
||||
try:
|
||||
from transmission_rpc import Client
|
||||
from urllib.parse import urlparse
|
||||
|
||||
url = CONFIG["transmission"]["url"]
|
||||
parsed = urlparse(url)
|
||||
|
||||
client = Client(
|
||||
host=parsed.hostname,
|
||||
port=parsed.port or 9091,
|
||||
username=CONFIG["transmission"]["username"],
|
||||
password=CONFIG["transmission"]["password"],
|
||||
)
|
||||
|
||||
# Test connection
|
||||
session = client.get_session()
|
||||
print(f" Connected to Transmission {session.version}")
|
||||
|
||||
# Get torrent list
|
||||
torrents = client.get_torrents()
|
||||
print(f" Active torrents: {len(torrents)}")
|
||||
|
||||
# Test adding a torrent (then remove it)
|
||||
print(" Testing add/remove torrent...")
|
||||
torrent = client.add_torrent(TEST_MAGNET, paused=True)
|
||||
print(f" Added test torrent: {torrent.name[:50]}...")
|
||||
|
||||
# Get status
|
||||
status = client.get_torrent(torrent.id)
|
||||
print(f" Status: {status.status} ({status.percent_done * 100:.1f}%)")
|
||||
|
||||
# Remove it
|
||||
client.remove_torrent(torrent.id, delete_data=True)
|
||||
print(" Removed test torrent")
|
||||
|
||||
print(" SUCCESS: Transmission is working!")
|
||||
return True
|
||||
|
||||
except ImportError:
|
||||
print(" ERROR: transmission-rpc not installed")
|
||||
print(" Run: pip install transmission-rpc")
|
||||
return False
|
||||
except Exception as e:
|
||||
print(f" ERROR: {e}")
|
||||
return False
|
||||
|
||||
|
||||
def test_deluge():
|
||||
"""Test Deluge connection."""
|
||||
print("\n" + "=" * 50)
|
||||
print("Testing Deluge")
|
||||
print("=" * 50)
|
||||
|
||||
try:
|
||||
from deluge_client import DelugeRPCClient
|
||||
|
||||
client = DelugeRPCClient(
|
||||
host=CONFIG["deluge"]["host"],
|
||||
port=CONFIG["deluge"]["port"],
|
||||
username=CONFIG["deluge"]["username"],
|
||||
password=CONFIG["deluge"]["password"],
|
||||
)
|
||||
|
||||
# Test connection
|
||||
client.connect()
|
||||
version = client.call("daemon.info")
|
||||
print(f" Connected to Deluge {version}")
|
||||
|
||||
# Get torrent list
|
||||
torrents = client.call("core.get_torrents_status", {}, ["name"])
|
||||
print(f" Active torrents: {len(torrents)}")
|
||||
|
||||
# Test adding a torrent (then remove it)
|
||||
print(" Testing add/remove torrent...")
|
||||
torrent_id = client.call("core.add_torrent_magnet", TEST_MAGNET, {"add_paused": True})
|
||||
|
||||
if torrent_id:
|
||||
print(f" Added test torrent: {torrent_id[:20]}...")
|
||||
|
||||
# Get status
|
||||
status = client.call("core.get_torrent_status", torrent_id, ["state", "progress"])
|
||||
state = status.get(b"state", b"unknown")
|
||||
if isinstance(state, bytes):
|
||||
state = state.decode()
|
||||
print(f" Status: {state}")
|
||||
|
||||
# Remove it
|
||||
client.call("core.remove_torrent", torrent_id, True)
|
||||
print(" Removed test torrent")
|
||||
else:
|
||||
print(" WARNING: Could not add test torrent")
|
||||
|
||||
print(" SUCCESS: Deluge is working!")
|
||||
return True
|
||||
|
||||
except ImportError:
|
||||
print(" ERROR: deluge-client not installed")
|
||||
print(" Run: pip install deluge-client")
|
||||
return False
|
||||
except Exception as e:
|
||||
print(f" ERROR: {e}")
|
||||
if "Connection refused" in str(e):
|
||||
print(" Is the container running? docker ps | grep deluge")
|
||||
elif "Bad login" in str(e) or "auth" in str(e).lower():
|
||||
print("\n Deluge auth setup required:")
|
||||
print(" 1. Add 'admin:admin:10' to .local/test-clients/deluge/config/auth")
|
||||
print(" 2. Restart: docker restart test-deluge")
|
||||
print(" 3. Or access Web UI at http://localhost:8112 (password: deluge)")
|
||||
return False
|
||||
|
||||
|
||||
def main():
|
||||
print("Download Client Test Suite")
|
||||
print("=" * 50)
|
||||
print("Make sure containers are running:")
|
||||
print(" docker compose -f docker-compose.test-clients.yml up -d")
|
||||
|
||||
results = {}
|
||||
|
||||
# Test usenet clients
|
||||
print("\n" + "=" * 50)
|
||||
print("USENET CLIENTS")
|
||||
print("=" * 50)
|
||||
results["nzbget"] = test_nzbget()
|
||||
results["sabnzbd"] = test_sabnzbd()
|
||||
|
||||
# Test torrent clients
|
||||
print("\n" + "=" * 50)
|
||||
print("TORRENT CLIENTS")
|
||||
print("=" * 50)
|
||||
results["qbittorrent"] = test_qbittorrent()
|
||||
results["transmission"] = test_transmission()
|
||||
results["deluge"] = test_deluge()
|
||||
|
||||
# Summary
|
||||
print("\n" + "=" * 50)
|
||||
print("SUMMARY")
|
||||
print("=" * 50)
|
||||
|
||||
for client, success in results.items():
|
||||
status = "PASS" if success else "FAIL"
|
||||
print(f" {client}: {status}")
|
||||
|
||||
passed = sum(results.values())
|
||||
total = len(results)
|
||||
print(f"\n Total: {passed}/{total} passed")
|
||||
|
||||
return 0 if passed == total else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1 @@
|
||||
"""Shelfmark - book search and download service."""
|
||||
@@ -0,0 +1,8 @@
|
||||
"""Package entry point for `python -m shelfmark`."""
|
||||
|
||||
from shelfmark.main import app, socketio
|
||||
from shelfmark.config.env import FLASK_HOST, FLASK_PORT
|
||||
from shelfmark.core.config import config
|
||||
|
||||
if __name__ == "__main__":
|
||||
socketio.run(app, host=FLASK_HOST, port=FLASK_PORT, debug=config.get("DEBUG", False))
|
||||
@@ -0,0 +1 @@
|
||||
"""API module - WebSocket handling."""
|
||||
@@ -0,0 +1,174 @@
|
||||
"""WebSocket manager for real-time status updates."""
|
||||
|
||||
import logging
|
||||
import threading
|
||||
from typing import Optional, Dict, Any, Callable, List
|
||||
|
||||
from flask_socketio import SocketIO
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class WebSocketManager:
|
||||
"""Manages WebSocket connections and broadcasts."""
|
||||
|
||||
def __init__(self):
|
||||
self.socketio: Optional[SocketIO] = None
|
||||
self._enabled = False
|
||||
self._connection_count = 0
|
||||
self._connection_lock = threading.Lock()
|
||||
self._on_first_connect_callbacks: List[Callable[[], None]] = []
|
||||
self._on_all_disconnect_callbacks: List[Callable[[], None]] = []
|
||||
self._needs_rewarm = False # Flag to trigger warmup callbacks on next connect
|
||||
|
||||
def init_app(self, app, socketio: SocketIO):
|
||||
"""Initialize the WebSocket manager with Flask-SocketIO instance."""
|
||||
self.socketio = socketio
|
||||
self._enabled = True
|
||||
logger.info("WebSocket manager initialized")
|
||||
|
||||
def register_on_first_connect(self, callback: Callable[[], None]):
|
||||
"""Register a callback for when the first client connects."""
|
||||
self._on_first_connect_callbacks.append(callback)
|
||||
logger.debug(f"Registered on_first_connect callback: {callback.__name__}")
|
||||
|
||||
def register_on_all_disconnect(self, callback: Callable[[], None]):
|
||||
"""Register a callback for when all clients disconnect."""
|
||||
self._on_all_disconnect_callbacks.append(callback)
|
||||
logger.debug(f"Registered on_all_disconnect callback: {callback.__name__}")
|
||||
|
||||
def request_warmup_on_next_connect(self):
|
||||
"""Request warmup callbacks on the next client connect (e.g., after idle shutdown)."""
|
||||
with self._connection_lock:
|
||||
self._needs_rewarm = True
|
||||
logger.debug("Warmup requested for next client connect")
|
||||
|
||||
def client_connected(self):
|
||||
"""Track a new client connection. Call this from the connect event handler."""
|
||||
with self._connection_lock:
|
||||
was_zero = self._connection_count == 0
|
||||
needs_rewarm = self._needs_rewarm
|
||||
self._connection_count += 1
|
||||
current_count = self._connection_count
|
||||
# Clear rewarm flag if we're going to trigger warmup
|
||||
if was_zero or needs_rewarm:
|
||||
self._needs_rewarm = False
|
||||
|
||||
logger.debug(f"Client connected. Active connections: {current_count}")
|
||||
|
||||
# Trigger warmup callbacks if this is the first connection OR if rewarm was requested
|
||||
# (rewarm is requested when bypasser shuts down due to idle while clients are connected)
|
||||
if was_zero or needs_rewarm:
|
||||
reason = "First client connected" if was_zero else "Rewarm requested after idle shutdown"
|
||||
logger.info(f"{reason}, triggering warmup callbacks...")
|
||||
for callback in self._on_first_connect_callbacks:
|
||||
try:
|
||||
# Run callbacks in a separate thread to not block the connection
|
||||
thread = threading.Thread(target=callback, daemon=True)
|
||||
thread.start()
|
||||
except Exception as e:
|
||||
logger.error(f"Error in on_first_connect callback {callback.__name__}: {e}")
|
||||
|
||||
def client_disconnected(self):
|
||||
"""Track a client disconnection. Call this from the disconnect event handler."""
|
||||
with self._connection_lock:
|
||||
self._connection_count = max(0, self._connection_count - 1)
|
||||
current_count = self._connection_count
|
||||
is_now_zero = current_count == 0
|
||||
|
||||
logger.debug(f"Client disconnected. Active connections: {current_count}")
|
||||
|
||||
# If all clients have disconnected, trigger cleanup callbacks
|
||||
if is_now_zero:
|
||||
logger.info("All clients disconnected, triggering disconnect callbacks...")
|
||||
for callback in self._on_all_disconnect_callbacks:
|
||||
try:
|
||||
callback()
|
||||
except Exception as e:
|
||||
logger.error(f"Error in on_all_disconnect callback {callback.__name__}: {e}")
|
||||
|
||||
def get_connection_count(self) -> int:
|
||||
"""Get the current number of active WebSocket connections."""
|
||||
with self._connection_lock:
|
||||
return self._connection_count
|
||||
|
||||
def has_active_connections(self) -> bool:
|
||||
"""Check if there are any active WebSocket connections."""
|
||||
return self.get_connection_count() > 0
|
||||
|
||||
def is_enabled(self) -> bool:
|
||||
"""Check if WebSocket is enabled and ready."""
|
||||
return self._enabled and self.socketio is not None
|
||||
|
||||
def broadcast_status_update(self, status_data: Dict[str, Any]):
|
||||
"""Broadcast status update to all connected clients."""
|
||||
if not self.is_enabled():
|
||||
return
|
||||
|
||||
try:
|
||||
# When calling socketio.emit() outside event handlers, it broadcasts by default
|
||||
self.socketio.emit('status_update', status_data)
|
||||
logger.debug(f"Broadcasted status update to all clients")
|
||||
except Exception as e:
|
||||
logger.error(f"Error broadcasting status update: {e}")
|
||||
|
||||
def broadcast_download_progress(self, book_id: str, progress: float, status: str):
|
||||
"""Broadcast download progress update for a specific book."""
|
||||
if not self.is_enabled():
|
||||
return
|
||||
|
||||
try:
|
||||
data = {
|
||||
'book_id': book_id,
|
||||
'progress': progress,
|
||||
'status': status
|
||||
}
|
||||
# When calling socketio.emit() outside event handlers, it broadcasts by default
|
||||
self.socketio.emit('download_progress', data)
|
||||
logger.debug(f"Broadcasted progress for book {book_id}: {progress}%")
|
||||
except Exception as e:
|
||||
logger.error(f"Error broadcasting download progress: {e}")
|
||||
|
||||
def broadcast_notification(self, message: str, notification_type: str = 'info'):
|
||||
"""Broadcast a notification message to all clients."""
|
||||
if not self.is_enabled():
|
||||
return
|
||||
|
||||
try:
|
||||
data = {
|
||||
'message': message,
|
||||
'type': notification_type
|
||||
}
|
||||
# When calling socketio.emit() outside event handlers, it broadcasts by default
|
||||
self.socketio.emit('notification', data)
|
||||
logger.debug(f"Broadcasted notification: {message}")
|
||||
except Exception as e:
|
||||
logger.error(f"Error broadcasting notification: {e}")
|
||||
|
||||
def broadcast_search_status(
|
||||
self,
|
||||
source: str,
|
||||
provider: str,
|
||||
book_id: str,
|
||||
message: str,
|
||||
phase: str = 'searching'
|
||||
):
|
||||
"""Broadcast search status update for a release source search."""
|
||||
if not self.is_enabled():
|
||||
return
|
||||
|
||||
try:
|
||||
data = {
|
||||
'source': source,
|
||||
'provider': provider,
|
||||
'book_id': book_id,
|
||||
'message': message,
|
||||
'phase': phase,
|
||||
}
|
||||
self.socketio.emit('search_status', data)
|
||||
except Exception as e:
|
||||
logger.error(f"Error broadcasting search status: {e}")
|
||||
|
||||
|
||||
# Global WebSocket manager instance
|
||||
ws_manager = WebSocketManager()
|
||||
@@ -0,0 +1,5 @@
|
||||
"""Cloudflare bypass utilities."""
|
||||
|
||||
|
||||
class BypassCancelledException(Exception):
|
||||
"""Raised when a bypass operation is cancelled."""
|
||||
@@ -0,0 +1,126 @@
|
||||
"""External Cloudflare bypasser using FlareSolverr."""
|
||||
|
||||
import random
|
||||
import time
|
||||
from threading import Event
|
||||
from typing import TYPE_CHECKING, Optional
|
||||
|
||||
import requests
|
||||
|
||||
from shelfmark.bypass import BypassCancelledException
|
||||
from shelfmark.core.config import config
|
||||
from shelfmark.core.logger import setup_logger
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from shelfmark.download import network
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
# Timeout constants (seconds)
|
||||
CONNECT_TIMEOUT = 10
|
||||
MAX_READ_TIMEOUT = 120
|
||||
READ_TIMEOUT_BUFFER = 15
|
||||
|
||||
# Retry settings
|
||||
MAX_RETRY = 5
|
||||
BACKOFF_BASE = 1.0
|
||||
BACKOFF_CAP = 10.0
|
||||
|
||||
|
||||
def _fetch_via_bypasser(target_url: str) -> Optional[str]:
|
||||
"""Make a single request to the external bypasser service. Returns HTML or None."""
|
||||
bypasser_url = config.get("EXT_BYPASSER_URL", "http://flaresolverr:8191")
|
||||
bypasser_path = config.get("EXT_BYPASSER_PATH", "/v1")
|
||||
bypasser_timeout = config.get("EXT_BYPASSER_TIMEOUT", 60000)
|
||||
|
||||
if not bypasser_url or not bypasser_path:
|
||||
logger.error("External bypasser not configured. Check EXT_BYPASSER_URL and EXT_BYPASSER_PATH.")
|
||||
return None
|
||||
|
||||
read_timeout = min((bypasser_timeout / 1000) + READ_TIMEOUT_BUFFER, MAX_READ_TIMEOUT)
|
||||
|
||||
try:
|
||||
response = requests.post(
|
||||
f"{bypasser_url}{bypasser_path}",
|
||||
headers={"Content-Type": "application/json"},
|
||||
json={"cmd": "request.get", "url": target_url, "maxTimeout": bypasser_timeout},
|
||||
timeout=(CONNECT_TIMEOUT, read_timeout)
|
||||
)
|
||||
response.raise_for_status()
|
||||
result = response.json()
|
||||
|
||||
status = result.get('status', 'unknown')
|
||||
message = result.get('message', '')
|
||||
logger.debug(f"External bypasser response for '{target_url}': {status} - {message}")
|
||||
|
||||
if status != 'ok':
|
||||
logger.warning(f"External bypasser failed for '{target_url}': {status} - {message}")
|
||||
return None
|
||||
|
||||
solution = result.get('solution')
|
||||
html = solution.get('response', '') if solution else ''
|
||||
|
||||
if not html:
|
||||
logger.warning(f"External bypasser returned empty response for '{target_url}'")
|
||||
return None
|
||||
|
||||
return html
|
||||
|
||||
except requests.exceptions.Timeout:
|
||||
logger.warning(f"External bypasser timed out for '{target_url}' (connect: {CONNECT_TIMEOUT}s, read: {read_timeout:.0f}s)")
|
||||
except requests.exceptions.RequestException as e:
|
||||
logger.warning(f"External bypasser request failed for '{target_url}': {e}")
|
||||
except (KeyError, TypeError, ValueError) as e:
|
||||
logger.warning(f"External bypasser returned malformed response for '{target_url}': {e}")
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def _check_cancelled(cancel_flag: Optional[Event], context: str) -> None:
|
||||
"""Check if operation was cancelled and raise exception if so."""
|
||||
if cancel_flag and cancel_flag.is_set():
|
||||
logger.info(f"External bypasser cancelled {context}")
|
||||
raise BypassCancelledException("Bypass cancelled")
|
||||
|
||||
|
||||
def _sleep_with_cancellation(seconds: float, cancel_flag: Optional[Event]) -> None:
|
||||
"""Sleep for the specified duration, checking for cancellation each second."""
|
||||
for _ in range(int(seconds)):
|
||||
_check_cancelled(cancel_flag, "during backoff")
|
||||
time.sleep(1)
|
||||
remaining = seconds - int(seconds)
|
||||
if remaining > 0:
|
||||
time.sleep(remaining)
|
||||
|
||||
|
||||
def get_bypassed_page(
|
||||
url: str,
|
||||
selector: Optional["network.AAMirrorSelector"] = None,
|
||||
cancel_flag: Optional[Event] = None
|
||||
) -> Optional[str]:
|
||||
"""Fetch HTML via external bypasser with retries and mirror rotation."""
|
||||
from shelfmark.download import network as network_module
|
||||
|
||||
sel = selector or network_module.AAMirrorSelector()
|
||||
|
||||
for attempt in range(1, MAX_RETRY + 1):
|
||||
_check_cancelled(cancel_flag, "by user")
|
||||
|
||||
attempt_url = sel.rewrite(url)
|
||||
result = _fetch_via_bypasser(attempt_url)
|
||||
if result:
|
||||
return result
|
||||
|
||||
if attempt == MAX_RETRY:
|
||||
break
|
||||
|
||||
delay = min(BACKOFF_CAP, BACKOFF_BASE * (2 ** (attempt - 1))) + random.random()
|
||||
logger.info(f"External bypasser attempt {attempt}/{MAX_RETRY} failed, retrying in {delay:.1f}s")
|
||||
|
||||
_sleep_with_cancellation(delay, cancel_flag)
|
||||
|
||||
new_base, action = sel.next_mirror_or_rotate_dns()
|
||||
if action in ("mirror", "dns") and new_base:
|
||||
logger.info(f"Rotated {action} for retry")
|
||||
|
||||
return None
|
||||
@@ -0,0 +1,57 @@
|
||||
"""Browser fingerprint profile management for bypass stealth."""
|
||||
|
||||
import random
|
||||
from typing import Optional
|
||||
|
||||
from shelfmark.core.logger import setup_logger
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
COMMON_RESOLUTIONS = [
|
||||
(1920, 1080, 0.35),
|
||||
(1366, 768, 0.18),
|
||||
(1536, 864, 0.10),
|
||||
(1440, 900, 0.08),
|
||||
(1280, 720, 0.07),
|
||||
(1600, 900, 0.06),
|
||||
(1280, 800, 0.05),
|
||||
(2560, 1440, 0.04),
|
||||
(1680, 1050, 0.04),
|
||||
(1920, 1200, 0.03),
|
||||
]
|
||||
|
||||
# Current screen size (module-level singleton)
|
||||
_current_screen_size: Optional[tuple[int, int]] = None
|
||||
|
||||
|
||||
def get_screen_size() -> tuple[int, int]:
|
||||
global _current_screen_size
|
||||
if _current_screen_size is None:
|
||||
_current_screen_size = _generate_screen_size()
|
||||
logger.debug(f"Generated initial screen size: {_current_screen_size[0]}x{_current_screen_size[1]}")
|
||||
return _current_screen_size
|
||||
|
||||
|
||||
def rotate_screen_size() -> tuple[int, int]:
|
||||
global _current_screen_size
|
||||
old_size = _current_screen_size
|
||||
_current_screen_size = _generate_screen_size()
|
||||
width, height = _current_screen_size
|
||||
|
||||
if old_size:
|
||||
logger.info(f"Rotated screen size: {old_size[0]}x{old_size[1]} -> {width}x{height}")
|
||||
else:
|
||||
logger.info(f"Generated screen size: {width}x{height}")
|
||||
|
||||
return _current_screen_size
|
||||
|
||||
|
||||
def clear_screen_size() -> None:
|
||||
global _current_screen_size
|
||||
_current_screen_size = None
|
||||
|
||||
|
||||
def _generate_screen_size() -> tuple[int, int]:
|
||||
resolutions = [(w, h) for w, h, _ in COMMON_RESOLUTIONS]
|
||||
weights = [weight for _, _, weight in COMMON_RESOLUTIONS]
|
||||
return random.choices(resolutions, weights=weights)[0]
|
||||
@@ -0,0 +1 @@
|
||||
"""Configuration module - environment variables and settings."""
|
||||
@@ -0,0 +1,153 @@
|
||||
"""Bootstrap environment variables. No local dependencies - import first."""
|
||||
|
||||
import json
|
||||
import os
|
||||
import shutil
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def string_to_bool(s: str) -> bool:
|
||||
"""Convert string to boolean."""
|
||||
return s.lower() in ["true", "yes", "1", "y"]
|
||||
|
||||
|
||||
def _read_debug_from_config() -> bool:
|
||||
"""Read DEBUG from env var or config file (import-time safe)."""
|
||||
env_debug = os.environ.get("DEBUG")
|
||||
if env_debug is not None:
|
||||
return string_to_bool(env_debug)
|
||||
|
||||
# Try to read from config file
|
||||
config_dir = Path(os.getenv("CONFIG_DIR", "/config"))
|
||||
config_file = config_dir / "plugins" / "advanced.json"
|
||||
|
||||
if config_file.exists():
|
||||
try:
|
||||
with open(config_file, "r") as f:
|
||||
config = json.load(f)
|
||||
if "DEBUG" in config:
|
||||
return bool(config["DEBUG"])
|
||||
except (json.JSONDecodeError, OSError):
|
||||
pass
|
||||
|
||||
return False
|
||||
|
||||
|
||||
def _is_sqlite_file(path: Path) -> bool:
|
||||
"""Check if a file is a valid SQLite database by reading magic bytes."""
|
||||
try:
|
||||
with open(path, "rb") as f:
|
||||
header = f.read(16)
|
||||
return header[:16] == b"SQLite format 3\x00"
|
||||
except (OSError, PermissionError):
|
||||
return False
|
||||
|
||||
|
||||
def _resolve_cwa_db_path() -> Path | None:
|
||||
"""Resolve CWA database path from env var or default location."""
|
||||
env_path = os.getenv("CWA_DB_PATH")
|
||||
if env_path:
|
||||
path = Path(env_path)
|
||||
if path.exists() and path.is_file() and _is_sqlite_file(path):
|
||||
return path
|
||||
|
||||
# Check default mount path
|
||||
default_path = Path("/auth/app.db")
|
||||
if default_path.exists() and default_path.is_file() and _is_sqlite_file(default_path):
|
||||
return default_path
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def _is_config_dir_writable() -> bool:
|
||||
"""Check if the config directory exists and is writable."""
|
||||
try:
|
||||
if not CONFIG_DIR.exists() or not CONFIG_DIR.is_dir():
|
||||
return False
|
||||
test_file = CONFIG_DIR / ".write_test"
|
||||
test_file.touch()
|
||||
test_file.unlink()
|
||||
return True
|
||||
except (OSError, PermissionError):
|
||||
return False
|
||||
|
||||
|
||||
def is_covers_cache_enabled() -> bool:
|
||||
"""Check if cover caching is enabled (requires setting + writable config dir)."""
|
||||
from shelfmark.core.config import config
|
||||
setting_enabled = config.get("COVERS_CACHE_ENABLED", True)
|
||||
return setting_enabled and _is_config_dir_writable()
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Bootstrap paths - needed before settings registry is available
|
||||
# =============================================================================
|
||||
|
||||
CONFIG_DIR = Path(os.getenv("CONFIG_DIR", "/config"))
|
||||
LOG_ROOT = Path(os.getenv("LOG_ROOT", "/var/log/"))
|
||||
LOG_DIR = LOG_ROOT / "shelfmark"
|
||||
LOG_FILE = LOG_DIR / "shelfmark.log"
|
||||
TMP_DIR = Path(os.getenv("TMP_DIR", "/tmp/shelfmark"))
|
||||
INGEST_DIR = Path(os.getenv("INGEST_DIR", "/books"))
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Logger configuration - needed before settings registry is available
|
||||
# =============================================================================
|
||||
|
||||
DEBUG = _read_debug_from_config()
|
||||
LOG_LEVEL = "DEBUG" if DEBUG else "INFO"
|
||||
ENABLE_LOGGING = string_to_bool(os.getenv("ENABLE_LOGGING", "true"))
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Flask configuration - needed before app starts
|
||||
# =============================================================================
|
||||
|
||||
FLASK_HOST = os.getenv("FLASK_HOST", "0.0.0.0")
|
||||
FLASK_PORT = int(os.getenv("FLASK_PORT", "8084"))
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Authentication
|
||||
# =============================================================================
|
||||
|
||||
SESSION_COOKIE_SECURE_ENV = os.getenv("SESSION_COOKIE_SECURE", "false")
|
||||
CWA_DB_PATH = _resolve_cwa_db_path()
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Version information from Docker build
|
||||
# =============================================================================
|
||||
|
||||
BUILD_VERSION = os.getenv("BUILD_VERSION", "N/A")
|
||||
RELEASE_VERSION = os.getenv("RELEASE_VERSION", "N/A")
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Capability detection - runtime checks, not user-configurable
|
||||
# =============================================================================
|
||||
|
||||
DOCKERMODE = string_to_bool(os.getenv("DOCKERMODE", "false"))
|
||||
TOR_VARIANT_AVAILABLE = shutil.which("tor") is not None
|
||||
USING_TOR = string_to_bool(os.getenv("USING_TOR", "false"))
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Debug/development settings
|
||||
# =============================================================================
|
||||
|
||||
# Debug: skip specific download sources for testing fallback chains
|
||||
# Comma-separated values: aa-fast, aa-slow-nowait, aa-slow-wait, libgen, zlib, welib
|
||||
_DEBUG_SKIP_SOURCES_RAW = os.getenv("DEBUG_SKIP_SOURCES", "").strip().lower()
|
||||
DEBUG_SKIP_SOURCES = set(s.strip() for s in _DEBUG_SKIP_SOURCES_RAW.split(",") if s.strip())
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Legacy migration support - will be removed in future version
|
||||
# =============================================================================
|
||||
|
||||
# Legacy welib settings - replaced by SOURCE_PRIORITY OrderableListField
|
||||
# Kept for migration: if set, used to build initial SOURCE_PRIORITY config
|
||||
_LEGACY_PRIORITIZE_WELIB = string_to_bool(os.getenv("PRIORITIZE_WELIB", "false"))
|
||||
_LEGACY_ALLOW_USE_WELIB = string_to_bool(os.getenv("ALLOW_USE_WELIB", "true"))
|
||||
@@ -0,0 +1,166 @@
|
||||
"""Authentication settings registration."""
|
||||
|
||||
from typing import Any, Dict
|
||||
|
||||
from werkzeug.security import generate_password_hash
|
||||
|
||||
from shelfmark.core.logger import setup_logger
|
||||
from shelfmark.core.settings_registry import (
|
||||
register_settings,
|
||||
register_on_save,
|
||||
load_config_file,
|
||||
TextField,
|
||||
PasswordField,
|
||||
CheckboxField,
|
||||
ActionButton,
|
||||
)
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
|
||||
def _clear_builtin_credentials() -> Dict[str, Any]:
|
||||
"""Clear built-in credentials to allow public access."""
|
||||
import json
|
||||
from shelfmark.core.settings_registry import _get_config_file_path, _ensure_config_dir
|
||||
|
||||
try:
|
||||
config = load_config_file("security")
|
||||
config.pop("BUILTIN_USERNAME", None)
|
||||
config.pop("BUILTIN_PASSWORD_HASH", None)
|
||||
|
||||
_ensure_config_dir("security")
|
||||
config_path = _get_config_file_path("security")
|
||||
with open(config_path, 'w') as f:
|
||||
json.dump(config, f, indent=2)
|
||||
|
||||
logger.info("Cleared credentials")
|
||||
return {"success": True, "message": "Credentials cleared. The app is now publicly accessible."}
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to clear credentials: {e}")
|
||||
return {"success": False, "message": f"Failed to clear credentials: {str(e)}"}
|
||||
|
||||
|
||||
def _on_save_security(values: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"""
|
||||
Custom save handler for security settings.
|
||||
|
||||
Handles password validation and hashing:
|
||||
- If new password is provided, validate confirmation and hash it
|
||||
- If password fields are empty, preserve existing hash
|
||||
- Never store raw passwords
|
||||
- Ensure username is present if password is set
|
||||
|
||||
Returns:
|
||||
Dict with processed values to save and any validation errors.
|
||||
"""
|
||||
password = values.get("BUILTIN_PASSWORD", "")
|
||||
password_confirm = values.get("BUILTIN_PASSWORD_CONFIRM", "")
|
||||
|
||||
# Remove raw password fields - they should never be persisted
|
||||
values.pop("BUILTIN_PASSWORD", None)
|
||||
values.pop("BUILTIN_PASSWORD_CONFIRM", None)
|
||||
|
||||
# If password is provided, validate and hash it
|
||||
if password:
|
||||
if not values.get("BUILTIN_USERNAME"):
|
||||
return {
|
||||
"error": True,
|
||||
"message": "Username cannot be empty",
|
||||
"values": values
|
||||
}
|
||||
|
||||
if password != password_confirm:
|
||||
return {
|
||||
"error": True,
|
||||
"message": "Passwords do not match",
|
||||
"values": values
|
||||
}
|
||||
|
||||
if len(password) < 4:
|
||||
return {
|
||||
"error": True,
|
||||
"message": "Password must be at least 4 characters",
|
||||
"values": values
|
||||
}
|
||||
|
||||
# Hash the password
|
||||
values["BUILTIN_PASSWORD_HASH"] = generate_password_hash(password)
|
||||
logger.info("Password hash updated")
|
||||
|
||||
# If no password provided but username is being set, preserve existing hash
|
||||
elif "BUILTIN_USERNAME" in values:
|
||||
existing = load_config_file("security")
|
||||
if "BUILTIN_PASSWORD_HASH" in existing:
|
||||
values["BUILTIN_PASSWORD_HASH"] = existing["BUILTIN_PASSWORD_HASH"]
|
||||
|
||||
return {"error": False, "values": values}
|
||||
|
||||
|
||||
@register_settings("security", "Security", icon="shield", order=5)
|
||||
def security_settings():
|
||||
"""Security and authentication settings."""
|
||||
from shelfmark.config.env import CWA_DB_PATH
|
||||
|
||||
cwa_db_available = CWA_DB_PATH is not None and CWA_DB_PATH.exists()
|
||||
|
||||
fields = [
|
||||
TextField(
|
||||
key="BUILTIN_USERNAME",
|
||||
label="Username",
|
||||
description="Set a username and password to require login. Leave both empty for public access.",
|
||||
placeholder="Enter username",
|
||||
env_supported=False,
|
||||
disabled_when={"field": "USE_CWA_AUTH", "value": True, "reason": "Using Calibre-Web database for authentication."},
|
||||
),
|
||||
PasswordField(
|
||||
key="BUILTIN_PASSWORD",
|
||||
label="Set Password",
|
||||
description="Fill in to set or change the password.",
|
||||
placeholder="Enter new password",
|
||||
env_supported=False,
|
||||
disabled_when={"field": "USE_CWA_AUTH", "value": True, "reason": "Using Calibre-Web database for authentication."},
|
||||
),
|
||||
PasswordField(
|
||||
key="BUILTIN_PASSWORD_CONFIRM",
|
||||
label="Confirm Password",
|
||||
placeholder="Confirm new password",
|
||||
env_supported=False,
|
||||
disabled_when={"field": "USE_CWA_AUTH", "value": True, "reason": "Using Calibre-Web database for authentication."},
|
||||
),
|
||||
ActionButton(
|
||||
key="clear_credentials",
|
||||
label="Clear Credentials",
|
||||
description="Remove login requirement and make the app publicly accessible.",
|
||||
style="danger",
|
||||
callback=_clear_builtin_credentials,
|
||||
disabled_when={"field": "USE_CWA_AUTH", "value": True, "reason": "Using Calibre-Web database for authentication."},
|
||||
),
|
||||
CheckboxField(
|
||||
key="USE_CWA_AUTH",
|
||||
label="Use Calibre-Web Database",
|
||||
description=(
|
||||
"Use your existing Calibre-Web user credentials for authentication."
|
||||
),
|
||||
default=False,
|
||||
env_supported=False,
|
||||
disabled=not cwa_db_available,
|
||||
disabled_reason="Mount your Calibre-Web app.db to /auth/app.db in docker compose to enable.",
|
||||
),
|
||||
CheckboxField(
|
||||
key="RESTRICT_SETTINGS_TO_ADMIN",
|
||||
label="Restrict Settings to Admins",
|
||||
description=(
|
||||
"Only users with admin role in Calibre-Web can access settings."
|
||||
),
|
||||
default=False,
|
||||
env_supported=False,
|
||||
show_when={"field": "USE_CWA_AUTH", "value": True},
|
||||
),
|
||||
]
|
||||
|
||||
return fields
|
||||
|
||||
|
||||
# Register the on_save handler for this tab
|
||||
register_on_save("security", _on_save_security)
|
||||
@@ -0,0 +1,5 @@
|
||||
"""Core module - shared models, queue, and utilities."""
|
||||
|
||||
from shelfmark.core.models import BookInfo, QueueItem, SearchFilters, QueueStatus
|
||||
from shelfmark.core.queue import BookQueue, book_queue
|
||||
from shelfmark.core.logger import setup_logger
|
||||
@@ -0,0 +1,172 @@
|
||||
"""Thread-safe in-memory cache with TTL support."""
|
||||
|
||||
import threading
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from functools import wraps
|
||||
from typing import Any, Callable, Dict, Optional, TypeVar
|
||||
|
||||
from shelfmark.core.logger import setup_logger
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
T = TypeVar("T")
|
||||
|
||||
|
||||
@dataclass
|
||||
class CacheEntry:
|
||||
"""A cached value with expiration time."""
|
||||
value: Any
|
||||
expires_at: float
|
||||
|
||||
|
||||
class CacheService:
|
||||
"""Thread-safe in-memory cache with TTL support."""
|
||||
|
||||
def __init__(self, max_size: int = 1000):
|
||||
"""Initialize cache with max_size entries before eviction."""
|
||||
self._cache: Dict[str, CacheEntry] = {}
|
||||
self._lock = threading.Lock()
|
||||
self._max_size = max_size
|
||||
|
||||
def get(self, key: str) -> Optional[Any]:
|
||||
"""Get cached value if not expired."""
|
||||
with self._lock:
|
||||
entry = self._cache.get(key)
|
||||
if entry is None:
|
||||
return None
|
||||
|
||||
if time.time() > entry.expires_at:
|
||||
del self._cache[key]
|
||||
return None
|
||||
|
||||
return entry.value
|
||||
|
||||
def set(self, key: str, value: Any, ttl: int) -> None:
|
||||
"""Cache value with TTL in seconds."""
|
||||
with self._lock:
|
||||
# Evict oldest entries if at capacity
|
||||
if len(self._cache) >= self._max_size:
|
||||
self._evict_oldest()
|
||||
|
||||
self._cache[key] = CacheEntry(
|
||||
value=value,
|
||||
expires_at=time.time() + ttl
|
||||
)
|
||||
|
||||
def invalidate(self, key: str) -> bool:
|
||||
"""Remove specific cache entry. Returns True if found."""
|
||||
with self._lock:
|
||||
if key in self._cache:
|
||||
del self._cache[key]
|
||||
return True
|
||||
return False
|
||||
|
||||
def clear(self) -> None:
|
||||
"""Clear all cache entries."""
|
||||
with self._lock:
|
||||
self._cache.clear()
|
||||
|
||||
def cleanup_expired(self) -> int:
|
||||
"""Remove all expired entries. Returns count removed."""
|
||||
with self._lock:
|
||||
now = time.time()
|
||||
expired_keys = [
|
||||
key for key, entry in self._cache.items()
|
||||
if entry.expires_at < now
|
||||
]
|
||||
for key in expired_keys:
|
||||
del self._cache[key]
|
||||
return len(expired_keys)
|
||||
|
||||
def _evict_oldest(self) -> None:
|
||||
"""Evict ~10% of oldest entries. Called with lock held."""
|
||||
if not self._cache:
|
||||
return
|
||||
|
||||
# Remove ~10% of entries, oldest first
|
||||
entries_to_remove = max(1, len(self._cache) // 10)
|
||||
sorted_entries = sorted(
|
||||
self._cache.items(),
|
||||
key=lambda x: x[1].expires_at
|
||||
)
|
||||
|
||||
for key, _ in sorted_entries[:entries_to_remove]:
|
||||
del self._cache[key]
|
||||
|
||||
def stats(self) -> Dict[str, int]:
|
||||
"""Get cache statistics (size, max_size)."""
|
||||
with self._lock:
|
||||
return {
|
||||
"size": len(self._cache),
|
||||
"max_size": self._max_size
|
||||
}
|
||||
|
||||
|
||||
# Global cache instance for metadata providers
|
||||
_metadata_cache = CacheService(max_size=1000)
|
||||
|
||||
|
||||
def get_metadata_cache() -> CacheService:
|
||||
"""Get the global metadata cache instance."""
|
||||
return _metadata_cache
|
||||
|
||||
|
||||
def cache_key(*args, **kwargs) -> str:
|
||||
"""Generate cache key from arguments."""
|
||||
parts = [str(arg) for arg in args]
|
||||
parts.extend(f"{k}={v}" for k, v in sorted(kwargs.items()))
|
||||
return ":".join(parts)
|
||||
|
||||
|
||||
def cacheable(
|
||||
ttl: Optional[int] = None,
|
||||
ttl_key: Optional[str] = None,
|
||||
ttl_default: int = 300,
|
||||
key_prefix: str = ""
|
||||
):
|
||||
"""Decorator for caching function results. Use ttl (static) or ttl_key (from config)."""
|
||||
def decorator(func: Callable[..., T]) -> Callable[..., T]:
|
||||
@wraps(func)
|
||||
def wrapper(*args, **kwargs) -> T:
|
||||
# Check if metadata caching is enabled
|
||||
from shelfmark.core.config import config
|
||||
|
||||
if not config.get("METADATA_CACHE_ENABLED", True):
|
||||
# Caching disabled, execute function directly
|
||||
return func(*args, **kwargs)
|
||||
|
||||
# Determine TTL: static or from config
|
||||
if ttl is not None:
|
||||
effective_ttl = ttl
|
||||
elif ttl_key:
|
||||
effective_ttl = config.get(ttl_key, ttl_default)
|
||||
else:
|
||||
effective_ttl = ttl_default
|
||||
|
||||
# Generate cache key from function name and arguments
|
||||
# Skip 'self' argument if present (first arg of method)
|
||||
cache_args = args[1:] if args and hasattr(args[0], func.__name__) else args
|
||||
|
||||
key = cache_key(
|
||||
key_prefix or func.__name__,
|
||||
*cache_args,
|
||||
**kwargs
|
||||
)
|
||||
|
||||
# Check cache
|
||||
cached = _metadata_cache.get(key)
|
||||
if cached is not None:
|
||||
return cached
|
||||
|
||||
# Execute function and cache result
|
||||
result = func(*args, **kwargs)
|
||||
|
||||
# Only cache non-None results
|
||||
if result is not None:
|
||||
_metadata_cache.set(key, result, effective_ttl)
|
||||
|
||||
return result
|
||||
|
||||
return wrapper
|
||||
return decorator
|
||||
@@ -0,0 +1,183 @@
|
||||
"""Configuration singleton with ENV > config file > default resolution."""
|
||||
|
||||
from threading import Lock
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
# Import lazily to avoid circular imports
|
||||
_registry_module = None
|
||||
_env_module = None
|
||||
|
||||
|
||||
def _get_registry():
|
||||
"""Lazy import of settings registry to avoid circular imports."""
|
||||
global _registry_module
|
||||
if _registry_module is None:
|
||||
from shelfmark.core import settings_registry
|
||||
_registry_module = settings_registry
|
||||
return _registry_module
|
||||
|
||||
|
||||
def _get_env():
|
||||
"""Lazy import of env module for fallback values."""
|
||||
global _env_module
|
||||
if _env_module is None:
|
||||
from shelfmark.config import env
|
||||
_env_module = env
|
||||
return _env_module
|
||||
|
||||
|
||||
class Config:
|
||||
"""
|
||||
Dynamic configuration singleton that provides live settings access.
|
||||
|
||||
Settings are resolved with priority: ENV var > config file > default.
|
||||
Values are cached for performance and can be refreshed when settings change.
|
||||
"""
|
||||
|
||||
_instance: Optional['Config'] = None
|
||||
_lock = Lock()
|
||||
|
||||
def __new__(cls) -> 'Config':
|
||||
if cls._instance is None:
|
||||
with cls._lock:
|
||||
if cls._instance is None:
|
||||
cls._instance = super().__new__(cls)
|
||||
cls._instance._initialized = False
|
||||
return cls._instance
|
||||
|
||||
def __init__(self):
|
||||
if self._initialized:
|
||||
return
|
||||
self._cache: Dict[str, Any] = {}
|
||||
self._field_map: Dict[str, tuple] = {} # key -> (field, tab_name)
|
||||
self._cache_lock = Lock()
|
||||
self._initialized = True
|
||||
self._loaded = False
|
||||
|
||||
def _ensure_loaded(self) -> None:
|
||||
"""Ensure settings are loaded from the registry."""
|
||||
if self._loaded:
|
||||
return
|
||||
with self._cache_lock:
|
||||
if self._loaded:
|
||||
return
|
||||
self._load_settings()
|
||||
|
||||
def _load_settings(self) -> None:
|
||||
"""Load all settings from the registry."""
|
||||
# Ensure all settings modules are imported before loading
|
||||
# This handles cases where config is accessed before settings are registered
|
||||
try:
|
||||
import shelfmark.config.settings # noqa: F401 - main app settings
|
||||
import shelfmark.release_sources # noqa: F401 - plugin settings
|
||||
import shelfmark.metadata_providers # noqa: F401 - plugin settings
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
registry = _get_registry()
|
||||
|
||||
# On first load, sync ENV values to config files
|
||||
# This ensures ENV values persist even if ENV vars are later removed
|
||||
if not hasattr(self, '_env_synced'):
|
||||
registry.sync_env_to_config()
|
||||
self._env_synced = True
|
||||
|
||||
# Build field map from all registered tabs
|
||||
self._field_map.clear()
|
||||
self._cache.clear()
|
||||
|
||||
for tab in registry.get_all_settings_tabs():
|
||||
for field in tab.fields:
|
||||
# Skip action buttons and headings - they don't have values
|
||||
if isinstance(field, (registry.ActionButton, registry.HeadingField)):
|
||||
continue
|
||||
|
||||
key = field.key
|
||||
self._field_map[key] = (field, tab.name)
|
||||
|
||||
# Load current value
|
||||
value = registry.get_setting_value(field, tab.name)
|
||||
self._cache[key] = value
|
||||
|
||||
self._loaded = True
|
||||
|
||||
def refresh(self) -> None:
|
||||
"""
|
||||
Refresh all cached settings from config files.
|
||||
|
||||
Call this after settings are updated via the UI to ensure
|
||||
the config singleton reflects the new values.
|
||||
"""
|
||||
with self._cache_lock:
|
||||
self._loaded = False
|
||||
self._load_settings()
|
||||
|
||||
def get(self, key: str, default: Any = None) -> Any:
|
||||
"""
|
||||
Get a setting value by key.
|
||||
|
||||
Args:
|
||||
key: The setting key (e.g., 'MAX_RETRY')
|
||||
default: Default value if setting not found
|
||||
|
||||
Returns:
|
||||
The setting value, or default if not found
|
||||
"""
|
||||
self._ensure_loaded()
|
||||
return self._cache.get(key, default)
|
||||
|
||||
def __getattr__(self, name: str) -> Any:
|
||||
"""
|
||||
Allow attribute-style access to settings.
|
||||
|
||||
Example: config.MAX_RETRY instead of config.get('MAX_RETRY')
|
||||
"""
|
||||
# Avoid recursion for internal attributes
|
||||
if name.startswith('_'):
|
||||
raise AttributeError(f"'{type(self).__name__}' object has no attribute '{name}'")
|
||||
|
||||
self._ensure_loaded()
|
||||
|
||||
if name in self._cache:
|
||||
return self._cache[name]
|
||||
|
||||
# Fallback to env module for settings not in registry
|
||||
# This ensures backward compatibility during migration
|
||||
env = _get_env()
|
||||
if hasattr(env, name):
|
||||
return getattr(env, name)
|
||||
|
||||
raise AttributeError(f"Setting '{name}' not found in config or env")
|
||||
|
||||
def is_from_env(self, key: str) -> bool:
|
||||
"""
|
||||
Check if a setting's value comes from an environment variable.
|
||||
|
||||
Args:
|
||||
key: The setting key
|
||||
|
||||
Returns:
|
||||
True if the value is set via ENV var, False otherwise
|
||||
"""
|
||||
self._ensure_loaded()
|
||||
|
||||
if key not in self._field_map:
|
||||
return False
|
||||
|
||||
field, _ = self._field_map[key]
|
||||
registry = _get_registry()
|
||||
return registry.is_value_from_env(field)
|
||||
|
||||
def get_all(self) -> Dict[str, Any]:
|
||||
"""
|
||||
Get all cached settings as a dictionary.
|
||||
|
||||
Returns:
|
||||
Dict of all setting keys to their current values
|
||||
"""
|
||||
self._ensure_loaded()
|
||||
return dict(self._cache)
|
||||
|
||||
|
||||
# Global singleton instance
|
||||
config = Config()
|
||||
@@ -0,0 +1,569 @@
|
||||
"""Disk-based image cache with LRU eviction."""
|
||||
|
||||
import json
|
||||
import os
|
||||
import threading
|
||||
import time
|
||||
from io import BytesIO
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, Optional, Tuple
|
||||
|
||||
import requests
|
||||
|
||||
from shelfmark.core.logger import setup_logger
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
# Image type detection via magic bytes
|
||||
IMAGE_SIGNATURES = {
|
||||
b'\xff\xd8\xff': ('image/jpeg', 'jpg'),
|
||||
b'\x89PNG\r\n\x1a\n': ('image/png', 'png'),
|
||||
b'GIF87a': ('image/gif', 'gif'),
|
||||
b'GIF89a': ('image/gif', 'gif'),
|
||||
b'RIFF': ('image/webp', 'webp'), # WebP starts with RIFF
|
||||
}
|
||||
|
||||
# HTTP headers for image fetching
|
||||
FETCH_HEADERS = {
|
||||
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) Chrome/129.0.0.0 Safari/537.36',
|
||||
'Accept': 'image/webp,image/apng,image/*,*/*;q=0.8',
|
||||
'Accept-Language': 'en-US,en;q=0.5',
|
||||
}
|
||||
|
||||
# Maximum image size to fetch (5 MB)
|
||||
MAX_IMAGE_SIZE = 5 * 1024 * 1024
|
||||
|
||||
# Negative cache TTL (for failed fetches) - 1 hour
|
||||
NEGATIVE_CACHE_TTL = 3600
|
||||
|
||||
# Transient failure cache TTL (for timeouts/connection errors) - 60 seconds
|
||||
# Short enough to retry soon, long enough to prevent spam during one page view
|
||||
TRANSIENT_CACHE_TTL = 60
|
||||
|
||||
|
||||
def _detect_image_type(data: bytes) -> Optional[Tuple[str, str]]:
|
||||
"""Detect image type from magic bytes.
|
||||
|
||||
Args:
|
||||
data: Image data bytes
|
||||
|
||||
Returns:
|
||||
Tuple of (content_type, extension) or None if not recognized
|
||||
"""
|
||||
for signature, (content_type, ext) in IMAGE_SIGNATURES.items():
|
||||
if data.startswith(signature):
|
||||
return content_type, ext
|
||||
|
||||
# Special case for WebP - check for WEBP after RIFF
|
||||
if data.startswith(b'RIFF') and len(data) > 12 and data[8:12] == b'WEBP':
|
||||
return 'image/webp', 'webp'
|
||||
|
||||
return None
|
||||
|
||||
|
||||
class ImageCacheService:
|
||||
"""Persistent image cache with LRU eviction and TTL support."""
|
||||
|
||||
def __init__(self, cache_dir: Path, max_size_mb: int = 500, ttl_seconds: int = 0):
|
||||
"""Initialize the image cache.
|
||||
|
||||
Args:
|
||||
cache_dir: Directory to store cached images
|
||||
max_size_mb: Maximum cache size in megabytes
|
||||
ttl_seconds: Time-to-live in seconds (0 = forever)
|
||||
"""
|
||||
self.cache_dir = cache_dir
|
||||
self.max_size_bytes = max_size_mb * 1024 * 1024
|
||||
self.ttl_seconds = ttl_seconds
|
||||
self.index_path = cache_dir / "cache_index.json"
|
||||
self._lock = threading.RLock()
|
||||
self._index: Dict[str, Dict[str, Any]] = {}
|
||||
|
||||
# Stats tracking
|
||||
self._hits = 0
|
||||
self._misses = 0
|
||||
|
||||
# Ensure cache directory exists
|
||||
self.cache_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# Load existing index and sync with files on disk (once at startup)
|
||||
self._load_index()
|
||||
self._sync_index_with_files()
|
||||
|
||||
def _load_index(self) -> None:
|
||||
"""Load cache index from disk."""
|
||||
if not self.index_path.exists():
|
||||
self._index = {}
|
||||
return
|
||||
|
||||
try:
|
||||
with open(self.index_path, 'r') as f:
|
||||
self._index = json.load(f)
|
||||
except (json.JSONDecodeError, IOError):
|
||||
self._index = {}
|
||||
|
||||
def _sync_index_with_files(self) -> None:
|
||||
"""Sync cache index with actual files on disk.
|
||||
|
||||
- Adds entries for files that exist but aren't in index
|
||||
- Removes entries for files that no longer exist (non-negative only)
|
||||
- Preserves negative cache entries (they have no files)
|
||||
"""
|
||||
image_extensions = {'.jpg', '.jpeg', '.png', '.gif', '.webp'}
|
||||
added_count = 0
|
||||
removed_count = 0
|
||||
|
||||
# Build set of files that exist on disk
|
||||
existing_files: Dict[str, Path] = {}
|
||||
for file_path in self.cache_dir.iterdir():
|
||||
if not file_path.is_file():
|
||||
continue
|
||||
if file_path.suffix.lower() not in image_extensions:
|
||||
continue
|
||||
existing_files[file_path.stem] = file_path
|
||||
|
||||
# Add files that aren't in the index
|
||||
for cache_id, file_path in existing_files.items():
|
||||
if cache_id in self._index:
|
||||
continue
|
||||
|
||||
ext = file_path.suffix.lstrip('.')
|
||||
stat = file_path.stat()
|
||||
|
||||
# Detect content type
|
||||
try:
|
||||
with open(file_path, 'rb') as f:
|
||||
header = f.read(16)
|
||||
detected = _detect_image_type(header)
|
||||
content_type = detected[0] if detected else f'image/{ext}'
|
||||
except IOError:
|
||||
content_type = f'image/{ext}'
|
||||
|
||||
self._index[cache_id] = {
|
||||
'ext': ext,
|
||||
'content_type': content_type,
|
||||
'size': stat.st_size,
|
||||
'cached_at': stat.st_mtime,
|
||||
'accessed_at': stat.st_mtime,
|
||||
}
|
||||
added_count += 1
|
||||
|
||||
# Remove index entries for missing files (skip negative cache entries)
|
||||
stale_entries = []
|
||||
for cache_id, entry in self._index.items():
|
||||
if entry.get('negative', False):
|
||||
continue # Negative entries don't have files
|
||||
if cache_id not in existing_files:
|
||||
stale_entries.append(cache_id)
|
||||
|
||||
for cache_id in stale_entries:
|
||||
del self._index[cache_id]
|
||||
removed_count += 1
|
||||
|
||||
if added_count > 0 or removed_count > 0:
|
||||
self._save_index()
|
||||
|
||||
def _save_index(self) -> None:
|
||||
"""Save cache index to disk."""
|
||||
try:
|
||||
# Write to temp file first, then rename for atomicity
|
||||
temp_path = self.index_path.with_suffix('.tmp')
|
||||
with open(temp_path, 'w') as f:
|
||||
json.dump(self._index, f)
|
||||
temp_path.rename(self.index_path)
|
||||
except IOError:
|
||||
pass
|
||||
|
||||
def _get_image_path(self, cache_id: str, ext: str) -> Path:
|
||||
"""Get the file path for a cached image."""
|
||||
return self.cache_dir / f"{cache_id}.{ext}"
|
||||
|
||||
def _is_expired(self, entry: Dict[str, Any]) -> bool:
|
||||
"""Check if a cache entry is expired."""
|
||||
if self.ttl_seconds == 0:
|
||||
return False
|
||||
return (time.time() - entry.get('cached_at', 0)) > self.ttl_seconds
|
||||
|
||||
def _is_negative_expired(self, entry: Dict[str, Any]) -> bool:
|
||||
"""Check if a negative cache entry is expired.
|
||||
|
||||
Transient failures (timeouts) expire after TRANSIENT_CACHE_TTL (60s).
|
||||
Permanent failures (404s) expire after NEGATIVE_CACHE_TTL (1 hour).
|
||||
"""
|
||||
if not entry.get('negative', False):
|
||||
return False
|
||||
|
||||
cached_at = entry.get('cached_at', 0)
|
||||
ttl = TRANSIENT_CACHE_TTL if entry.get('transient', False) else NEGATIVE_CACHE_TTL
|
||||
return (time.time() - cached_at) > ttl
|
||||
|
||||
def _calculate_total_size(self) -> int:
|
||||
"""Calculate total size of cached images."""
|
||||
return sum(entry.get('size', 0) for entry in self._index.values())
|
||||
|
||||
def _evict_if_needed(self, required_space: int = 0) -> None:
|
||||
"""Evict old entries if cache is over size limit.
|
||||
|
||||
Uses LRU eviction based on accessed_at timestamp.
|
||||
"""
|
||||
current_size = self._calculate_total_size()
|
||||
target_size = self.max_size_bytes - required_space
|
||||
|
||||
if current_size <= target_size:
|
||||
return
|
||||
|
||||
# Sort entries by accessed_at (oldest first)
|
||||
sorted_entries = sorted(
|
||||
self._index.items(),
|
||||
key=lambda x: x[1].get('accessed_at', 0)
|
||||
)
|
||||
|
||||
evicted_count = 0
|
||||
for cache_id, entry in sorted_entries:
|
||||
if current_size <= target_size:
|
||||
break
|
||||
|
||||
# Delete the image file
|
||||
ext = entry.get('ext', 'jpg')
|
||||
image_path = self._get_image_path(cache_id, ext)
|
||||
try:
|
||||
if image_path.exists():
|
||||
image_path.unlink()
|
||||
except IOError:
|
||||
pass
|
||||
|
||||
# Update tracking
|
||||
current_size -= entry.get('size', 0)
|
||||
del self._index[cache_id]
|
||||
evicted_count += 1
|
||||
|
||||
if evicted_count > 0:
|
||||
self._save_index()
|
||||
|
||||
def get(self, cache_id: str) -> Optional[Tuple[bytes, str]]:
|
||||
"""Get a cached image.
|
||||
|
||||
Args:
|
||||
cache_id: Cache key (book ID or composite key)
|
||||
|
||||
Returns:
|
||||
Tuple of (image_data, content_type) or None if not cached/expired
|
||||
"""
|
||||
with self._lock:
|
||||
entry = self._index.get(cache_id)
|
||||
|
||||
# Try reloading from disk if not found (handles multiprocess case)
|
||||
if not entry:
|
||||
self._load_index()
|
||||
entry = self._index.get(cache_id)
|
||||
if not entry:
|
||||
self._misses += 1
|
||||
return None
|
||||
|
||||
# Check for negative cache (failed fetch)
|
||||
if entry.get('negative', False):
|
||||
if self._is_negative_expired(entry):
|
||||
# Negative cache expired, allow retry
|
||||
del self._index[cache_id]
|
||||
self._save_index()
|
||||
self._misses += 1
|
||||
return None
|
||||
# Still in negative cache, return None (don't retry)
|
||||
return None
|
||||
|
||||
# Check for expired entry
|
||||
if self._is_expired(entry):
|
||||
# Remove expired entry
|
||||
ext = entry.get('ext', 'jpg')
|
||||
image_path = self._get_image_path(cache_id, ext)
|
||||
try:
|
||||
if image_path.exists():
|
||||
image_path.unlink()
|
||||
except IOError:
|
||||
pass
|
||||
del self._index[cache_id]
|
||||
self._save_index()
|
||||
self._misses += 1
|
||||
return None
|
||||
|
||||
# Try to read the cached image
|
||||
ext = entry.get('ext', 'jpg')
|
||||
content_type = entry.get('content_type', 'image/jpeg')
|
||||
image_path = self._get_image_path(cache_id, ext)
|
||||
|
||||
try:
|
||||
if not image_path.exists():
|
||||
# File missing, remove from index
|
||||
del self._index[cache_id]
|
||||
self._save_index()
|
||||
self._misses += 1
|
||||
return None
|
||||
|
||||
with open(image_path, 'rb') as f:
|
||||
data = f.read()
|
||||
|
||||
# Update accessed time
|
||||
entry['accessed_at'] = time.time()
|
||||
self._save_index()
|
||||
|
||||
self._hits += 1
|
||||
return data, content_type
|
||||
|
||||
except IOError:
|
||||
self._misses += 1
|
||||
return None
|
||||
|
||||
def put(self, cache_id: str, data: bytes, content_type: str) -> bool:
|
||||
"""Store an image in the cache.
|
||||
|
||||
Args:
|
||||
cache_id: Cache key
|
||||
data: Image data bytes
|
||||
content_type: MIME type of the image
|
||||
|
||||
Returns:
|
||||
True if stored successfully
|
||||
"""
|
||||
with self._lock:
|
||||
# Detect image type for extension
|
||||
detected = _detect_image_type(data)
|
||||
if detected:
|
||||
content_type, ext = detected
|
||||
else:
|
||||
# Fall back to content-type header
|
||||
if 'jpeg' in content_type or 'jpg' in content_type:
|
||||
ext = 'jpg'
|
||||
elif 'png' in content_type:
|
||||
ext = 'png'
|
||||
elif 'gif' in content_type:
|
||||
ext = 'gif'
|
||||
elif 'webp' in content_type:
|
||||
ext = 'webp'
|
||||
else:
|
||||
ext = 'jpg' # Default
|
||||
|
||||
image_size = len(data)
|
||||
|
||||
# Evict if needed to make room
|
||||
self._evict_if_needed(image_size)
|
||||
|
||||
# Write image to disk
|
||||
image_path = self._get_image_path(cache_id, ext)
|
||||
try:
|
||||
with open(image_path, 'wb') as f:
|
||||
f.write(data)
|
||||
except IOError:
|
||||
return False
|
||||
|
||||
# Update index
|
||||
now = time.time()
|
||||
self._index[cache_id] = {
|
||||
'ext': ext,
|
||||
'content_type': content_type,
|
||||
'size': image_size,
|
||||
'cached_at': now,
|
||||
'accessed_at': now,
|
||||
'negative': False,
|
||||
}
|
||||
self._save_index()
|
||||
return True
|
||||
|
||||
def put_negative(self, cache_id: str, transient: bool = False) -> None:
|
||||
"""Store a negative cache entry (failed fetch).
|
||||
|
||||
Args:
|
||||
cache_id: Cache key
|
||||
transient: If True, uses shorter TTL (for timeouts/connection errors)
|
||||
"""
|
||||
with self._lock:
|
||||
self._index[cache_id] = {
|
||||
'negative': True,
|
||||
'transient': transient,
|
||||
'cached_at': time.time(),
|
||||
}
|
||||
self._save_index()
|
||||
|
||||
def delete(self, cache_id: str) -> bool:
|
||||
"""Delete a single cache entry.
|
||||
|
||||
Args:
|
||||
cache_id: Cache key
|
||||
|
||||
Returns:
|
||||
True if entry existed and was deleted
|
||||
"""
|
||||
with self._lock:
|
||||
entry = self._index.get(cache_id)
|
||||
if not entry:
|
||||
return False
|
||||
|
||||
# Delete file if it exists
|
||||
if not entry.get('negative', False):
|
||||
ext = entry.get('ext', 'jpg')
|
||||
image_path = self._get_image_path(cache_id, ext)
|
||||
try:
|
||||
if image_path.exists():
|
||||
image_path.unlink()
|
||||
except IOError:
|
||||
pass
|
||||
|
||||
del self._index[cache_id]
|
||||
self._save_index()
|
||||
return True
|
||||
|
||||
def clear(self) -> int:
|
||||
"""Clear all cached images.
|
||||
|
||||
Returns:
|
||||
Number of entries cleared
|
||||
"""
|
||||
with self._lock:
|
||||
count = len(self._index)
|
||||
|
||||
# Delete all image files
|
||||
for cache_id, entry in self._index.items():
|
||||
if not entry.get('negative', False):
|
||||
ext = entry.get('ext', 'jpg')
|
||||
image_path = self._get_image_path(cache_id, ext)
|
||||
try:
|
||||
if image_path.exists():
|
||||
image_path.unlink()
|
||||
except IOError:
|
||||
pass
|
||||
|
||||
# Clear index
|
||||
self._index = {}
|
||||
self._save_index()
|
||||
|
||||
# Reset stats
|
||||
self._hits = 0
|
||||
self._misses = 0
|
||||
|
||||
return count
|
||||
|
||||
def stats(self) -> Dict[str, Any]:
|
||||
"""Get cache statistics.
|
||||
|
||||
Returns:
|
||||
Dict with size, count, hit rate, etc.
|
||||
"""
|
||||
with self._lock:
|
||||
total_size = self._calculate_total_size()
|
||||
entry_count = len(self._index)
|
||||
negative_count = sum(1 for e in self._index.values() if e.get('negative', False))
|
||||
total_requests = self._hits + self._misses
|
||||
hit_rate = (self._hits / total_requests * 100) if total_requests > 0 else 0
|
||||
|
||||
return {
|
||||
'entry_count': entry_count,
|
||||
'negative_count': negative_count,
|
||||
'total_size_bytes': total_size,
|
||||
'total_size_mb': round(total_size / (1024 * 1024), 2),
|
||||
'max_size_mb': self.max_size_bytes / (1024 * 1024),
|
||||
'hits': self._hits,
|
||||
'misses': self._misses,
|
||||
'hit_rate': round(hit_rate, 1),
|
||||
}
|
||||
|
||||
def fetch_and_cache(self, cache_id: str, url: str) -> Optional[Tuple[bytes, str]]:
|
||||
"""Fetch an image from URL and cache it.
|
||||
|
||||
Args:
|
||||
cache_id: Cache key
|
||||
url: URL to fetch from
|
||||
|
||||
Returns:
|
||||
Tuple of (image_data, content_type) or None on failure
|
||||
"""
|
||||
try:
|
||||
|
||||
response = requests.get(
|
||||
url,
|
||||
timeout=(5, 10),
|
||||
headers=FETCH_HEADERS,
|
||||
stream=True,
|
||||
)
|
||||
response.raise_for_status()
|
||||
|
||||
# Validate content type
|
||||
content_type = response.headers.get('content-type', '')
|
||||
if not content_type.startswith('image/'):
|
||||
self.put_negative(cache_id)
|
||||
return None
|
||||
|
||||
# Read with size limit
|
||||
data = BytesIO()
|
||||
for chunk in response.iter_content(chunk_size=8192):
|
||||
data.write(chunk)
|
||||
if data.tell() > MAX_IMAGE_SIZE:
|
||||
self.put_negative(cache_id)
|
||||
return None
|
||||
|
||||
image_data = data.getvalue()
|
||||
|
||||
if not image_data:
|
||||
self.put_negative(cache_id)
|
||||
return None
|
||||
|
||||
# Store in cache
|
||||
if self.put(cache_id, image_data, content_type):
|
||||
# Get the actual content type from detection
|
||||
detected = _detect_image_type(image_data)
|
||||
if detected:
|
||||
content_type = detected[0]
|
||||
return image_data, content_type
|
||||
|
||||
return None
|
||||
|
||||
except requests.exceptions.Timeout:
|
||||
self.put_negative(cache_id, transient=True)
|
||||
return None
|
||||
except requests.exceptions.ConnectionError:
|
||||
self.put_negative(cache_id, transient=True)
|
||||
return None
|
||||
except requests.exceptions.HTTPError as e:
|
||||
is_404 = e.response is not None and e.response.status_code == 404
|
||||
self.put_negative(cache_id, transient=not is_404)
|
||||
return None
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
# Singleton instance (initialized lazily when config is available)
|
||||
_instance: Optional[ImageCacheService] = None
|
||||
_instance_lock = threading.Lock()
|
||||
|
||||
|
||||
def get_image_cache() -> ImageCacheService:
|
||||
"""Get the singleton image cache instance.
|
||||
|
||||
Lazily initializes using config values.
|
||||
"""
|
||||
global _instance
|
||||
|
||||
if _instance is None:
|
||||
with _instance_lock:
|
||||
if _instance is None:
|
||||
from shelfmark.core.config import config
|
||||
from shelfmark.config.env import CONFIG_DIR
|
||||
|
||||
cache_dir = CONFIG_DIR / "covers"
|
||||
max_size_mb = config.get("COVERS_CACHE_MAX_SIZE_MB", 500)
|
||||
ttl_days = config.get("COVERS_CACHE_TTL", 0)
|
||||
ttl_seconds = ttl_days * 86400 if ttl_days > 0 else 0
|
||||
|
||||
_instance = ImageCacheService(
|
||||
cache_dir=cache_dir,
|
||||
max_size_mb=max_size_mb,
|
||||
ttl_seconds=ttl_seconds,
|
||||
)
|
||||
logger.debug(f"Initialized image cache: {cache_dir} (max {max_size_mb}MB, TTL {ttl_days} days)")
|
||||
|
||||
return _instance
|
||||
|
||||
|
||||
def reset_image_cache() -> None:
|
||||
"""Reset the singleton instance (for testing or config changes)."""
|
||||
global _instance
|
||||
with _instance_lock:
|
||||
_instance = None
|
||||
@@ -1,72 +1,80 @@
|
||||
"""Centralized logging configuration for the book downloader application."""
|
||||
"""Logging configuration and custom logger with error tracing."""
|
||||
|
||||
import logging
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from logging.handlers import RotatingFileHandler
|
||||
from env import LOG_FILE, ENABLE_LOGGING, LOG_LEVEL
|
||||
from typing import Any
|
||||
|
||||
from shelfmark.config.env import LOG_FILE, ENABLE_LOGGING, LOG_LEVEL
|
||||
|
||||
|
||||
class CustomLogger(logging.Logger):
|
||||
"""Custom logger class with additional error_trace method."""
|
||||
|
||||
|
||||
def error_trace(self, msg: Any, *args: Any, **kwargs: Any) -> None:
|
||||
"""Log an error message with full stack trace."""
|
||||
self.log_resource_usage()
|
||||
kwargs.pop('exc_info', None)
|
||||
self.error(msg, *args, exc_info=True, **kwargs)
|
||||
|
||||
def warning_trace(self, msg: Any, *args: Any, **kwargs: Any) -> None:
|
||||
"""Log a warning message with full stack trace."""
|
||||
self.log_resource_usage()
|
||||
kwargs.pop('exc_info', None)
|
||||
self.warning(msg, *args, exc_info=True, **kwargs)
|
||||
|
||||
|
||||
def info_trace(self, msg: Any, *args: Any, **kwargs: Any) -> None:
|
||||
"""Log an info message with full stack trace."""
|
||||
self.log_resource_usage()
|
||||
self.info(msg, *args, exc_info=True, **kwargs)
|
||||
|
||||
"""Log an info message (stack trace only if exception active)."""
|
||||
kwargs.pop('exc_info', None)
|
||||
# Only include exc_info if there's actually an exception
|
||||
has_exception = sys.exc_info()[0] is not None
|
||||
self.info(msg, *args, exc_info=has_exception, **kwargs)
|
||||
|
||||
def debug_trace(self, msg: Any, *args: Any, **kwargs: Any) -> None:
|
||||
"""Log a debug message with full stack trace."""
|
||||
self.log_resource_usage()
|
||||
self.debug(msg, *args, exc_info=True, **kwargs)
|
||||
|
||||
"""Log a debug message (stack trace only if exception active)."""
|
||||
kwargs.pop('exc_info', None)
|
||||
# Only include exc_info if there's actually an exception
|
||||
has_exception = sys.exc_info()[0] is not None
|
||||
self.debug(msg, *args, exc_info=has_exception, **kwargs)
|
||||
|
||||
def log_resource_usage(self):
|
||||
import psutil
|
||||
|
||||
# Sum RSS of all processes for actual app memory
|
||||
app_memory_mb = 0
|
||||
for proc in psutil.process_iter(['memory_info']):
|
||||
try:
|
||||
if proc.info['memory_info']:
|
||||
app_memory_mb += proc.info['memory_info'].rss / (1024 * 1024)
|
||||
except (psutil.NoSuchProcess, psutil.AccessDenied):
|
||||
continue
|
||||
|
||||
memory = psutil.virtual_memory()
|
||||
system_used_mb = memory.used / (1024 * 1024)
|
||||
available_mb = memory.available / (1024 * 1024)
|
||||
memory_used_mb = memory.used / (1024 * 1024)
|
||||
cpu_percent = psutil.cpu_percent()
|
||||
self.debug(f"Container Memory: Available={available_mb:.2f} MB, Used={memory_used_mb:.2f} MB, CPU: {cpu_percent:.2f}%")
|
||||
self.debug(f"Container Memory: App={app_memory_mb:.2f} MB, System={system_used_mb:.2f} MB, Available={available_mb:.2f} MB, CPU: {cpu_percent:.2f}%")
|
||||
|
||||
|
||||
def setup_logger(name: str, log_file: Path = LOG_FILE) -> CustomLogger:
|
||||
"""Set up and configure a logger instance.
|
||||
|
||||
|
||||
Args:
|
||||
name: The name of the logger instance
|
||||
log_file: Optional path to log file. If None, logs only to stdout/stderr
|
||||
|
||||
|
||||
Returns:
|
||||
CustomLogger: Configured logger instance with error_trace method
|
||||
"""
|
||||
# Register our custom logger class
|
||||
logging.setLoggerClass(CustomLogger)
|
||||
|
||||
|
||||
# Create logger as CustomLogger instance
|
||||
logger = CustomLogger(name)
|
||||
log_level = logging.INFO
|
||||
if LOG_LEVEL == "DEBUG":
|
||||
log_level = logging.DEBUG
|
||||
elif LOG_LEVEL == "INFO":
|
||||
log_level = logging.INFO
|
||||
elif LOG_LEVEL == "WARNING":
|
||||
log_level = logging.WARNING
|
||||
elif LOG_LEVEL == "ERROR":
|
||||
log_level = logging.ERROR
|
||||
elif LOG_LEVEL == "CRITICAL":
|
||||
log_level = logging.CRITICAL
|
||||
log_level = getattr(logging, LOG_LEVEL, logging.INFO)
|
||||
logger.setLevel(log_level)
|
||||
|
||||
|
||||
formatter = logging.Formatter(
|
||||
'%(asctime)s - %(name)s - %(levelname)s - %(filename)s:%(lineno)d - %(message)s'
|
||||
)
|
||||
@@ -77,13 +85,13 @@ def setup_logger(name: str, log_file: Path = LOG_FILE) -> CustomLogger:
|
||||
console_handler.setLevel(log_level)
|
||||
console_handler.addFilter(lambda record: record.levelno < logging.ERROR) # Only allow logs below ERROR to stdout
|
||||
logger.addHandler(console_handler)
|
||||
|
||||
|
||||
# Error handler for stderr
|
||||
error_handler = logging.StreamHandler(sys.stderr)
|
||||
error_handler.setLevel(logging.ERROR) # Error and above go to stderr
|
||||
error_handler.setFormatter(formatter)
|
||||
logger.addHandler(error_handler)
|
||||
|
||||
|
||||
# File handler if log file is specified
|
||||
try:
|
||||
if ENABLE_LOGGING:
|
||||
@@ -101,4 +109,3 @@ def setup_logger(name: str, log_file: Path = LOG_FILE) -> CustomLogger:
|
||||
logger.error_trace(f"Failed to create log file: {e}", exc_info=True)
|
||||
|
||||
return logger
|
||||
|
||||
@@ -0,0 +1,213 @@
|
||||
"""Centralized mirror configuration for all download sources."""
|
||||
|
||||
from typing import List
|
||||
|
||||
# Lazy import to avoid circular imports
|
||||
_config_module = None
|
||||
|
||||
|
||||
def _get_config():
|
||||
"""Lazy import of config module to avoid circular imports."""
|
||||
global _config_module
|
||||
if _config_module is None:
|
||||
from shelfmark.core.config import config
|
||||
_config_module = config
|
||||
return _config_module
|
||||
|
||||
|
||||
# Default mirror lists (hardcoded fallbacks)
|
||||
DEFAULT_AA_MIRRORS = [
|
||||
"https://annas-archive.se",
|
||||
"https://annas-archive.li",
|
||||
"https://annas-archive.pm",
|
||||
"https://annas-archive.in",
|
||||
]
|
||||
|
||||
DEFAULT_LIBGEN_MIRRORS = [
|
||||
"https://libgen.gl",
|
||||
"https://libgen.li",
|
||||
"https://libgen.bz",
|
||||
"https://libgen.la",
|
||||
"https://libgen.vg",
|
||||
]
|
||||
|
||||
DEFAULT_ZLIB_MIRRORS = [
|
||||
"https://z-lib.fm",
|
||||
"https://z-lib.gs",
|
||||
"https://z-lib.id",
|
||||
"https://z-library.sk",
|
||||
"https://zlibrary-global.se",
|
||||
]
|
||||
|
||||
DEFAULT_WELIB_MIRRORS = [
|
||||
"https://welib.org",
|
||||
]
|
||||
|
||||
|
||||
def get_aa_mirrors() -> List[str]:
|
||||
"""
|
||||
Get Anna's Archive mirrors from config + defaults.
|
||||
|
||||
Returns:
|
||||
List of AA mirror URLs, starting with defaults then custom additions.
|
||||
"""
|
||||
mirrors = list(DEFAULT_AA_MIRRORS)
|
||||
config = _get_config()
|
||||
|
||||
additional = config.get("AA_ADDITIONAL_URLS", "")
|
||||
if additional:
|
||||
for url in additional.split(","):
|
||||
url = url.strip()
|
||||
if url and url not in mirrors:
|
||||
mirrors.append(url)
|
||||
|
||||
return mirrors
|
||||
|
||||
|
||||
def get_libgen_mirrors() -> List[str]:
|
||||
"""
|
||||
Get LibGen mirrors: defaults + any additional from config.
|
||||
|
||||
Returns:
|
||||
List of LibGen mirror URLs (defaults first, then custom additions).
|
||||
"""
|
||||
mirrors = list(DEFAULT_LIBGEN_MIRRORS)
|
||||
config = _get_config()
|
||||
|
||||
additional = config.get("LIBGEN_ADDITIONAL_URLS", "")
|
||||
if additional:
|
||||
for url in additional.split(","):
|
||||
url = url.strip()
|
||||
if url and url not in mirrors:
|
||||
mirrors.append(url)
|
||||
|
||||
return mirrors
|
||||
|
||||
|
||||
def get_zlib_mirrors() -> List[str]:
|
||||
"""
|
||||
Get Z-Library mirrors, with primary first.
|
||||
|
||||
Returns:
|
||||
List of Z-Library mirror URLs, primary first.
|
||||
"""
|
||||
config = _get_config()
|
||||
|
||||
primary = config.get("ZLIB_PRIMARY_URL", DEFAULT_ZLIB_MIRRORS[0])
|
||||
mirrors = [primary]
|
||||
|
||||
# Add other defaults (excluding primary)
|
||||
for url in DEFAULT_ZLIB_MIRRORS:
|
||||
if url != primary:
|
||||
mirrors.append(url)
|
||||
|
||||
# Add custom mirrors
|
||||
additional = config.get("ZLIB_ADDITIONAL_URLS", "")
|
||||
if additional:
|
||||
for url in additional.split(","):
|
||||
url = url.strip()
|
||||
if url and url not in mirrors:
|
||||
mirrors.append(url)
|
||||
|
||||
return mirrors
|
||||
|
||||
|
||||
def get_zlib_primary_url() -> str:
|
||||
"""
|
||||
Get the primary Z-Library mirror URL.
|
||||
|
||||
Returns:
|
||||
Primary Z-Library mirror URL.
|
||||
"""
|
||||
config = _get_config()
|
||||
return config.get("ZLIB_PRIMARY_URL", DEFAULT_ZLIB_MIRRORS[0])
|
||||
|
||||
|
||||
def get_zlib_url_template() -> str:
|
||||
"""
|
||||
Get Z-Library URL template using configured primary mirror.
|
||||
|
||||
Returns:
|
||||
URL template with {md5} placeholder.
|
||||
"""
|
||||
primary = get_zlib_primary_url()
|
||||
return f"{primary}/md5/{{md5}}"
|
||||
|
||||
|
||||
def get_welib_mirrors() -> List[str]:
|
||||
"""
|
||||
Get Welib mirrors, with primary first.
|
||||
|
||||
Returns:
|
||||
List of Welib mirror URLs, primary first.
|
||||
"""
|
||||
config = _get_config()
|
||||
|
||||
primary = config.get("WELIB_PRIMARY_URL", DEFAULT_WELIB_MIRRORS[0])
|
||||
mirrors = [primary]
|
||||
|
||||
# Add other defaults (excluding primary)
|
||||
for url in DEFAULT_WELIB_MIRRORS:
|
||||
if url != primary:
|
||||
mirrors.append(url)
|
||||
|
||||
# Add custom mirrors
|
||||
additional = config.get("WELIB_ADDITIONAL_URLS", "")
|
||||
if additional:
|
||||
for url in additional.split(","):
|
||||
url = url.strip()
|
||||
if url and url not in mirrors:
|
||||
mirrors.append(url)
|
||||
|
||||
return mirrors
|
||||
|
||||
|
||||
def get_welib_primary_url() -> str:
|
||||
"""
|
||||
Get the primary Welib mirror URL.
|
||||
|
||||
Returns:
|
||||
Primary Welib mirror URL.
|
||||
"""
|
||||
config = _get_config()
|
||||
return config.get("WELIB_PRIMARY_URL", DEFAULT_WELIB_MIRRORS[0])
|
||||
|
||||
|
||||
def get_welib_url_template() -> str:
|
||||
"""
|
||||
Get Welib URL template using configured primary mirror.
|
||||
|
||||
Returns:
|
||||
URL template with {md5} placeholder.
|
||||
"""
|
||||
primary = get_welib_primary_url()
|
||||
return f"{primary}/md5/{{md5}}"
|
||||
|
||||
|
||||
def get_zlib_cookie_domains() -> set:
|
||||
"""
|
||||
Get set of Z-Library domains that need full cookie handling.
|
||||
|
||||
Used by internal_bypasser for CF bypass cookie management.
|
||||
|
||||
Returns:
|
||||
Set of domain strings (without protocol).
|
||||
"""
|
||||
domains = set()
|
||||
|
||||
# Add all default domains
|
||||
for url in DEFAULT_ZLIB_MIRRORS:
|
||||
domain = url.replace("https://", "").replace("http://", "").split("/")[0]
|
||||
domains.add(domain)
|
||||
|
||||
# Add custom domains
|
||||
config = _get_config()
|
||||
additional = config.get("ZLIB_ADDITIONAL_URLS", "")
|
||||
if additional:
|
||||
for url in additional.split(","):
|
||||
url = url.strip()
|
||||
if url:
|
||||
domain = url.replace("https://", "").replace("http://", "").split("/")[0]
|
||||
domains.add(domain)
|
||||
|
||||
return domains
|
||||
@@ -0,0 +1,170 @@
|
||||
"""Data structures and models used across the application."""
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Dict, List, Optional
|
||||
from enum import Enum
|
||||
import re
|
||||
import time
|
||||
|
||||
|
||||
def build_filename(
|
||||
title: str,
|
||||
author: Optional[str] = None,
|
||||
year: Optional[str] = None,
|
||||
fmt: Optional[str] = None,
|
||||
) -> str:
|
||||
parts = []
|
||||
if author:
|
||||
parts.append(author)
|
||||
parts.append(" - ")
|
||||
parts.append(title)
|
||||
if year:
|
||||
parts.append(f" ({year})")
|
||||
|
||||
filename = "".join(parts)
|
||||
filename = re.sub(r'[\\/:*?"<>|]', '_', filename.strip())[:245]
|
||||
|
||||
if fmt:
|
||||
filename = f"{filename}.{fmt}"
|
||||
|
||||
return filename
|
||||
|
||||
|
||||
class QueueStatus(str, Enum):
|
||||
"""Enum for possible book queue statuses."""
|
||||
QUEUED = "queued"
|
||||
RESOLVING = "resolving"
|
||||
DOWNLOADING = "downloading"
|
||||
COMPLETE = "complete"
|
||||
AVAILABLE = "available"
|
||||
ERROR = "error"
|
||||
DONE = "done"
|
||||
CANCELLED = "cancelled"
|
||||
|
||||
|
||||
class SearchMode(str, Enum):
|
||||
DIRECT = "direct"
|
||||
UNIVERSAL = "universal"
|
||||
|
||||
|
||||
@dataclass
|
||||
class QueueItem:
|
||||
"""Queue item with priority and metadata."""
|
||||
book_id: str
|
||||
priority: int
|
||||
added_time: float
|
||||
|
||||
def __lt__(self, other):
|
||||
"""Compare items for priority queue (lower priority number = higher precedence)."""
|
||||
if self.priority != other.priority:
|
||||
return self.priority < other.priority
|
||||
return self.added_time < other.added_time
|
||||
|
||||
|
||||
@dataclass
|
||||
class DownloadTask:
|
||||
task_id: str # Unique ID (e.g., AA MD5 hash, Prowlarr GUID)
|
||||
source: str # Handler name ("direct_download", "prowlarr")
|
||||
title: str # Display title for queue sidebar
|
||||
|
||||
# Display info for queue sidebar
|
||||
author: Optional[str] = None
|
||||
year: Optional[str] = None
|
||||
format: Optional[str] = None
|
||||
size: Optional[str] = None
|
||||
preview: Optional[str] = None
|
||||
content_type: Optional[str] = None # "book (fiction)", "audiobook", "magazine", etc.
|
||||
|
||||
# Series info (for library naming templates)
|
||||
series_name: Optional[str] = None
|
||||
series_position: Optional[float] = None # Float for novellas (e.g., 1.5)
|
||||
subtitle: Optional[str] = None # Book subtitle for naming templates
|
||||
|
||||
# Hardlinking support
|
||||
original_download_path: Optional[str] = None # Path in download client (for hardlinking)
|
||||
|
||||
# Search mode - determines post-download processing behavior
|
||||
# See SearchMode enum for behavioral differences
|
||||
search_mode: Optional[SearchMode] = None
|
||||
|
||||
# Runtime state
|
||||
priority: int = 0
|
||||
added_time: float = field(default_factory=time.time)
|
||||
progress: float = 0.0
|
||||
status: QueueStatus = QueueStatus.QUEUED
|
||||
status_message: Optional[str] = None
|
||||
download_path: Optional[str] = None
|
||||
|
||||
def __lt__(self, other):
|
||||
"""Compare tasks for priority queue (lower priority number = higher precedence)."""
|
||||
if self.priority != other.priority:
|
||||
return self.priority < other.priority
|
||||
return self.added_time < other.added_time
|
||||
|
||||
def get_filename(self) -> str:
|
||||
"""Build sanitized filename from task metadata."""
|
||||
if self.download_path:
|
||||
return Path(self.download_path).name
|
||||
return build_filename(self.title, self.author, self.year, self.format)
|
||||
|
||||
|
||||
@dataclass
|
||||
class BookInfo:
|
||||
"""Data class representing book information."""
|
||||
id: str
|
||||
title: str
|
||||
preview: Optional[str] = None
|
||||
author: Optional[str] = None
|
||||
publisher: Optional[str] = None
|
||||
year: Optional[str] = None
|
||||
language: Optional[str] = None
|
||||
content: Optional[str] = None
|
||||
format: Optional[str] = None
|
||||
size: Optional[str] = None
|
||||
info: Optional[Dict[str, List[str]]] = None
|
||||
description: Optional[str] = None
|
||||
download_urls: List[str] = field(default_factory=list)
|
||||
download_path: Optional[str] = None
|
||||
priority: int = 0
|
||||
progress: Optional[float] = None
|
||||
status_message: Optional[str] = None # Detailed status message for UI display
|
||||
added_time: Optional[float] = None # Timestamp when added to queue
|
||||
source: str = "direct_download" # Release source handler to use for downloads
|
||||
source_url: Optional[str] = None # Link to source page (e.g., Anna's Archive)
|
||||
|
||||
def get_filename(self, fallback_url: Optional[str] = None) -> str:
|
||||
"""Build sanitized filename: 'Author - Title (Year).format'
|
||||
|
||||
Resolves format from self.format, download_urls, or fallback_url.
|
||||
|
||||
Args:
|
||||
fallback_url: URL to extract format from if not already known
|
||||
|
||||
Returns:
|
||||
Sanitized filename safe for filesystem use
|
||||
"""
|
||||
# Resolve format if needed
|
||||
if not self.format:
|
||||
urls = [self.download_urls[0]] if self.download_urls else []
|
||||
if fallback_url:
|
||||
urls.append(fallback_url)
|
||||
for url in urls:
|
||||
ext = url.split(".")[-1].lower()
|
||||
if ext and len(ext) <= 5 and ext.isalnum():
|
||||
self.format = ext
|
||||
break
|
||||
|
||||
return build_filename(self.title, self.author, self.year, self.format)
|
||||
|
||||
|
||||
@dataclass
|
||||
class SearchFilters:
|
||||
"""Filters for book search queries."""
|
||||
isbn: Optional[List[str]] = None
|
||||
author: Optional[List[str]] = None
|
||||
title: Optional[List[str]] = None
|
||||
lang: Optional[List[str]] = None
|
||||
sort: Optional[str] = None
|
||||
content: Optional[List[str]] = None
|
||||
format: Optional[List[str]] = None
|
||||
@@ -0,0 +1,197 @@
|
||||
"""Template-based naming for library organization."""
|
||||
|
||||
import os
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import Dict, Optional, Union
|
||||
|
||||
from shelfmark.core.logger import setup_logger
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
|
||||
TOKEN_PATTERN = re.compile(
|
||||
r'\{([- ._/\[(]*)' # prefix: space, dash, dot, underscore, slash, brackets
|
||||
r'([A-Za-z]+)' # token name
|
||||
r'([- ._/\])]*)\}' # suffix: space, dash, dot, underscore, slash, brackets
|
||||
)
|
||||
|
||||
# Characters that are invalid in filenames on various filesystems
|
||||
INVALID_CHARS = re.compile(r'[\\:*?"<>|]')
|
||||
|
||||
|
||||
def _sanitize(name: str, max_length: int = 245) -> str:
|
||||
"""Sanitize a string for filesystem use."""
|
||||
if not name:
|
||||
return ""
|
||||
|
||||
sanitized = INVALID_CHARS.sub('_', name)
|
||||
sanitized = re.sub(r'^[\s.]+|[\s.]+$', '', sanitized) # Strip whitespace and dots
|
||||
sanitized = re.sub(r'_+', '_', sanitized) # Collapse underscores
|
||||
return sanitized[:max_length]
|
||||
|
||||
|
||||
def sanitize_filename(name: str, max_length: int = 245) -> str:
|
||||
"""Sanitize a string for use as a filename or path component."""
|
||||
return _sanitize(name, max_length)
|
||||
|
||||
|
||||
# Alias for backwards compatibility
|
||||
sanitize_path_component = sanitize_filename
|
||||
|
||||
|
||||
def format_series_position(position: Optional[Union[int, float]]) -> str:
|
||||
if position is None:
|
||||
return ""
|
||||
|
||||
# Display as integer if whole number
|
||||
if isinstance(position, float) and position.is_integer():
|
||||
return str(int(position))
|
||||
|
||||
return str(position)
|
||||
|
||||
|
||||
# Pads numbers to 9 digits for natural sorting (e.g., "Part 2" -> "Part 000000002")
|
||||
PAD_NUMBERS_PATTERN = re.compile(r'\d+')
|
||||
|
||||
|
||||
def natural_sort_key(path: Union[str, Path]) -> str:
|
||||
"""Generate a sort key with padded numbers for natural sorting."""
|
||||
filename = Path(path).name.lower()
|
||||
return PAD_NUMBERS_PATTERN.sub(lambda m: m.group().zfill(9), filename)
|
||||
|
||||
|
||||
def assign_part_numbers(
|
||||
files: list[Path],
|
||||
zero_pad_width: int = 2,
|
||||
) -> list[tuple[Path, str]]:
|
||||
"""Sort files naturally and assign sequential part numbers (1, 2, 3...)."""
|
||||
if not files:
|
||||
return []
|
||||
|
||||
sorted_files = sorted(files, key=natural_sort_key)
|
||||
return [
|
||||
(file_path, str(part_num).zfill(zero_pad_width))
|
||||
for part_num, file_path in enumerate(sorted_files, start=1)
|
||||
]
|
||||
|
||||
|
||||
def parse_naming_template(
|
||||
template: str,
|
||||
metadata: Dict[str, Optional[Union[str, int, float]]],
|
||||
) -> str:
|
||||
if not template:
|
||||
return ""
|
||||
|
||||
# Normalize metadata keys to lowercase for case-insensitive matching
|
||||
normalized = {k.lower(): v for k, v in metadata.items()}
|
||||
|
||||
def replace_token(match: re.Match) -> str:
|
||||
prefix = match.group(1)
|
||||
token_name = match.group(2).lower()
|
||||
suffix = match.group(3)
|
||||
|
||||
# Get the value for this token
|
||||
value = normalized.get(token_name)
|
||||
|
||||
# Special handling for series position
|
||||
if token_name == 'seriesposition':
|
||||
value = format_series_position(value)
|
||||
|
||||
# Convert to string
|
||||
if value is None:
|
||||
value = ""
|
||||
else:
|
||||
value = str(value).strip()
|
||||
|
||||
# If value is empty, return empty string (no prefix/suffix)
|
||||
if not value:
|
||||
return ""
|
||||
|
||||
# Sanitize the value
|
||||
value = sanitize_filename(value)
|
||||
|
||||
return f"{prefix}{value}{suffix}"
|
||||
|
||||
# Replace all tokens
|
||||
result = TOKEN_PATTERN.sub(replace_token, template)
|
||||
|
||||
# Clean up any double slashes that might result from empty tokens
|
||||
result = re.sub(r'/+', '/', result)
|
||||
|
||||
# Remove leading/trailing slashes
|
||||
result = result.strip('/')
|
||||
|
||||
# Clean up any orphaned separators (e.g., " - " at start/end, or " - - ")
|
||||
result = re.sub(r'^[\s\-_.]+', '', result)
|
||||
result = re.sub(r'[\s\-_.]+$', '', result)
|
||||
result = re.sub(r'(\s*-\s*){2,}', ' - ', result)
|
||||
|
||||
# Clean up empty parentheses/brackets
|
||||
result = re.sub(r'\(\s*\)', '', result)
|
||||
result = re.sub(r'\[\s*\]', '', result)
|
||||
|
||||
# Final trim of any trailing separators left after cleanup
|
||||
result = re.sub(r'[\s\-_.]+$', '', result)
|
||||
|
||||
return result
|
||||
|
||||
|
||||
def build_library_path(
|
||||
base_path: str,
|
||||
template: str,
|
||||
metadata: Dict[str, Optional[Union[str, int, float]]],
|
||||
extension: Optional[str] = None,
|
||||
) -> Path:
|
||||
relative = parse_naming_template(template, metadata)
|
||||
|
||||
if not relative:
|
||||
# Fallback to title if template produces empty result
|
||||
title = metadata.get('Title') or metadata.get('title') or 'Unknown'
|
||||
relative = sanitize_filename(str(title))
|
||||
|
||||
# Remove any path traversal attempts
|
||||
relative = relative.replace('..', '')
|
||||
|
||||
base = Path(base_path).resolve()
|
||||
full_path = (base / relative).resolve()
|
||||
|
||||
# Verify the path is within the base directory
|
||||
try:
|
||||
full_path.relative_to(base)
|
||||
except ValueError:
|
||||
raise ValueError(f"Path traversal detected: template would escape library directory")
|
||||
|
||||
if extension:
|
||||
ext = extension.lstrip('.')
|
||||
# Don't use with_suffix() - it replaces everything after the first dot
|
||||
# e.g., "2.5 - Title" would become "2.epub" instead of "2.5 - Title.epub"
|
||||
full_path = Path(f"{full_path}.{ext}")
|
||||
|
||||
return full_path
|
||||
|
||||
|
||||
def same_filesystem(path1: Union[str, Path], path2: Union[str, Path]) -> bool:
|
||||
"""Check if two paths are on the same filesystem."""
|
||||
path1 = Path(path1)
|
||||
path2 = Path(path2)
|
||||
|
||||
def get_device(p: Path) -> Optional[int]:
|
||||
try:
|
||||
while not p.exists():
|
||||
p = p.parent
|
||||
if p == p.parent:
|
||||
break
|
||||
return os.stat(p).st_dev
|
||||
except (OSError, PermissionError) as e:
|
||||
logger.debug(f"Cannot stat {p}: {e}")
|
||||
return None
|
||||
|
||||
dev1 = get_device(path1)
|
||||
dev2 = get_device(path2)
|
||||
|
||||
if dev1 is None or dev2 is None:
|
||||
logger.warning(f"Cannot determine filesystem for hardlink check, falling back to copy")
|
||||
return False
|
||||
|
||||
return dev1 == dev2
|
||||
@@ -0,0 +1,296 @@
|
||||
"""Thread-safe download queue manager with priority support and cancellation."""
|
||||
|
||||
import queue
|
||||
import time
|
||||
from datetime import datetime, timedelta
|
||||
from pathlib import Path
|
||||
from threading import Lock, Event
|
||||
from typing import Dict, List, Optional, Tuple, Any
|
||||
|
||||
from shelfmark.core.config import config as app_config
|
||||
from shelfmark.core.models import QueueStatus, QueueItem, DownloadTask
|
||||
|
||||
|
||||
class BookQueue:
|
||||
"""Thread-safe download queue manager with priority support and cancellation."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self._queue: queue.PriorityQueue[QueueItem] = queue.PriorityQueue()
|
||||
self._lock = Lock()
|
||||
self._status: dict[str, QueueStatus] = {}
|
||||
self._task_data: dict[str, DownloadTask] = {}
|
||||
self._status_timestamps: dict[str, datetime] = {} # Track when each status was last updated
|
||||
self._cancel_flags: dict[str, Event] = {} # Cancellation flags for active downloads
|
||||
self._active_downloads: dict[str, bool] = {} # Track currently downloading tasks
|
||||
|
||||
@property
|
||||
def _status_timeout(self) -> timedelta:
|
||||
"""Get status timeout from config (allows live updates)."""
|
||||
return timedelta(seconds=app_config.get("STATUS_TIMEOUT", 3600))
|
||||
|
||||
def add(self, task: DownloadTask) -> bool:
|
||||
"""Add a download task to the queue. Returns False if already exists."""
|
||||
with self._lock:
|
||||
task_id = task.task_id
|
||||
|
||||
# Don't add if already exists and not in error/done state
|
||||
if task_id in self._status and self._status[task_id] not in [QueueStatus.ERROR, QueueStatus.DONE, QueueStatus.CANCELLED]:
|
||||
return False
|
||||
|
||||
# Ensure added_time is set
|
||||
if task.added_time == 0:
|
||||
task.added_time = time.time()
|
||||
|
||||
queue_item = QueueItem(task_id, task.priority, task.added_time)
|
||||
self._queue.put(queue_item)
|
||||
self._task_data[task_id] = task
|
||||
self._update_status(task_id, QueueStatus.QUEUED)
|
||||
return True
|
||||
|
||||
def get_next(self) -> Optional[Tuple[str, Event]]:
|
||||
"""Get next task ID from queue with cancellation flag."""
|
||||
# Use iterative approach to avoid stack overflow if many items are cancelled
|
||||
while True:
|
||||
try:
|
||||
queue_item = self._queue.get_nowait()
|
||||
task_id = queue_item.book_id # QueueItem uses book_id as the ID field
|
||||
|
||||
with self._lock:
|
||||
# Check if task was cancelled while in queue
|
||||
if task_id in self._status and self._status[task_id] == QueueStatus.CANCELLED:
|
||||
continue # Skip cancelled items, try next
|
||||
|
||||
# Create cancellation flag for this download
|
||||
cancel_flag = Event()
|
||||
self._cancel_flags[task_id] = cancel_flag
|
||||
self._active_downloads[task_id] = True
|
||||
|
||||
return task_id, cancel_flag
|
||||
except queue.Empty:
|
||||
return None
|
||||
|
||||
def get_task(self, task_id: str) -> Optional[DownloadTask]:
|
||||
"""Get a task by its ID."""
|
||||
with self._lock:
|
||||
return self._task_data.get(task_id)
|
||||
|
||||
def _update_status(self, book_id: str, status: QueueStatus) -> None:
|
||||
"""Internal method to update status and timestamp."""
|
||||
self._status[book_id] = status
|
||||
self._status_timestamps[book_id] = datetime.now()
|
||||
|
||||
def update_status(self, book_id: str, status: QueueStatus) -> None:
|
||||
"""Update status of a book in the queue."""
|
||||
with self._lock:
|
||||
self._update_status(book_id, status)
|
||||
|
||||
# Clean up active download tracking when finished
|
||||
if status in [QueueStatus.COMPLETE, QueueStatus.AVAILABLE, QueueStatus.ERROR, QueueStatus.DONE, QueueStatus.CANCELLED]:
|
||||
self._active_downloads.pop(book_id, None)
|
||||
self._cancel_flags.pop(book_id, None)
|
||||
|
||||
def update_download_path(self, task_id: str, download_path: str) -> None:
|
||||
"""Update the download path of a task in the queue."""
|
||||
with self._lock:
|
||||
if task_id in self._task_data:
|
||||
self._task_data[task_id].download_path = download_path
|
||||
|
||||
def update_progress(self, task_id: str, progress: float) -> None:
|
||||
"""Update download progress for a task."""
|
||||
with self._lock:
|
||||
if task_id in self._task_data:
|
||||
self._task_data[task_id].progress = progress
|
||||
|
||||
def update_status_message(self, task_id: str, message: str) -> None:
|
||||
"""Update detailed status message for a task."""
|
||||
with self._lock:
|
||||
if task_id in self._task_data:
|
||||
self._task_data[task_id].status_message = message
|
||||
|
||||
def get_status(self) -> Dict[QueueStatus, Dict[str, DownloadTask]]:
|
||||
"""Get current queue status grouped by status."""
|
||||
self.refresh()
|
||||
with self._lock:
|
||||
result: Dict[QueueStatus, Dict[str, DownloadTask]] = {status: {} for status in QueueStatus}
|
||||
for task_id, status in self._status.items():
|
||||
if task_id in self._task_data:
|
||||
result[status][task_id] = self._task_data[task_id]
|
||||
return result
|
||||
|
||||
def get_queue_order(self) -> List[Dict[str, Any]]:
|
||||
"""Get current queue order for display."""
|
||||
with self._lock:
|
||||
queue_items = []
|
||||
|
||||
# Get items from priority queue without removing them
|
||||
temp_items = []
|
||||
while not self._queue.empty():
|
||||
try:
|
||||
item = self._queue.get_nowait()
|
||||
temp_items.append(item)
|
||||
task_id = item.book_id # QueueItem uses book_id as the ID field
|
||||
if task_id in self._task_data:
|
||||
task = self._task_data[task_id]
|
||||
queue_items.append({
|
||||
'id': task_id,
|
||||
'title': task.title,
|
||||
'author': task.author,
|
||||
'priority': item.priority,
|
||||
'added_time': item.added_time,
|
||||
'status': self._status.get(task_id, QueueStatus.QUEUED)
|
||||
})
|
||||
except queue.Empty:
|
||||
break
|
||||
|
||||
# Put items back in queue
|
||||
for item in temp_items:
|
||||
self._queue.put(item)
|
||||
|
||||
return sorted(queue_items, key=lambda x: (x['priority'], x['added_time']))
|
||||
|
||||
def cancel_download(self, task_id: str) -> bool:
|
||||
"""Cancel a download or clear a completed/errored item."""
|
||||
with self._lock:
|
||||
current_status = self._status.get(task_id)
|
||||
|
||||
# Allow cancellation during any active state
|
||||
if current_status in [QueueStatus.RESOLVING, QueueStatus.DOWNLOADING]:
|
||||
# Signal active download to stop
|
||||
if task_id in self._cancel_flags:
|
||||
self._cancel_flags[task_id].set()
|
||||
self._update_status(task_id, QueueStatus.CANCELLED)
|
||||
return True
|
||||
elif current_status == QueueStatus.QUEUED:
|
||||
# Remove from queue and mark as cancelled
|
||||
self._update_status(task_id, QueueStatus.CANCELLED)
|
||||
return True
|
||||
elif current_status in [QueueStatus.COMPLETE, QueueStatus.DONE, QueueStatus.AVAILABLE, QueueStatus.ERROR, QueueStatus.CANCELLED]:
|
||||
# Clear completed/errored/cancelled items from tracking
|
||||
self._status.pop(task_id, None)
|
||||
self._status_timestamps.pop(task_id, None)
|
||||
self._task_data.pop(task_id, None)
|
||||
self._cancel_flags.pop(task_id, None)
|
||||
self._active_downloads.pop(task_id, None)
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
def set_priority(self, task_id: str, new_priority: int) -> bool:
|
||||
"""Change the priority of a queued task (lower = higher priority)."""
|
||||
with self._lock:
|
||||
if task_id not in self._status or self._status[task_id] != QueueStatus.QUEUED:
|
||||
return False
|
||||
|
||||
# Remove task from queue and re-add with new priority
|
||||
temp_items = []
|
||||
found = False
|
||||
|
||||
while not self._queue.empty():
|
||||
try:
|
||||
item = self._queue.get_nowait()
|
||||
if item.book_id == task_id: # QueueItem uses book_id as the ID field
|
||||
# Create new item with updated priority
|
||||
new_item = QueueItem(task_id, new_priority, item.added_time)
|
||||
temp_items.append(new_item)
|
||||
found = True
|
||||
# Update task data priority
|
||||
if task_id in self._task_data:
|
||||
self._task_data[task_id].priority = new_priority
|
||||
else:
|
||||
temp_items.append(item)
|
||||
except queue.Empty:
|
||||
break
|
||||
|
||||
# Put all items back
|
||||
for item in temp_items:
|
||||
self._queue.put(item)
|
||||
|
||||
return found
|
||||
|
||||
def reorder_queue(self, task_priorities: Dict[str, int]) -> bool:
|
||||
"""Bulk reorder queue by mapping task_id to new priority."""
|
||||
with self._lock:
|
||||
# Extract all items from queue
|
||||
all_items = []
|
||||
while not self._queue.empty():
|
||||
try:
|
||||
item = self._queue.get_nowait()
|
||||
task_id = item.book_id # QueueItem uses book_id as the ID field
|
||||
# Update priority if specified
|
||||
if task_id in task_priorities:
|
||||
new_priority = task_priorities[task_id]
|
||||
item = QueueItem(task_id, new_priority, item.added_time)
|
||||
# Update task data priority
|
||||
if task_id in self._task_data:
|
||||
self._task_data[task_id].priority = new_priority
|
||||
all_items.append(item)
|
||||
except queue.Empty:
|
||||
break
|
||||
|
||||
# Put all items back with updated priorities
|
||||
for item in all_items:
|
||||
self._queue.put(item)
|
||||
|
||||
return True
|
||||
|
||||
def get_active_downloads(self) -> List[str]:
|
||||
"""Get list of currently active download task IDs."""
|
||||
with self._lock:
|
||||
return list(self._active_downloads.keys())
|
||||
|
||||
def has_pending_work(self) -> bool:
|
||||
"""Check if there are any active downloads or queued items."""
|
||||
with self._lock:
|
||||
if self._active_downloads:
|
||||
return True
|
||||
return any(status == QueueStatus.QUEUED for status in self._status.values())
|
||||
|
||||
def clear_completed(self) -> int:
|
||||
"""Remove all completed, errored, or cancelled tasks from tracking."""
|
||||
terminal_statuses = {QueueStatus.COMPLETE, QueueStatus.DONE, QueueStatus.AVAILABLE, QueueStatus.ERROR, QueueStatus.CANCELLED}
|
||||
with self._lock:
|
||||
to_remove = [task_id for task_id, status in self._status.items() if status in terminal_statuses]
|
||||
|
||||
for task_id in to_remove:
|
||||
self._status.pop(task_id, None)
|
||||
self._status_timestamps.pop(task_id, None)
|
||||
self._task_data.pop(task_id, None)
|
||||
self._cancel_flags.pop(task_id, None)
|
||||
self._active_downloads.pop(task_id, None)
|
||||
|
||||
return len(to_remove)
|
||||
|
||||
def refresh(self) -> None:
|
||||
"""Remove any tasks that are done downloading or have stale status."""
|
||||
terminal_statuses = {QueueStatus.COMPLETE, QueueStatus.DONE, QueueStatus.ERROR, QueueStatus.AVAILABLE, QueueStatus.CANCELLED}
|
||||
with self._lock:
|
||||
current_time = datetime.now()
|
||||
to_remove = []
|
||||
|
||||
for task_id, status in self._status.items():
|
||||
task = self._task_data.get(task_id)
|
||||
if not task:
|
||||
continue
|
||||
|
||||
# Clear stale download paths
|
||||
if task.download_path and not Path(task.download_path).exists():
|
||||
task.download_path = None
|
||||
|
||||
# Mark available downloads as done if file is gone
|
||||
if status == QueueStatus.AVAILABLE and not task.download_path:
|
||||
self._update_status(task_id, QueueStatus.DONE)
|
||||
|
||||
# Check for stale status entries
|
||||
last_update = self._status_timestamps.get(task_id)
|
||||
if last_update and (current_time - last_update) > self._status_timeout:
|
||||
if status in terminal_statuses:
|
||||
to_remove.append(task_id)
|
||||
|
||||
# Remove stale entries
|
||||
for task_id in to_remove:
|
||||
self._status.pop(task_id, None)
|
||||
self._status_timestamps.pop(task_id, None)
|
||||
self._task_data.pop(task_id, None)
|
||||
|
||||
# Global instance of BookQueue
|
||||
book_queue = BookQueue()
|
||||
@@ -0,0 +1,767 @@
|
||||
"""Plugin settings registry with config file persistence."""
|
||||
|
||||
import json
|
||||
import os
|
||||
from dataclasses import dataclass, field, asdict
|
||||
from pathlib import Path
|
||||
from typing import Any, Callable, Dict, List, Optional, Type, Union
|
||||
from threading import Lock
|
||||
|
||||
from shelfmark.core.logger import setup_logger
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class FieldBase:
|
||||
"""Base class for all settings fields."""
|
||||
key: str # Environment variable / config key
|
||||
label: str # Display label in UI
|
||||
description: str = "" # Help text
|
||||
default: Any = None # Default value if not set
|
||||
required: bool = False # Whether field must have a value
|
||||
env_var: Optional[str] = None # Override env var name (defaults to key)
|
||||
env_supported: bool = True # Whether this setting can be set via ENV var (False = UI-only)
|
||||
disabled: bool = False # Whether field is disabled/greyed out
|
||||
disabled_reason: str = "" # Explanation shown when disabled
|
||||
show_when: Optional[Dict[str, Any]] = None # Conditional visibility: {"field": "key", "value": "expected"} or {"field": "key", "notEmpty": True}
|
||||
disabled_when: Optional[Dict[str, Any]] = None # Conditional disable: {"field": "key", "value": "expected", "reason": "..."}
|
||||
requires_restart: bool = False # Whether changing this setting requires a container restart
|
||||
universal_only: bool = False # Only show in Universal search mode (hide in Direct mode)
|
||||
|
||||
def get_env_var_name(self) -> str:
|
||||
"""Get the environment variable name for this field."""
|
||||
return self.env_var or self.key
|
||||
|
||||
def get_field_type(self) -> str:
|
||||
"""Get the field type name for serialization."""
|
||||
return self.__class__.__name__
|
||||
|
||||
|
||||
@dataclass
|
||||
class TextField(FieldBase):
|
||||
"""Single-line text input."""
|
||||
placeholder: str = ""
|
||||
max_length: Optional[int] = None
|
||||
|
||||
|
||||
@dataclass
|
||||
class PasswordField(FieldBase):
|
||||
"""Password input (masked in UI, not returned in API responses)."""
|
||||
placeholder: str = ""
|
||||
|
||||
|
||||
@dataclass
|
||||
class NumberField(FieldBase):
|
||||
"""Numeric input."""
|
||||
min_value: Optional[float] = None
|
||||
max_value: Optional[float] = None
|
||||
step: float = 1
|
||||
default: float = 0
|
||||
|
||||
|
||||
@dataclass
|
||||
class CheckboxField(FieldBase):
|
||||
"""Boolean checkbox."""
|
||||
default: bool = False
|
||||
|
||||
|
||||
@dataclass
|
||||
class SelectField(FieldBase):
|
||||
"""Single-choice dropdown."""
|
||||
# Options can be a list or a callable that returns a list (for lazy evaluation)
|
||||
options: Any = field(default_factory=list) # [{value: "", label: ""}] or callable
|
||||
|
||||
|
||||
@dataclass
|
||||
class MultiSelectField(FieldBase):
|
||||
"""Multiple-choice selection."""
|
||||
# Options can be a list or a callable that returns a list (for lazy evaluation)
|
||||
options: Any = field(default_factory=list) # [{value: "", label: ""}] or callable
|
||||
default: List[str] = field(default_factory=list)
|
||||
variant: str = "pills" # "pills" (default) or "dropdown" for checkbox dropdown style
|
||||
|
||||
|
||||
@dataclass
|
||||
class OrderableListField(FieldBase):
|
||||
# Options can be a list or a callable that returns a list (for lazy evaluation)
|
||||
# Each option: {id, label, description?, disabledReason?, isLocked?, section?, isPinned?}
|
||||
# - isLocked: toggle is disabled (can't enable/disable)
|
||||
# - isPinned: can't be reordered (but toggle may still work if not also isLocked)
|
||||
options: Any = field(default_factory=list)
|
||||
# Default value: [{id, enabled}, ...] in priority order
|
||||
default: List[Dict[str, Any]] = field(default_factory=list)
|
||||
|
||||
|
||||
@dataclass
|
||||
class ActionButton:
|
||||
key: str # Action identifier
|
||||
label: str # Button text
|
||||
description: str = "" # Help text
|
||||
style: str = "default" # "default", "primary", "danger"
|
||||
callback: Optional[Callable[[], Dict[str, Any]]] = None # Returns {"success": bool, "message": str}
|
||||
disabled: bool = False # Whether button is disabled/greyed out
|
||||
disabled_reason: str = "" # Explanation shown when disabled
|
||||
show_when: Optional[Dict[str, Any]] = None # Conditional visibility: {"field": "key", "value": "expected"} or {"field": "key", "notEmpty": True}
|
||||
disabled_when: Optional[Dict[str, Any]] = None # Conditional disable: {"field": "key", "value": "expected", "reason": "..."}
|
||||
|
||||
def get_field_type(self) -> str:
|
||||
return "ActionButton"
|
||||
|
||||
|
||||
@dataclass
|
||||
class HeadingField:
|
||||
"""
|
||||
Display-only heading with title and description.
|
||||
|
||||
Used to add section titles and descriptive text to settings pages.
|
||||
Not an input field - purely for display.
|
||||
"""
|
||||
key: str # Unique identifier
|
||||
title: str # Heading title
|
||||
description: str = "" # Description text (supports markdown-style links)
|
||||
link_url: str = "" # Optional URL for a link
|
||||
link_text: str = "" # Text for the link (defaults to URL if not provided)
|
||||
show_when: Optional[Dict[str, Any]] = None # Conditional visibility: {"field": "key", "value": "expected"} or {"field": "key", "notEmpty": True}
|
||||
universal_only: bool = False # Only show in Universal search mode (hide in Direct mode)
|
||||
|
||||
def get_field_type(self) -> str:
|
||||
return "HeadingField"
|
||||
|
||||
|
||||
# Type alias for all field types
|
||||
SettingsField = Union[TextField, PasswordField, NumberField, CheckboxField, SelectField, MultiSelectField, OrderableListField, ActionButton, HeadingField]
|
||||
|
||||
|
||||
@dataclass
|
||||
class SettingsTab:
|
||||
"""A tab/section in the settings UI."""
|
||||
name: str # Internal name (used in URLs)
|
||||
display_name: str # Display name in UI
|
||||
fields: List[SettingsField] = field(default_factory=list)
|
||||
icon: Optional[str] = None # Icon name for UI
|
||||
order: int = 100 # Sort order (lower = earlier)
|
||||
group: Optional[str] = None # Group name this tab belongs to
|
||||
|
||||
|
||||
@dataclass
|
||||
class SettingsGroup:
|
||||
"""A collapsible group of settings tabs in the UI."""
|
||||
name: str # Internal name
|
||||
display_name: str # Display name in UI
|
||||
icon: Optional[str] = None # Icon name for UI
|
||||
order: int = 100 # Sort order (lower = earlier)
|
||||
|
||||
|
||||
_SETTINGS_REGISTRY: Dict[str, SettingsTab] = {}
|
||||
_GROUPS_REGISTRY: Dict[str, SettingsGroup] = {}
|
||||
_ON_SAVE_HANDLERS: Dict[str, Callable[[Dict[str, Any]], Dict[str, Any]]] = {}
|
||||
_REGISTRY_LOCK = Lock()
|
||||
|
||||
|
||||
def register_group(
|
||||
name: str,
|
||||
display_name: str,
|
||||
icon: Optional[str] = None,
|
||||
order: int = 100
|
||||
) -> None:
|
||||
with _REGISTRY_LOCK:
|
||||
group = SettingsGroup(
|
||||
name=name,
|
||||
display_name=display_name,
|
||||
icon=icon,
|
||||
order=order,
|
||||
)
|
||||
_GROUPS_REGISTRY[name] = group
|
||||
logger.debug(f"Registered settings group: {name}")
|
||||
|
||||
|
||||
def register_settings(
|
||||
name: str,
|
||||
display_name: str,
|
||||
icon: Optional[str] = None,
|
||||
order: int = 100,
|
||||
group: Optional[str] = None
|
||||
):
|
||||
def decorator(func: Callable[[], List[SettingsField]]):
|
||||
with _REGISTRY_LOCK:
|
||||
fields = func()
|
||||
tab = SettingsTab(
|
||||
name=name,
|
||||
display_name=display_name,
|
||||
fields=fields,
|
||||
icon=icon,
|
||||
order=order,
|
||||
group=group,
|
||||
)
|
||||
_SETTINGS_REGISTRY[name] = tab
|
||||
logger.debug(f"Registered settings tab: {name} ({len(fields)} fields)" +
|
||||
(f" in group {group}" if group else ""))
|
||||
return func
|
||||
return decorator
|
||||
|
||||
|
||||
def register_on_save(
|
||||
tab_name: str,
|
||||
handler: Callable[[Dict[str, Any]], Dict[str, Any]]
|
||||
) -> None:
|
||||
with _REGISTRY_LOCK:
|
||||
_ON_SAVE_HANDLERS[tab_name] = handler
|
||||
logger.debug(f"Registered on_save handler for tab: {tab_name}")
|
||||
|
||||
|
||||
def get_on_save_handler(tab_name: str) -> Optional[Callable[[Dict[str, Any]], Dict[str, Any]]]:
|
||||
"""Get the on_save handler for a settings tab, if any."""
|
||||
return _ON_SAVE_HANDLERS.get(tab_name)
|
||||
|
||||
|
||||
def get_settings_tab(name: str) -> Optional[SettingsTab]:
|
||||
"""Get a specific settings tab by name."""
|
||||
return _SETTINGS_REGISTRY.get(name)
|
||||
|
||||
|
||||
def get_all_settings_tabs() -> List[SettingsTab]:
|
||||
"""Get all registered settings tabs, sorted by order."""
|
||||
return sorted(_SETTINGS_REGISTRY.values(), key=lambda t: (t.order, t.name))
|
||||
|
||||
|
||||
def list_registered_settings() -> List[str]:
|
||||
"""List all registered settings tab names."""
|
||||
return list(_SETTINGS_REGISTRY.keys())
|
||||
|
||||
|
||||
def _get_config_dir() -> Path:
|
||||
"""Get the config directory path."""
|
||||
from shelfmark.config.env import CONFIG_DIR
|
||||
return Path(CONFIG_DIR)
|
||||
|
||||
|
||||
def _get_config_file_path(tab_name: str) -> Path:
|
||||
"""Get the config file path for a settings tab."""
|
||||
config_dir = _get_config_dir()
|
||||
# Core settings tabs share the main settings.json file
|
||||
if tab_name in ("general", "search_mode"):
|
||||
return config_dir / "settings.json"
|
||||
return config_dir / "plugins" / f"{tab_name}.json"
|
||||
|
||||
|
||||
def _ensure_config_dir(tab_name: str) -> None:
|
||||
"""Ensure the config directory exists."""
|
||||
config_path = _get_config_file_path(tab_name)
|
||||
config_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
|
||||
def load_config_file(tab_name: str) -> Dict[str, Any]:
|
||||
config_path = _get_config_file_path(tab_name)
|
||||
|
||||
if not config_path.exists():
|
||||
return {}
|
||||
|
||||
try:
|
||||
with open(config_path, 'r') as f:
|
||||
return json.load(f)
|
||||
except json.JSONDecodeError as e:
|
||||
logger.error(f"Invalid JSON in config file {config_path}: {e}")
|
||||
return {}
|
||||
|
||||
|
||||
def save_config_file(tab_name: str, values: Dict[str, Any]) -> bool:
|
||||
try:
|
||||
_ensure_config_dir(tab_name)
|
||||
config_path = _get_config_file_path(tab_name)
|
||||
|
||||
# Load existing config and merge
|
||||
existing = load_config_file(tab_name)
|
||||
existing.update(values)
|
||||
|
||||
with open(config_path, 'w') as f:
|
||||
json.dump(existing, f, indent=2)
|
||||
|
||||
logger.info(f"Saved settings to {config_path}")
|
||||
return True
|
||||
except Exception as e:
|
||||
logger.error(f"Error saving config file for {tab_name}: {e}")
|
||||
return False
|
||||
|
||||
|
||||
def sync_env_to_config() -> None:
|
||||
for tab in get_all_settings_tabs():
|
||||
values_to_sync = {}
|
||||
|
||||
for field in tab.fields:
|
||||
# Skip non-value fields
|
||||
if isinstance(field, (ActionButton, HeadingField)):
|
||||
continue
|
||||
|
||||
# Skip fields that don't support ENV vars
|
||||
if not getattr(field, 'env_supported', True):
|
||||
continue
|
||||
|
||||
# Check if ENV var is set
|
||||
env_var_name = field.get_env_var_name()
|
||||
env_value = os.environ.get(env_var_name)
|
||||
|
||||
if env_value is not None:
|
||||
# Parse the ENV value to the appropriate type
|
||||
parsed_value = _parse_env_value(env_value, field)
|
||||
values_to_sync[field.key] = parsed_value
|
||||
|
||||
# Save synced values to config file (merge with existing)
|
||||
if values_to_sync:
|
||||
save_config_file(tab.name, values_to_sync)
|
||||
logger.debug(f"Synced {len(values_to_sync)} ENV values to {tab.name} config: {list(values_to_sync.keys())}")
|
||||
|
||||
migrate_legacy_settings()
|
||||
|
||||
|
||||
def migrate_legacy_settings() -> None:
|
||||
"""Migrate legacy settings to new unified file destination format.
|
||||
|
||||
Maps old settings to new:
|
||||
- PROCESSING_MODE + USE_BOOK_TITLE -> FILE_ORGANIZATION
|
||||
- INGEST_DIR / LIBRARY_PATH -> DESTINATION
|
||||
- LIBRARY_TEMPLATE -> TEMPLATE
|
||||
- USE_CONTENT_TYPE_DIRECTORIES -> AA_CONTENT_TYPE_ROUTING
|
||||
- INGEST_DIR_* -> AA_CONTENT_TYPE_DIR_*
|
||||
- TORRENT_HARDLINK -> HARDLINK_TORRENTS / HARDLINK_TORRENTS_AUDIOBOOK
|
||||
"""
|
||||
# Load existing downloads config
|
||||
downloads_config = load_config_file("downloads")
|
||||
source_config = load_config_file("download_sources")
|
||||
|
||||
# Skip migration if already using new settings
|
||||
if "FILE_ORGANIZATION" in downloads_config or "DESTINATION" in downloads_config:
|
||||
return
|
||||
|
||||
migrated_downloads = {}
|
||||
migrated_sources = {}
|
||||
|
||||
# === BOOKS MIGRATION ===
|
||||
old_mode = downloads_config.get("PROCESSING_MODE", "ingest")
|
||||
old_ingest_dir = downloads_config.get("INGEST_DIR", "/cwa-book-ingest")
|
||||
old_library_path = downloads_config.get("LIBRARY_PATH", "")
|
||||
old_use_book_title = downloads_config.get("USE_BOOK_TITLE", True)
|
||||
old_library_template = downloads_config.get("LIBRARY_TEMPLATE", "{Author}/{Title}")
|
||||
|
||||
# Map PROCESSING_MODE + USE_BOOK_TITLE -> FILE_ORGANIZATION
|
||||
if old_mode == "library":
|
||||
migrated_downloads["FILE_ORGANIZATION"] = "organize"
|
||||
migrated_downloads["DESTINATION"] = old_library_path or "/books"
|
||||
migrated_downloads["TEMPLATE"] = old_library_template
|
||||
else:
|
||||
if old_use_book_title:
|
||||
migrated_downloads["FILE_ORGANIZATION"] = "rename"
|
||||
migrated_downloads["TEMPLATE"] = "{Author} - {Title} ({Year})"
|
||||
else:
|
||||
migrated_downloads["FILE_ORGANIZATION"] = "none"
|
||||
migrated_downloads["DESTINATION"] = old_ingest_dir
|
||||
|
||||
# === AUDIOBOOKS MIGRATION ===
|
||||
old_mode_ab = downloads_config.get("PROCESSING_MODE_AUDIOBOOK", "ingest")
|
||||
old_ingest_dir_ab = downloads_config.get("INGEST_DIR_AUDIOBOOK", "")
|
||||
old_library_path_ab = downloads_config.get("LIBRARY_PATH_AUDIOBOOK", "")
|
||||
old_library_template_ab = downloads_config.get("LIBRARY_TEMPLATE_AUDIOBOOK", "{Author}/{Title}")
|
||||
|
||||
if old_mode_ab == "library":
|
||||
migrated_downloads["FILE_ORGANIZATION_AUDIOBOOK"] = "organize"
|
||||
migrated_downloads["DESTINATION_AUDIOBOOK"] = old_library_path_ab or ""
|
||||
migrated_downloads["TEMPLATE_AUDIOBOOK"] = old_library_template_ab
|
||||
else:
|
||||
migrated_downloads["FILE_ORGANIZATION_AUDIOBOOK"] = "rename"
|
||||
migrated_downloads["TEMPLATE_AUDIOBOOK"] = "{Author} - {Title}"
|
||||
if old_ingest_dir_ab:
|
||||
migrated_downloads["DESTINATION_AUDIOBOOK"] = old_ingest_dir_ab
|
||||
|
||||
# === HARDLINK MIGRATION ===
|
||||
old_torrent_hardlink = downloads_config.get("TORRENT_HARDLINK")
|
||||
if old_torrent_hardlink is not None:
|
||||
# Books default to False (ingest folder use case)
|
||||
# Audiobooks default to True (library folder use case)
|
||||
# But if explicitly set, apply to both
|
||||
migrated_downloads["HARDLINK_TORRENTS"] = old_torrent_hardlink
|
||||
migrated_downloads["HARDLINK_TORRENTS_AUDIOBOOK"] = old_torrent_hardlink
|
||||
|
||||
# === CONTENT-TYPE ROUTING MIGRATION ===
|
||||
old_use_content_type = downloads_config.get("USE_CONTENT_TYPE_DIRECTORIES", False)
|
||||
if old_use_content_type:
|
||||
migrated_sources["AA_CONTENT_TYPE_ROUTING"] = True
|
||||
|
||||
# Map old keys to new keys
|
||||
content_type_mapping = {
|
||||
"INGEST_DIR_BOOK_FICTION": "AA_CONTENT_TYPE_DIR_FICTION",
|
||||
"INGEST_DIR_BOOK_NON_FICTION": "AA_CONTENT_TYPE_DIR_NON_FICTION",
|
||||
"INGEST_DIR_BOOK_UNKNOWN": "AA_CONTENT_TYPE_DIR_UNKNOWN",
|
||||
"INGEST_DIR_MAGAZINE": "AA_CONTENT_TYPE_DIR_MAGAZINE",
|
||||
"INGEST_DIR_COMIC_BOOK": "AA_CONTENT_TYPE_DIR_COMIC",
|
||||
"INGEST_DIR_STANDARDS_DOCUMENT": "AA_CONTENT_TYPE_DIR_STANDARDS",
|
||||
"INGEST_DIR_MUSICAL_SCORE": "AA_CONTENT_TYPE_DIR_MUSICAL_SCORE",
|
||||
"INGEST_DIR_OTHER": "AA_CONTENT_TYPE_DIR_OTHER",
|
||||
}
|
||||
|
||||
for old_key, new_key in content_type_mapping.items():
|
||||
old_value = downloads_config.get(old_key, "")
|
||||
if old_value:
|
||||
migrated_sources[new_key] = old_value
|
||||
|
||||
# Save migrated settings
|
||||
if migrated_downloads:
|
||||
save_config_file("downloads", migrated_downloads)
|
||||
logger.info(f"Migrated download settings: {list(migrated_downloads.keys())}")
|
||||
|
||||
if migrated_sources:
|
||||
save_config_file("download_sources", migrated_sources)
|
||||
logger.info(f"Migrated content-type routing settings: {list(migrated_sources.keys())}")
|
||||
|
||||
|
||||
def get_setting_value(field: SettingsField, tab_name: str) -> Any:
|
||||
if isinstance(field, (ActionButton, HeadingField)):
|
||||
return None # Actions and headings don't have values
|
||||
|
||||
# 1. Check environment variable (if supported for this field)
|
||||
if field.env_supported:
|
||||
env_var_name = field.get_env_var_name()
|
||||
env_value = os.environ.get(env_var_name)
|
||||
if env_value is not None:
|
||||
return _parse_env_value(env_value, field)
|
||||
|
||||
# 2. Check config file
|
||||
config = load_config_file(tab_name)
|
||||
if field.key in config:
|
||||
return config[field.key]
|
||||
|
||||
# 3. Return default
|
||||
return field.default
|
||||
|
||||
|
||||
def _parse_env_value(value: str, field: SettingsField) -> Any:
|
||||
"""Parse an environment variable value to the appropriate type."""
|
||||
if isinstance(field, CheckboxField):
|
||||
return value.lower() in ('true', '1', 'yes', 'on')
|
||||
elif isinstance(field, NumberField):
|
||||
try:
|
||||
if '.' in value:
|
||||
return float(value)
|
||||
return int(value)
|
||||
except ValueError:
|
||||
return field.default
|
||||
elif isinstance(field, MultiSelectField):
|
||||
return [v.strip() for v in value.split(',') if v.strip()]
|
||||
elif isinstance(field, OrderableListField):
|
||||
# Parse JSON array: [{"id": "...", "enabled": true}, ...]
|
||||
try:
|
||||
return json.loads(value)
|
||||
except json.JSONDecodeError:
|
||||
logger.warning(f"Invalid JSON for {field.key}, using default")
|
||||
return field.default
|
||||
else:
|
||||
return value
|
||||
|
||||
|
||||
def is_value_from_env(field: SettingsField) -> bool:
|
||||
"""Check if a field's value comes from an environment variable."""
|
||||
if isinstance(field, (ActionButton, HeadingField)):
|
||||
return False
|
||||
# UI-only settings never come from ENV (env_supported=False)
|
||||
if not getattr(field, 'env_supported', True):
|
||||
return False
|
||||
return field.get_env_var_name() in os.environ
|
||||
|
||||
|
||||
def serialize_field(field: SettingsField, tab_name: str, include_value: bool = True) -> Dict[str, Any]:
|
||||
"""
|
||||
Serialize a field for API response.
|
||||
|
||||
Args:
|
||||
field: The settings field.
|
||||
tab_name: The settings tab name.
|
||||
include_value: Whether to include the current value.
|
||||
|
||||
Returns:
|
||||
Dict representation of the field.
|
||||
"""
|
||||
# HeadingField has a different structure - handle separately
|
||||
if isinstance(field, HeadingField):
|
||||
result = {
|
||||
"key": field.key,
|
||||
"type": field.get_field_type(),
|
||||
"title": field.title,
|
||||
"description": field.description,
|
||||
}
|
||||
if field.link_url:
|
||||
result["linkUrl"] = field.link_url
|
||||
result["linkText"] = field.link_text or field.link_url
|
||||
if field.show_when:
|
||||
result["showWhen"] = field.show_when
|
||||
if field.universal_only:
|
||||
result["universalOnly"] = True
|
||||
return result
|
||||
|
||||
result = {
|
||||
"key": field.key,
|
||||
"label": field.label,
|
||||
"type": field.get_field_type(),
|
||||
"description": getattr(field, 'description', ''),
|
||||
"required": getattr(field, 'required', False),
|
||||
"disabled": getattr(field, 'disabled', False),
|
||||
"disabledReason": getattr(field, 'disabled_reason', ''),
|
||||
"requiresRestart": getattr(field, 'requires_restart', False),
|
||||
}
|
||||
|
||||
# Add optional properties if set
|
||||
if getattr(field, 'show_when', None):
|
||||
result["showWhen"] = field.show_when
|
||||
if getattr(field, 'disabled_when', None):
|
||||
result["disabledWhen"] = field.disabled_when
|
||||
if getattr(field, 'universal_only', False):
|
||||
result["universalOnly"] = True
|
||||
|
||||
# Add type-specific properties
|
||||
if isinstance(field, TextField):
|
||||
result["placeholder"] = field.placeholder
|
||||
if field.max_length:
|
||||
result["maxLength"] = field.max_length
|
||||
elif isinstance(field, PasswordField):
|
||||
result["placeholder"] = field.placeholder
|
||||
elif isinstance(field, NumberField):
|
||||
result["min"] = field.min_value
|
||||
result["max"] = field.max_value
|
||||
result["step"] = field.step
|
||||
elif isinstance(field, SelectField):
|
||||
# Support callable options for lazy evaluation (avoids circular imports)
|
||||
options = field.options() if callable(field.options) else field.options
|
||||
result["options"] = options
|
||||
if field.default is not None:
|
||||
result["default"] = field.default
|
||||
elif isinstance(field, MultiSelectField):
|
||||
# Support callable options for lazy evaluation (avoids circular imports)
|
||||
options = field.options() if callable(field.options) else field.options
|
||||
result["options"] = options
|
||||
result["variant"] = field.variant
|
||||
elif isinstance(field, OrderableListField):
|
||||
# Support callable options for lazy evaluation (avoids circular imports)
|
||||
options = field.options() if callable(field.options) else field.options
|
||||
result["options"] = options
|
||||
elif isinstance(field, ActionButton):
|
||||
result["style"] = field.style
|
||||
result["description"] = field.description
|
||||
|
||||
if include_value and not isinstance(field, (ActionButton, HeadingField)):
|
||||
value = get_setting_value(field, tab_name)
|
||||
result["value"] = value if value is not None else ""
|
||||
result["fromEnv"] = is_value_from_env(field)
|
||||
|
||||
return result
|
||||
|
||||
|
||||
def serialize_tab(tab: SettingsTab, include_values: bool = True) -> Dict[str, Any]:
|
||||
"""Serialize a settings tab for API response."""
|
||||
return {
|
||||
"name": tab.name,
|
||||
"displayName": tab.display_name,
|
||||
"icon": tab.icon,
|
||||
"order": tab.order,
|
||||
"group": tab.group,
|
||||
"fields": [serialize_field(f, tab.name, include_values) for f in tab.fields],
|
||||
}
|
||||
|
||||
|
||||
def serialize_group(group: SettingsGroup) -> Dict[str, Any]:
|
||||
"""Serialize a settings group for API response."""
|
||||
return {
|
||||
"name": group.name,
|
||||
"displayName": group.display_name,
|
||||
"icon": group.icon,
|
||||
"order": group.order,
|
||||
}
|
||||
|
||||
|
||||
def get_all_groups() -> List[SettingsGroup]:
|
||||
"""Get all registered settings groups, sorted by order."""
|
||||
return sorted(_GROUPS_REGISTRY.values(), key=lambda g: (g.order, g.name))
|
||||
|
||||
|
||||
def serialize_all_settings(include_values: bool = True) -> Dict[str, Any]:
|
||||
"""Serialize all settings for API response."""
|
||||
tabs = get_all_settings_tabs()
|
||||
groups = get_all_groups()
|
||||
return {
|
||||
"tabs": [serialize_tab(t, include_values) for t in tabs],
|
||||
"groups": [serialize_group(g) for g in groups],
|
||||
}
|
||||
|
||||
|
||||
def execute_action(tab_name: str, action_key: str, current_values: Optional[Dict[str, Any]] = None) -> Dict[str, Any]:
|
||||
"""
|
||||
Execute an action button's callback.
|
||||
|
||||
Args:
|
||||
tab_name: The settings tab name.
|
||||
action_key: The action key to execute.
|
||||
current_values: Optional dict of current form values (unsaved).
|
||||
Passed to callbacks that accept it.
|
||||
|
||||
Returns:
|
||||
Dict with "success" (bool) and "message" (str).
|
||||
"""
|
||||
import inspect
|
||||
|
||||
tab = get_settings_tab(tab_name)
|
||||
if not tab:
|
||||
return {"success": False, "message": f"Unknown settings tab: {tab_name}"}
|
||||
|
||||
for field in tab.fields:
|
||||
if isinstance(field, ActionButton) and field.key == action_key:
|
||||
if field.callback:
|
||||
try:
|
||||
# Check if callback accepts current_values parameter
|
||||
sig = inspect.signature(field.callback)
|
||||
if 'current_values' in sig.parameters:
|
||||
return field.callback(current_values=current_values or {})
|
||||
else:
|
||||
return field.callback()
|
||||
except Exception as e:
|
||||
logger.error(f"Action {action_key} failed: {e}")
|
||||
return {"success": False, "message": str(e)}
|
||||
else:
|
||||
return {"success": False, "message": "Action has no callback defined"}
|
||||
|
||||
return {"success": False, "message": f"Unknown action: {action_key}"}
|
||||
|
||||
|
||||
def _sync_metadata_provider_selection() -> None:
|
||||
"""
|
||||
Sync the METADATA_PROVIDER setting based on enabled providers.
|
||||
|
||||
Called after saving metadata provider settings to auto-select
|
||||
the first enabled provider if the current selection is invalid.
|
||||
"""
|
||||
try:
|
||||
from shelfmark.metadata_providers import sync_metadata_provider_selection
|
||||
sync_metadata_provider_selection()
|
||||
except ImportError:
|
||||
pass # Metadata providers module not available
|
||||
|
||||
|
||||
def _apply_dns_settings(config) -> None:
|
||||
"""
|
||||
Apply DNS settings changes to the network module.
|
||||
|
||||
This ensures DNS changes take effect immediately without requiring
|
||||
a container restart.
|
||||
"""
|
||||
try:
|
||||
from shelfmark.download import network
|
||||
|
||||
provider = config.get("CUSTOM_DNS", "auto")
|
||||
use_doh = config.get("USE_DOH", False)
|
||||
manual_servers = None
|
||||
|
||||
if provider == "manual":
|
||||
manual_dns = config.get("CUSTOM_DNS_MANUAL", "")
|
||||
if manual_dns:
|
||||
# Parse comma-separated server list
|
||||
manual_servers = [s.strip() for s in manual_dns.split(",") if s.strip()]
|
||||
|
||||
network.set_dns_provider(provider, manual_servers, use_doh=use_doh)
|
||||
except ImportError:
|
||||
pass # Network module not available
|
||||
except Exception as e:
|
||||
logger.warning(f"Failed to apply DNS settings: {e}")
|
||||
|
||||
|
||||
def update_settings(tab_name: str, values: Dict[str, Any]) -> Dict[str, Any]:
|
||||
tab = get_settings_tab(tab_name)
|
||||
if not tab:
|
||||
return {"success": False, "message": f"Unknown settings tab: {tab_name}", "updated": [], "requiresRestart": False}
|
||||
|
||||
# Build a map of field keys to fields (exclude non-value fields)
|
||||
field_map = {f.key: f for f in tab.fields if not isinstance(f, (ActionButton, HeadingField))}
|
||||
|
||||
# Filter out values that are set via env vars or unknown
|
||||
values_to_save = {}
|
||||
skipped_env = []
|
||||
skipped_unknown = []
|
||||
restart_required_keys = []
|
||||
|
||||
for key, value in values.items():
|
||||
if key not in field_map:
|
||||
skipped_unknown.append(key)
|
||||
continue
|
||||
|
||||
field = field_map[key]
|
||||
if is_value_from_env(field):
|
||||
skipped_env.append(key)
|
||||
continue
|
||||
|
||||
# Handle password fields - only update if a new value is provided
|
||||
if isinstance(field, PasswordField) and not value:
|
||||
continue
|
||||
|
||||
values_to_save[key] = value
|
||||
|
||||
# Track if this field requires restart
|
||||
if getattr(field, 'requires_restart', False):
|
||||
restart_required_keys.append(key)
|
||||
|
||||
if not values_to_save:
|
||||
message = "No settings to update"
|
||||
if skipped_env:
|
||||
message += f". Skipped (set via env): {', '.join(skipped_env)}"
|
||||
return {"success": True, "message": message, "updated": [], "requiresRestart": False}
|
||||
|
||||
# Call on_save handler if registered (for custom validation/transformation)
|
||||
on_save_handler = get_on_save_handler(tab_name)
|
||||
if on_save_handler:
|
||||
try:
|
||||
result = on_save_handler(values_to_save.copy())
|
||||
if result.get("error"):
|
||||
return {
|
||||
"success": False,
|
||||
"message": result.get("message", "Validation failed"),
|
||||
"updated": [],
|
||||
"requiresRestart": False
|
||||
}
|
||||
# Use the transformed values
|
||||
values_to_save = result.get("values", values_to_save)
|
||||
except Exception as e:
|
||||
logger.error(f"on_save handler for {tab_name} failed: {e}")
|
||||
return {
|
||||
"success": False,
|
||||
"message": f"Save handler error: {str(e)}",
|
||||
"updated": [],
|
||||
"requiresRestart": False
|
||||
}
|
||||
|
||||
# Save to config file
|
||||
if save_config_file(tab_name, values_to_save):
|
||||
# Refresh the config singleton so live settings take effect immediately
|
||||
try:
|
||||
from shelfmark.core.config import config
|
||||
config.refresh()
|
||||
except ImportError:
|
||||
pass # Config module not yet available during initial setup
|
||||
|
||||
# Apply DNS settings changes live (network tab)
|
||||
dns_keys = {"CUSTOM_DNS", "CUSTOM_DNS_MANUAL", "USE_DOH"}
|
||||
if tab_name == "network" and dns_keys.intersection(values_to_save.keys()):
|
||||
_apply_dns_settings(config)
|
||||
|
||||
# Sync metadata provider selection when a provider's enabled state changes
|
||||
tab = get_settings_tab(tab_name)
|
||||
if tab and tab.group == "metadata_providers":
|
||||
_sync_metadata_provider_selection()
|
||||
|
||||
message = f"Updated {len(values_to_save)} setting(s)"
|
||||
if skipped_env:
|
||||
message += f". Skipped (set via env): {', '.join(skipped_env)}"
|
||||
|
||||
requires_restart = len(restart_required_keys) > 0
|
||||
return {
|
||||
"success": True,
|
||||
"message": message,
|
||||
"updated": list(values_to_save.keys()),
|
||||
"requiresRestart": requires_restart,
|
||||
"restartRequiredFor": restart_required_keys,
|
||||
}
|
||||
else:
|
||||
return {"success": False, "message": "Failed to save settings", "updated": [], "requiresRestart": False}
|
||||
@@ -0,0 +1,127 @@
|
||||
"""Shared utility functions for the Shelfmark."""
|
||||
|
||||
import base64
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
|
||||
def is_audiobook(content_type: Optional[str]) -> bool:
|
||||
"""Check if content type indicates an audiobook."""
|
||||
return bool(content_type and "audiobook" in content_type.lower())
|
||||
|
||||
|
||||
CONTENT_TYPES = [
|
||||
"book (fiction)",
|
||||
"book (non-fiction)",
|
||||
"book (unknown)",
|
||||
"magazine",
|
||||
"comic book",
|
||||
"audiobook",
|
||||
"standards document",
|
||||
"musical score",
|
||||
"other",
|
||||
]
|
||||
|
||||
# Maps AA content types to their config keys for content-type routing
|
||||
# Used when AA_CONTENT_TYPE_ROUTING is enabled
|
||||
_AA_CONTENT_TYPE_TO_CONFIG_KEY = {
|
||||
"book (fiction)": "AA_CONTENT_TYPE_DIR_FICTION",
|
||||
"book (non-fiction)": "AA_CONTENT_TYPE_DIR_NON_FICTION",
|
||||
"book (unknown)": "AA_CONTENT_TYPE_DIR_UNKNOWN",
|
||||
"magazine": "AA_CONTENT_TYPE_DIR_MAGAZINE",
|
||||
"comic book": "AA_CONTENT_TYPE_DIR_COMIC",
|
||||
"audiobook": "AA_CONTENT_TYPE_DIR_AUDIOBOOK",
|
||||
"standards document": "AA_CONTENT_TYPE_DIR_STANDARDS",
|
||||
"musical score": "AA_CONTENT_TYPE_DIR_MUSICAL_SCORE",
|
||||
"other": "AA_CONTENT_TYPE_DIR_OTHER",
|
||||
}
|
||||
|
||||
# Legacy mapping - kept for backwards compatibility during migration
|
||||
_LEGACY_CONTENT_TYPE_TO_CONFIG_KEY = {
|
||||
"book (fiction)": "INGEST_DIR_BOOK_FICTION",
|
||||
"book (non-fiction)": "INGEST_DIR_BOOK_NON_FICTION",
|
||||
"book (unknown)": "INGEST_DIR_BOOK_UNKNOWN",
|
||||
"magazine": "INGEST_DIR_MAGAZINE",
|
||||
"comic book": "INGEST_DIR_COMIC_BOOK",
|
||||
"audiobook": "INGEST_DIR_AUDIOBOOK",
|
||||
"standards document": "INGEST_DIR_STANDARDS_DOCUMENT",
|
||||
"musical score": "INGEST_DIR_MUSICAL_SCORE",
|
||||
"other": "INGEST_DIR_OTHER",
|
||||
}
|
||||
|
||||
|
||||
def get_destination(is_audiobook: bool = False) -> Path:
|
||||
"""Get base destination directory. Audiobooks fall back to main destination."""
|
||||
from shelfmark.core.config import config
|
||||
|
||||
if is_audiobook:
|
||||
# Audiobook destination with fallback to main destination
|
||||
audiobook_dest = config.get("DESTINATION_AUDIOBOOK", "")
|
||||
if audiobook_dest:
|
||||
return Path(audiobook_dest)
|
||||
|
||||
# Main destination (also fallback for audiobooks)
|
||||
# Check new setting first, then legacy INGEST_DIR
|
||||
destination = config.get("DESTINATION", "") or config.get("INGEST_DIR", "/books")
|
||||
return Path(destination)
|
||||
|
||||
|
||||
def get_aa_content_type_dir(content_type: Optional[str] = None) -> Optional[Path]:
|
||||
"""Get override directory for AA content-type routing if configured."""
|
||||
from shelfmark.core.config import config
|
||||
|
||||
# Check if content-type routing is enabled (new or legacy setting)
|
||||
if not config.get("AA_CONTENT_TYPE_ROUTING", False) and not config.get("USE_CONTENT_TYPE_DIRECTORIES", False):
|
||||
return None
|
||||
|
||||
if not content_type:
|
||||
return None
|
||||
|
||||
content_type_lower = content_type.lower().strip()
|
||||
|
||||
# Try new AA-specific config keys first, then legacy keys
|
||||
for mapping in (_AA_CONTENT_TYPE_TO_CONFIG_KEY, _LEGACY_CONTENT_TYPE_TO_CONFIG_KEY):
|
||||
config_key = mapping.get(content_type_lower)
|
||||
if config_key:
|
||||
custom_dir = config.get(config_key, "")
|
||||
if custom_dir:
|
||||
return Path(custom_dir)
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def get_ingest_dir(content_type: Optional[str] = None) -> Path:
|
||||
"""DEPRECATED: Use get_destination() and get_aa_content_type_dir() instead."""
|
||||
from shelfmark.core.config import config
|
||||
|
||||
# Check new DESTINATION setting first, then legacy INGEST_DIR
|
||||
default_ingest_dir = Path(config.get("DESTINATION", "") or config.get("INGEST_DIR", "/books"))
|
||||
|
||||
if not content_type:
|
||||
return default_ingest_dir
|
||||
|
||||
# Check for content-type override
|
||||
override_dir = get_aa_content_type_dir(content_type)
|
||||
if override_dir:
|
||||
return override_dir
|
||||
|
||||
return default_ingest_dir
|
||||
|
||||
|
||||
def transform_cover_url(cover_url: Optional[str], cache_id: str) -> Optional[str]:
|
||||
"""Transform external cover URL to local proxy URL when caching is enabled."""
|
||||
if not cover_url:
|
||||
return cover_url
|
||||
|
||||
# Skip if already a local URL (starts with /)
|
||||
if cover_url.startswith('/'):
|
||||
return cover_url
|
||||
|
||||
# Check if cover caching is enabled
|
||||
from shelfmark.config.env import is_covers_cache_enabled
|
||||
if not is_covers_cache_enabled():
|
||||
return cover_url
|
||||
|
||||
# Encode the original URL and create a proxy URL
|
||||
encoded_url = base64.urlsafe_b64encode(cover_url.encode()).decode()
|
||||
return f"/api/covers/{cache_id}?url={encoded_url}"
|
||||
@@ -0,0 +1 @@
|
||||
"""Download module - HTTP downloads, network, and orchestration."""
|
||||
@@ -0,0 +1,449 @@
|
||||
"""Archive extraction utilities for downloaded book archives."""
|
||||
|
||||
import os
|
||||
import shutil
|
||||
import zipfile
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import List, Optional, Tuple
|
||||
|
||||
from shelfmark.core.logger import setup_logger
|
||||
from shelfmark.core.config import config
|
||||
from shelfmark.core.naming import parse_naming_template, sanitize_filename
|
||||
from shelfmark.core.utils import is_audiobook as check_audiobook
|
||||
from shelfmark.download.fs import atomic_write, atomic_move
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
|
||||
def _get_supported_formats() -> List[str]:
|
||||
"""Get current supported formats from config singleton."""
|
||||
formats = config.get("SUPPORTED_FORMATS", ["epub", "mobi", "azw3", "fb2", "djvu", "cbz", "cbr"])
|
||||
# Handle both list (from MultiSelectField) and comma-separated string (legacy/env)
|
||||
if isinstance(formats, str):
|
||||
return [fmt.strip().lower() for fmt in formats.split(",") if fmt.strip()]
|
||||
return [fmt.lower() for fmt in formats]
|
||||
|
||||
|
||||
def _get_supported_audiobook_formats() -> List[str]:
|
||||
"""Get current supported audiobook formats from config singleton."""
|
||||
formats = config.get("SUPPORTED_AUDIOBOOK_FORMATS", ["m4b", "mp3"])
|
||||
# Handle both list (from MultiSelectField) and comma-separated string (legacy/env)
|
||||
if isinstance(formats, str):
|
||||
return [fmt.strip().lower() for fmt in formats.split(",") if fmt.strip()]
|
||||
return [fmt.lower() for fmt in formats]
|
||||
|
||||
|
||||
def _get_file_organization(is_audiobook: bool) -> str:
|
||||
"""Get the file organization mode for the content type."""
|
||||
key = "FILE_ORGANIZATION_AUDIOBOOK" if is_audiobook else "FILE_ORGANIZATION"
|
||||
mode = config.get(key, "rename")
|
||||
|
||||
# Handle legacy settings migration
|
||||
if mode not in ("none", "rename", "organize"):
|
||||
legacy_key = "PROCESSING_MODE_AUDIOBOOK" if is_audiobook else "PROCESSING_MODE"
|
||||
legacy_mode = config.get(legacy_key, "ingest")
|
||||
if legacy_mode == "library":
|
||||
return "organize"
|
||||
if config.get("USE_BOOK_TITLE", True):
|
||||
return "rename"
|
||||
return "none"
|
||||
|
||||
return mode
|
||||
|
||||
|
||||
def _get_template(is_audiobook: bool, organization_mode: str) -> str:
|
||||
"""Get the template for the content type and organization mode."""
|
||||
# Determine the correct key based on content type and organization mode
|
||||
if is_audiobook:
|
||||
if organization_mode == "organize":
|
||||
key = "TEMPLATE_AUDIOBOOK_ORGANIZE"
|
||||
else:
|
||||
key = "TEMPLATE_AUDIOBOOK_RENAME"
|
||||
else:
|
||||
if organization_mode == "organize":
|
||||
key = "TEMPLATE_ORGANIZE"
|
||||
else:
|
||||
key = "TEMPLATE_RENAME"
|
||||
|
||||
template = config.get(key, "")
|
||||
|
||||
# Fallback to legacy keys if new keys are empty
|
||||
if not template:
|
||||
legacy_key = "TEMPLATE_AUDIOBOOK" if is_audiobook else "TEMPLATE"
|
||||
template = config.get(legacy_key, "")
|
||||
|
||||
if not template:
|
||||
legacy_key = "LIBRARY_TEMPLATE_AUDIOBOOK" if is_audiobook else "LIBRARY_TEMPLATE"
|
||||
template = config.get(legacy_key, "")
|
||||
|
||||
if not template:
|
||||
if organization_mode == "organize":
|
||||
return "{Author}/{Title} ({Year})"
|
||||
return "{Author} - {Title} ({Year})"
|
||||
|
||||
return template
|
||||
|
||||
|
||||
def _build_filename_from_task(task, extension: str, organization_mode: str) -> str:
|
||||
"""Build a filename from task metadata using the configured template."""
|
||||
is_audiobook = check_audiobook(task.content_type)
|
||||
|
||||
template = _get_template(is_audiobook, organization_mode)
|
||||
metadata = {
|
||||
"Author": task.author,
|
||||
"Title": task.title,
|
||||
"Subtitle": getattr(task, 'subtitle', None),
|
||||
"Year": task.year,
|
||||
"Series": getattr(task, 'series_name', None),
|
||||
"SeriesPosition": getattr(task, 'series_position', None),
|
||||
}
|
||||
|
||||
filename = parse_naming_template(template, metadata)
|
||||
if filename:
|
||||
return f"{sanitize_filename(filename)}.{extension}"
|
||||
return ""
|
||||
|
||||
# Check for rarfile availability at module load
|
||||
try:
|
||||
import rarfile
|
||||
|
||||
RAR_AVAILABLE = True
|
||||
except ImportError:
|
||||
RAR_AVAILABLE = False
|
||||
logger.warning("rarfile not installed - RAR extraction disabled")
|
||||
|
||||
|
||||
class ArchiveExtractionError(Exception):
|
||||
"""Raised when archive extraction fails."""
|
||||
|
||||
pass
|
||||
|
||||
|
||||
class PasswordProtectedError(ArchiveExtractionError):
|
||||
"""Raised when archive requires a password."""
|
||||
|
||||
pass
|
||||
|
||||
|
||||
class CorruptedArchiveError(ArchiveExtractionError):
|
||||
"""Raised when archive is corrupted."""
|
||||
|
||||
pass
|
||||
|
||||
|
||||
def is_archive(file_path: Path) -> bool:
|
||||
"""Check if file is a supported archive format."""
|
||||
suffix = file_path.suffix.lower().lstrip(".")
|
||||
return suffix in ("zip", "rar")
|
||||
|
||||
|
||||
def _is_supported_file(file_path: Path, content_type: Optional[str] = None) -> bool:
|
||||
"""Check if file matches user's supported formats setting based on content type."""
|
||||
ext = file_path.suffix.lower().lstrip(".")
|
||||
if check_audiobook(content_type):
|
||||
supported_formats = _get_supported_audiobook_formats()
|
||||
else:
|
||||
supported_formats = _get_supported_formats()
|
||||
return ext in supported_formats
|
||||
|
||||
|
||||
# All known ebook extensions (superset of what user might enable)
|
||||
ALL_EBOOK_EXTENSIONS = {'.pdf', '.epub', '.mobi', '.azw', '.azw3', '.fb2', '.djvu', '.cbz', '.cbr', '.doc', '.docx', '.rtf', '.txt'}
|
||||
|
||||
# All known audio extensions (superset of what user might enable for audiobooks)
|
||||
ALL_AUDIO_EXTENSIONS = {'.m4b', '.mp3', '.m4a', '.aac', '.flac', '.ogg', '.wma', '.wav', '.opus'}
|
||||
|
||||
|
||||
def _filter_files(
|
||||
extracted_files: List[Path],
|
||||
content_type: Optional[str] = None,
|
||||
) -> Tuple[List[Path], List[Path], List[Path]]:
|
||||
"""Filter files by content type. Returns (matched, rejected_format, other)."""
|
||||
is_audiobook = check_audiobook(content_type)
|
||||
known_extensions = ALL_AUDIO_EXTENSIONS if is_audiobook else ALL_EBOOK_EXTENSIONS
|
||||
|
||||
matched_files = []
|
||||
rejected_format_files = []
|
||||
other_files = []
|
||||
|
||||
for file_path in extracted_files:
|
||||
if _is_supported_file(file_path, content_type):
|
||||
matched_files.append(file_path)
|
||||
elif file_path.suffix.lower() in known_extensions:
|
||||
rejected_format_files.append(file_path)
|
||||
else:
|
||||
other_files.append(file_path)
|
||||
|
||||
return matched_files, rejected_format_files, other_files
|
||||
|
||||
|
||||
def extract_archive(
|
||||
archive_path: Path,
|
||||
output_dir: Path,
|
||||
content_type: Optional[str] = None,
|
||||
) -> Tuple[List[Path], List[str], List[Path]]:
|
||||
"""Extract archive and filter by content type. Returns (matched, warnings, rejected)."""
|
||||
suffix = archive_path.suffix.lower().lstrip(".")
|
||||
|
||||
if suffix == "zip":
|
||||
extracted_files, warnings = _extract_zip(archive_path, output_dir)
|
||||
elif suffix == "rar":
|
||||
extracted_files, warnings = _extract_rar(archive_path, output_dir)
|
||||
else:
|
||||
raise ArchiveExtractionError(f"Unsupported archive format: {suffix}")
|
||||
|
||||
is_audiobook = check_audiobook(content_type)
|
||||
file_type_label = "audiobook" if is_audiobook else "book"
|
||||
|
||||
# Filter files based on content type
|
||||
matched_files, rejected_files, other_files = _filter_files(extracted_files, content_type)
|
||||
|
||||
# Delete rejected files (valid formats but not enabled by user)
|
||||
for rejected_file in rejected_files:
|
||||
try:
|
||||
rejected_file.unlink()
|
||||
logger.debug(f"Deleted rejected {file_type_label} file: {rejected_file.name}")
|
||||
except OSError as e:
|
||||
logger.warning(f"Failed to delete rejected {file_type_label} file {rejected_file}: {e}")
|
||||
|
||||
if rejected_files:
|
||||
rejected_exts = sorted(set(f.suffix.lower() for f in rejected_files))
|
||||
warnings.append(f"Skipped {len(rejected_files)} {file_type_label}(s) with unsupported format: {', '.join(rejected_exts)}")
|
||||
|
||||
# Delete other files (images, html, etc)
|
||||
for other_file in other_files:
|
||||
try:
|
||||
other_file.unlink()
|
||||
logger.debug(f"Deleted non-{file_type_label} file: {other_file.name}")
|
||||
except OSError as e:
|
||||
logger.warning(f"Failed to delete non-{file_type_label} file {other_file}: {e}")
|
||||
|
||||
if other_files:
|
||||
warnings.append(f"Skipped {len(other_files)} non-{file_type_label} file(s)")
|
||||
|
||||
return matched_files, warnings, rejected_files
|
||||
|
||||
|
||||
def _extract_files_from_archive(archive, output_dir: Path) -> List[Path]:
|
||||
"""Extract files from ZipFile or RarFile to output_dir with security checks."""
|
||||
extracted_files = []
|
||||
|
||||
for info in archive.infolist():
|
||||
if info.is_dir():
|
||||
continue
|
||||
|
||||
# Use only filename, strip directory path (security: prevent path traversal)
|
||||
filename = Path(info.filename).name
|
||||
if not filename:
|
||||
continue
|
||||
|
||||
# Security: reject filenames with null bytes or path separators
|
||||
# Check both / and \ since archives may be created on different OSes
|
||||
if "\x00" in filename or "/" in filename or "\\" in filename:
|
||||
logger.warning(f"Skipping suspicious filename in archive: {info.filename!r}")
|
||||
continue
|
||||
|
||||
# Extract to output_dir with flat structure
|
||||
target_path = output_dir / filename
|
||||
|
||||
# Security: verify resolved path stays within output directory (defense-in-depth)
|
||||
try:
|
||||
target_path.resolve().relative_to(output_dir.resolve())
|
||||
except ValueError:
|
||||
logger.warning(f"Path traversal attempt blocked: {info.filename!r}")
|
||||
continue
|
||||
|
||||
with archive.open(info) as src:
|
||||
data = src.read()
|
||||
final_path = atomic_write(target_path, data)
|
||||
extracted_files.append(final_path)
|
||||
logger.debug(f"Extracted: {filename}")
|
||||
|
||||
return extracted_files
|
||||
|
||||
|
||||
def _extract_zip(archive_path: Path, output_dir: Path) -> Tuple[List[Path], List[str]]:
|
||||
"""Extract files from a ZIP archive."""
|
||||
try:
|
||||
with zipfile.ZipFile(archive_path, "r") as zf:
|
||||
# Check for password protection
|
||||
for info in zf.infolist():
|
||||
if info.flag_bits & 0x1: # Encrypted flag
|
||||
raise PasswordProtectedError("ZIP archive is password protected")
|
||||
|
||||
# Test archive integrity
|
||||
bad_file = zf.testzip()
|
||||
if bad_file:
|
||||
raise CorruptedArchiveError(f"Corrupted file in archive: {bad_file}")
|
||||
|
||||
return _extract_files_from_archive(zf, output_dir), []
|
||||
|
||||
except zipfile.BadZipFile as e:
|
||||
raise CorruptedArchiveError(f"Invalid or corrupted ZIP: {e}")
|
||||
except PermissionError as e:
|
||||
raise ArchiveExtractionError(f"Permission denied: {e}")
|
||||
|
||||
|
||||
def _extract_rar(archive_path: Path, output_dir: Path) -> Tuple[List[Path], List[str]]:
|
||||
"""Extract files from a RAR archive."""
|
||||
if not RAR_AVAILABLE:
|
||||
raise ArchiveExtractionError("RAR extraction not available - rarfile library not installed")
|
||||
|
||||
try:
|
||||
with rarfile.RarFile(archive_path, "r") as rf:
|
||||
# Check for password protection
|
||||
if rf.needs_password():
|
||||
raise PasswordProtectedError("RAR archive is password protected")
|
||||
|
||||
# Test archive integrity
|
||||
rf.testrar()
|
||||
|
||||
return _extract_files_from_archive(rf, output_dir), []
|
||||
|
||||
except rarfile.BadRarFile as e:
|
||||
raise CorruptedArchiveError(f"Invalid or corrupted RAR: {e}")
|
||||
except rarfile.RarCannotExec:
|
||||
raise ArchiveExtractionError("unrar binary not found - install unrar package")
|
||||
except PermissionError as e:
|
||||
raise ArchiveExtractionError(f"Permission denied: {e}")
|
||||
|
||||
|
||||
@dataclass
|
||||
class ArchiveResult:
|
||||
"""Result of archive processing."""
|
||||
|
||||
success: bool
|
||||
final_paths: List[Path]
|
||||
message: str
|
||||
error: Optional[str] = None
|
||||
|
||||
|
||||
def process_archive(
|
||||
archive_path: Path,
|
||||
temp_dir: Path,
|
||||
ingest_dir: Path,
|
||||
archive_id: str,
|
||||
task: Optional["DownloadTask"] = None,
|
||||
) -> ArchiveResult:
|
||||
"""Extract archive, filter to supported formats, move to ingest directory."""
|
||||
extract_dir = temp_dir / f"extract_{archive_id}"
|
||||
content_type = task.content_type if task else None
|
||||
is_audiobook = check_audiobook(content_type)
|
||||
file_type_label = "audiobook" if is_audiobook else "book"
|
||||
|
||||
try:
|
||||
# Create temp extraction directory
|
||||
os.makedirs(extract_dir, exist_ok=True)
|
||||
os.makedirs(ingest_dir, exist_ok=True)
|
||||
|
||||
# Extract to temp directory (filters based on content type)
|
||||
extracted_files, warnings, rejected_files = extract_archive(archive_path, extract_dir, content_type)
|
||||
|
||||
if not extracted_files:
|
||||
# Clean up and return error
|
||||
shutil.rmtree(extract_dir, ignore_errors=True)
|
||||
archive_path.unlink(missing_ok=True)
|
||||
|
||||
if rejected_files:
|
||||
# Found files but they weren't in supported formats
|
||||
rejected_exts = sorted(set(f.suffix.lower() for f in rejected_files))
|
||||
rejected_list = ", ".join(rejected_exts)
|
||||
supported_formats = _get_supported_audiobook_formats() if is_audiobook else _get_supported_formats()
|
||||
logger.warning(
|
||||
f"Found {len(rejected_files)} {file_type_label}(s) in archive but format not supported. "
|
||||
f"Rejected: {rejected_list}. Supported: {', '.join(sorted(supported_formats))}"
|
||||
)
|
||||
return ArchiveResult(
|
||||
success=False,
|
||||
final_paths=[],
|
||||
message="",
|
||||
error=f"Found {len(rejected_files)} {file_type_label}(s) but format not supported ({rejected_list}). Enable in Settings > Formats.",
|
||||
)
|
||||
|
||||
return ArchiveResult(
|
||||
success=False,
|
||||
final_paths=[],
|
||||
message="",
|
||||
error=f"No {file_type_label} files found in archive",
|
||||
)
|
||||
|
||||
for warning in warnings:
|
||||
logger.debug(warning)
|
||||
|
||||
logger.info(f"Extracted {len(extracted_files)} {file_type_label} file(s) from archive")
|
||||
|
||||
# Move book files to ingest folder
|
||||
final_paths = []
|
||||
|
||||
# Determine file organization mode
|
||||
is_audiobook = check_audiobook(task.content_type) if task else False
|
||||
organization_mode = _get_file_organization(is_audiobook) if task else "none"
|
||||
|
||||
for extracted_file in extracted_files:
|
||||
# For multi-file archives (book packs, series), always preserve original filenames
|
||||
# since metadata title only applies to the searched book, not the whole pack.
|
||||
# For single files, respect FILE_ORGANIZATION setting.
|
||||
if len(extracted_files) == 1 and organization_mode != "none" and task:
|
||||
# Use the extracted file's actual extension, not the archive's extension
|
||||
extracted_format = extracted_file.suffix.lower().lstrip('.')
|
||||
filename = _build_filename_from_task(task, extracted_format, organization_mode)
|
||||
if not filename:
|
||||
filename = extracted_file.name
|
||||
else:
|
||||
filename = extracted_file.name
|
||||
|
||||
dest_path = ingest_dir / filename
|
||||
final_path = atomic_move(extracted_file, dest_path)
|
||||
final_paths.append(final_path)
|
||||
logger.debug(f"Moved to ingest: {final_path.name}")
|
||||
|
||||
# Clean up temp extraction directory and archive
|
||||
shutil.rmtree(extract_dir, ignore_errors=True)
|
||||
archive_path.unlink(missing_ok=True)
|
||||
|
||||
# Build success message with format info
|
||||
formats = [p.suffix.lstrip(".").upper() for p in final_paths]
|
||||
if len(formats) == 1:
|
||||
message = f"Complete ({formats[0]})"
|
||||
else:
|
||||
message = f"Complete ({len(formats)} files)"
|
||||
|
||||
return ArchiveResult(
|
||||
success=True,
|
||||
final_paths=final_paths,
|
||||
message=message,
|
||||
)
|
||||
|
||||
except PasswordProtectedError:
|
||||
logger.error(f"Password-protected archive: {archive_path.name}")
|
||||
shutil.rmtree(extract_dir, ignore_errors=True)
|
||||
archive_path.unlink(missing_ok=True)
|
||||
return ArchiveResult(
|
||||
success=False,
|
||||
final_paths=[],
|
||||
message="",
|
||||
error="Archive is password protected",
|
||||
)
|
||||
|
||||
except CorruptedArchiveError as e:
|
||||
logger.error(f"Corrupted archive: {e}")
|
||||
shutil.rmtree(extract_dir, ignore_errors=True)
|
||||
archive_path.unlink(missing_ok=True)
|
||||
return ArchiveResult(
|
||||
success=False,
|
||||
final_paths=[],
|
||||
message="",
|
||||
error=f"Corrupted archive: {e}",
|
||||
)
|
||||
|
||||
except ArchiveExtractionError as e:
|
||||
logger.error(f"Archive extraction failed: {e}")
|
||||
shutil.rmtree(extract_dir, ignore_errors=True)
|
||||
archive_path.unlink(missing_ok=True)
|
||||
return ArchiveResult(
|
||||
success=False,
|
||||
final_paths=[],
|
||||
message="",
|
||||
error=f"Extraction failed: {e}",
|
||||
)
|
||||
@@ -0,0 +1,191 @@
|
||||
"""Atomic filesystem operations for concurrent-safe file handling.
|
||||
|
||||
These utilities handle file collisions atomically, avoiding TOCTOU race conditions
|
||||
when multiple workers may try to write to the same path simultaneously.
|
||||
"""
|
||||
|
||||
import errno
|
||||
import os
|
||||
import shutil
|
||||
from pathlib import Path
|
||||
|
||||
from shelfmark.core.logger import setup_logger
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
|
||||
def atomic_write(dest_path: Path, data: bytes, max_attempts: int = 100) -> Path:
|
||||
"""Write data to a file with atomic collision detection.
|
||||
|
||||
If the destination already exists, retries with counter suffix (_1, _2, etc.)
|
||||
until a unique path is found.
|
||||
|
||||
Args:
|
||||
dest_path: Desired destination path
|
||||
data: Bytes to write
|
||||
max_attempts: Maximum collision retries before raising error
|
||||
|
||||
Returns:
|
||||
Path where file was actually written (may differ from dest_path)
|
||||
|
||||
Raises:
|
||||
RuntimeError: If no unique path found after max_attempts
|
||||
"""
|
||||
base = dest_path.stem
|
||||
ext = dest_path.suffix
|
||||
parent = dest_path.parent
|
||||
|
||||
for attempt in range(max_attempts):
|
||||
try_path = dest_path if attempt == 0 else parent / f"{base}_{attempt}{ext}"
|
||||
try:
|
||||
# O_CREAT | O_EXCL fails atomically if file exists
|
||||
fd = os.open(str(try_path), os.O_CREAT | os.O_EXCL | os.O_WRONLY)
|
||||
try:
|
||||
os.write(fd, data)
|
||||
finally:
|
||||
os.close(fd)
|
||||
if attempt > 0:
|
||||
logger.info(f"File collision resolved: {try_path.name}")
|
||||
return try_path
|
||||
except FileExistsError:
|
||||
continue
|
||||
|
||||
raise RuntimeError(f"Could not write file after {max_attempts} attempts: {dest_path}")
|
||||
|
||||
|
||||
def atomic_move(source_path: Path, dest_path: Path, max_attempts: int = 100) -> Path:
|
||||
"""Move a file with collision detection.
|
||||
|
||||
Uses os.rename() for same-filesystem moves (atomic, triggers inotify events),
|
||||
falls back to exclusive create + shutil.move for cross-filesystem moves.
|
||||
|
||||
Note: We use os.rename() instead of hardlink+unlink because os.rename()
|
||||
triggers proper inotify IN_MOVED_TO events that file watchers (like Calibre's
|
||||
auto-add) rely on to detect new files.
|
||||
|
||||
Args:
|
||||
source_path: Source file to move
|
||||
dest_path: Desired destination path
|
||||
max_attempts: Maximum collision retries before raising error
|
||||
|
||||
Returns:
|
||||
Path where file was actually moved (may differ from dest_path)
|
||||
|
||||
Raises:
|
||||
RuntimeError: If no unique path found after max_attempts
|
||||
"""
|
||||
base = dest_path.stem
|
||||
ext = dest_path.suffix
|
||||
parent = dest_path.parent
|
||||
|
||||
for attempt in range(max_attempts):
|
||||
try_path = dest_path if attempt == 0 else parent / f"{base}_{attempt}{ext}"
|
||||
|
||||
# Check for existing file (os.rename would overwrite on Unix)
|
||||
if try_path.exists():
|
||||
continue
|
||||
|
||||
try:
|
||||
# os.rename is atomic on same filesystem and triggers inotify events
|
||||
os.rename(str(source_path), str(try_path))
|
||||
if attempt > 0:
|
||||
logger.info(f"File collision resolved: {try_path.name}")
|
||||
return try_path
|
||||
except FileExistsError:
|
||||
# Race condition: file created between exists() check and rename()
|
||||
continue
|
||||
except OSError as e:
|
||||
# Cross-filesystem - fall back to exclusive create + move
|
||||
if e.errno != errno.EXDEV:
|
||||
raise
|
||||
try:
|
||||
fd = os.open(str(try_path), os.O_CREAT | os.O_EXCL | os.O_WRONLY)
|
||||
os.close(fd)
|
||||
try:
|
||||
shutil.move(str(source_path), str(try_path))
|
||||
if attempt > 0:
|
||||
logger.info(f"File collision resolved: {try_path.name}")
|
||||
return try_path
|
||||
except Exception:
|
||||
try_path.unlink(missing_ok=True)
|
||||
raise
|
||||
except FileExistsError:
|
||||
continue
|
||||
|
||||
raise RuntimeError(f"Could not move file after {max_attempts} attempts: {dest_path}")
|
||||
|
||||
|
||||
def atomic_hardlink(source_path: Path, dest_path: Path, max_attempts: int = 100) -> Path:
|
||||
"""Create a hardlink with atomic collision detection.
|
||||
|
||||
Args:
|
||||
source_path: Source file to link from
|
||||
dest_path: Desired destination path for the link
|
||||
max_attempts: Maximum collision retries before raising error
|
||||
|
||||
Returns:
|
||||
Path where link was actually created (may differ from dest_path)
|
||||
|
||||
Raises:
|
||||
RuntimeError: If no unique path found after max_attempts
|
||||
"""
|
||||
base = dest_path.stem
|
||||
ext = dest_path.suffix
|
||||
parent = dest_path.parent
|
||||
|
||||
for attempt in range(max_attempts):
|
||||
try_path = dest_path if attempt == 0 else parent / f"{base}_{attempt}{ext}"
|
||||
try:
|
||||
os.link(str(source_path), str(try_path))
|
||||
if attempt > 0:
|
||||
logger.info(f"File collision resolved: {try_path.name}")
|
||||
return try_path
|
||||
except FileExistsError:
|
||||
continue
|
||||
|
||||
raise RuntimeError(f"Could not create hardlink after {max_attempts} attempts: {dest_path}")
|
||||
|
||||
|
||||
def atomic_copy(source_path: Path, dest_path: Path, max_attempts: int = 100) -> Path:
|
||||
"""Copy a file with atomic collision detection.
|
||||
|
||||
Uses exclusive create to claim destination, then copies via temp file
|
||||
to avoid partial files on failure.
|
||||
|
||||
Args:
|
||||
source_path: Source file to copy
|
||||
dest_path: Desired destination path
|
||||
max_attempts: Maximum collision retries before raising error
|
||||
|
||||
Returns:
|
||||
Path where file was actually copied (may differ from dest_path)
|
||||
|
||||
Raises:
|
||||
RuntimeError: If no unique path found after max_attempts
|
||||
"""
|
||||
base = dest_path.stem
|
||||
ext = dest_path.suffix
|
||||
parent = dest_path.parent
|
||||
|
||||
for attempt in range(max_attempts):
|
||||
try_path = dest_path if attempt == 0 else parent / f"{base}_{attempt}{ext}"
|
||||
try:
|
||||
# Atomically claim the destination by creating an exclusive file
|
||||
fd = os.open(str(try_path), os.O_CREAT | os.O_EXCL | os.O_WRONLY)
|
||||
os.close(fd)
|
||||
# Copy to temp file first, then replace to avoid partial files
|
||||
temp_path = try_path.parent / f".{try_path.name}.tmp"
|
||||
try:
|
||||
shutil.copy2(str(source_path), str(temp_path))
|
||||
temp_path.replace(try_path)
|
||||
if attempt > 0:
|
||||
logger.info(f"File collision resolved: {try_path.name}")
|
||||
return try_path
|
||||
except Exception:
|
||||
try_path.unlink(missing_ok=True)
|
||||
temp_path.unlink(missing_ok=True)
|
||||
raise
|
||||
except FileExistsError:
|
||||
continue
|
||||
|
||||
raise RuntimeError(f"Could not copy file after {max_attempts} attempts: {dest_path}")
|
||||
@@ -0,0 +1,450 @@
|
||||
"""HTTP download with retry, resume, and Cloudflare bypass support."""
|
||||
|
||||
import random
|
||||
import time
|
||||
from io import BytesIO
|
||||
from threading import Event
|
||||
from typing import Callable, Optional
|
||||
from urllib.parse import urlparse
|
||||
|
||||
import requests
|
||||
from tqdm import tqdm
|
||||
|
||||
from shelfmark.download import network
|
||||
from shelfmark.download.network import get_proxies
|
||||
from shelfmark.core.config import config as app_config
|
||||
from shelfmark.core.logger import setup_logger
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
# Bypasser modules are imported lazily to support dynamic selection based on config
|
||||
_internal_bypasser = None
|
||||
_external_bypasser = None
|
||||
|
||||
|
||||
def _get_internal_bypasser():
|
||||
"""Lazy import of internal bypasser module."""
|
||||
global _internal_bypasser
|
||||
if _internal_bypasser is None:
|
||||
try:
|
||||
from shelfmark.bypass import internal_bypasser
|
||||
_internal_bypasser = internal_bypasser
|
||||
except ImportError as e:
|
||||
raise RuntimeError(
|
||||
f"Failed to import internal bypasser: {e}. "
|
||||
"Check that all dependencies are installed. "
|
||||
"You may need to disable CF bypass or use the external bypasser."
|
||||
) from e
|
||||
return _internal_bypasser
|
||||
|
||||
|
||||
def _get_external_bypasser():
|
||||
"""Lazy import of external bypasser module."""
|
||||
global _external_bypasser
|
||||
if _external_bypasser is None:
|
||||
try:
|
||||
from shelfmark.bypass import external_bypasser
|
||||
_external_bypasser = external_bypasser
|
||||
except ImportError as e:
|
||||
raise RuntimeError(
|
||||
f"Failed to import external bypasser: {e}. "
|
||||
"Check that the external bypasser is properly configured."
|
||||
) from e
|
||||
return _external_bypasser
|
||||
|
||||
|
||||
def _is_using_external_bypasser() -> bool:
|
||||
"""Check if external bypasser is configured (reads from config, not just env)."""
|
||||
return app_config.get("USING_EXTERNAL_BYPASSER", False)
|
||||
|
||||
|
||||
def _is_cf_bypass_enabled() -> bool:
|
||||
"""Check if Cloudflare bypass is enabled."""
|
||||
return app_config.get("USE_CF_BYPASS", True)
|
||||
|
||||
|
||||
def get_bypassed_page(url, selector=None, cancel_flag=None):
|
||||
"""Wrapper that delegates to the appropriate bypasser based on config."""
|
||||
if _is_using_external_bypasser():
|
||||
return _get_external_bypasser().get_bypassed_page(url, selector, cancel_flag)
|
||||
return _get_internal_bypasser().get_bypassed_page(url, selector, cancel_flag)
|
||||
|
||||
|
||||
def get_cf_cookies_for_domain(domain):
|
||||
"""Get CF cookies - only available with internal bypasser."""
|
||||
if _is_using_external_bypasser():
|
||||
logger.debug(f"External bypasser in use, CF cookies not available for {domain}")
|
||||
return {}
|
||||
return _get_internal_bypasser().get_cf_cookies_for_domain(domain)
|
||||
|
||||
|
||||
def get_cf_user_agent_for_domain(domain):
|
||||
"""Get CF user agent - only available with internal bypasser."""
|
||||
if _is_using_external_bypasser():
|
||||
logger.debug(f"External bypasser in use, CF user agent not available for {domain}")
|
||||
return None
|
||||
return _get_internal_bypasser().get_cf_user_agent_for_domain(domain)
|
||||
|
||||
|
||||
def _apply_cf_bypass(url: str, headers: dict) -> dict:
|
||||
"""Apply CF bypass cookies and user agent if available.
|
||||
|
||||
Modifies headers in-place with the stored user agent (if available).
|
||||
Returns cookies dict to use with the request.
|
||||
"""
|
||||
if not _is_cf_bypass_enabled():
|
||||
return {}
|
||||
|
||||
parsed = urlparse(url)
|
||||
hostname = parsed.hostname or ""
|
||||
cookies = get_cf_cookies_for_domain(hostname)
|
||||
stored_ua = get_cf_user_agent_for_domain(hostname)
|
||||
if stored_ua:
|
||||
headers['User-Agent'] = stored_ua
|
||||
return cookies
|
||||
|
||||
|
||||
# Network settings
|
||||
REQUEST_TIMEOUT = (5, 10) # (connect, read)
|
||||
MAX_DOWNLOAD_RETRIES = 2
|
||||
MAX_RESUME_ATTEMPTS = 3
|
||||
|
||||
RETRYABLE_CODES = (429, 500, 502, 503, 504)
|
||||
CONNECTION_ERRORS = (requests.exceptions.ConnectionError, requests.exceptions.Timeout,
|
||||
requests.exceptions.SSLError, requests.exceptions.ChunkedEncodingError)
|
||||
DOWNLOAD_HEADERS = {
|
||||
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/129.0.0.0 Safari/537.36',
|
||||
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8',
|
||||
'Accept-Language': 'en-US,en;q=0.5',
|
||||
'Accept-Encoding': 'gzip, deflate, br',
|
||||
'Connection': 'keep-alive',
|
||||
'Upgrade-Insecure-Requests': '1',
|
||||
}
|
||||
|
||||
|
||||
def parse_size_string(size: str) -> Optional[float]:
|
||||
"""Parse a human-readable size string (e.g., '10.5 MB') into bytes."""
|
||||
if not size:
|
||||
return None
|
||||
try:
|
||||
normalized = size.strip().replace(" ", "").replace(",", ".").upper()
|
||||
multipliers = {"GB": 1024**3, "MB": 1024**2, "KB": 1024}
|
||||
for suffix, mult in multipliers.items():
|
||||
if normalized.endswith(suffix):
|
||||
return float(normalized[:-2]) * mult
|
||||
return float(normalized)
|
||||
except (ValueError, IndexError):
|
||||
return None
|
||||
|
||||
def _backoff_delay(attempt: int, base: float = 0.25, cap: float = 3.0) -> float:
|
||||
"""Exponential backoff with jitter."""
|
||||
return min(cap, base * (2 ** (attempt - 1))) + random.random() * base
|
||||
|
||||
|
||||
def _get_status_code(e: Exception) -> Optional[int]:
|
||||
"""Extract HTTP status code from an exception, or None if not applicable."""
|
||||
if isinstance(e, requests.exceptions.HTTPError) and e.response is not None:
|
||||
return e.response.status_code
|
||||
return None
|
||||
|
||||
def _is_retryable_error(e: Exception) -> bool:
|
||||
"""Check if error is retryable (connection error or retryable HTTP status)."""
|
||||
if isinstance(e, CONNECTION_ERRORS):
|
||||
return True
|
||||
status = _get_status_code(e)
|
||||
return status is not None and status in RETRYABLE_CODES
|
||||
|
||||
|
||||
def _try_rotation(original_url: str, current_url: str, selector: network.AAMirrorSelector) -> Optional[str]:
|
||||
"""Try mirror/DNS rotation. Returns new URL or None."""
|
||||
if current_url.startswith(network.get_aa_base_url()):
|
||||
new_base, action = selector.next_mirror_or_rotate_dns()
|
||||
if action in ("mirror", "dns") and new_base:
|
||||
new_url = selector.rewrite(original_url)
|
||||
logger.info(f"[{action}] switching to: {new_url}")
|
||||
return new_url
|
||||
elif network.should_rotate_dns_for_url(current_url) and network.rotate_dns_provider():
|
||||
logger.info(f"[dns-rotate] retrying: {original_url}")
|
||||
return original_url
|
||||
return None
|
||||
|
||||
|
||||
def html_get_page(
|
||||
url: str,
|
||||
retry: Optional[int] = None,
|
||||
use_bypasser: bool = False,
|
||||
selector: Optional[network.AAMirrorSelector] = None,
|
||||
cancel_flag: Optional[Event] = None,
|
||||
status_callback: Optional[Callable[[str, Optional[str]], None]] = None,
|
||||
) -> str:
|
||||
"""Fetch HTML content from a URL with retry mechanism."""
|
||||
retry = retry if retry is not None else app_config.MAX_RETRY
|
||||
selector = selector or network.AAMirrorSelector()
|
||||
original_url = url
|
||||
current_url = selector.rewrite(original_url)
|
||||
use_bypasser_now = use_bypasser
|
||||
|
||||
for attempt in range(1, retry + 1):
|
||||
# Check for cancellation before each attempt
|
||||
if cancel_flag and cancel_flag.is_set():
|
||||
logger.info(f"html_get_page cancelled before attempt {attempt}")
|
||||
return ""
|
||||
|
||||
try:
|
||||
if use_bypasser_now and _is_cf_bypass_enabled():
|
||||
logger.debug(f"GET (bypasser): {current_url}")
|
||||
if status_callback:
|
||||
status_callback("resolving", "Bypassing protection")
|
||||
try:
|
||||
result = get_bypassed_page(current_url, selector, cancel_flag)
|
||||
return result or ""
|
||||
except Exception as e:
|
||||
logger.warning(f"Bypasser error: {type(e).__name__}: {e}")
|
||||
return ""
|
||||
|
||||
logger.debug(f"GET: {current_url}")
|
||||
# Try with CF cookies/UA if available (from previous bypass)
|
||||
headers = {}
|
||||
cookies = _apply_cf_bypass(current_url, headers)
|
||||
response = requests.get(current_url, proxies=get_proxies(), timeout=REQUEST_TIMEOUT, cookies=cookies, headers=headers)
|
||||
response.raise_for_status()
|
||||
time.sleep(1)
|
||||
return response.text
|
||||
|
||||
except Exception as e:
|
||||
status = _get_status_code(e)
|
||||
|
||||
# 403 = Cloudflare/DDoS-Guard protection
|
||||
if status == 403:
|
||||
if _is_cf_bypass_enabled() and not use_bypasser_now:
|
||||
# Before switching to bypasser, check if cookies have become available
|
||||
# (another concurrent download may have completed bypass and extracted cookies)
|
||||
parsed = urlparse(current_url)
|
||||
fresh_cookies = get_cf_cookies_for_domain(parsed.hostname or "")
|
||||
if fresh_cookies and not cookies:
|
||||
# Cookies are now available - retry with cookies before using bypasser
|
||||
logger.debug(f"403 but cookies now available - retrying with cookies: {current_url}")
|
||||
continue
|
||||
logger.info(f"403 detected; switching to bypasser: {current_url}")
|
||||
if status_callback:
|
||||
status_callback("resolving", "Bypassing protection...")
|
||||
use_bypasser_now = True
|
||||
continue
|
||||
logger.warning(f"403 error, giving up: {current_url}")
|
||||
return ""
|
||||
|
||||
# 404 = Not found
|
||||
if status == 404:
|
||||
logger.warning(f"404 error: {current_url}")
|
||||
return ""
|
||||
|
||||
# Try mirror/DNS rotation on retryable errors
|
||||
if _is_retryable_error(e):
|
||||
new_url = _try_rotation(original_url, current_url, selector)
|
||||
if new_url:
|
||||
current_url = new_url
|
||||
continue
|
||||
|
||||
# Retry with backoff
|
||||
if attempt < retry:
|
||||
logger.warning(f"Retry {attempt}/{retry} for {current_url}: {type(e).__name__}: {e}")
|
||||
time.sleep(_backoff_delay(attempt))
|
||||
else:
|
||||
logger.error(f"Giving up after {retry} attempts: {current_url}")
|
||||
|
||||
return ""
|
||||
|
||||
|
||||
def download_url(
|
||||
link: str,
|
||||
size: str = "",
|
||||
progress_callback: Optional[Callable[[float], None]] = None,
|
||||
cancel_flag: Optional[Event] = None,
|
||||
_selector: Optional[network.AAMirrorSelector] = None,
|
||||
status_callback: Optional[Callable[[str, Optional[str]], None]] = None,
|
||||
referer: Optional[str] = None,
|
||||
) -> Optional[BytesIO]:
|
||||
"""Download content from URL with automatic retry and resume support."""
|
||||
selector = _selector or network.AAMirrorSelector()
|
||||
current_url = selector.rewrite(link)
|
||||
|
||||
# Build headers with optional referer
|
||||
headers = DOWNLOAD_HEADERS.copy()
|
||||
if referer:
|
||||
headers['Referer'] = referer
|
||||
total_size = parse_size_string(size) or 0
|
||||
|
||||
attempt = 0
|
||||
zlib_cookie_refresh_attempted = False
|
||||
|
||||
while attempt < MAX_DOWNLOAD_RETRIES:
|
||||
if cancel_flag and cancel_flag.is_set():
|
||||
return None
|
||||
|
||||
buffer = BytesIO()
|
||||
bytes_downloaded = 0
|
||||
|
||||
try:
|
||||
if attempt > 0 and status_callback:
|
||||
status_callback("resolving", f"Connecting (Attempt {attempt + 1}/{MAX_DOWNLOAD_RETRIES})")
|
||||
|
||||
logger.info(f"Downloading: {current_url} (attempt {attempt + 1}/{MAX_DOWNLOAD_RETRIES})")
|
||||
# Try with CF cookies/UA if available
|
||||
cookies = _apply_cf_bypass(current_url, headers)
|
||||
response = requests.get(current_url, stream=True, proxies=get_proxies(), timeout=REQUEST_TIMEOUT, cookies=cookies, headers=headers)
|
||||
response.raise_for_status()
|
||||
|
||||
if status_callback:
|
||||
status_callback("downloading", "")
|
||||
|
||||
total_size = total_size or float(response.headers.get('content-length', 0))
|
||||
pbar = tqdm(total=total_size, unit='B', unit_scale=True, desc='Downloading')
|
||||
|
||||
for chunk in response.iter_content(chunk_size=8192):
|
||||
if chunk:
|
||||
buffer.write(chunk)
|
||||
bytes_downloaded += len(chunk)
|
||||
pbar.update(len(chunk))
|
||||
if progress_callback and total_size > 0:
|
||||
progress_callback(bytes_downloaded * 100.0 / total_size)
|
||||
if cancel_flag and cancel_flag.is_set():
|
||||
pbar.close()
|
||||
return None
|
||||
pbar.close()
|
||||
|
||||
# Validate - check we didn't get HTML instead of file
|
||||
if total_size > 0 and bytes_downloaded < total_size * 0.9:
|
||||
if response.headers.get('content-type', '').startswith('text/html'):
|
||||
logger.warning(f"Received HTML instead of file: {current_url}")
|
||||
return None
|
||||
|
||||
logger.debug(f"Download completed: {bytes_downloaded} bytes")
|
||||
return buffer
|
||||
|
||||
except requests.exceptions.RequestException as e:
|
||||
status = _get_status_code(e)
|
||||
retryable = _is_retryable_error(e)
|
||||
|
||||
# Z-Library 403 - try refreshing cookies via bypasser once before giving up
|
||||
if status == 403 and _is_cf_bypass_enabled() and not zlib_cookie_refresh_attempted:
|
||||
parsed = urlparse(current_url)
|
||||
if parsed.hostname and 'z-lib' in parsed.hostname and referer:
|
||||
zlib_cookie_refresh_attempted = True
|
||||
logger.info(f"Z-Library 403 - refreshing cookies via referer: {referer}")
|
||||
try:
|
||||
get_bypassed_page(referer, selector, cancel_flag)
|
||||
time.sleep(0.5)
|
||||
# Retry with fresh cookies (don't increment attempt)
|
||||
continue
|
||||
except Exception as cookie_err:
|
||||
logger.warning(f"Z-Library cookie refresh failed: {cookie_err}")
|
||||
|
||||
# Non-retryable errors
|
||||
if status in (403, 404):
|
||||
logger.warning(f"Download failed ({status}): {current_url}")
|
||||
return None
|
||||
|
||||
# Rate limited - skip to next source immediately
|
||||
# (waiting doesn't help with concurrent downloads hitting the same server)
|
||||
if status == 429:
|
||||
logger.info(f"Rate limited (429) - trying next source")
|
||||
if status_callback:
|
||||
status_callback("resolving", "Server busy, trying next")
|
||||
return None
|
||||
|
||||
# Timeout - don't retry, server likely overloaded
|
||||
if isinstance(e, requests.exceptions.Timeout):
|
||||
logger.warning(f"Timeout: {current_url} - skipping to next source")
|
||||
if status_callback:
|
||||
status_callback("resolving", "Server timed out, trying next")
|
||||
return None
|
||||
|
||||
# Try to resume if we got some data
|
||||
if bytes_downloaded > 0 and retryable:
|
||||
resumed = _try_resume(current_url, buffer, bytes_downloaded, total_size, progress_callback, cancel_flag, headers)
|
||||
if resumed:
|
||||
return resumed
|
||||
|
||||
# Try mirror/DNS rotation if nothing downloaded yet
|
||||
if bytes_downloaded == 0 and retryable:
|
||||
new_url = _try_rotation(link, current_url, selector)
|
||||
if new_url:
|
||||
current_url = new_url
|
||||
attempt += 1
|
||||
continue
|
||||
|
||||
logger.warning(f"Download error: {type(e).__name__}: {e}")
|
||||
if attempt < MAX_DOWNLOAD_RETRIES - 1:
|
||||
time.sleep(_backoff_delay(attempt + 1))
|
||||
attempt += 1
|
||||
|
||||
logger.error(f"Download failed after {MAX_DOWNLOAD_RETRIES} attempts: {link}")
|
||||
return None
|
||||
|
||||
|
||||
def _try_resume(
|
||||
url: str,
|
||||
buffer: BytesIO,
|
||||
start_byte: int,
|
||||
total_size: float,
|
||||
progress_callback: Optional[Callable[[float], None]],
|
||||
cancel_flag: Optional[Event],
|
||||
base_headers: Optional[dict] = None,
|
||||
) -> Optional[BytesIO]:
|
||||
"""Try to resume an interrupted download."""
|
||||
for attempt in range(MAX_RESUME_ATTEMPTS):
|
||||
logger.info(f"Resuming from {start_byte} bytes (attempt {attempt + 1}/{MAX_RESUME_ATTEMPTS})")
|
||||
time.sleep(_backoff_delay(attempt + 1, base=0.5, cap=5.0))
|
||||
|
||||
try:
|
||||
# Try with CF cookies/UA if available
|
||||
resume_headers = {**(base_headers or DOWNLOAD_HEADERS), 'Range': f'bytes={start_byte}-'}
|
||||
cookies = _apply_cf_bypass(url, resume_headers)
|
||||
response = requests.get(
|
||||
url, stream=True, proxies=get_proxies(), timeout=REQUEST_TIMEOUT,
|
||||
headers=resume_headers, cookies=cookies
|
||||
)
|
||||
|
||||
# Check resume support
|
||||
if response.status_code == 200: # Server doesn't support resume
|
||||
logger.info("Server doesn't support resume")
|
||||
return None
|
||||
if response.status_code == 416: # Range not satisfiable
|
||||
logger.warning("Range not satisfiable")
|
||||
return None
|
||||
if response.status_code != 206:
|
||||
response.raise_for_status()
|
||||
|
||||
pbar = tqdm(total=total_size, initial=start_byte, unit='B', unit_scale=True, desc='Resuming')
|
||||
for chunk in response.iter_content(chunk_size=8192):
|
||||
if chunk:
|
||||
buffer.write(chunk)
|
||||
start_byte += len(chunk)
|
||||
pbar.update(len(chunk))
|
||||
if progress_callback and total_size > 0:
|
||||
progress_callback(start_byte * 100.0 / total_size)
|
||||
if cancel_flag and cancel_flag.is_set():
|
||||
pbar.close()
|
||||
return None
|
||||
pbar.close()
|
||||
|
||||
logger.info(f"Resume completed: {start_byte} bytes")
|
||||
return buffer
|
||||
|
||||
except requests.exceptions.RequestException as e:
|
||||
logger.debug(f"Resume attempt {attempt + 1} failed: {e}")
|
||||
|
||||
logger.warning(f"Resume failed after {MAX_RESUME_ATTEMPTS} attempts")
|
||||
return None
|
||||
|
||||
|
||||
def get_absolute_url(base_url: str, url: str) -> str:
|
||||
"""Convert a relative URL to absolute using the base URL."""
|
||||
url = url.strip()
|
||||
if not url or url == "#" or url.startswith("http"):
|
||||
return url if url.startswith("http") else ""
|
||||
parsed = urlparse(url)
|
||||
base = urlparse(base_url)
|
||||
if not parsed.netloc or not parsed.scheme:
|
||||
parsed = parsed._replace(netloc=base.netloc, scheme=base.scheme)
|
||||
return parsed.geturl()
|
||||
@@ -0,0 +1,318 @@
|
||||
# Metadata Providers
|
||||
|
||||
This module provides a plugin architecture for fetching book metadata from various sources with a unified interface.
|
||||
|
||||
## Overview
|
||||
|
||||
Metadata providers allow searching for books and retrieving detailed metadata (title, authors, cover images, descriptions, etc.) from external services. The system uses a decorator-based registration pattern, making it easy to add new providers.
|
||||
|
||||
## Available Providers
|
||||
|
||||
| Provider | Auth Required | Description |
|
||||
|----------|---------------|-------------|
|
||||
| **Hardcover** | Yes (API key) | Modern book tracking platform with GraphQL API. Get your key at [hardcover.app/account/api](https://hardcover.app/account/api) |
|
||||
| **Open Library** | No | Free, open-source library catalog from the Internet Archive. Rate limited to ~100 requests/minute |
|
||||
|
||||
## Core Components
|
||||
|
||||
### BookMetadata
|
||||
|
||||
Dataclass representing a book from a metadata provider:
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class BookMetadata:
|
||||
provider: str # Internal provider name (e.g., "hardcover")
|
||||
provider_id: str # ID in that provider's system
|
||||
title: str
|
||||
|
||||
# Optional fields
|
||||
provider_display_name: str # Human-readable name (e.g., "Hardcover")
|
||||
authors: List[str]
|
||||
isbn_10: str
|
||||
isbn_13: str
|
||||
cover_url: str
|
||||
description: str
|
||||
publisher: str
|
||||
publish_year: int
|
||||
language: str
|
||||
genres: List[str]
|
||||
source_url: str # Link to book on provider's site
|
||||
display_fields: List[DisplayField] # Provider-specific display data
|
||||
```
|
||||
|
||||
### DisplayField
|
||||
|
||||
Provider-specific metadata for UI cards (ratings, page counts, reader counts, etc.):
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class DisplayField:
|
||||
label: str # e.g., "Rating", "Pages", "Readers"
|
||||
value: str # e.g., "4.5", "496", "8,041"
|
||||
icon: str # Icon name: "star", "book", "users", "editions"
|
||||
```
|
||||
|
||||
### MetadataSearchOptions
|
||||
|
||||
Unified search options that work across all providers:
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class MetadataSearchOptions:
|
||||
query: str
|
||||
search_type: SearchType = SearchType.GENERAL # GENERAL, TITLE, AUTHOR, ISBN
|
||||
language: str = None # ISO 639-1 code (e.g., "en")
|
||||
sort: SortOrder = SortOrder.RELEVANCE
|
||||
limit: int = 40
|
||||
page: int = 1
|
||||
```
|
||||
|
||||
### SortOrder
|
||||
|
||||
Available sort options (provider support varies):
|
||||
|
||||
| Sort Order | Description | Hardcover | Open Library |
|
||||
|------------|-------------|-----------|--------------|
|
||||
| `RELEVANCE` | Best match first (default) | ✓ | ✓ |
|
||||
| `POPULARITY` | Most popular first | ✓ | ✗ |
|
||||
| `RATING` | Highest rated first | ✓ | ✗ |
|
||||
| `NEWEST` | Most recently published | ✓ | ✓ |
|
||||
| `OLDEST` | Oldest published first | ✓ | ✓ |
|
||||
|
||||
### MetadataProvider (Abstract Base Class)
|
||||
|
||||
All providers must implement this interface:
|
||||
|
||||
```python
|
||||
class MetadataProvider(ABC):
|
||||
name: str # Internal identifier
|
||||
display_name: str # Human-readable name
|
||||
requires_auth: bool # True if API key required
|
||||
supported_sorts: List[SortOrder] # Supported sort options
|
||||
|
||||
@abstractmethod
|
||||
def search(self, options: MetadataSearchOptions) -> List[BookMetadata]:
|
||||
"""Search for books using the provided options."""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def get_book(self, book_id: str) -> Optional[BookMetadata]:
|
||||
"""Get a specific book by provider ID."""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def search_by_isbn(self, isbn: str) -> Optional[BookMetadata]:
|
||||
"""Search for a book by ISBN."""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def is_available(self) -> bool:
|
||||
"""Check if this provider is configured and available."""
|
||||
pass
|
||||
```
|
||||
|
||||
## Registry Functions
|
||||
|
||||
### Provider Registration
|
||||
|
||||
```python
|
||||
from shelfmark.metadata_providers import register_provider
|
||||
|
||||
@register_provider("my_provider")
|
||||
class MyProvider(MetadataProvider):
|
||||
...
|
||||
```
|
||||
|
||||
### Getting Providers
|
||||
|
||||
```python
|
||||
from shelfmark.metadata_providers import (
|
||||
get_provider,
|
||||
get_configured_provider,
|
||||
get_provider_kwargs,
|
||||
list_providers,
|
||||
is_provider_registered,
|
||||
)
|
||||
|
||||
# Get specific provider with kwargs
|
||||
provider = get_provider("hardcover", api_key="...")
|
||||
|
||||
# Get currently configured provider (from settings)
|
||||
provider = get_configured_provider()
|
||||
|
||||
# Get provider-specific kwargs from config
|
||||
kwargs = get_provider_kwargs("hardcover") # {"api_key": "..."}
|
||||
|
||||
# List all registered providers
|
||||
providers = list_providers()
|
||||
# [{"name": "hardcover", "display_name": "Hardcover", "requires_auth": True}, ...]
|
||||
|
||||
# Check if provider exists
|
||||
exists = is_provider_registered("hardcover") # True
|
||||
```
|
||||
|
||||
### Sort Options
|
||||
|
||||
```python
|
||||
from shelfmark.metadata_providers import get_provider_sort_options
|
||||
|
||||
# Get sort options for a specific provider
|
||||
options = get_provider_sort_options("hardcover")
|
||||
# [{"value": "relevance", "label": "Most relevant"}, ...]
|
||||
|
||||
# Get sort options for configured provider
|
||||
options = get_provider_sort_options() # Uses METADATA_PROVIDER from config
|
||||
```
|
||||
|
||||
## Creating a New Provider
|
||||
|
||||
1. Create a new file in `shelfmark/metadata_providers/` (e.g., `my_provider.py`)
|
||||
|
||||
2. Implement the provider:
|
||||
|
||||
```python
|
||||
from shelfmark.metadata_providers import (
|
||||
BookMetadata,
|
||||
DisplayField,
|
||||
MetadataProvider,
|
||||
MetadataSearchOptions,
|
||||
SearchType,
|
||||
SortOrder,
|
||||
register_provider,
|
||||
)
|
||||
from shelfmark.core.settings_registry import (
|
||||
register_settings,
|
||||
HeadingField,
|
||||
PasswordField,
|
||||
ActionButton,
|
||||
)
|
||||
from shelfmark.core.config import config
|
||||
|
||||
|
||||
@register_provider("my_provider")
|
||||
class MyProvider(MetadataProvider):
|
||||
name = "my_provider"
|
||||
display_name = "My Provider"
|
||||
requires_auth = True
|
||||
supported_sorts = [SortOrder.RELEVANCE, SortOrder.NEWEST]
|
||||
|
||||
def __init__(self, api_key: str = None):
|
||||
self.api_key = api_key or config.get("MY_PROVIDER_API_KEY", "")
|
||||
|
||||
def is_available(self) -> bool:
|
||||
return bool(self.api_key)
|
||||
|
||||
def search(self, options: MetadataSearchOptions) -> List[BookMetadata]:
|
||||
# Handle ISBN search separately
|
||||
if options.search_type == SearchType.ISBN:
|
||||
result = self.search_by_isbn(options.query)
|
||||
return [result] if result else []
|
||||
|
||||
# Implement search logic...
|
||||
return []
|
||||
|
||||
def get_book(self, book_id: str) -> Optional[BookMetadata]:
|
||||
# Implement get book logic...
|
||||
return None
|
||||
|
||||
def search_by_isbn(self, isbn: str) -> Optional[BookMetadata]:
|
||||
# Implement ISBN search logic...
|
||||
return None
|
||||
|
||||
|
||||
# Settings for the UI
|
||||
@register_settings("my_provider", "My Provider", icon="book", order=53, group="metadata_providers")
|
||||
def my_provider_settings():
|
||||
return [
|
||||
HeadingField(
|
||||
key="my_provider_heading",
|
||||
title="My Provider",
|
||||
description="Description of your provider",
|
||||
link_url="https://myprovider.com",
|
||||
link_text="myprovider.com",
|
||||
),
|
||||
PasswordField(
|
||||
key="MY_PROVIDER_API_KEY",
|
||||
label="API Key",
|
||||
description="Your API key",
|
||||
required=True,
|
||||
),
|
||||
ActionButton(
|
||||
key="test_connection",
|
||||
label="Test Connection",
|
||||
style="primary",
|
||||
callback=_test_connection,
|
||||
),
|
||||
]
|
||||
```
|
||||
|
||||
3. Import your provider in `__init__.py`:
|
||||
|
||||
```python
|
||||
try:
|
||||
from shelfmark.metadata_providers import my_provider # noqa: F401
|
||||
except ImportError:
|
||||
pass # Provider is optional
|
||||
```
|
||||
|
||||
4. Add provider kwargs to `get_provider_kwargs()` in `__init__.py`:
|
||||
|
||||
```python
|
||||
def get_provider_kwargs(provider_name: str) -> Dict:
|
||||
kwargs: Dict = {}
|
||||
if provider_name == "hardcover":
|
||||
kwargs["api_key"] = app_config.get("HARDCOVER_API_KEY", "")
|
||||
elif provider_name == "my_provider":
|
||||
kwargs["api_key"] = app_config.get("MY_PROVIDER_API_KEY", "")
|
||||
return kwargs
|
||||
```
|
||||
|
||||
## Caching
|
||||
|
||||
Providers should use the `@cacheable` decorator for API calls:
|
||||
|
||||
```python
|
||||
from shelfmark.core.cache import cacheable
|
||||
from shelfmark.config.env import (
|
||||
METADATA_CACHE_SEARCH_TTL,
|
||||
METADATA_CACHE_BOOK_TTL,
|
||||
)
|
||||
|
||||
@cacheable(ttl=METADATA_CACHE_SEARCH_TTL, key_prefix="myprovider:search")
|
||||
def _search_cached(self, cache_key: str, options: MetadataSearchOptions):
|
||||
# Cached search implementation
|
||||
pass
|
||||
|
||||
@cacheable(ttl=METADATA_CACHE_BOOK_TTL, key_prefix="myprovider:book")
|
||||
def get_book(self, book_id: str):
|
||||
# Cached book lookup
|
||||
pass
|
||||
```
|
||||
|
||||
## Rate Limiting
|
||||
|
||||
For providers with rate limits (like Open Library), implement a rate limiter:
|
||||
|
||||
```python
|
||||
from shelfmark.metadata_providers.openlibrary import RateLimiter
|
||||
|
||||
# 90 requests per 60 seconds
|
||||
rate_limiter = RateLimiter(max_requests=90, window_seconds=60)
|
||||
|
||||
def make_request(self):
|
||||
rate_limiter.wait_if_needed() # Blocks if rate limited
|
||||
# ... make request
|
||||
```
|
||||
|
||||
## Configuration
|
||||
|
||||
Provider settings are stored in `CONFIG_DIR/plugins/<provider_name>.json` and managed via the Settings UI. See [Plugin Settings Guide](../../docs/plugin-settings.md) for detailed documentation on adding settings to your provider.
|
||||
|
||||
## Environment Variables
|
||||
|
||||
| Variable | Default | Description |
|
||||
|----------|---------|-------------|
|
||||
| `METADATA_PROVIDER` | `""` | Active metadata provider name |
|
||||
| `METADATA_CACHE_SEARCH_TTL` | `3600` | Search cache TTL in seconds |
|
||||
| `METADATA_CACHE_BOOK_TTL` | `86400` | Book lookup cache TTL in seconds |
|
||||
@@ -0,0 +1,426 @@
|
||||
"""Metadata provider plugin system - base classes and registry."""
|
||||
|
||||
from abc import ABC, abstractmethod
|
||||
from dataclasses import dataclass, field
|
||||
from enum import Enum
|
||||
from typing import Any, Dict, List, Optional, Type, Union
|
||||
|
||||
|
||||
class SearchType(str, Enum):
|
||||
"""Type of search to perform."""
|
||||
GENERAL = "general" # Search all fields (title, author, ISBN, etc.)
|
||||
TITLE = "title" # Search by title only
|
||||
AUTHOR = "author" # Search by author only
|
||||
ISBN = "isbn" # Search by ISBN
|
||||
|
||||
|
||||
class SortOrder(str, Enum):
|
||||
"""Sort order for search results."""
|
||||
RELEVANCE = "relevance" # Best match first (default)
|
||||
POPULARITY = "popularity" # Most popular first
|
||||
RATING = "rating" # Highest rated first
|
||||
NEWEST = "newest" # Most recently published first
|
||||
OLDEST = "oldest" # Oldest published first
|
||||
SERIES_ORDER = "series_order" # By series position (requires series field)
|
||||
|
||||
|
||||
# Display labels for sort options
|
||||
SORT_LABELS: Dict[SortOrder, str] = {
|
||||
SortOrder.RELEVANCE: "Most relevant",
|
||||
SortOrder.POPULARITY: "Most popular",
|
||||
SortOrder.RATING: "Highest rated",
|
||||
SortOrder.NEWEST: "Newest",
|
||||
SortOrder.OLDEST: "Oldest",
|
||||
SortOrder.SERIES_ORDER: "Series order",
|
||||
}
|
||||
|
||||
|
||||
@dataclass
|
||||
class TextSearchField:
|
||||
"""Text input search field."""
|
||||
key: str # Field identifier (e.g., "author", "publisher")
|
||||
label: str # Display label in UI
|
||||
placeholder: str = "" # Placeholder text
|
||||
description: str = "" # Help text
|
||||
|
||||
|
||||
@dataclass
|
||||
class NumberSearchField:
|
||||
"""Numeric input search field."""
|
||||
key: str
|
||||
label: str
|
||||
placeholder: str = ""
|
||||
description: str = ""
|
||||
min_value: Optional[int] = None
|
||||
max_value: Optional[int] = None
|
||||
step: int = 1
|
||||
|
||||
|
||||
@dataclass
|
||||
class SelectSearchField:
|
||||
"""Single-choice dropdown search field."""
|
||||
key: str
|
||||
label: str
|
||||
options: List[Dict[str, str]] = field(default_factory=list) # [{value: "", label: ""}]
|
||||
placeholder: str = ""
|
||||
description: str = ""
|
||||
|
||||
|
||||
@dataclass
|
||||
class CheckboxSearchField:
|
||||
"""Boolean checkbox search field."""
|
||||
key: str
|
||||
label: str
|
||||
description: str = ""
|
||||
default: bool = False
|
||||
|
||||
|
||||
# Type alias for all search field types
|
||||
SearchField = Union[TextSearchField, NumberSearchField, SelectSearchField, CheckboxSearchField]
|
||||
|
||||
|
||||
def serialize_search_field(search_field: SearchField) -> Dict[str, Any]:
|
||||
"""Serialize a search field to dict for API response."""
|
||||
result: Dict[str, Any] = {
|
||||
"key": search_field.key,
|
||||
"label": search_field.label,
|
||||
"type": search_field.__class__.__name__,
|
||||
"placeholder": getattr(search_field, 'placeholder', ''),
|
||||
"description": getattr(search_field, 'description', ''),
|
||||
}
|
||||
|
||||
# Add type-specific properties
|
||||
if isinstance(search_field, NumberSearchField):
|
||||
result["min"] = search_field.min_value
|
||||
result["max"] = search_field.max_value
|
||||
result["step"] = search_field.step
|
||||
elif isinstance(search_field, SelectSearchField):
|
||||
result["options"] = search_field.options
|
||||
elif isinstance(search_field, CheckboxSearchField):
|
||||
result["default"] = search_field.default
|
||||
|
||||
return result
|
||||
|
||||
|
||||
@dataclass
|
||||
class MetadataSearchOptions:
|
||||
"""Options for metadata search queries across all providers."""
|
||||
query: str
|
||||
search_type: SearchType = SearchType.GENERAL
|
||||
language: Optional[str] = None # ISO 639-1 code (e.g., "en", "fr")
|
||||
sort: SortOrder = SortOrder.RELEVANCE
|
||||
limit: int = 40
|
||||
page: int = 1
|
||||
fields: Dict[str, Any] = field(default_factory=dict) # Custom search field values
|
||||
|
||||
|
||||
@dataclass
|
||||
class DisplayField:
|
||||
"""A display field for metadata cards (ratings, page counts, etc.)."""
|
||||
label: str # e.g., "Rating", "Pages", "Readers"
|
||||
value: str # e.g., "4.5", "496", "8,041"
|
||||
icon: Optional[str] = None # Icon name: "star", "book", "users", "editions"
|
||||
|
||||
|
||||
@dataclass
|
||||
class BookMetadata:
|
||||
"""Book from metadata provider (not a specific release)."""
|
||||
provider: str # Which provider this came from (internal name)
|
||||
provider_id: str # ID in that provider's system
|
||||
title: str
|
||||
|
||||
# Provider display name for UI (e.g., "Open Library" instead of "openlibrary")
|
||||
provider_display_name: Optional[str] = None
|
||||
|
||||
# Optional - not all providers have all fields
|
||||
authors: List[str] = field(default_factory=list)
|
||||
isbn_10: Optional[str] = None
|
||||
isbn_13: Optional[str] = None
|
||||
cover_url: Optional[str] = None
|
||||
description: Optional[str] = None
|
||||
publisher: Optional[str] = None
|
||||
publish_year: Optional[int] = None
|
||||
language: Optional[str] = None
|
||||
genres: List[str] = field(default_factory=list)
|
||||
source_url: Optional[str] = None # Link to book on provider's site
|
||||
subtitle: Optional[str] = None # Book subtitle, if any
|
||||
|
||||
# Provider-specific display fields for cards/lists
|
||||
display_fields: List[DisplayField] = field(default_factory=list)
|
||||
|
||||
# Series info (if book is part of a series)
|
||||
series_name: Optional[str] = None # Name of the series
|
||||
series_position: Optional[float] = None # This book's position (e.g., 3, 1.5 for novellas)
|
||||
series_count: Optional[int] = None # Total books in the series
|
||||
|
||||
# Alternative titles by language (for localized searches)
|
||||
# Maps language code (e.g., "de", "German") to localized title
|
||||
titles_by_language: Dict[str, str] = field(default_factory=dict)
|
||||
|
||||
|
||||
@dataclass
|
||||
class SearchResult:
|
||||
"""Result from a metadata search with pagination info."""
|
||||
books: List[BookMetadata]
|
||||
page: int = 1
|
||||
total_found: int = 0 # Total matching results (if known)
|
||||
has_more: bool = False # True if more results available
|
||||
|
||||
|
||||
class MetadataProvider(ABC):
|
||||
"""Interface for metadata providers.
|
||||
|
||||
All metadata providers must implement this interface. The search method
|
||||
accepts MetadataSearchOptions for unified search across providers.
|
||||
|
||||
Attributes:
|
||||
name: Internal identifier (e.g., "hardcover")
|
||||
display_name: Human-readable name (e.g., "Hardcover")
|
||||
requires_auth: True if API key/authentication is required
|
||||
supported_sorts: List of SortOrder values this provider supports
|
||||
search_fields: List of provider-specific search fields
|
||||
"""
|
||||
name: str
|
||||
display_name: str
|
||||
requires_auth: bool
|
||||
supported_sorts: List[SortOrder] = [SortOrder.RELEVANCE]
|
||||
search_fields: List[SearchField] = []
|
||||
|
||||
@abstractmethod
|
||||
def search(self, options: MetadataSearchOptions) -> List[BookMetadata]:
|
||||
"""Search for books using the provided options."""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def get_book(self, book_id: str) -> Optional[BookMetadata]:
|
||||
"""Get a specific book by provider ID."""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def search_by_isbn(self, isbn: str) -> Optional[BookMetadata]:
|
||||
"""Search for a book by ISBN."""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def is_available(self) -> bool:
|
||||
"""Check if this provider is configured and available."""
|
||||
pass
|
||||
|
||||
def search_paginated(self, options: MetadataSearchOptions) -> SearchResult:
|
||||
"""Search with pagination info. Override for accurate pagination."""
|
||||
books = self.search(options)
|
||||
# Heuristic: if we got exactly limit results, there might be more
|
||||
has_more = len(books) >= options.limit
|
||||
return SearchResult(
|
||||
books=books,
|
||||
page=options.page,
|
||||
total_found=0, # Unknown without provider-specific implementation
|
||||
has_more=has_more
|
||||
)
|
||||
|
||||
|
||||
# Provider registry
|
||||
_PROVIDERS: Dict[str, Type[MetadataProvider]] = {}
|
||||
_PROVIDER_KWARGS_FACTORIES: Dict[str, Any] = {} # Callable[[], Dict]
|
||||
|
||||
|
||||
def register_provider(name: str):
|
||||
"""Decorator to register a metadata provider."""
|
||||
def decorator(cls):
|
||||
_PROVIDERS[name] = cls
|
||||
return cls
|
||||
return decorator
|
||||
|
||||
|
||||
def register_provider_kwargs(name: str):
|
||||
"""Decorator to register a provider's kwargs factory.
|
||||
|
||||
The decorated function should return a Dict of kwargs to pass to the
|
||||
provider constructor. This allows each provider to define its own
|
||||
configuration requirements without polluting the core module.
|
||||
|
||||
Example:
|
||||
@register_provider_kwargs("hardcover")
|
||||
def _hardcover_kwargs() -> Dict:
|
||||
from shelfmark.core.config import config
|
||||
return {"api_key": config.get("HARDCOVER_API_KEY", "")}
|
||||
"""
|
||||
def decorator(fn):
|
||||
_PROVIDER_KWARGS_FACTORIES[name] = fn
|
||||
return fn
|
||||
return decorator
|
||||
|
||||
|
||||
def get_provider(name: str, **kwargs) -> MetadataProvider:
|
||||
"""Factory - instantiate any registered provider."""
|
||||
if name not in _PROVIDERS:
|
||||
raise ValueError(f"Unknown metadata provider: {name}")
|
||||
return _PROVIDERS[name](**kwargs)
|
||||
|
||||
|
||||
def list_providers() -> List[dict]:
|
||||
"""For settings UI - list available providers with their requirements."""
|
||||
return [
|
||||
{"name": n, "display_name": c.display_name, "requires_auth": c.requires_auth}
|
||||
for n, c in _PROVIDERS.items()
|
||||
]
|
||||
|
||||
|
||||
def get_provider_kwargs(provider_name: str) -> Dict:
|
||||
"""Get provider-specific initialization kwargs from registered factory."""
|
||||
factory = _PROVIDER_KWARGS_FACTORIES.get(provider_name)
|
||||
if factory:
|
||||
return factory()
|
||||
return {}
|
||||
|
||||
|
||||
def is_provider_registered(provider_name: str) -> bool:
|
||||
"""Check if a provider is registered."""
|
||||
return provider_name in _PROVIDERS
|
||||
|
||||
|
||||
def is_provider_enabled(provider_name: str) -> bool:
|
||||
"""Check if a provider is enabled in settings."""
|
||||
from shelfmark.core.config import config as app_config
|
||||
|
||||
# Refresh config to get latest settings
|
||||
app_config.refresh()
|
||||
|
||||
# Check the provider-specific enabled flag
|
||||
enabled_key = f"{provider_name.upper()}_ENABLED"
|
||||
return app_config.get(enabled_key, False) is True
|
||||
|
||||
|
||||
def get_enabled_providers() -> List[str]:
|
||||
"""Get list of all enabled provider names."""
|
||||
return [name for name in _PROVIDERS if is_provider_enabled(name)]
|
||||
|
||||
|
||||
def get_configured_provider(content_type: str = "ebook") -> Optional[MetadataProvider]:
|
||||
"""Get the currently configured metadata provider for the content type."""
|
||||
from shelfmark.core.config import config as app_config
|
||||
|
||||
# Refresh config to ensure we have the latest saved settings
|
||||
app_config.refresh()
|
||||
|
||||
# For audiobooks, try audiobook-specific provider first, then fall back to main provider
|
||||
if content_type == "audiobook":
|
||||
metadata_provider = app_config.get("METADATA_PROVIDER_AUDIOBOOK", "")
|
||||
if not metadata_provider:
|
||||
metadata_provider = app_config.get("METADATA_PROVIDER", "")
|
||||
else:
|
||||
metadata_provider = app_config.get("METADATA_PROVIDER", "")
|
||||
|
||||
if not metadata_provider:
|
||||
return None
|
||||
|
||||
if metadata_provider not in _PROVIDERS:
|
||||
return None
|
||||
|
||||
# Check if the provider is enabled
|
||||
if not is_provider_enabled(metadata_provider):
|
||||
return None
|
||||
|
||||
kwargs = get_provider_kwargs(metadata_provider)
|
||||
return get_provider(metadata_provider, **kwargs)
|
||||
|
||||
|
||||
def _get_configured_provider_name() -> str:
|
||||
"""Get the currently configured metadata provider name from config."""
|
||||
from shelfmark.core.config import config as app_config
|
||||
app_config.refresh()
|
||||
return app_config.get("METADATA_PROVIDER", "")
|
||||
|
||||
|
||||
def get_provider_sort_options(provider_name: Optional[str] = None) -> List[Dict[str, str]]:
|
||||
"""Get sort options for a metadata provider as {value, label} dicts."""
|
||||
if provider_name is None:
|
||||
provider_name = _get_configured_provider_name()
|
||||
|
||||
if provider_name and provider_name in _PROVIDERS:
|
||||
provider_class = _PROVIDERS[provider_name]
|
||||
supported = getattr(provider_class, 'supported_sorts', [SortOrder.RELEVANCE])
|
||||
else:
|
||||
supported = [SortOrder.RELEVANCE]
|
||||
|
||||
return [
|
||||
{"value": sort.value, "label": SORT_LABELS.get(sort, sort.value.title())}
|
||||
for sort in supported
|
||||
]
|
||||
|
||||
|
||||
def get_provider_search_fields(provider_name: Optional[str] = None) -> List[Dict[str, Any]]:
|
||||
"""Get search fields for a metadata provider as serialized dicts."""
|
||||
if provider_name is None:
|
||||
provider_name = _get_configured_provider_name()
|
||||
|
||||
if provider_name and provider_name in _PROVIDERS:
|
||||
provider_class = _PROVIDERS[provider_name]
|
||||
fields = getattr(provider_class, 'search_fields', [])
|
||||
else:
|
||||
fields = []
|
||||
|
||||
return [serialize_search_field(f) for f in fields]
|
||||
|
||||
|
||||
def get_provider_default_sort(provider_name: Optional[str] = None) -> str:
|
||||
"""Get the default sort order for a metadata provider."""
|
||||
from shelfmark.core.config import config as app_config
|
||||
|
||||
if provider_name is None:
|
||||
provider_name = _get_configured_provider_name()
|
||||
|
||||
if not provider_name:
|
||||
return "relevance"
|
||||
|
||||
# Look up provider-specific default sort setting
|
||||
setting_key = f"{provider_name.upper()}_DEFAULT_SORT"
|
||||
return app_config.get(setting_key, "relevance")
|
||||
|
||||
|
||||
def sync_metadata_provider_selection() -> None:
|
||||
"""Sync the METADATA_PROVIDER setting based on enabled providers.
|
||||
|
||||
If the currently selected provider is not enabled (or nothing is selected),
|
||||
auto-select the first enabled provider. This should be called after
|
||||
enabling/disabling a provider.
|
||||
"""
|
||||
from shelfmark.core.config import config as app_config
|
||||
from shelfmark.core.settings_registry import save_config_file, load_config_file
|
||||
|
||||
app_config.refresh()
|
||||
|
||||
current_provider = app_config.get("METADATA_PROVIDER", "")
|
||||
enabled = get_enabled_providers()
|
||||
|
||||
# If current provider is valid and enabled, nothing to do
|
||||
if current_provider and current_provider in enabled:
|
||||
return
|
||||
|
||||
# Auto-select first enabled provider (or clear if none)
|
||||
new_provider = enabled[0] if enabled else ""
|
||||
|
||||
if new_provider != current_provider:
|
||||
# Update the general settings config
|
||||
general_config = load_config_file("general")
|
||||
general_config["METADATA_PROVIDER"] = new_provider
|
||||
save_config_file("general", general_config)
|
||||
app_config.refresh()
|
||||
|
||||
|
||||
# Import provider implementations to trigger registration
|
||||
# These must be imported AFTER the base classes and registry are defined
|
||||
try:
|
||||
from shelfmark.metadata_providers import hardcover # noqa: F401, E402
|
||||
except ImportError:
|
||||
pass # Hardcover provider is optional
|
||||
|
||||
try:
|
||||
from shelfmark.metadata_providers import openlibrary # noqa: F401, E402
|
||||
except ImportError:
|
||||
pass # Open Library provider is optional
|
||||
|
||||
try:
|
||||
from shelfmark.metadata_providers import googlebooks # noqa: F401, E402
|
||||
except ImportError:
|
||||
pass # Google Books provider is optional
|
||||
@@ -0,0 +1,452 @@
|
||||
"""Google Books metadata provider.
|
||||
|
||||
Uses the Google Books API v1 to search and retrieve book metadata.
|
||||
Requires a free API key from Google Cloud Console (~1000 requests/day quota).
|
||||
|
||||
API Documentation: https://developers.google.com/books/docs/v1/using
|
||||
"""
|
||||
|
||||
import requests
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from shelfmark.core.cache import cacheable
|
||||
from shelfmark.core.logger import setup_logger
|
||||
from shelfmark.core.settings_registry import (
|
||||
register_settings,
|
||||
CheckboxField,
|
||||
PasswordField,
|
||||
SelectField,
|
||||
ActionButton,
|
||||
HeadingField,
|
||||
)
|
||||
from shelfmark.core.config import config as app_config
|
||||
from shelfmark.metadata_providers import (
|
||||
BookMetadata,
|
||||
DisplayField,
|
||||
MetadataProvider,
|
||||
MetadataSearchOptions,
|
||||
SearchType,
|
||||
SortOrder,
|
||||
register_provider,
|
||||
register_provider_kwargs,
|
||||
TextSearchField,
|
||||
)
|
||||
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
GOOGLE_BOOKS_BASE_URL = "https://www.googleapis.com/books/v1"
|
||||
|
||||
# Sort mapping - Google only supports "relevance" and "newest"
|
||||
SORT_MAPPING: Dict[SortOrder, Optional[str]] = {
|
||||
SortOrder.RELEVANCE: None, # Default, no param needed
|
||||
SortOrder.NEWEST: "newest",
|
||||
# POPULARITY, RATING, OLDEST not supported - fall back to relevance
|
||||
}
|
||||
|
||||
|
||||
@register_provider_kwargs("googlebooks")
|
||||
def _googlebooks_kwargs() -> Dict[str, Any]:
|
||||
"""Provide Google Books-specific constructor kwargs."""
|
||||
return {"api_key": app_config.get("GOOGLEBOOKS_API_KEY", "")}
|
||||
|
||||
|
||||
@register_provider("googlebooks")
|
||||
class GoogleBooksProvider(MetadataProvider):
|
||||
"""Google Books metadata provider using REST API."""
|
||||
|
||||
name = "googlebooks"
|
||||
display_name = "Google Books"
|
||||
requires_auth = True
|
||||
supported_sorts = [SortOrder.RELEVANCE, SortOrder.NEWEST]
|
||||
search_fields = [
|
||||
TextSearchField(
|
||||
key="author",
|
||||
label="Author",
|
||||
description="Search by author name",
|
||||
),
|
||||
TextSearchField(
|
||||
key="title",
|
||||
label="Title",
|
||||
description="Search by book title",
|
||||
),
|
||||
]
|
||||
|
||||
def __init__(self, api_key: Optional[str] = None):
|
||||
"""Initialize provider with optional API key (falls back to config)."""
|
||||
self.api_key = api_key or app_config.get("GOOGLEBOOKS_API_KEY", "")
|
||||
self.session = requests.Session()
|
||||
|
||||
def is_available(self) -> bool:
|
||||
"""Check if provider is configured with an API key."""
|
||||
return bool(self.api_key)
|
||||
|
||||
def search(self, options: MetadataSearchOptions) -> List[BookMetadata]:
|
||||
"""Search for books using Google Books API."""
|
||||
if not self.api_key:
|
||||
logger.warning("Google Books API key not configured")
|
||||
return []
|
||||
|
||||
# Handle ISBN search separately
|
||||
if options.search_type == SearchType.ISBN:
|
||||
result = self.search_by_isbn(options.query)
|
||||
return [result] if result else []
|
||||
|
||||
# Build cache key from all options
|
||||
fields_key = ":".join(f"{k}={v}" for k, v in sorted(options.fields.items()))
|
||||
cache_key = (
|
||||
f"{options.query}:{options.search_type.value}:{options.sort.value}:"
|
||||
f"{options.language}:{options.limit}:{options.page}:{fields_key}"
|
||||
)
|
||||
return self._search_cached(cache_key, options)
|
||||
|
||||
@cacheable(
|
||||
ttl_key="METADATA_CACHE_SEARCH_TTL",
|
||||
ttl_default=300,
|
||||
key_prefix="googlebooks:search",
|
||||
)
|
||||
def _search_cached(
|
||||
self, cache_key: str, options: MetadataSearchOptions
|
||||
) -> List[BookMetadata]:
|
||||
"""Cached search implementation."""
|
||||
# Build query string with Google Books operators
|
||||
author_value = options.fields.get("author", "").strip()
|
||||
title_value = options.fields.get("title", "").strip()
|
||||
|
||||
query_parts = []
|
||||
|
||||
# Add field-specific operators
|
||||
if title_value:
|
||||
query_parts.append(f"intitle:{title_value}")
|
||||
elif options.search_type == SearchType.TITLE:
|
||||
query_parts.append(f"intitle:{options.query}")
|
||||
|
||||
if author_value:
|
||||
query_parts.append(f"inauthor:{author_value}")
|
||||
elif options.search_type == SearchType.AUTHOR:
|
||||
query_parts.append(f"inauthor:{options.query}")
|
||||
|
||||
# Fall back to general search if no specific fields
|
||||
if not query_parts:
|
||||
query_parts.append(options.query)
|
||||
|
||||
query = "+".join(query_parts)
|
||||
|
||||
# Build request params
|
||||
params: Dict[str, Any] = {
|
||||
"q": query,
|
||||
"maxResults": min(options.limit, 40), # Google max is 40
|
||||
"startIndex": (options.page - 1) * options.limit,
|
||||
"printType": "books", # Exclude magazines
|
||||
}
|
||||
|
||||
# Map sort order (Google only supports relevance and newest)
|
||||
sort = SORT_MAPPING.get(options.sort)
|
||||
if sort: # Only add if not default (relevance)
|
||||
params["orderBy"] = sort
|
||||
|
||||
# Add language filter if specified
|
||||
if options.language:
|
||||
params["langRestrict"] = options.language
|
||||
|
||||
try:
|
||||
result = self._make_request("/volumes", params)
|
||||
if not result:
|
||||
return []
|
||||
|
||||
items = result.get("items", [])
|
||||
books = []
|
||||
|
||||
for item in items:
|
||||
book = self._parse_volume(item)
|
||||
if book:
|
||||
books.append(book)
|
||||
|
||||
logger.info(f"Google Books search '{query}' returned {len(books)} results")
|
||||
return books
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Google Books search error: {e}")
|
||||
return []
|
||||
|
||||
@cacheable(
|
||||
ttl_key="METADATA_CACHE_BOOK_TTL",
|
||||
ttl_default=600,
|
||||
key_prefix="googlebooks:book",
|
||||
)
|
||||
def get_book(self, book_id: str) -> Optional[BookMetadata]:
|
||||
"""Get book details by Google Books volume ID."""
|
||||
try:
|
||||
result = self._make_request(f"/volumes/{book_id}", {})
|
||||
if not result:
|
||||
return None
|
||||
|
||||
return self._parse_volume(result)
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Google Books get_book error: {e}")
|
||||
return None
|
||||
|
||||
@cacheable(
|
||||
ttl_key="METADATA_CACHE_BOOK_TTL",
|
||||
ttl_default=600,
|
||||
key_prefix="googlebooks:isbn",
|
||||
)
|
||||
def search_by_isbn(self, isbn: str) -> Optional[BookMetadata]:
|
||||
"""Search for a book by ISBN-10 or ISBN-13."""
|
||||
# Clean ISBN (remove hyphens and spaces)
|
||||
clean_isbn = isbn.replace("-", "").replace(" ", "").strip()
|
||||
|
||||
# Use ISBN operator for precise lookup
|
||||
params: Dict[str, Any] = {
|
||||
"q": f"isbn:{clean_isbn}",
|
||||
"maxResults": 1,
|
||||
}
|
||||
|
||||
try:
|
||||
result = self._make_request("/volumes", params)
|
||||
if not result:
|
||||
return None
|
||||
|
||||
items = result.get("items", [])
|
||||
if not items:
|
||||
logger.debug(f"No Google Books result for ISBN: {isbn}")
|
||||
return None
|
||||
|
||||
return self._parse_volume(items[0])
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Google Books ISBN search error: {e}")
|
||||
return None
|
||||
|
||||
def _make_request(
|
||||
self, endpoint: str, params: Dict[str, Any]
|
||||
) -> Optional[Dict[str, Any]]:
|
||||
"""Make authenticated API request to endpoint."""
|
||||
if not self.api_key:
|
||||
logger.warning("Google Books API key not configured")
|
||||
return None
|
||||
|
||||
# Add API key to params
|
||||
params["key"] = self.api_key
|
||||
|
||||
url = f"{GOOGLE_BOOKS_BASE_URL}{endpoint}"
|
||||
|
||||
try:
|
||||
response = self.session.get(url, params=params, timeout=15)
|
||||
response.raise_for_status()
|
||||
return response.json()
|
||||
|
||||
except requests.Timeout:
|
||||
logger.warning("Google Books API request timed out")
|
||||
return None
|
||||
except requests.HTTPError as e:
|
||||
if e.response is not None:
|
||||
if e.response.status_code == 403:
|
||||
# Quota exceeded or invalid API key
|
||||
logger.error(
|
||||
"Google Books API: quota exceeded or invalid API key (HTTP 403)"
|
||||
)
|
||||
elif e.response.status_code == 400:
|
||||
logger.warning(f"Google Books API: bad request - {e}")
|
||||
elif e.response.status_code == 404:
|
||||
logger.debug("Google Books: volume not found")
|
||||
else:
|
||||
logger.error(f"Google Books API HTTP error: {e}")
|
||||
else:
|
||||
logger.error(f"Google Books API HTTP error: {e}")
|
||||
return None
|
||||
except Exception as e:
|
||||
logger.error(f"Google Books API request failed: {e}")
|
||||
return None
|
||||
|
||||
def _parse_volume(self, volume: Dict[str, Any]) -> Optional[BookMetadata]:
|
||||
"""Parse a volume object into BookMetadata."""
|
||||
try:
|
||||
volume_id = volume.get("id")
|
||||
volume_info = volume.get("volumeInfo", {})
|
||||
|
||||
title = volume_info.get("title")
|
||||
if not volume_id or not title:
|
||||
return None
|
||||
|
||||
# Authors (list)
|
||||
authors = volume_info.get("authors", [])
|
||||
|
||||
# ISBNs - extract from industryIdentifiers
|
||||
isbn_10 = None
|
||||
isbn_13 = None
|
||||
for identifier in volume_info.get("industryIdentifiers", []):
|
||||
id_type = identifier.get("type", "")
|
||||
id_value = identifier.get("identifier", "")
|
||||
if id_type == "ISBN_10" and not isbn_10:
|
||||
isbn_10 = id_value
|
||||
elif id_type == "ISBN_13" and not isbn_13:
|
||||
isbn_13 = id_value
|
||||
|
||||
# Cover URL - prefer larger images
|
||||
image_links = volume_info.get("imageLinks", {})
|
||||
cover_url = (
|
||||
image_links.get("large")
|
||||
or image_links.get("medium")
|
||||
or image_links.get("small")
|
||||
or image_links.get("thumbnail")
|
||||
or image_links.get("smallThumbnail")
|
||||
)
|
||||
# Remove edge=curl parameter and upgrade to https
|
||||
if cover_url:
|
||||
cover_url = cover_url.replace("&edge=curl", "").replace(
|
||||
"http://", "https://"
|
||||
)
|
||||
|
||||
# Publisher
|
||||
publisher = volume_info.get("publisher")
|
||||
|
||||
# Publish year - extract from publishedDate (YYYY-MM-DD or YYYY)
|
||||
publish_year = None
|
||||
published_date = volume_info.get("publishedDate", "")
|
||||
if published_date:
|
||||
try:
|
||||
publish_year = int(published_date[:4])
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
|
||||
# Language
|
||||
language = volume_info.get("language")
|
||||
|
||||
# Genres/categories (limit to 5)
|
||||
genres = volume_info.get("categories", [])[:5]
|
||||
|
||||
# Description (may contain HTML - leave as-is for UI to sanitize)
|
||||
description = volume_info.get("description")
|
||||
|
||||
# Source URL
|
||||
source_url = volume_info.get("infoLink")
|
||||
|
||||
# Build display fields - rating only
|
||||
display_fields: List[DisplayField] = []
|
||||
|
||||
average_rating = volume_info.get("averageRating")
|
||||
ratings_count = volume_info.get("ratingsCount")
|
||||
if average_rating is not None:
|
||||
rating_str = f"{average_rating:.1f}"
|
||||
if ratings_count:
|
||||
rating_str += f" ({ratings_count:,})"
|
||||
display_fields.append(
|
||||
DisplayField(label="Rating", value=rating_str, icon="star")
|
||||
)
|
||||
|
||||
return BookMetadata(
|
||||
provider="googlebooks",
|
||||
provider_id=volume_id,
|
||||
title=title,
|
||||
provider_display_name="Google Books",
|
||||
authors=authors,
|
||||
isbn_10=isbn_10,
|
||||
isbn_13=isbn_13,
|
||||
cover_url=cover_url,
|
||||
description=description,
|
||||
publisher=publisher,
|
||||
publish_year=publish_year,
|
||||
language=language,
|
||||
genres=genres,
|
||||
source_url=source_url,
|
||||
display_fields=display_fields,
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
logger.debug(f"Failed to parse Google Books volume: {e}")
|
||||
return None
|
||||
|
||||
|
||||
def _test_googlebooks_connection(current_values: Dict[str, Any] = None) -> Dict[str, Any]:
|
||||
"""Test the Google Books API connection using current form values."""
|
||||
current_values = current_values or {}
|
||||
|
||||
# Use current form values first, fall back to saved config
|
||||
api_key = current_values.get("GOOGLEBOOKS_API_KEY") or app_config.get("GOOGLEBOOKS_API_KEY", "")
|
||||
|
||||
if not api_key:
|
||||
return {
|
||||
"success": False,
|
||||
"message": "API key is required",
|
||||
}
|
||||
|
||||
try:
|
||||
provider = GoogleBooksProvider(api_key=api_key)
|
||||
# Simple test search
|
||||
result = provider._make_request("/volumes", {"q": "test", "maxResults": 1})
|
||||
|
||||
if result is not None and "items" in result:
|
||||
return {
|
||||
"success": True,
|
||||
"message": "Successfully connected to Google Books API",
|
||||
}
|
||||
elif result is not None:
|
||||
return {
|
||||
"success": True,
|
||||
"message": "API connected but returned no results for test query",
|
||||
}
|
||||
else:
|
||||
return {
|
||||
"success": False,
|
||||
"message": "API request failed - check your API key",
|
||||
}
|
||||
except Exception as e:
|
||||
logger.exception("Google Books connection test failed")
|
||||
return {"success": False, "message": f"Connection failed: {str(e)}"}
|
||||
|
||||
|
||||
# Sort options for settings UI
|
||||
_GOOGLEBOOKS_SORT_OPTIONS = [
|
||||
{"value": "relevance", "label": "Most relevant"},
|
||||
{"value": "newest", "label": "Newest"},
|
||||
]
|
||||
|
||||
|
||||
@register_settings(
|
||||
"googlebooks", "Google Books", icon="book", order=53, group="metadata_providers"
|
||||
)
|
||||
def googlebooks_settings():
|
||||
"""Google Books metadata provider settings."""
|
||||
return [
|
||||
HeadingField(
|
||||
key="googlebooks_heading",
|
||||
title="Google Books",
|
||||
description=(
|
||||
"Access Google's comprehensive book database. "
|
||||
"Requires a free API key with ~1000 requests/day quota."
|
||||
),
|
||||
link_url="https://console.cloud.google.com/apis/library/books.googleapis.com",
|
||||
link_text="Get API Key",
|
||||
),
|
||||
CheckboxField(
|
||||
key="GOOGLEBOOKS_ENABLED",
|
||||
label="Enable Google Books",
|
||||
description="Enable Google Books as a metadata provider for book searches",
|
||||
default=False,
|
||||
),
|
||||
PasswordField(
|
||||
key="GOOGLEBOOKS_API_KEY",
|
||||
label="API Key",
|
||||
description=(
|
||||
"Get your API key from Google Cloud Console "
|
||||
"(APIs & Services > Credentials)"
|
||||
),
|
||||
required=True,
|
||||
),
|
||||
ActionButton(
|
||||
key="test_connection",
|
||||
label="Test Connection",
|
||||
description="Verify your API key works",
|
||||
style="primary",
|
||||
callback=_test_googlebooks_connection,
|
||||
),
|
||||
SelectField(
|
||||
key="GOOGLEBOOKS_DEFAULT_SORT",
|
||||
label="Default Sort Order",
|
||||
description="Default sort order for Google Books search results.",
|
||||
options=_GOOGLEBOOKS_SORT_OPTIONS,
|
||||
default="relevance",
|
||||
),
|
||||
]
|
||||
@@ -0,0 +1,839 @@
|
||||
"""Hardcover.app metadata provider. Requires API key."""
|
||||
|
||||
import requests
|
||||
from datetime import datetime
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from shelfmark.core.cache import cacheable
|
||||
from shelfmark.core.logger import setup_logger
|
||||
from shelfmark.core.settings_registry import (
|
||||
register_settings,
|
||||
CheckboxField,
|
||||
PasswordField,
|
||||
SelectField,
|
||||
ActionButton,
|
||||
HeadingField,
|
||||
)
|
||||
from shelfmark.core.config import config as app_config
|
||||
from shelfmark.metadata_providers import (
|
||||
BookMetadata,
|
||||
DisplayField,
|
||||
MetadataProvider,
|
||||
MetadataSearchOptions,
|
||||
SearchResult,
|
||||
SearchType,
|
||||
SortOrder,
|
||||
register_provider,
|
||||
register_provider_kwargs,
|
||||
TextSearchField,
|
||||
)
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
HARDCOVER_API_URL = "https://api.hardcover.app/v1/graphql"
|
||||
HARDCOVER_PAGE_SIZE = 25 # Hardcover API returns max 25 results per page
|
||||
|
||||
|
||||
# Mapping from abstract sort order to Hardcover sort parameter
|
||||
# Note: release_year is more consistently populated than release_date_i
|
||||
SORT_MAPPING: Dict[SortOrder, str] = {
|
||||
SortOrder.RELEVANCE: "_text_match:desc,users_count:desc",
|
||||
SortOrder.POPULARITY: "users_count:desc",
|
||||
SortOrder.RATING: "rating:desc",
|
||||
SortOrder.NEWEST: "release_year:desc",
|
||||
SortOrder.OLDEST: "release_year:asc",
|
||||
}
|
||||
|
||||
# Mapping from abstract search type to Hardcover fields parameter
|
||||
SEARCH_TYPE_FIELDS: Dict[SearchType, str] = {
|
||||
SearchType.GENERAL: "title,isbns,series_names,author_names,alternative_titles",
|
||||
SearchType.TITLE: "title,alternative_titles",
|
||||
SearchType.AUTHOR: "author_names",
|
||||
# ISBN is handled separately via search_by_isbn()
|
||||
}
|
||||
|
||||
|
||||
def _combine_headline_description(headline: Optional[str], description: Optional[str]) -> Optional[str]:
|
||||
"""Combine headline (tagline) and description into a single description."""
|
||||
if headline and description:
|
||||
return f"{headline}\n\n{description}"
|
||||
return headline or description
|
||||
|
||||
|
||||
def _extract_cover_url(data: Dict, *keys: str) -> Optional[str]:
|
||||
"""Extract cover URL from data dict, trying multiple keys.
|
||||
|
||||
Handles both string URLs and dict with 'url' key.
|
||||
"""
|
||||
for key in keys:
|
||||
value = data.get(key)
|
||||
if value:
|
||||
if isinstance(value, str):
|
||||
return value
|
||||
if isinstance(value, dict):
|
||||
return value.get("url")
|
||||
return None
|
||||
|
||||
|
||||
def _extract_publish_year(data: Dict) -> Optional[int]:
|
||||
"""Extract publish year from release_year or release_date fields."""
|
||||
if data.get("release_year"):
|
||||
try:
|
||||
return int(data["release_year"])
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
if data.get("release_date"):
|
||||
try:
|
||||
return int(str(data["release_date"])[:4])
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
return None
|
||||
|
||||
|
||||
def _build_source_url(slug: str) -> Optional[str]:
|
||||
"""Build Hardcover source URL from book slug."""
|
||||
return f"https://hardcover.app/books/{slug}" if slug else None
|
||||
|
||||
|
||||
@register_provider_kwargs("hardcover")
|
||||
def _hardcover_kwargs() -> Dict[str, Any]:
|
||||
"""Provide Hardcover-specific constructor kwargs."""
|
||||
return {"api_key": app_config.get("HARDCOVER_API_KEY", "")}
|
||||
|
||||
|
||||
@register_provider("hardcover")
|
||||
class HardcoverProvider(MetadataProvider):
|
||||
"""Hardcover.app metadata provider using GraphQL API."""
|
||||
|
||||
name = "hardcover"
|
||||
display_name = "Hardcover"
|
||||
requires_auth = True
|
||||
supported_sorts = [
|
||||
SortOrder.RELEVANCE,
|
||||
SortOrder.POPULARITY,
|
||||
SortOrder.RATING,
|
||||
SortOrder.NEWEST,
|
||||
SortOrder.OLDEST,
|
||||
SortOrder.SERIES_ORDER,
|
||||
]
|
||||
search_fields = [
|
||||
TextSearchField(
|
||||
key="author",
|
||||
label="Author",
|
||||
description="Search by author name",
|
||||
),
|
||||
TextSearchField(
|
||||
key="title",
|
||||
label="Title",
|
||||
description="Search by book title",
|
||||
),
|
||||
TextSearchField(
|
||||
key="series",
|
||||
label="Series",
|
||||
description="Search by series name",
|
||||
),
|
||||
]
|
||||
|
||||
def __init__(self, api_key: Optional[str] = None):
|
||||
"""Initialize provider with optional API key (falls back to config)."""
|
||||
raw_key = api_key or app_config.get("HARDCOVER_API_KEY", "")
|
||||
# Strip "Bearer " prefix if user pasted the full auth header from Hardcover
|
||||
self.api_key = raw_key.removeprefix("Bearer ").strip() if raw_key else ""
|
||||
self.session = requests.Session()
|
||||
if self.api_key:
|
||||
self.session.headers.update({
|
||||
"Authorization": f"Bearer {self.api_key}",
|
||||
"Content-Type": "application/json",
|
||||
})
|
||||
|
||||
def is_available(self) -> bool:
|
||||
"""Check if provider is configured with an API key."""
|
||||
return bool(self.api_key)
|
||||
|
||||
def _build_search_params(
|
||||
self, default_query: str, author: str, title: str, series: str
|
||||
) -> tuple[str, Optional[str], Optional[str]]:
|
||||
"""Build search query, fields, and weights based on provided values.
|
||||
|
||||
Returns (query, fields, weights) tuple. Fields/weights are None for general search.
|
||||
"""
|
||||
if series and not author and not title:
|
||||
return series, "series_names", "1"
|
||||
if author and not title and not series:
|
||||
return author, "author_names", "1"
|
||||
if title and not author and not series:
|
||||
return title, "title,alternative_titles", "5,1"
|
||||
if author and title and not series:
|
||||
return f"{title} {author}", "title,alternative_titles,author_names", "5,1,3"
|
||||
if series:
|
||||
query = " ".join(p for p in [series, title, author] if p)
|
||||
return query, "series_names,title,alternative_titles,author_names", "5,3,1,2"
|
||||
return default_query, None, None
|
||||
|
||||
def search(self, options: MetadataSearchOptions) -> List[BookMetadata]:
|
||||
"""Search for books using Hardcover's search API."""
|
||||
return self.search_paginated(options).books
|
||||
|
||||
def search_paginated(self, options: MetadataSearchOptions) -> SearchResult:
|
||||
"""Search for books with pagination info."""
|
||||
if not self.api_key:
|
||||
logger.warning("Hardcover API key not configured")
|
||||
return SearchResult(books=[], page=options.page, total_found=0, has_more=False)
|
||||
|
||||
# Handle ISBN search separately
|
||||
if options.search_type == SearchType.ISBN:
|
||||
result = self.search_by_isbn(options.query)
|
||||
books = [result] if result else []
|
||||
return SearchResult(books=books, page=1, total_found=len(books), has_more=False)
|
||||
|
||||
# Build cache key from options (include fields and settings for cache differentiation)
|
||||
fields_key = ":".join(f"{k}={v}" for k, v in sorted(options.fields.items()))
|
||||
exclude_compilations = app_config.get("HARDCOVER_EXCLUDE_COMPILATIONS", False)
|
||||
exclude_unreleased = app_config.get("HARDCOVER_EXCLUDE_UNRELEASED", False)
|
||||
cache_key = f"{options.query}:{options.search_type.value}:{options.sort.value}:{options.limit}:{options.page}:{fields_key}:excl_comp={exclude_compilations}:excl_unrel={exclude_unreleased}"
|
||||
return self._search_cached(cache_key, options)
|
||||
|
||||
@cacheable(ttl_key="METADATA_CACHE_SEARCH_TTL", ttl_default=300, key_prefix="hardcover:search")
|
||||
def _search_cached(self, cache_key: str, options: MetadataSearchOptions) -> SearchResult:
|
||||
"""Cached search implementation."""
|
||||
# Determine query and fields based on custom search fields
|
||||
# Note: Hardcover API requires 'weights' when using 'fields' parameter
|
||||
author_value = options.fields.get("author", "").strip()
|
||||
title_value = options.fields.get("title", "").strip()
|
||||
series_value = options.fields.get("series", "").strip()
|
||||
|
||||
# Build query and field configuration based on which fields are provided
|
||||
query, search_fields, search_weights = self._build_search_params(
|
||||
options.query, author_value, title_value, series_value
|
||||
)
|
||||
|
||||
# Build GraphQL query - include fields/weights parameters only when needed
|
||||
if search_fields:
|
||||
graphql_query = """
|
||||
query SearchBooks($query: String!, $limit: Int!, $page: Int!, $sort: String, $fields: String, $weights: String) {
|
||||
search(query: $query, query_type: "Book", per_page: $limit, page: $page, sort: $sort, fields: $fields, weights: $weights) {
|
||||
results
|
||||
}
|
||||
}
|
||||
"""
|
||||
else:
|
||||
graphql_query = """
|
||||
query SearchBooks($query: String!, $limit: Int!, $page: Int!, $sort: String) {
|
||||
search(query: $query, query_type: "Book", per_page: $limit, page: $page, sort: $sort) {
|
||||
results
|
||||
}
|
||||
}
|
||||
"""
|
||||
|
||||
# Map abstract sort order to Hardcover's sort parameter
|
||||
sort_param = SORT_MAPPING.get(options.sort, SORT_MAPPING[SortOrder.RELEVANCE])
|
||||
|
||||
variables = {
|
||||
"query": query,
|
||||
"limit": options.limit,
|
||||
"page": options.page,
|
||||
"sort": sort_param,
|
||||
}
|
||||
|
||||
if search_fields:
|
||||
variables["fields"] = search_fields
|
||||
variables["weights"] = search_weights
|
||||
|
||||
try:
|
||||
result = self._execute_query(graphql_query, variables)
|
||||
if not result:
|
||||
logger.debug("Hardcover search: No result from API")
|
||||
return SearchResult(books=[], page=options.page, total_found=0, has_more=False)
|
||||
|
||||
# Extract hits from Typesense response
|
||||
results_obj = result.get("search", {}).get("results", {})
|
||||
if isinstance(results_obj, dict):
|
||||
hits = results_obj.get("hits", [])
|
||||
found_count = results_obj.get("found", 0)
|
||||
else:
|
||||
hits = results_obj if isinstance(results_obj, list) else []
|
||||
found_count = 0
|
||||
|
||||
# Parse hits, filtering compilations and unreleased books if enabled
|
||||
exclude_compilations = app_config.get("HARDCOVER_EXCLUDE_COMPILATIONS", False)
|
||||
exclude_unreleased = app_config.get("HARDCOVER_EXCLUDE_UNRELEASED", False)
|
||||
current_year = datetime.now().year
|
||||
books = []
|
||||
for hit in hits:
|
||||
item = hit.get("document", hit) if isinstance(hit, dict) else hit
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
if exclude_compilations and item.get("compilation"):
|
||||
continue
|
||||
if exclude_unreleased:
|
||||
release_year = item.get("release_year")
|
||||
if release_year is not None and release_year > current_year:
|
||||
continue
|
||||
book = self._parse_search_result(item)
|
||||
if book:
|
||||
books.append(book)
|
||||
|
||||
# If series order sort is selected and series field is provided,
|
||||
# filter to exact matches and sort by position
|
||||
if options.sort == SortOrder.SERIES_ORDER and series_value and books:
|
||||
books = self._apply_series_ordering(books, series_value)
|
||||
|
||||
logger.info(f"Hardcover search '{query}' (fields={search_fields}) returned {len(books)} results")
|
||||
|
||||
# Calculate if there are more results
|
||||
results_so_far = (options.page - 1) * HARDCOVER_PAGE_SIZE + len(hits)
|
||||
has_more = results_so_far < found_count
|
||||
|
||||
return SearchResult(
|
||||
books=books,
|
||||
page=options.page,
|
||||
total_found=found_count,
|
||||
has_more=has_more
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Hardcover search error: {e}")
|
||||
return SearchResult(books=[], page=options.page, total_found=0, has_more=False)
|
||||
|
||||
def _apply_series_ordering(self, books: List[BookMetadata], series_name: str) -> List[BookMetadata]:
|
||||
"""Filter books to exact series match and sort by series position."""
|
||||
series_name_lower = series_name.lower()
|
||||
books_with_position = []
|
||||
|
||||
for book in books:
|
||||
# Fetch full book details to get series info
|
||||
full_book = self.get_book(book.provider_id)
|
||||
if not full_book or not full_book.series_name:
|
||||
continue
|
||||
|
||||
# Exact match on series name
|
||||
if full_book.series_name.lower() != series_name_lower:
|
||||
continue
|
||||
|
||||
# Merge series info into the search result book
|
||||
book.series_name = full_book.series_name
|
||||
book.series_position = full_book.series_position
|
||||
book.series_count = full_book.series_count
|
||||
# Also grab description if search didn't have it
|
||||
if not book.description and full_book.description:
|
||||
book.description = full_book.description
|
||||
books_with_position.append(book)
|
||||
|
||||
# Sort by series position (books without position go last)
|
||||
books_with_position.sort(key=lambda b: (b.series_position is None, b.series_position or 0))
|
||||
|
||||
logger.debug(f"Series ordering: filtered {len(books)} -> {len(books_with_position)} books for '{series_name}'")
|
||||
return books_with_position
|
||||
|
||||
@cacheable(ttl_key="METADATA_CACHE_BOOK_TTL", ttl_default=600, key_prefix="hardcover:book")
|
||||
def get_book(self, book_id: str) -> Optional[BookMetadata]:
|
||||
"""Get book details by Hardcover ID."""
|
||||
if not self.api_key:
|
||||
logger.warning("Hardcover API key not configured")
|
||||
return None
|
||||
|
||||
# Query for specific book by ID
|
||||
# Use contributions with filter to get only primary authors (not translators/narrators)
|
||||
# Also include cached_contributors as fallback if contributions is empty
|
||||
# Include featured_book_series for series info
|
||||
# Include editions with titles and languages for localized search support
|
||||
graphql_query = """
|
||||
query GetBook($id: Int!) {
|
||||
books(where: {id: {_eq: $id}}, limit: 1) {
|
||||
id
|
||||
title
|
||||
subtitle
|
||||
slug
|
||||
release_date
|
||||
headline
|
||||
description
|
||||
pages
|
||||
cached_image
|
||||
cached_tags
|
||||
cached_contributors
|
||||
contributions(where: {contribution: {_eq: "Author"}}) {
|
||||
author {
|
||||
name
|
||||
}
|
||||
}
|
||||
default_physical_edition {
|
||||
isbn_10
|
||||
isbn_13
|
||||
}
|
||||
featured_book_series {
|
||||
position
|
||||
series {
|
||||
name
|
||||
primary_books_count
|
||||
}
|
||||
}
|
||||
editions(limit: 20, order_by: {users_count: desc}) {
|
||||
title
|
||||
language {
|
||||
language
|
||||
code2
|
||||
code3
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
"""
|
||||
|
||||
try:
|
||||
book_id_int = int(book_id)
|
||||
result = self._execute_query(graphql_query, {"id": book_id_int})
|
||||
if not result:
|
||||
return None
|
||||
|
||||
books = result.get("books", [])
|
||||
if not books:
|
||||
return None
|
||||
|
||||
return self._parse_book(books[0])
|
||||
|
||||
except ValueError:
|
||||
logger.error(f"Invalid book ID: {book_id}")
|
||||
return None
|
||||
except Exception as e:
|
||||
logger.error(f"Hardcover get_book error: {e}")
|
||||
return None
|
||||
|
||||
@cacheable(ttl_key="METADATA_CACHE_BOOK_TTL", ttl_default=600, key_prefix="hardcover:isbn")
|
||||
def search_by_isbn(self, isbn: str) -> Optional[BookMetadata]:
|
||||
"""Search for a book by ISBN-10 or ISBN-13."""
|
||||
if not self.api_key:
|
||||
logger.warning("Hardcover API key not configured")
|
||||
return None
|
||||
|
||||
# Clean ISBN (remove hyphens)
|
||||
clean_isbn = isbn.replace("-", "").strip()
|
||||
|
||||
# Search for editions with matching ISBN
|
||||
# Use contributions with filter to get only primary authors (not translators/narrators)
|
||||
graphql_query = """
|
||||
query SearchByISBN($isbn: String!) {
|
||||
editions(
|
||||
where: {
|
||||
_or: [
|
||||
{isbn_10: {_eq: $isbn}},
|
||||
{isbn_13: {_eq: $isbn}}
|
||||
]
|
||||
},
|
||||
limit: 1
|
||||
) {
|
||||
isbn_10
|
||||
isbn_13
|
||||
book {
|
||||
id
|
||||
title
|
||||
subtitle
|
||||
slug
|
||||
release_date
|
||||
headline
|
||||
description
|
||||
pages
|
||||
cached_image
|
||||
cached_tags
|
||||
contributions(where: {contribution: {_eq: "Author"}}) {
|
||||
author {
|
||||
name
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
"""
|
||||
|
||||
try:
|
||||
result = self._execute_query(graphql_query, {"isbn": clean_isbn})
|
||||
if not result:
|
||||
return None
|
||||
|
||||
editions = result.get("editions", [])
|
||||
if not editions:
|
||||
logger.debug(f"No Hardcover book found for ISBN: {isbn}")
|
||||
return None
|
||||
|
||||
edition = editions[0]
|
||||
book_data = edition.get("book", {})
|
||||
if not book_data:
|
||||
return None
|
||||
|
||||
# Add ISBN data from edition to book data
|
||||
book_data["isbn_10"] = edition.get("isbn_10")
|
||||
book_data["isbn_13"] = edition.get("isbn_13")
|
||||
|
||||
return self._parse_book(book_data)
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Hardcover ISBN search error: {e}")
|
||||
return None
|
||||
|
||||
def _execute_query(self, query: str, variables: Dict[str, Any]) -> Optional[Dict]:
|
||||
"""Execute a GraphQL query and return data or None on error."""
|
||||
try:
|
||||
response = self.session.post(
|
||||
HARDCOVER_API_URL,
|
||||
json={"query": query, "variables": variables},
|
||||
timeout=15
|
||||
)
|
||||
response.raise_for_status()
|
||||
|
||||
data = response.json()
|
||||
|
||||
if "errors" in data:
|
||||
logger.error(f"GraphQL errors: {data['errors']}")
|
||||
return None
|
||||
|
||||
return data.get("data")
|
||||
|
||||
except requests.Timeout:
|
||||
logger.warning("Hardcover API request timed out")
|
||||
return None
|
||||
except requests.HTTPError as e:
|
||||
if e.response.status_code == 401:
|
||||
logger.error("Hardcover API key is invalid")
|
||||
else:
|
||||
logger.error(f"Hardcover API HTTP error: {e}")
|
||||
return None
|
||||
except Exception as e:
|
||||
logger.error(f"Hardcover API request failed: {e}")
|
||||
return None
|
||||
|
||||
def _parse_search_result(self, item: Dict) -> Optional[BookMetadata]:
|
||||
"""Parse a search result item into BookMetadata."""
|
||||
try:
|
||||
book_id = item.get("id") or item.get("document", {}).get("id")
|
||||
title = item.get("title") or item.get("document", {}).get("title")
|
||||
|
||||
if not book_id or not title:
|
||||
return None
|
||||
|
||||
# Extract authors - use contribution_types to filter author_names if available
|
||||
authors = []
|
||||
|
||||
author_names = item.get("author_names", [])
|
||||
if isinstance(author_names, str):
|
||||
author_names = [author_names]
|
||||
|
||||
contribution_types = item.get("contribution_types", [])
|
||||
|
||||
# If we have parallel arrays, filter to only "Author" contributions
|
||||
if contribution_types and len(contribution_types) == len(author_names):
|
||||
for name, contrib_type in zip(author_names, contribution_types):
|
||||
if contrib_type == "Author":
|
||||
authors.append(name)
|
||||
elif author_names:
|
||||
# No contribution_types or length mismatch - use all names as fallback
|
||||
authors = author_names
|
||||
|
||||
# Normalize whitespace in author names (some API data has multiple spaces)
|
||||
authors = [" ".join(name.split()) for name in authors]
|
||||
|
||||
cover_url = _extract_cover_url(item, "image")
|
||||
publish_year = _extract_publish_year(item)
|
||||
source_url = _build_source_url(item.get("slug", ""))
|
||||
|
||||
# Build display fields from Hardcover-specific data
|
||||
display_fields = []
|
||||
|
||||
# Rating (e.g., "4.5 (3,764)")
|
||||
rating = item.get("rating")
|
||||
ratings_count = item.get("ratings_count")
|
||||
if rating is not None:
|
||||
rating_str = f"{rating:.1f}"
|
||||
if ratings_count:
|
||||
rating_str += f" ({ratings_count:,})"
|
||||
display_fields.append(DisplayField(label="Rating", value=rating_str, icon="star"))
|
||||
|
||||
# Readers (users who have this book)
|
||||
users_count = item.get("users_count")
|
||||
if users_count:
|
||||
display_fields.append(DisplayField(label="Readers", value=f"{users_count:,}", icon="users"))
|
||||
|
||||
# Combine headline and description if both present
|
||||
headline = item.get("headline")
|
||||
description = item.get("description")
|
||||
full_description = _combine_headline_description(headline, description)
|
||||
|
||||
# Extract subtitle if available in search results
|
||||
subtitle = item.get("subtitle")
|
||||
|
||||
return BookMetadata(
|
||||
provider="hardcover",
|
||||
provider_id=str(book_id),
|
||||
title=title,
|
||||
subtitle=subtitle,
|
||||
provider_display_name="Hardcover",
|
||||
authors=authors,
|
||||
cover_url=cover_url,
|
||||
description=full_description,
|
||||
publish_year=publish_year,
|
||||
source_url=source_url,
|
||||
display_fields=display_fields,
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
logger.debug(f"Failed to parse Hardcover search result: {e}")
|
||||
return None
|
||||
|
||||
def _parse_book(self, book: Dict) -> BookMetadata:
|
||||
"""Parse a book object into BookMetadata."""
|
||||
# Extract authors - try contributions first (filtered), fall back to cached_contributors
|
||||
authors = []
|
||||
contributions = book.get("contributions") or []
|
||||
cached_contributors = book.get("cached_contributors") or []
|
||||
|
||||
# Try contributions first (filtered to "Author" role only - cleaner data)
|
||||
for contrib in contributions:
|
||||
author = contrib.get("author", {})
|
||||
if author and author.get("name"):
|
||||
authors.append(author["name"])
|
||||
|
||||
# Fallback to cached_contributors if no authors found
|
||||
if not authors:
|
||||
for contrib in cached_contributors:
|
||||
if isinstance(contrib, dict):
|
||||
# Handle nested structure: {"author": {"name": "..."}, "contribution": ...}
|
||||
if contrib.get("author", {}).get("name"):
|
||||
authors.append(contrib["author"]["name"])
|
||||
# Handle flat structure: {"name": "..."}
|
||||
elif contrib.get("name"):
|
||||
authors.append(contrib["name"])
|
||||
elif isinstance(contrib, str):
|
||||
authors.append(contrib)
|
||||
|
||||
# Normalize whitespace in author names (some API data has multiple spaces)
|
||||
authors = [" ".join(name.split()) for name in authors]
|
||||
|
||||
cover_url = _extract_cover_url(book, "cached_image", "image")
|
||||
publish_year = _extract_publish_year(book)
|
||||
|
||||
# Extract genres from cached_tags
|
||||
genres = []
|
||||
for tag in book.get("cached_tags", []):
|
||||
if isinstance(tag, dict) and tag.get("tag"):
|
||||
genres.append(tag["tag"])
|
||||
elif isinstance(tag, str):
|
||||
genres.append(tag)
|
||||
|
||||
# Get ISBN from direct fields, default_physical_edition, or editions
|
||||
isbn_10 = book.get("isbn_10")
|
||||
isbn_13 = book.get("isbn_13")
|
||||
|
||||
if not isbn_10 and not isbn_13:
|
||||
# Try default_physical_edition first
|
||||
edition = book.get("default_physical_edition")
|
||||
if edition:
|
||||
isbn_10 = edition.get("isbn_10")
|
||||
isbn_13 = edition.get("isbn_13")
|
||||
|
||||
# Fallback to editions array
|
||||
if not isbn_10 and not isbn_13 and book.get("editions"):
|
||||
for ed in book["editions"]:
|
||||
if not isbn_10 and ed.get("isbn_10"):
|
||||
isbn_10 = ed["isbn_10"]
|
||||
if not isbn_13 and ed.get("isbn_13"):
|
||||
isbn_13 = ed["isbn_13"]
|
||||
if isbn_10 and isbn_13:
|
||||
break
|
||||
|
||||
source_url = _build_source_url(book.get("slug", ""))
|
||||
|
||||
# Combine headline and description if both present
|
||||
headline = book.get("headline")
|
||||
description = book.get("description")
|
||||
full_description = _combine_headline_description(headline, description)
|
||||
|
||||
# Extract series info from featured_book_series
|
||||
series_name = None
|
||||
series_position = None
|
||||
series_count = None
|
||||
featured_series = book.get("featured_book_series")
|
||||
if featured_series:
|
||||
series_position = featured_series.get("position")
|
||||
series_data = featured_series.get("series")
|
||||
if series_data:
|
||||
series_name = series_data.get("name")
|
||||
series_count = series_data.get("primary_books_count")
|
||||
|
||||
# Extract titles by language from editions
|
||||
# This allows searching with localized titles when language filter is active
|
||||
titles_by_language: Dict[str, str] = {}
|
||||
editions = book.get("editions", [])
|
||||
for edition in editions:
|
||||
edition_title = edition.get("title")
|
||||
lang_data = edition.get("language")
|
||||
if edition_title and lang_data:
|
||||
# Store by various language identifiers for flexible matching
|
||||
# Language name (e.g., "German", "English")
|
||||
lang_name = lang_data.get("language")
|
||||
# 2-letter code (e.g., "de", "en")
|
||||
code2 = lang_data.get("code2")
|
||||
# 3-letter code (e.g., "deu", "eng")
|
||||
code3 = lang_data.get("code3")
|
||||
|
||||
# Store with all available keys (first title wins for each language)
|
||||
if lang_name and lang_name not in titles_by_language:
|
||||
titles_by_language[lang_name] = edition_title
|
||||
if code2 and code2 not in titles_by_language:
|
||||
titles_by_language[code2] = edition_title
|
||||
if code3 and code3 not in titles_by_language:
|
||||
titles_by_language[code3] = edition_title
|
||||
|
||||
return BookMetadata(
|
||||
provider="hardcover",
|
||||
provider_id=str(book["id"]),
|
||||
title=book["title"],
|
||||
subtitle=book.get("subtitle"),
|
||||
provider_display_name="Hardcover",
|
||||
authors=authors,
|
||||
isbn_10=isbn_10,
|
||||
isbn_13=isbn_13,
|
||||
cover_url=cover_url,
|
||||
description=full_description,
|
||||
publish_year=publish_year,
|
||||
genres=genres,
|
||||
source_url=source_url,
|
||||
series_name=series_name,
|
||||
series_position=series_position,
|
||||
series_count=series_count,
|
||||
titles_by_language=titles_by_language,
|
||||
)
|
||||
|
||||
|
||||
def _test_hardcover_connection(current_values: Optional[Dict[str, Any]] = None) -> Dict[str, Any]:
|
||||
"""Test the Hardcover API connection using current form values."""
|
||||
from shelfmark.core.config import config as app_config
|
||||
|
||||
current_values = current_values or {}
|
||||
|
||||
# Use current form values first, fall back to saved config
|
||||
raw_key = current_values.get("HARDCOVER_API_KEY") or app_config.get("HARDCOVER_API_KEY", "")
|
||||
# Strip "Bearer " prefix if user pasted the full auth header from Hardcover
|
||||
api_key = raw_key.removeprefix("Bearer ").strip() if raw_key else ""
|
||||
|
||||
key_len = len(api_key) if api_key else 0
|
||||
logger.debug(f"Hardcover test: key length={key_len}")
|
||||
|
||||
if not api_key:
|
||||
# Clear any stored username since there's no key
|
||||
_save_connected_username(None)
|
||||
return {"success": False, "message": "API key is required"}
|
||||
|
||||
if key_len < 100:
|
||||
return {"success": False, "message": f"API key seems too short ({key_len} chars). Expected 500+ chars."}
|
||||
|
||||
try:
|
||||
provider = HardcoverProvider(api_key=api_key)
|
||||
# Use the 'me' query to test connection (recommended by API docs)
|
||||
result = provider._execute_query(
|
||||
"query { me { id, username } }",
|
||||
{}
|
||||
)
|
||||
if result is not None:
|
||||
# Handle both single object and array response formats
|
||||
me_data = result.get("me", {})
|
||||
if isinstance(me_data, list) and me_data:
|
||||
me_data = me_data[0]
|
||||
username = me_data.get("username", "Unknown") if isinstance(me_data, dict) else "Unknown"
|
||||
|
||||
# Save the username for persistent display
|
||||
_save_connected_username(username)
|
||||
|
||||
return {"success": True, "message": f"Connected as: {username}"}
|
||||
else:
|
||||
_save_connected_username(None)
|
||||
return {"success": False, "message": "API request failed - check your API key"}
|
||||
except Exception as e:
|
||||
logger.exception("Hardcover connection test failed")
|
||||
_save_connected_username(None)
|
||||
return {"success": False, "message": f"Connection failed: {str(e)}"}
|
||||
|
||||
|
||||
def _save_connected_username(username: Optional[str]) -> None:
|
||||
"""Save or clear the connected username in config."""
|
||||
from shelfmark.core.settings_registry import save_config_file, load_config_file
|
||||
|
||||
config = load_config_file("hardcover")
|
||||
if username:
|
||||
config["_connected_username"] = username
|
||||
else:
|
||||
config.pop("_connected_username", None)
|
||||
save_config_file("hardcover", config)
|
||||
|
||||
|
||||
def _get_connected_username() -> Optional[str]:
|
||||
"""Get the stored connected username."""
|
||||
from shelfmark.core.settings_registry import load_config_file
|
||||
|
||||
config = load_config_file("hardcover")
|
||||
return config.get("_connected_username")
|
||||
|
||||
|
||||
# Hardcover sort options for settings UI
|
||||
_HARDCOVER_SORT_OPTIONS = [
|
||||
{"value": "relevance", "label": "Most relevant"},
|
||||
{"value": "popularity", "label": "Most popular"},
|
||||
{"value": "rating", "label": "Highest rated"},
|
||||
{"value": "newest", "label": "Newest"},
|
||||
{"value": "oldest", "label": "Oldest"},
|
||||
]
|
||||
|
||||
|
||||
@register_settings("hardcover", "Hardcover", icon="book", order=51, group="metadata_providers")
|
||||
def hardcover_settings():
|
||||
"""Hardcover metadata provider settings."""
|
||||
# Check for connected username to show status
|
||||
connected_user = _get_connected_username()
|
||||
test_button_description = f"Connected as: {connected_user}" if connected_user else "Verify your API key works"
|
||||
|
||||
return [
|
||||
HeadingField(
|
||||
key="hardcover_heading",
|
||||
title="Hardcover",
|
||||
description="A modern book tracking and discovery platform with a comprehensive API.",
|
||||
link_url="https://hardcover.app",
|
||||
link_text="hardcover.app",
|
||||
),
|
||||
CheckboxField(
|
||||
key="HARDCOVER_ENABLED",
|
||||
label="Enable Hardcover",
|
||||
description="Enable Hardcover as a metadata provider for book searches",
|
||||
default=False,
|
||||
),
|
||||
PasswordField(
|
||||
key="HARDCOVER_API_KEY",
|
||||
label="API Key",
|
||||
description="Get your API key from hardcover.app/account/api",
|
||||
required=True,
|
||||
env_supported=False, # UI-only setting, no ENV var support
|
||||
),
|
||||
ActionButton(
|
||||
key="test_connection",
|
||||
label="Test Connection",
|
||||
description=test_button_description,
|
||||
style="primary",
|
||||
callback=_test_hardcover_connection,
|
||||
),
|
||||
SelectField(
|
||||
key="HARDCOVER_DEFAULT_SORT",
|
||||
label="Default Sort Order",
|
||||
description="Default sort order for Hardcover search results.",
|
||||
options=_HARDCOVER_SORT_OPTIONS,
|
||||
default="relevance",
|
||||
env_supported=False, # UI-only setting
|
||||
),
|
||||
CheckboxField(
|
||||
key="HARDCOVER_EXCLUDE_COMPILATIONS",
|
||||
label="Exclude Compilations",
|
||||
description="Filter out compilations, anthologies, and omnibus editions from search results",
|
||||
default=False,
|
||||
),
|
||||
CheckboxField(
|
||||
key="HARDCOVER_EXCLUDE_UNRELEASED",
|
||||
label="Exclude Unreleased Books",
|
||||
description="Filter out books with a release year in the future",
|
||||
default=False,
|
||||
),
|
||||
]
|
||||
@@ -0,0 +1,563 @@
|
||||
"""Open Library metadata provider. No API key required, rate limited."""
|
||||
|
||||
import re
|
||||
import time
|
||||
import threading
|
||||
from collections import deque
|
||||
from typing import Any, Deque, Dict, List, Optional
|
||||
|
||||
import requests
|
||||
|
||||
from shelfmark.core.cache import cacheable
|
||||
from shelfmark.core.logger import setup_logger
|
||||
from shelfmark.core.settings_registry import (
|
||||
register_settings,
|
||||
CheckboxField,
|
||||
SelectField,
|
||||
ActionButton,
|
||||
HeadingField,
|
||||
)
|
||||
from shelfmark.metadata_providers import (
|
||||
BookMetadata,
|
||||
DisplayField,
|
||||
MetadataProvider,
|
||||
MetadataSearchOptions,
|
||||
SearchType,
|
||||
SortOrder,
|
||||
register_provider,
|
||||
TextSearchField,
|
||||
)
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
OPENLIBRARY_BASE_URL = "https://openlibrary.org"
|
||||
COVERS_BASE_URL = "https://covers.openlibrary.org"
|
||||
|
||||
# Rate limiting: Open Library allows ~100 requests per minute
|
||||
# We use a sliding window with 90 requests per 60 seconds for safety margin
|
||||
RATE_LIMIT_REQUESTS = 90
|
||||
RATE_LIMIT_WINDOW_SECONDS = 60
|
||||
|
||||
|
||||
class RateLimiter:
|
||||
"""Simple sliding window rate limiter."""
|
||||
|
||||
def __init__(self, max_requests: int, window_seconds: int):
|
||||
"""Initialize rate limiter with max requests per time window."""
|
||||
self.max_requests = max_requests
|
||||
self.window_seconds = window_seconds
|
||||
self.timestamps: Deque[float] = deque()
|
||||
self.lock = threading.Lock()
|
||||
|
||||
def wait_if_needed(self) -> None:
|
||||
"""Block until a request is allowed (thread-safe)."""
|
||||
wait_time = 0
|
||||
|
||||
# Calculate wait time with lock held
|
||||
with self.lock:
|
||||
now = time.time()
|
||||
cutoff = now - self.window_seconds
|
||||
|
||||
# Remove timestamps outside the window
|
||||
while self.timestamps and self.timestamps[0] < cutoff:
|
||||
self.timestamps.popleft()
|
||||
|
||||
if len(self.timestamps) >= self.max_requests:
|
||||
# Calculate wait time until oldest request falls outside window
|
||||
wait_time = self.timestamps[0] + self.window_seconds - now
|
||||
|
||||
# Sleep outside the lock to avoid blocking other threads
|
||||
if wait_time > 0:
|
||||
logger.debug(f"Rate limited, waiting {wait_time:.2f}s")
|
||||
time.sleep(wait_time)
|
||||
|
||||
# Re-acquire lock and record request
|
||||
with self.lock:
|
||||
# Re-clean timestamps after sleeping
|
||||
now = time.time()
|
||||
cutoff = now - self.window_seconds
|
||||
while self.timestamps and self.timestamps[0] < cutoff:
|
||||
self.timestamps.popleft()
|
||||
|
||||
# Record this request
|
||||
self.timestamps.append(time.time())
|
||||
|
||||
|
||||
# Global rate limiter for Open Library
|
||||
_rate_limiter = RateLimiter(RATE_LIMIT_REQUESTS, RATE_LIMIT_WINDOW_SECONDS)
|
||||
|
||||
|
||||
# Mapping from abstract sort order to Open Library sort parameter
|
||||
# Note: Open Library only supports relevance (default), new, old, random
|
||||
SORT_MAPPING: Dict[str, Optional[str]] = {
|
||||
SortOrder.RELEVANCE: None, # Default (no sort param)
|
||||
SortOrder.NEWEST: "new",
|
||||
SortOrder.OLDEST: "old",
|
||||
# POPULARITY and RATING not supported - will fall back to relevance
|
||||
}
|
||||
|
||||
|
||||
@register_provider("openlibrary")
|
||||
class OpenLibraryProvider(MetadataProvider):
|
||||
"""Open Library metadata provider using REST API."""
|
||||
|
||||
name = "openlibrary"
|
||||
display_name = "Open Library"
|
||||
requires_auth = False
|
||||
supported_sorts = [
|
||||
SortOrder.RELEVANCE,
|
||||
SortOrder.NEWEST,
|
||||
SortOrder.OLDEST,
|
||||
]
|
||||
search_fields = [
|
||||
TextSearchField(
|
||||
key="author",
|
||||
label="Author",
|
||||
description="Search by author name",
|
||||
),
|
||||
TextSearchField(
|
||||
key="title",
|
||||
label="Title",
|
||||
description="Search by book title",
|
||||
),
|
||||
]
|
||||
|
||||
def __init__(self):
|
||||
"""Initialize provider."""
|
||||
self.session = requests.Session()
|
||||
|
||||
def is_available(self) -> bool:
|
||||
"""Open Library is always available (no auth required)."""
|
||||
return True
|
||||
|
||||
def search(self, options: MetadataSearchOptions) -> List[BookMetadata]:
|
||||
"""Search for books using Open Library's search API."""
|
||||
# Handle ISBN search separately
|
||||
if options.search_type == SearchType.ISBN:
|
||||
result = self.search_by_isbn(options.query)
|
||||
return [result] if result else []
|
||||
|
||||
# Build cache key from options (include fields for cache differentiation)
|
||||
fields_key = ":".join(f"{k}={v}" for k, v in sorted(options.fields.items()))
|
||||
cache_key = f"{options.query}:{options.search_type.value}:{options.sort.value}:{options.language}:{options.limit}:{options.page}:{fields_key}"
|
||||
return self._search_cached(cache_key, options)
|
||||
|
||||
@cacheable(ttl_key="METADATA_CACHE_SEARCH_TTL", ttl_default=300, key_prefix="openlibrary:search")
|
||||
def _search_cached(self, cache_key: str, options: MetadataSearchOptions) -> List[BookMetadata]:
|
||||
"""Cached search implementation."""
|
||||
_rate_limiter.wait_if_needed()
|
||||
|
||||
# Build query params
|
||||
params: Dict[str, Any] = {
|
||||
"limit": options.limit,
|
||||
"page": options.page,
|
||||
"fields": "key,title,author_name,first_publish_year,cover_i,isbn,publisher,language,subject,ratings_average,ratings_count",
|
||||
}
|
||||
|
||||
# Field-first search: use custom field values when provided
|
||||
author_value = options.fields.get("author", "").strip()
|
||||
title_value = options.fields.get("title", "").strip()
|
||||
|
||||
if author_value or title_value:
|
||||
# Use field-specific search params (Open Library supports both simultaneously)
|
||||
if author_value:
|
||||
params["author"] = author_value
|
||||
if title_value:
|
||||
params["title"] = title_value
|
||||
# Also add general query if provided (for additional filtering)
|
||||
if options.query.strip():
|
||||
params["q"] = options.query
|
||||
elif options.search_type == SearchType.TITLE:
|
||||
params["title"] = options.query
|
||||
elif options.search_type == SearchType.AUTHOR:
|
||||
params["author"] = options.query
|
||||
else:
|
||||
# General search
|
||||
params["q"] = options.query
|
||||
|
||||
# Add sort if supported (fallback to relevance/default if not)
|
||||
sort = SORT_MAPPING.get(options.sort)
|
||||
if sort:
|
||||
params["sort"] = sort
|
||||
|
||||
# Add language preference if specified
|
||||
if options.language:
|
||||
params["lang"] = options.language
|
||||
|
||||
try:
|
||||
response = self.session.get(
|
||||
f"{OPENLIBRARY_BASE_URL}/search.json",
|
||||
params=params,
|
||||
timeout=15
|
||||
)
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
|
||||
books = []
|
||||
for doc in data.get("docs", []):
|
||||
book = self._parse_search_doc(doc)
|
||||
if book:
|
||||
books.append(book)
|
||||
|
||||
logger.info(f"Open Library search '{options.query}' returned {len(books)} results")
|
||||
return books
|
||||
|
||||
except requests.Timeout:
|
||||
logger.warning("Open Library search timed out")
|
||||
return []
|
||||
except requests.HTTPError as e:
|
||||
if e.response.status_code == 503:
|
||||
logger.warning("Open Library service unavailable (503)")
|
||||
else:
|
||||
logger.error(f"Open Library HTTP error: {e}")
|
||||
return []
|
||||
except Exception as e:
|
||||
logger.error(f"Open Library search error: {e}")
|
||||
return []
|
||||
|
||||
@cacheable(ttl_key="METADATA_CACHE_BOOK_TTL", ttl_default=600, key_prefix="openlibrary:book")
|
||||
def get_book(self, book_id: str) -> Optional[BookMetadata]:
|
||||
"""Get book details by Open Library work ID (e.g., 'OL12345W')."""
|
||||
_rate_limiter.wait_if_needed()
|
||||
|
||||
# Normalize the book_id format
|
||||
if not book_id.startswith("OL"):
|
||||
book_id = f"OL{book_id}"
|
||||
if not book_id.endswith("W"):
|
||||
book_id = f"{book_id}W"
|
||||
|
||||
try:
|
||||
response = self.session.get(
|
||||
f"{OPENLIBRARY_BASE_URL}/works/{book_id}.json",
|
||||
timeout=15
|
||||
)
|
||||
response.raise_for_status()
|
||||
work = response.json()
|
||||
|
||||
return self._parse_work(work, book_id)
|
||||
|
||||
except requests.Timeout:
|
||||
logger.warning("Open Library get_book timed out")
|
||||
return None
|
||||
except requests.HTTPError as e:
|
||||
if e.response.status_code == 404:
|
||||
logger.debug(f"Open Library work not found: {book_id}")
|
||||
else:
|
||||
logger.error(f"Open Library HTTP error: {e}")
|
||||
return None
|
||||
except Exception as e:
|
||||
logger.error(f"Open Library get_book error: {e}")
|
||||
return None
|
||||
|
||||
@cacheable(ttl_key="METADATA_CACHE_BOOK_TTL", ttl_default=600, key_prefix="openlibrary:isbn")
|
||||
def search_by_isbn(self, isbn: str) -> Optional[BookMetadata]:
|
||||
"""Search for a book by ISBN-10 or ISBN-13."""
|
||||
# Clean ISBN
|
||||
clean_isbn = isbn.replace("-", "").strip()
|
||||
|
||||
_rate_limiter.wait_if_needed()
|
||||
|
||||
try:
|
||||
# First try the ISBN API which returns edition data
|
||||
response = self.session.get(
|
||||
f"{OPENLIBRARY_BASE_URL}/isbn/{clean_isbn}.json",
|
||||
timeout=15
|
||||
)
|
||||
response.raise_for_status()
|
||||
edition = response.json()
|
||||
|
||||
# Get the work key for full book info
|
||||
works = edition.get("works", [])
|
||||
if works:
|
||||
work_key = works[0].get("key", "")
|
||||
work_id = work_key.split("/")[-1] if work_key else None
|
||||
|
||||
if work_id:
|
||||
# Fetch full work data
|
||||
book = self.get_book(work_id)
|
||||
if book:
|
||||
# Update with ISBN from edition if not present
|
||||
# Use dataclasses.replace() to avoid mutating cached object
|
||||
from dataclasses import replace
|
||||
updates = {}
|
||||
if not book.isbn_10:
|
||||
isbn_10_list = edition.get("isbn_10", [])
|
||||
if isbn_10_list:
|
||||
updates["isbn_10"] = isbn_10_list[0]
|
||||
if not book.isbn_13:
|
||||
isbn_13_list = edition.get("isbn_13", [])
|
||||
if isbn_13_list:
|
||||
updates["isbn_13"] = isbn_13_list[0]
|
||||
if updates:
|
||||
return replace(book, **updates)
|
||||
return book
|
||||
|
||||
# Fallback: parse edition data directly
|
||||
return self._parse_edition(edition, clean_isbn)
|
||||
|
||||
except requests.HTTPError as e:
|
||||
if e.response.status_code == 404:
|
||||
logger.debug(f"Open Library ISBN not found: {isbn}")
|
||||
else:
|
||||
logger.error(f"Open Library ISBN search HTTP error: {e}")
|
||||
return None
|
||||
except Exception as e:
|
||||
logger.error(f"Open Library ISBN search error: {e}")
|
||||
return None
|
||||
|
||||
def _parse_search_doc(self, doc: dict) -> Optional[BookMetadata]:
|
||||
"""Parse a search document into BookMetadata."""
|
||||
try:
|
||||
# Extract work ID from key
|
||||
key = doc.get("key", "")
|
||||
work_id = key.split("/")[-1] if key else None
|
||||
|
||||
if not work_id or not doc.get("title"):
|
||||
return None
|
||||
|
||||
# Get authors
|
||||
authors = doc.get("author_name", [])
|
||||
if not isinstance(authors, list):
|
||||
authors = [authors] if authors else []
|
||||
|
||||
# Get ISBNs - find first ISBN-10 and ISBN-13
|
||||
isbns = doc.get("isbn", [])
|
||||
isbn_10 = next((i for i in isbns if len(i) == 10), None)
|
||||
isbn_13 = next((i for i in isbns if len(i) == 13), None)
|
||||
|
||||
# Get cover URL
|
||||
cover_id = doc.get("cover_i")
|
||||
cover_url = f"{COVERS_BASE_URL}/b/id/{cover_id}-L.jpg" if cover_id else None
|
||||
|
||||
# Get publishers (take first one)
|
||||
publishers = doc.get("publisher", [])
|
||||
publisher = publishers[0] if publishers else None
|
||||
|
||||
# Get languages (take first one)
|
||||
languages = doc.get("language", [])
|
||||
language = languages[0] if languages else None
|
||||
|
||||
# Get subjects as genres (take first 5)
|
||||
subjects = doc.get("subject", [])
|
||||
genres = subjects[:5] if subjects else []
|
||||
|
||||
# Build display fields from Open Library-specific data
|
||||
display_fields = []
|
||||
|
||||
# Rating (if available - not always present)
|
||||
ratings_avg = doc.get("ratings_average")
|
||||
ratings_count = doc.get("ratings_count")
|
||||
if ratings_avg is not None and ratings_avg > 0:
|
||||
rating_str = f"{ratings_avg:.1f}"
|
||||
if ratings_count:
|
||||
rating_str += f" ({ratings_count:,})"
|
||||
display_fields.append(DisplayField(label="Rating", value=rating_str, icon="star"))
|
||||
|
||||
return BookMetadata(
|
||||
provider="openlibrary",
|
||||
provider_id=work_id,
|
||||
title=doc["title"],
|
||||
provider_display_name="Open Library",
|
||||
authors=authors,
|
||||
isbn_10=isbn_10,
|
||||
isbn_13=isbn_13,
|
||||
cover_url=cover_url,
|
||||
publisher=publisher,
|
||||
publish_year=doc.get("first_publish_year"),
|
||||
language=language,
|
||||
genres=genres,
|
||||
source_url=f"{OPENLIBRARY_BASE_URL}/works/{work_id}",
|
||||
display_fields=display_fields,
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
logger.debug(f"Failed to parse Open Library search doc: {e}")
|
||||
return None
|
||||
|
||||
def _parse_work(self, work: dict, work_id: str) -> Optional[BookMetadata]:
|
||||
"""Parse a work object into BookMetadata."""
|
||||
try:
|
||||
title = work.get("title")
|
||||
if not title:
|
||||
return None
|
||||
|
||||
# Get description
|
||||
description = work.get("description")
|
||||
if isinstance(description, dict):
|
||||
description = description.get("value")
|
||||
|
||||
# Get authors (requires additional API calls)
|
||||
authors = []
|
||||
for author_ref in work.get("authors", []):
|
||||
author_key = None
|
||||
if isinstance(author_ref, dict):
|
||||
author_key = author_ref.get("author", {}).get("key")
|
||||
if author_key:
|
||||
author_name = self._get_author_name(author_key)
|
||||
if author_name:
|
||||
authors.append(author_name)
|
||||
|
||||
# Get cover URL from covers array
|
||||
cover_url = None
|
||||
covers = work.get("covers", [])
|
||||
if covers:
|
||||
cover_id = covers[0]
|
||||
cover_url = f"{COVERS_BASE_URL}/b/id/{cover_id}-L.jpg"
|
||||
|
||||
# Get subjects as genres
|
||||
subjects = work.get("subjects", [])
|
||||
genres = subjects[:5] if subjects else []
|
||||
|
||||
return BookMetadata(
|
||||
provider="openlibrary",
|
||||
provider_id=work_id,
|
||||
title=title,
|
||||
provider_display_name="Open Library",
|
||||
authors=authors,
|
||||
cover_url=cover_url,
|
||||
description=description,
|
||||
genres=genres,
|
||||
source_url=f"{OPENLIBRARY_BASE_URL}/works/{work_id}",
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
logger.debug(f"Failed to parse Open Library work: {e}")
|
||||
return None
|
||||
|
||||
def _parse_edition(self, edition: dict, isbn: str) -> Optional[BookMetadata]:
|
||||
"""Parse an edition object into BookMetadata (fallback for ISBN lookup)."""
|
||||
try:
|
||||
title = edition.get("title")
|
||||
if not title:
|
||||
return None
|
||||
|
||||
# Get the edition key as ID
|
||||
key = edition.get("key", "")
|
||||
edition_id = key.split("/")[-1] if key else isbn
|
||||
|
||||
# Get ISBNs
|
||||
isbn_10_list = edition.get("isbn_10", [])
|
||||
isbn_13_list = edition.get("isbn_13", [])
|
||||
isbn_10 = isbn_10_list[0] if isbn_10_list else None
|
||||
isbn_13 = isbn_13_list[0] if isbn_13_list else None
|
||||
|
||||
# Get publishers
|
||||
publishers = edition.get("publishers", [])
|
||||
publisher = publishers[0] if publishers else None
|
||||
|
||||
# Get cover URL
|
||||
cover_url = None
|
||||
covers = edition.get("covers", [])
|
||||
if covers:
|
||||
cover_id = covers[0]
|
||||
cover_url = f"{COVERS_BASE_URL}/b/id/{cover_id}-L.jpg"
|
||||
|
||||
# Get publish date and try to extract year
|
||||
publish_year = None
|
||||
publish_date = edition.get("publish_date", "")
|
||||
if publish_date:
|
||||
# Try to extract year from various formats
|
||||
year_match = re.search(r'\b(19|20)\d{2}\b', publish_date)
|
||||
if year_match:
|
||||
publish_year = int(year_match.group())
|
||||
|
||||
return BookMetadata(
|
||||
provider="openlibrary",
|
||||
provider_id=edition_id,
|
||||
title=title,
|
||||
provider_display_name="Open Library",
|
||||
isbn_10=isbn_10,
|
||||
isbn_13=isbn_13,
|
||||
cover_url=cover_url,
|
||||
publisher=publisher,
|
||||
publish_year=publish_year,
|
||||
source_url=f"{OPENLIBRARY_BASE_URL}{key}" if key else None,
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
logger.debug(f"Failed to parse Open Library edition: {e}")
|
||||
return None
|
||||
|
||||
def _get_author_name(self, author_key: str) -> Optional[str]:
|
||||
"""Get author name from author key (e.g., '/authors/OL123A')."""
|
||||
_rate_limiter.wait_if_needed()
|
||||
|
||||
try:
|
||||
response = self.session.get(
|
||||
f"{OPENLIBRARY_BASE_URL}{author_key}.json",
|
||||
timeout=10
|
||||
)
|
||||
response.raise_for_status()
|
||||
author = response.json()
|
||||
return author.get("name")
|
||||
|
||||
except Exception:
|
||||
# Don't log errors for author lookups - they're supplementary
|
||||
return None
|
||||
|
||||
|
||||
def _test_openlibrary_connection() -> Dict[str, Any]:
|
||||
"""Test the Open Library API connection."""
|
||||
try:
|
||||
provider = OpenLibraryProvider()
|
||||
# Simple API call to test connectivity
|
||||
response = provider.session.get(
|
||||
f"{OPENLIBRARY_BASE_URL}/search.json",
|
||||
params={"q": "test", "limit": 1},
|
||||
timeout=10
|
||||
)
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
if "docs" in data:
|
||||
return {"success": True, "message": "Successfully connected to Open Library API"}
|
||||
else:
|
||||
return {"success": False, "message": "Unexpected response from API"}
|
||||
except requests.Timeout:
|
||||
return {"success": False, "message": "Connection timed out"}
|
||||
except requests.RequestException as e:
|
||||
return {"success": False, "message": f"Connection failed: {str(e)}"}
|
||||
except Exception as e:
|
||||
return {"success": False, "message": f"Error: {str(e)}"}
|
||||
|
||||
|
||||
# Open Library sort options for settings UI
|
||||
_OPENLIBRARY_SORT_OPTIONS = [
|
||||
{"value": "relevance", "label": "Most relevant"},
|
||||
{"value": "newest", "label": "Newest"},
|
||||
{"value": "oldest", "label": "Oldest"},
|
||||
]
|
||||
|
||||
|
||||
@register_settings("openlibrary", "Open Library", icon="library", order=52, group="metadata_providers")
|
||||
def openlibrary_settings():
|
||||
"""Open Library metadata provider settings."""
|
||||
return [
|
||||
HeadingField(
|
||||
key="openlibrary_heading",
|
||||
title="Open Library",
|
||||
description="An initiative of the Internet Archive. A free, open-source library catalog with millions of books. No API key required.",
|
||||
link_url="https://openlibrary.org",
|
||||
link_text="openlibrary.org",
|
||||
),
|
||||
CheckboxField(
|
||||
key="OPENLIBRARY_ENABLED",
|
||||
label="Enable Open Library",
|
||||
description="Enable Open Library as a metadata provider for book searches",
|
||||
default=False,
|
||||
),
|
||||
ActionButton(
|
||||
key="test_connection",
|
||||
label="Test Connection",
|
||||
description="Verify Open Library API is accessible",
|
||||
style="primary",
|
||||
callback=_test_openlibrary_connection,
|
||||
),
|
||||
SelectField(
|
||||
key="OPENLIBRARY_DEFAULT_SORT",
|
||||
label="Default Sort Order",
|
||||
description="Default sort order for Open Library search results.",
|
||||
options=_OPENLIBRARY_SORT_OPTIONS,
|
||||
default="relevance",
|
||||
env_supported=False, # UI-only setting
|
||||
),
|
||||
]
|
||||
@@ -0,0 +1,338 @@
|
||||
"""Release source plugin system - base classes and registry."""
|
||||
|
||||
from abc import ABC, abstractmethod
|
||||
from dataclasses import dataclass, field, asdict
|
||||
from enum import Enum
|
||||
from threading import Event
|
||||
from typing import List, Optional, Dict, Type, Callable, Literal, Any
|
||||
|
||||
from shelfmark.core.models import DownloadTask
|
||||
from shelfmark.metadata_providers import BookMetadata
|
||||
|
||||
|
||||
class ReleaseProtocol(str, Enum):
|
||||
"""Protocol for downloading a release."""
|
||||
HTTP = "http" # Direct HTTP download
|
||||
TORRENT = "torrent" # BitTorrent
|
||||
NZB = "nzb" # Usenet NZB
|
||||
DCC = "dcc" # IRC DCC
|
||||
|
||||
|
||||
@dataclass
|
||||
class Release:
|
||||
"""A downloadable release - all sources return this same structure."""
|
||||
source: str # "direct", "prowlarr", "irc", etc.
|
||||
source_id: str # ID within that source
|
||||
title: str
|
||||
format: Optional[str] = None
|
||||
language: Optional[str] = None # ISO 639-1 code (e.g., "en", "de", "fr")
|
||||
size: Optional[str] = None
|
||||
size_bytes: Optional[int] = None
|
||||
download_url: Optional[str] = None
|
||||
info_url: Optional[str] = None # Link to release info page (e.g., tracker) - makes title clickable
|
||||
protocol: Optional[ReleaseProtocol] = None
|
||||
indexer: Optional[str] = None # Source name for display
|
||||
seeders: Optional[int] = None # For torrents
|
||||
peers: Optional[str] = None # For torrents: "seeders/leechers" display string
|
||||
content_type: Optional[str] = None # "ebook" or "audiobook" - preserved from search
|
||||
extra: Dict = field(default_factory=dict) # Source-specific metadata
|
||||
|
||||
|
||||
@dataclass
|
||||
class DownloadProgress:
|
||||
"""DEPRECATED: Use progress_callback and status_callback instead."""
|
||||
status: str # "queued", "resolving", "downloading", "complete", "failed"
|
||||
progress: float # 0-100
|
||||
status_message: Optional[str] = None
|
||||
download_speed: Optional[int] = None
|
||||
eta: Optional[int] = None
|
||||
save_path: Optional[str] = None
|
||||
|
||||
|
||||
# --- Column Schema for Plugin-Driven UI ---
|
||||
|
||||
class ColumnRenderType(str, Enum):
|
||||
"""How the frontend should render the column value."""
|
||||
TEXT = "text" # Plain text
|
||||
BADGE = "badge" # Colored badge (format, language)
|
||||
SIZE = "size" # File size formatting
|
||||
NUMBER = "number" # Numeric value
|
||||
PEERS = "peers" # Peers display: "S/L" with color based on seeder count
|
||||
|
||||
|
||||
class ColumnAlign(str, Enum):
|
||||
"""Column alignment options."""
|
||||
LEFT = "left"
|
||||
CENTER = "center"
|
||||
RIGHT = "right"
|
||||
|
||||
|
||||
@dataclass
|
||||
class ColumnColorHint:
|
||||
"""Color hint for badge-type columns."""
|
||||
type: Literal["map", "static"] # "map" uses frontend colorMaps, "static" is fixed class
|
||||
value: str # Map name ("format", "language") or Tailwind class
|
||||
|
||||
|
||||
@dataclass
|
||||
class ColumnSchema:
|
||||
"""Definition for a single column in the release list."""
|
||||
key: str # Data path (e.g., "format", "extra.language")
|
||||
label: str # Accessibility label
|
||||
render_type: ColumnRenderType = ColumnRenderType.TEXT
|
||||
align: ColumnAlign = ColumnAlign.LEFT
|
||||
width: str = "auto" # CSS width (e.g., "80px", "minmax(0,2fr)")
|
||||
hide_mobile: bool = False # Hide on small screens
|
||||
color_hint: Optional[ColumnColorHint] = None # For BADGE render type
|
||||
fallback: str = "-" # Value to show when data is missing
|
||||
uppercase: bool = False # Force uppercase display
|
||||
sortable: bool = False # Show in sort dropdown (opt-in)
|
||||
sort_key: Optional[str] = None # Field to sort by (defaults to `key` if None)
|
||||
|
||||
|
||||
class LeadingCellType(str, Enum):
|
||||
"""Type of leading cell to display in release rows."""
|
||||
THUMBNAIL = "thumbnail" # Show book cover image
|
||||
BADGE = "badge" # Show colored badge (e.g., "Torrent", "Usenet")
|
||||
NONE = "none" # No leading cell
|
||||
|
||||
|
||||
@dataclass
|
||||
class LeadingCellConfig:
|
||||
"""Configuration for the leading cell in release rows."""
|
||||
type: LeadingCellType = LeadingCellType.THUMBNAIL
|
||||
key: Optional[str] = None # Field path for data (e.g., "extra.preview" or "extra.download_type")
|
||||
color_hint: Optional[ColumnColorHint] = None # For badge type - maps values to colors
|
||||
uppercase: bool = False # Force uppercase for badge text
|
||||
|
||||
|
||||
@dataclass
|
||||
class SourceActionButton:
|
||||
"""Action button configuration for a release source."""
|
||||
label: str # Button text (e.g., "Refresh search")
|
||||
action: str = "expand" # Action type: "expand" triggers expand_search
|
||||
|
||||
|
||||
@dataclass
|
||||
class ReleaseColumnConfig:
|
||||
"""Complete column configuration for a release source."""
|
||||
columns: List[ColumnSchema]
|
||||
grid_template: str = "minmax(0,2fr) 60px 80px 80px" # CSS grid-template-columns
|
||||
leading_cell: Optional[LeadingCellConfig] = None # Defaults to thumbnail mode if None
|
||||
online_servers: Optional[List[str]] = None # For IRC: list of currently online server nicks
|
||||
cache_ttl_seconds: Optional[int] = None # How long to cache results (default: 5 min)
|
||||
supported_filters: Optional[List[str]] = None # Which filters this source supports: ["format", "language"]
|
||||
action_button: Optional[SourceActionButton] = None # Custom action button (replaces default expand search)
|
||||
|
||||
|
||||
def serialize_column_config(config: ReleaseColumnConfig) -> Dict[str, Any]:
|
||||
"""Serialize column configuration for API response."""
|
||||
result: Dict[str, Any] = {
|
||||
"columns": [
|
||||
{
|
||||
"key": col.key,
|
||||
"label": col.label,
|
||||
"render_type": col.render_type.value,
|
||||
"align": col.align.value,
|
||||
"width": col.width,
|
||||
"hide_mobile": col.hide_mobile,
|
||||
"color_hint": {
|
||||
"type": col.color_hint.type,
|
||||
"value": col.color_hint.value
|
||||
} if col.color_hint else None,
|
||||
"fallback": col.fallback,
|
||||
"uppercase": col.uppercase,
|
||||
"sortable": col.sortable,
|
||||
"sort_key": col.sort_key,
|
||||
}
|
||||
for col in config.columns
|
||||
],
|
||||
"grid_template": config.grid_template,
|
||||
}
|
||||
|
||||
# Include leading_cell config if specified
|
||||
if config.leading_cell:
|
||||
result["leading_cell"] = {
|
||||
"type": config.leading_cell.type.value,
|
||||
"key": config.leading_cell.key,
|
||||
"color_hint": {
|
||||
"type": config.leading_cell.color_hint.type,
|
||||
"value": config.leading_cell.color_hint.value
|
||||
} if config.leading_cell.color_hint else None,
|
||||
"uppercase": config.leading_cell.uppercase,
|
||||
}
|
||||
|
||||
# Include online_servers if provided (e.g., for IRC source)
|
||||
if config.online_servers is not None:
|
||||
result["online_servers"] = config.online_servers
|
||||
|
||||
# Include cache TTL if specified (sources can request longer caching)
|
||||
if config.cache_ttl_seconds is not None:
|
||||
result["cache_ttl_seconds"] = config.cache_ttl_seconds
|
||||
|
||||
# Include supported filters (sources declare which filters they support)
|
||||
if config.supported_filters is not None:
|
||||
result["supported_filters"] = config.supported_filters
|
||||
|
||||
# Include action button if specified (replaces default expand search)
|
||||
if config.action_button is not None:
|
||||
result["action_button"] = {
|
||||
"label": config.action_button.label,
|
||||
"action": config.action_button.action,
|
||||
}
|
||||
|
||||
return result
|
||||
|
||||
|
||||
def _default_column_config() -> ReleaseColumnConfig:
|
||||
"""Default column configuration used when source doesn't define its own."""
|
||||
return ReleaseColumnConfig(
|
||||
columns=[
|
||||
ColumnSchema(
|
||||
key="extra.language",
|
||||
label="Language",
|
||||
render_type=ColumnRenderType.BADGE,
|
||||
align=ColumnAlign.CENTER,
|
||||
width="60px",
|
||||
hide_mobile=False, # Language shown on mobile
|
||||
color_hint=ColumnColorHint(type="map", value="language"),
|
||||
uppercase=True,
|
||||
),
|
||||
ColumnSchema(
|
||||
key="format",
|
||||
label="Format",
|
||||
render_type=ColumnRenderType.BADGE,
|
||||
align=ColumnAlign.CENTER,
|
||||
width="80px",
|
||||
hide_mobile=False, # Format shown on mobile
|
||||
color_hint=ColumnColorHint(type="map", value="format"),
|
||||
uppercase=True,
|
||||
),
|
||||
ColumnSchema(
|
||||
key="size",
|
||||
label="Size",
|
||||
render_type=ColumnRenderType.SIZE,
|
||||
align=ColumnAlign.CENTER,
|
||||
width="80px",
|
||||
hide_mobile=False, # Size shown on mobile
|
||||
),
|
||||
],
|
||||
grid_template="minmax(0,2fr) 60px 80px 80px",
|
||||
supported_filters=["format", "language"], # Default: both filters available
|
||||
)
|
||||
|
||||
|
||||
class ReleaseSource(ABC):
|
||||
"""Interface for searching a release source."""
|
||||
name: str # "direct", "prowlarr"
|
||||
display_name: str # "Direct Download", "Prowlarr"
|
||||
supported_content_types: List[str] = ["ebook", "audiobook"] # Content types this source supports
|
||||
can_be_default: bool = True # Whether this source can be selected as default in settings
|
||||
|
||||
@abstractmethod
|
||||
def search(
|
||||
self,
|
||||
book: BookMetadata,
|
||||
expand_search: bool = False,
|
||||
languages: Optional[List[str]] = None,
|
||||
content_type: str = "ebook"
|
||||
) -> List[Release]:
|
||||
"""Search for releases of a book."""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def is_available(self) -> bool:
|
||||
"""Check if this source is configured and reachable."""
|
||||
pass
|
||||
|
||||
@classmethod
|
||||
def get_column_config(cls) -> ReleaseColumnConfig:
|
||||
"""Get column configuration for release list UI. Override for custom columns."""
|
||||
return _default_column_config()
|
||||
|
||||
|
||||
class DownloadHandler(ABC):
|
||||
"""Interface for executing downloads. Handlers stage files to TMP_DIR;
|
||||
orchestrator handles post-processing and move to INGEST_DIR.
|
||||
"""
|
||||
|
||||
@abstractmethod
|
||||
def download(
|
||||
self,
|
||||
task: DownloadTask,
|
||||
cancel_flag: Event,
|
||||
progress_callback: Callable[[float], None],
|
||||
status_callback: Callable[[str, Optional[str]], None]
|
||||
) -> Optional[str]:
|
||||
"""Execute download and return path to staged file in TMP_DIR."""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def cancel(self, task_id: str) -> bool:
|
||||
"""Cancel an in-progress download."""
|
||||
pass
|
||||
|
||||
|
||||
# --- Registry ---
|
||||
|
||||
_SOURCES: Dict[str, Type[ReleaseSource]] = {}
|
||||
_HANDLERS: Dict[str, Type[DownloadHandler]] = {}
|
||||
|
||||
|
||||
def register_source(name: str):
|
||||
"""Decorator to register a release source."""
|
||||
def decorator(cls):
|
||||
_SOURCES[name] = cls
|
||||
return cls
|
||||
return decorator
|
||||
|
||||
|
||||
def register_handler(name: str):
|
||||
"""Decorator to register a download handler."""
|
||||
def decorator(cls):
|
||||
_HANDLERS[name] = cls
|
||||
return cls
|
||||
return decorator
|
||||
|
||||
|
||||
def get_source(name: str) -> ReleaseSource:
|
||||
"""Get a release source instance by name."""
|
||||
if name not in _SOURCES:
|
||||
raise ValueError(f"Unknown release source: {name}")
|
||||
return _SOURCES[name]()
|
||||
|
||||
|
||||
def get_handler(name: str) -> DownloadHandler:
|
||||
"""Get a download handler instance by name."""
|
||||
if name not in _HANDLERS:
|
||||
raise ValueError(f"Unknown download handler: {name}")
|
||||
return _HANDLERS[name]()
|
||||
|
||||
|
||||
def list_available_sources() -> List[dict]:
|
||||
"""List all registered sources with their availability status."""
|
||||
result = []
|
||||
for name, src_class in _SOURCES.items():
|
||||
instance = src_class()
|
||||
result.append({
|
||||
"name": name,
|
||||
"display_name": instance.display_name,
|
||||
"enabled": instance.is_available(),
|
||||
"supported_content_types": getattr(instance, 'supported_content_types', ["ebook", "audiobook"]),
|
||||
"can_be_default": getattr(instance, 'can_be_default', True),
|
||||
})
|
||||
return result
|
||||
|
||||
|
||||
def get_source_display_name(name: str) -> str:
|
||||
"""Get display name for a source by its identifier."""
|
||||
if name in _SOURCES:
|
||||
return _SOURCES[name]().display_name
|
||||
return name.replace('_', ' ').title()
|
||||
|
||||
|
||||
# Import source implementations to trigger registration
|
||||
# These must be imported AFTER the base classes and registry are defined
|
||||
from shelfmark.release_sources import direct_download # noqa: F401, E402
|
||||
from shelfmark.release_sources import prowlarr # noqa: F401, E402
|
||||
from shelfmark.release_sources import irc # noqa: F401, E402
|
||||
@@ -0,0 +1,11 @@
|
||||
"""IRC release source plugin.
|
||||
|
||||
Searches and downloads ebooks from IRC channels via DCC protocol.
|
||||
Available when IRC server, channel, and nickname are configured in settings.
|
||||
|
||||
Based on OpenBooks (https://github.com/evan-buss/openbooks).
|
||||
"""
|
||||
|
||||
from shelfmark.release_sources.irc import source # noqa: F401
|
||||
from shelfmark.release_sources.irc import handler # noqa: F401
|
||||
from shelfmark.release_sources.irc import settings # noqa: F401
|
||||
@@ -0,0 +1,289 @@
|
||||
"""Persistent file-based cache for IRC search results.
|
||||
|
||||
Stores search results in CONFIG_DIR to survive container restarts.
|
||||
IRC searches are slow and resource-intensive, so we cache aggressively.
|
||||
"""
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import time
|
||||
from dataclasses import asdict
|
||||
from pathlib import Path
|
||||
from threading import Lock
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from shelfmark.config import env
|
||||
from shelfmark.core.logger import setup_logger
|
||||
from shelfmark.release_sources import Release, ReleaseProtocol
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
# Cache file location
|
||||
CACHE_FILE = Path(env.CONFIG_DIR) / "irc_cache.json"
|
||||
|
||||
# Default TTL: 30 days (in seconds)
|
||||
DEFAULT_CACHE_TTL = 30 * 24 * 60 * 60
|
||||
|
||||
# Lock for thread-safe file access
|
||||
_cache_lock = Lock()
|
||||
|
||||
|
||||
def _generate_cache_key(provider: str, provider_id: str) -> str:
|
||||
"""Generate a cache key from provider and provider_id."""
|
||||
return f"{provider}:{provider_id}"
|
||||
|
||||
|
||||
def _load_cache() -> Dict[str, Any]:
|
||||
"""Load cache from disk."""
|
||||
try:
|
||||
if CACHE_FILE.exists():
|
||||
return json.loads(CACHE_FILE.read_text())
|
||||
except (json.JSONDecodeError, IOError) as e:
|
||||
logger.warning(f"Failed to load IRC cache: {e}")
|
||||
return {"entries": {}, "version": 1}
|
||||
|
||||
|
||||
def _save_cache(cache: Dict[str, Any]) -> None:
|
||||
"""Save cache to disk."""
|
||||
try:
|
||||
CACHE_FILE.write_text(json.dumps(cache, indent=2))
|
||||
except IOError as e:
|
||||
logger.error(f"Failed to save IRC cache: {e}")
|
||||
|
||||
|
||||
def _release_to_dict(release: Release) -> Dict[str, Any]:
|
||||
"""Convert Release to a JSON-serializable dict."""
|
||||
data = asdict(release)
|
||||
# Convert enum to string
|
||||
if data.get("protocol"):
|
||||
data["protocol"] = data["protocol"].value if hasattr(data["protocol"], "value") else str(data["protocol"])
|
||||
return data
|
||||
|
||||
|
||||
def _dict_to_release(data: Dict[str, Any]) -> Release:
|
||||
"""Convert dict back to Release object."""
|
||||
# Convert protocol string back to enum
|
||||
if data.get("protocol"):
|
||||
try:
|
||||
data["protocol"] = ReleaseProtocol(data["protocol"])
|
||||
except (ValueError, KeyError):
|
||||
data["protocol"] = None
|
||||
return Release(**data)
|
||||
|
||||
|
||||
def get_cached_results(
|
||||
provider: str,
|
||||
provider_id: str,
|
||||
ttl_seconds: Optional[int] = None
|
||||
) -> Optional[Dict[str, Any]]:
|
||||
"""
|
||||
Get cached search results for a book.
|
||||
|
||||
Args:
|
||||
provider: Metadata provider name (e.g., "hardcover", "openlibrary")
|
||||
provider_id: Book ID in the provider's system
|
||||
ttl_seconds: Cache TTL in seconds (from settings)
|
||||
|
||||
Returns:
|
||||
Dict with 'releases' (List[Release]) and 'online_servers' (List[str]),
|
||||
or None if not cached or expired
|
||||
"""
|
||||
from shelfmark.core.config import config
|
||||
|
||||
if ttl_seconds is None:
|
||||
ttl_value = config.get("IRC_CACHE_TTL", DEFAULT_CACHE_TTL)
|
||||
# Config values are stored as strings, convert to int
|
||||
ttl_seconds = int(ttl_value) if ttl_value else DEFAULT_CACHE_TTL
|
||||
|
||||
# TTL of 0 means cache forever
|
||||
if ttl_seconds == 0:
|
||||
ttl_seconds = float('inf')
|
||||
|
||||
cache_key = _generate_cache_key(provider, provider_id)
|
||||
|
||||
with _cache_lock:
|
||||
cache = _load_cache()
|
||||
entry = cache.get("entries", {}).get(cache_key)
|
||||
|
||||
if not entry:
|
||||
return None
|
||||
|
||||
# Check expiration
|
||||
cached_at = entry.get("cached_at", 0)
|
||||
age = time.time() - cached_at
|
||||
|
||||
if age > ttl_seconds:
|
||||
logger.debug(f"IRC cache expired for '{title}' (age: {age:.0f}s > TTL: {ttl_seconds}s)")
|
||||
# Don't delete here - let cleanup handle it
|
||||
return None
|
||||
|
||||
# Convert dicts back to Release objects
|
||||
releases = [_dict_to_release(r) for r in entry.get("releases", [])]
|
||||
online_servers = entry.get("online_servers", [])
|
||||
title = entry.get("title", "")
|
||||
|
||||
logger.info(f"IRC cache hit for '{title}' ({len(releases)} releases, age: {age:.0f}s)")
|
||||
|
||||
return {
|
||||
"releases": releases,
|
||||
"online_servers": online_servers,
|
||||
"cached_at": cached_at,
|
||||
}
|
||||
|
||||
|
||||
def cache_results(
|
||||
provider: str,
|
||||
provider_id: str,
|
||||
title: str,
|
||||
releases: List[Release],
|
||||
online_servers: Optional[List[str]] = None
|
||||
) -> None:
|
||||
"""
|
||||
Cache search results for a book.
|
||||
|
||||
Args:
|
||||
provider: Metadata provider name
|
||||
provider_id: Book ID in the provider's system
|
||||
title: Book title (for logging/display)
|
||||
releases: List of Release objects from search
|
||||
online_servers: List of online server nicks (optional)
|
||||
"""
|
||||
cache_key = _generate_cache_key(provider, provider_id)
|
||||
|
||||
with _cache_lock:
|
||||
cache = _load_cache()
|
||||
|
||||
if "entries" not in cache:
|
||||
cache["entries"] = {}
|
||||
|
||||
cache["entries"][cache_key] = {
|
||||
"provider": provider,
|
||||
"provider_id": provider_id,
|
||||
"title": title,
|
||||
"releases": [_release_to_dict(r) for r in releases],
|
||||
"online_servers": list(online_servers) if online_servers else [],
|
||||
"cached_at": time.time(),
|
||||
}
|
||||
|
||||
_save_cache(cache)
|
||||
logger.info(f"Cached {len(releases)} IRC releases for '{title}'")
|
||||
|
||||
|
||||
def invalidate_cache(provider: str, provider_id: str) -> bool:
|
||||
"""
|
||||
Remove a specific entry from the cache.
|
||||
|
||||
Args:
|
||||
provider: Metadata provider name
|
||||
provider_id: Book ID in the provider's system
|
||||
|
||||
Returns:
|
||||
True if entry was found and removed
|
||||
"""
|
||||
cache_key = _generate_cache_key(provider, provider_id)
|
||||
|
||||
with _cache_lock:
|
||||
cache = _load_cache()
|
||||
entry = cache.get("entries", {}).get(cache_key)
|
||||
title = entry.get("title", cache_key) if entry else cache_key
|
||||
|
||||
if cache_key in cache.get("entries", {}):
|
||||
del cache["entries"][cache_key]
|
||||
_save_cache(cache)
|
||||
logger.info(f"Invalidated IRC cache for '{title}'")
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
|
||||
def clear_cache() -> int:
|
||||
"""
|
||||
Clear all cached entries.
|
||||
|
||||
Returns:
|
||||
Number of entries cleared
|
||||
"""
|
||||
with _cache_lock:
|
||||
cache = _load_cache()
|
||||
count = len(cache.get("entries", {}))
|
||||
cache["entries"] = {}
|
||||
_save_cache(cache)
|
||||
logger.info(f"Cleared {count} IRC cache entries")
|
||||
return count
|
||||
|
||||
|
||||
def cleanup_expired(ttl_seconds: Optional[int] = None) -> int:
|
||||
"""
|
||||
Remove all expired entries from the cache.
|
||||
|
||||
Returns:
|
||||
Number of entries removed
|
||||
"""
|
||||
from shelfmark.core.config import config
|
||||
|
||||
if ttl_seconds is None:
|
||||
ttl_value = config.get("IRC_CACHE_TTL", DEFAULT_CACHE_TTL)
|
||||
# Config values are stored as strings, convert to int
|
||||
ttl_seconds = int(ttl_value) if ttl_value else DEFAULT_CACHE_TTL
|
||||
|
||||
current_time = time.time()
|
||||
removed = 0
|
||||
|
||||
with _cache_lock:
|
||||
cache = _load_cache()
|
||||
entries = cache.get("entries", {})
|
||||
|
||||
expired_keys = [
|
||||
key for key, entry in entries.items()
|
||||
if current_time - entry.get("cached_at", 0) > ttl_seconds
|
||||
]
|
||||
|
||||
for key in expired_keys:
|
||||
del entries[key]
|
||||
removed += 1
|
||||
|
||||
if removed:
|
||||
_save_cache(cache)
|
||||
logger.info(f"Cleaned up {removed} expired IRC cache entries")
|
||||
|
||||
return removed
|
||||
|
||||
|
||||
def get_cache_stats() -> Dict[str, Any]:
|
||||
"""
|
||||
Get cache statistics.
|
||||
|
||||
Returns:
|
||||
Dict with cache stats
|
||||
"""
|
||||
from shelfmark.core.config import config
|
||||
|
||||
ttl_value = config.get("IRC_CACHE_TTL", DEFAULT_CACHE_TTL)
|
||||
# Config values are stored as strings, convert to int
|
||||
ttl_seconds = int(ttl_value) if ttl_value else DEFAULT_CACHE_TTL
|
||||
current_time = time.time()
|
||||
|
||||
with _cache_lock:
|
||||
cache = _load_cache()
|
||||
entries = cache.get("entries", {})
|
||||
|
||||
total = len(entries)
|
||||
expired = sum(
|
||||
1 for entry in entries.values()
|
||||
if current_time - entry.get("cached_at", 0) > ttl_seconds
|
||||
)
|
||||
|
||||
# Calculate total releases cached
|
||||
total_releases = sum(
|
||||
len(entry.get("releases", []))
|
||||
for entry in entries.values()
|
||||
)
|
||||
|
||||
return {
|
||||
"total_entries": total,
|
||||
"expired_entries": expired,
|
||||
"valid_entries": total - expired,
|
||||
"total_releases": total_releases,
|
||||
"ttl_seconds": ttl_seconds,
|
||||
"cache_file": str(CACHE_FILE),
|
||||
}
|
||||
@@ -0,0 +1,400 @@
|
||||
"""IRC client implementation using raw sockets.
|
||||
|
||||
Minimal IRC client for ebook searches.
|
||||
"""
|
||||
|
||||
import re
|
||||
import socket
|
||||
import ssl
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from enum import Enum, auto
|
||||
from typing import Iterator, Optional
|
||||
|
||||
from shelfmark.core.logger import setup_logger
|
||||
|
||||
from .dcc import DCCOffer, parse_dcc_send
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
|
||||
# Timing
|
||||
POST_CONNECT_DELAY = 2.0 # Seconds to wait after connect before joining
|
||||
SOCKET_TIMEOUT = 300.0 # 5 minutes - long because we wait for DCC offers
|
||||
RECV_BUFFER = 4096
|
||||
|
||||
# IRC channel user prefixes that indicate elevated status (ops, voice, etc.)
|
||||
# These are the download bots/servers
|
||||
ELEVATED_PREFIXES = frozenset({'~', '&', '@', '%', '+'})
|
||||
|
||||
|
||||
class IRCEvent(Enum):
|
||||
"""Events detected from IRC messages."""
|
||||
MESSAGE = auto() # Generic message
|
||||
SEARCH_RESULT = auto() # DCC SEND with "_results_for"
|
||||
BOOK_RESULT = auto() # DCC SEND for actual book
|
||||
NO_RESULTS = auto() # "Sorry" notice
|
||||
BAD_SERVER = auto() # "try another server" notice
|
||||
SEARCH_ACCEPTED = auto() # "has been accepted" notice
|
||||
MATCHES_FOUND = auto() # "X matches" notice
|
||||
SERVER_LIST = auto() # User list (353/366)
|
||||
PING = auto() # Server PING
|
||||
VERSION = auto() # CTCP VERSION request
|
||||
|
||||
|
||||
@dataclass
|
||||
class IRCMessage:
|
||||
"""Parsed IRC message."""
|
||||
raw: str
|
||||
prefix: Optional[str] = None
|
||||
command: str = ""
|
||||
params: list[str] = field(default_factory=list)
|
||||
trailing: Optional[str] = None
|
||||
event: IRCEvent = IRCEvent.MESSAGE
|
||||
|
||||
|
||||
class IRCError(Exception):
|
||||
"""Base IRC error."""
|
||||
pass
|
||||
|
||||
|
||||
class IRCConnectionError(IRCError):
|
||||
"""Connection failed."""
|
||||
pass
|
||||
|
||||
|
||||
class IRCClient:
|
||||
"""Minimal IRC client for per-request ebook searches."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
nick: str,
|
||||
server: str,
|
||||
port: int,
|
||||
use_tls: bool = True,
|
||||
version: str = "Shelfmark 1.0",
|
||||
):
|
||||
if not nick:
|
||||
raise IRCError("IRC nickname is required")
|
||||
if not server:
|
||||
raise IRCError("IRC server is required")
|
||||
if not port:
|
||||
raise IRCError("IRC port is required")
|
||||
self.nick = nick
|
||||
self.server = server
|
||||
self.port = port
|
||||
self.use_tls = use_tls
|
||||
self.version = version
|
||||
|
||||
self._socket: Optional[socket.socket] = None
|
||||
self._buffer = ""
|
||||
self._connected = False
|
||||
|
||||
# Track online servers (elevated users in channel)
|
||||
self.online_servers: set[str] = set()
|
||||
|
||||
def connect(self) -> None:
|
||||
"""Connect to IRC server, send USER/NICK, and wait for welcome."""
|
||||
logger.info(f"Connecting to {self.server}:{self.port} (TLS={self.use_tls})")
|
||||
|
||||
try:
|
||||
# Create socket
|
||||
sock = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
|
||||
sock.settimeout(SOCKET_TIMEOUT)
|
||||
|
||||
# Wrap with TLS if needed
|
||||
if self.use_tls:
|
||||
context = ssl.create_default_context()
|
||||
# Skip verification for self-signed certs common on IRC servers
|
||||
context.check_hostname = False
|
||||
context.verify_mode = ssl.CERT_NONE
|
||||
sock = context.wrap_socket(sock, server_hostname=self.server)
|
||||
|
||||
sock.connect((self.server, self.port))
|
||||
self._socket = sock
|
||||
|
||||
except socket.error as e:
|
||||
raise IRCConnectionError(f"Failed to connect: {e}")
|
||||
|
||||
# Send authentication (USER before NICK per IRC protocol)
|
||||
self._send(f"USER {self.nick} 0 * :{self.nick}")
|
||||
self._send(f"NICK {self.nick}")
|
||||
|
||||
# Wait for server to process welcome messages
|
||||
logger.debug(f"Waiting {POST_CONNECT_DELAY}s for server welcome")
|
||||
time.sleep(POST_CONNECT_DELAY)
|
||||
|
||||
self._connected = True
|
||||
logger.info(f"Connected as {self.nick}")
|
||||
|
||||
def disconnect(self) -> None:
|
||||
"""Gracefully disconnect from server."""
|
||||
if self._socket:
|
||||
try:
|
||||
self._send("QUIT :Goodbye")
|
||||
except Exception:
|
||||
pass # Best effort
|
||||
|
||||
try:
|
||||
self._socket.close()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
self._socket = None
|
||||
self._connected = False
|
||||
logger.info("Disconnected from IRC")
|
||||
|
||||
def join_channel(self, channel: str, wait_for_join: bool = True) -> None:
|
||||
"""Join an IRC channel (without # prefix) and capture online servers."""
|
||||
self._send(f"JOIN #{channel}")
|
||||
logger.debug(f"Sent JOIN #{channel}")
|
||||
|
||||
# Clear any existing server list before joining
|
||||
self.online_servers.clear()
|
||||
|
||||
if wait_for_join:
|
||||
# Wait for end of NAMES list (366) which confirms we're in the channel
|
||||
start = time.time()
|
||||
timeout = 10.0 # 10 seconds should be plenty
|
||||
|
||||
for line in self._recv_lines():
|
||||
if time.time() - start > timeout:
|
||||
logger.warning(f"Timeout waiting for JOIN confirmation on #{channel}")
|
||||
break
|
||||
|
||||
msg = self._parse_message(line)
|
||||
|
||||
# Handle PING during join wait
|
||||
if msg.event == IRCEvent.PING:
|
||||
self._handle_ping(msg)
|
||||
continue
|
||||
|
||||
# 353 = RPL_NAMREPLY - parse the names list
|
||||
if msg.command == "353":
|
||||
self._parse_names_list(msg.raw)
|
||||
continue
|
||||
|
||||
# 366 = RPL_ENDOFNAMES - channel join is complete
|
||||
if msg.command == "366":
|
||||
logger.info(f"Joined #{channel} - {len(self.online_servers)} servers online")
|
||||
return
|
||||
|
||||
# Check for errors (e.g., banned, channel doesn't exist)
|
||||
if msg.command in ("473", "474", "475", "403"):
|
||||
logger.error(f"Cannot join #{channel}: {msg.trailing}")
|
||||
return
|
||||
|
||||
logger.warning(f"Joined #{channel} (no confirmation received)")
|
||||
|
||||
def send_message(self, target: str, message: str) -> None:
|
||||
"""Send a PRIVMSG to a channel or user."""
|
||||
self._send(f"PRIVMSG {target} :{message}")
|
||||
logger.debug(f"Sent to {target}: {message[:50]}...")
|
||||
|
||||
def send_notice(self, target: str, message: str) -> None:
|
||||
"""Send a NOTICE to a user."""
|
||||
self._send(f"NOTICE {target} :{message}")
|
||||
|
||||
def request_names(self, channel: str) -> None:
|
||||
"""Request user list for a channel (without # prefix)."""
|
||||
self._send(f"NAMES #{channel}")
|
||||
|
||||
def _parse_names_list(self, names_data: str) -> None:
|
||||
"""Parse 353 NAMES reply and extract elevated users (download servers)."""
|
||||
# Extract the trailing part after the last colon (the actual names)
|
||||
if ' :' in names_data:
|
||||
names_part = names_data.split(' :')[-1]
|
||||
else:
|
||||
names_part = names_data
|
||||
|
||||
for name in names_part.split():
|
||||
# Check if user has an elevated prefix
|
||||
if name[0] in ELEVATED_PREFIXES:
|
||||
# Strip the prefix to get the actual nick
|
||||
self.online_servers.add(name[1:])
|
||||
# Note: we only care about elevated users for server status
|
||||
|
||||
def _send(self, message: str) -> None:
|
||||
"""Send raw IRC message."""
|
||||
if not self._socket:
|
||||
raise IRCError("Not connected")
|
||||
|
||||
data = f"{message}\r\n".encode('utf-8')
|
||||
self._socket.sendall(data)
|
||||
|
||||
def _recv_lines(self) -> Iterator[str]:
|
||||
"""Receive and yield complete CRLF-delimited IRC lines."""
|
||||
while True:
|
||||
# Check if we have a complete line in buffer
|
||||
while '\r\n' in self._buffer:
|
||||
line, self._buffer = self._buffer.split('\r\n', 1)
|
||||
if line:
|
||||
yield line
|
||||
|
||||
# Read more data
|
||||
try:
|
||||
data = self._socket.recv(RECV_BUFFER)
|
||||
if not data:
|
||||
return # Connection closed
|
||||
self._buffer += data.decode('utf-8', errors='replace')
|
||||
except socket.timeout:
|
||||
continue # Keep waiting
|
||||
except socket.error as e:
|
||||
logger.warning(f"Socket error: {e}")
|
||||
return # Connection error
|
||||
|
||||
def _parse_message(self, line: str) -> IRCMessage:
|
||||
"""Parse an IRC message line into components.
|
||||
|
||||
Format: [:prefix] COMMAND [params] [:trailing]
|
||||
"""
|
||||
msg = IRCMessage(raw=line)
|
||||
|
||||
# Extract prefix if present
|
||||
if line.startswith(':'):
|
||||
space_idx = line.find(' ')
|
||||
if space_idx != -1:
|
||||
msg.prefix = line[1:space_idx]
|
||||
line = line[space_idx + 1:]
|
||||
|
||||
# Extract trailing if present
|
||||
if ' :' in line:
|
||||
idx = line.find(' :')
|
||||
msg.trailing = line[idx + 2:]
|
||||
line = line[:idx]
|
||||
|
||||
# Split remaining into command and params
|
||||
parts = line.split()
|
||||
if parts:
|
||||
msg.command = parts[0]
|
||||
msg.params = parts[1:]
|
||||
|
||||
# Classify event type based on message content
|
||||
msg.event = self._classify_event(msg)
|
||||
|
||||
return msg
|
||||
|
||||
def _classify_event(self, msg: IRCMessage) -> IRCEvent:
|
||||
"""Classify message into event type using string containment checks."""
|
||||
raw = msg.raw
|
||||
trailing = msg.trailing or ""
|
||||
|
||||
# DCC SEND detection
|
||||
if "DCC SEND" in raw:
|
||||
if "_results_for" in raw:
|
||||
return IRCEvent.SEARCH_RESULT
|
||||
return IRCEvent.BOOK_RESULT
|
||||
|
||||
# NOTICE messages
|
||||
if msg.command == "NOTICE" or "NOTICE" in raw:
|
||||
if "Sorry" in trailing:
|
||||
return IRCEvent.NO_RESULTS
|
||||
if "try another server" in trailing:
|
||||
return IRCEvent.BAD_SERVER
|
||||
if "has been accepted" in trailing:
|
||||
return IRCEvent.SEARCH_ACCEPTED
|
||||
if "matches" in trailing:
|
||||
return IRCEvent.MATCHES_FOUND
|
||||
|
||||
# User list (RPL_NAMREPLY and RPL_ENDOFNAMES)
|
||||
if msg.command in ("353", "366"):
|
||||
return IRCEvent.SERVER_LIST
|
||||
|
||||
# Server PING
|
||||
if msg.command == "PING":
|
||||
return IRCEvent.PING
|
||||
|
||||
# CTCP VERSION
|
||||
if "\x01VERSION\x01" in raw:
|
||||
return IRCEvent.VERSION
|
||||
|
||||
return IRCEvent.MESSAGE
|
||||
|
||||
def _handle_ping(self, msg: IRCMessage) -> None:
|
||||
"""Respond to server PING with PONG."""
|
||||
# PING message format: PING :server
|
||||
server = msg.trailing or self.server
|
||||
self._send(f"PONG :{server}")
|
||||
logger.debug(f"PONG {server}")
|
||||
|
||||
def _handle_version(self, msg: IRCMessage) -> None:
|
||||
"""Respond to CTCP VERSION request."""
|
||||
if msg.prefix:
|
||||
# Extract nick from prefix (nick!user@host)
|
||||
sender = msg.prefix.split('!')[0]
|
||||
self.send_notice(sender, f"\x01VERSION {self.version}\x01")
|
||||
logger.debug(f"Sent VERSION to {sender}")
|
||||
|
||||
def read_messages(self, auto_handle: bool = True) -> Iterator[IRCMessage]:
|
||||
"""Read and yield IRC messages, optionally auto-handling PING/VERSION."""
|
||||
for line in self._recv_lines():
|
||||
msg = self._parse_message(line)
|
||||
|
||||
# Auto-handle certain events
|
||||
if auto_handle:
|
||||
if msg.event == IRCEvent.PING:
|
||||
self._handle_ping(msg)
|
||||
continue # Don't yield PING messages
|
||||
|
||||
if msg.event == IRCEvent.VERSION:
|
||||
self._handle_version(msg)
|
||||
continue # Don't yield VERSION messages
|
||||
|
||||
yield msg
|
||||
|
||||
def wait_for_dcc(
|
||||
self,
|
||||
timeout: float = 60.0,
|
||||
result_type: bool = False,
|
||||
) -> Optional[DCCOffer]:
|
||||
"""Wait for a DCC SEND offer. Returns None on timeout or no results."""
|
||||
target_event = IRCEvent.SEARCH_RESULT if result_type else IRCEvent.BOOK_RESULT
|
||||
start = time.time()
|
||||
|
||||
for msg in self.read_messages():
|
||||
if time.time() - start > timeout:
|
||||
logger.warning("Timeout waiting for DCC offer")
|
||||
return None
|
||||
|
||||
if msg.event == target_event:
|
||||
try:
|
||||
offer = parse_dcc_send(msg.raw)
|
||||
logger.info(f"Received DCC offer: {offer.filename}")
|
||||
return offer
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to parse DCC: {e}")
|
||||
return None
|
||||
|
||||
# Log other events for debugging
|
||||
if msg.event == IRCEvent.NO_RESULTS:
|
||||
logger.info("Server reports no results")
|
||||
return None
|
||||
elif msg.event == IRCEvent.BAD_SERVER:
|
||||
logger.warning("Server unavailable")
|
||||
return None
|
||||
elif msg.event == IRCEvent.SEARCH_ACCEPTED:
|
||||
logger.info("Search accepted, waiting for results...")
|
||||
elif msg.event == IRCEvent.MATCHES_FOUND:
|
||||
# Extract count from "returned X matches"
|
||||
if msg.trailing and "returned" in msg.trailing:
|
||||
try:
|
||||
match = re.search(r'returned\s+(\d+)\s+matches', msg.trailing)
|
||||
if match:
|
||||
count = match.group(1)
|
||||
logger.info(f"Found {count} matches")
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return None
|
||||
|
||||
@property
|
||||
def is_connected(self) -> bool:
|
||||
"""Check if currently connected."""
|
||||
return self._connected and self._socket is not None
|
||||
|
||||
def __enter__(self):
|
||||
self.connect()
|
||||
return self
|
||||
|
||||
def __exit__(self, *args):
|
||||
self.disconnect()
|
||||
@@ -0,0 +1,144 @@
|
||||
"""DCC (Direct Client-to-Client) protocol implementation.
|
||||
|
||||
Handles DCC SEND file transfers used by IRC bots to send files.
|
||||
"""
|
||||
|
||||
import re
|
||||
import socket
|
||||
import struct
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from threading import Event
|
||||
from typing import Callable, Optional
|
||||
|
||||
from shelfmark.core.logger import setup_logger
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
# Regex to parse DCC SEND messages - handles quoted filenames
|
||||
# Format: DCC SEND "filename.epub" 2760158537 2050 2321788
|
||||
# | | | |
|
||||
# filename IP(int) port size
|
||||
DCC_REGEX = re.compile(r'DCC SEND "?(.+[^"])"?\s(\d+)\s+(\d+)\s+(\d+)\s*')
|
||||
|
||||
# Buffer size for DCC transfers - 4096 bytes provides good performance
|
||||
BUFFER_SIZE = 4096
|
||||
|
||||
|
||||
@dataclass
|
||||
class DCCOffer:
|
||||
"""Parsed DCC SEND offer."""
|
||||
filename: str
|
||||
ip: str
|
||||
port: int
|
||||
size: int
|
||||
|
||||
@property
|
||||
def address(self) -> tuple[str, int]:
|
||||
"""Return (ip, port) tuple for socket.connect()."""
|
||||
return (self.ip, self.port)
|
||||
|
||||
|
||||
class DCCError(Exception):
|
||||
"""Base exception for DCC operations."""
|
||||
pass
|
||||
|
||||
|
||||
class DCCParseError(DCCError):
|
||||
"""Failed to parse DCC SEND string."""
|
||||
pass
|
||||
|
||||
|
||||
class DCCSizeError(DCCError):
|
||||
"""Downloaded size doesn't match expected size."""
|
||||
pass
|
||||
|
||||
|
||||
class DCCConnectionError(DCCError):
|
||||
"""Failed to connect to DCC sender."""
|
||||
pass
|
||||
|
||||
|
||||
def int_to_ip(ip_int: int) -> str:
|
||||
"""Convert 32-bit integer (DCC format) to dotted IP notation."""
|
||||
packed = struct.pack('>I', ip_int)
|
||||
return '.'.join(str(b) for b in packed)
|
||||
|
||||
|
||||
def parse_dcc_send(text: str) -> DCCOffer:
|
||||
"""Parse a DCC SEND message into a DCCOffer. Raises DCCParseError on failure."""
|
||||
match = DCC_REGEX.search(text)
|
||||
if not match:
|
||||
raise DCCParseError(f"Invalid DCC SEND format: {text[:100]}")
|
||||
|
||||
filename = match.group(1).strip('"')
|
||||
ip_int = int(match.group(2))
|
||||
port = int(match.group(3))
|
||||
size = int(match.group(4))
|
||||
|
||||
return DCCOffer(
|
||||
filename=filename,
|
||||
ip=int_to_ip(ip_int),
|
||||
port=port,
|
||||
size=size,
|
||||
)
|
||||
|
||||
|
||||
def download_dcc(
|
||||
offer: DCCOffer,
|
||||
dest_path: Path,
|
||||
progress_callback: Optional[Callable[[float], None]] = None,
|
||||
cancel_flag: Optional[Event] = None,
|
||||
timeout: float = 30.0,
|
||||
) -> None:
|
||||
"""Download file via DCC protocol to dest_path. Raises DCCError on failure."""
|
||||
logger.info(f"DCC connecting to {offer.ip}:{offer.port} for {offer.filename}")
|
||||
|
||||
try:
|
||||
sock = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
|
||||
sock.settimeout(timeout)
|
||||
sock.connect(offer.address)
|
||||
except socket.error as e:
|
||||
raise DCCConnectionError(f"Failed to connect to {offer.ip}:{offer.port}: {e}")
|
||||
|
||||
try:
|
||||
received = 0
|
||||
last_progress = -1
|
||||
|
||||
with open(dest_path, 'wb') as f:
|
||||
while received < offer.size:
|
||||
# Check for cancellation
|
||||
if cancel_flag and cancel_flag.is_set():
|
||||
logger.info("DCC download cancelled")
|
||||
return
|
||||
|
||||
# Read chunk
|
||||
try:
|
||||
chunk = sock.recv(BUFFER_SIZE)
|
||||
except socket.timeout:
|
||||
raise DCCError(f"Timeout reading from {offer.ip}:{offer.port}")
|
||||
|
||||
if not chunk:
|
||||
# Connection closed prematurely
|
||||
break
|
||||
|
||||
f.write(chunk)
|
||||
received += len(chunk)
|
||||
|
||||
# Report progress (every 1%)
|
||||
if progress_callback:
|
||||
progress = int((received / offer.size) * 100)
|
||||
if progress != last_progress:
|
||||
progress_callback(progress)
|
||||
last_progress = progress
|
||||
|
||||
# Verify downloaded size matches expected
|
||||
if received != offer.size:
|
||||
raise DCCSizeError(
|
||||
f"Size mismatch: expected {offer.size} bytes, got {received}"
|
||||
)
|
||||
|
||||
logger.info(f"DCC download complete: {received} bytes")
|
||||
|
||||
finally:
|
||||
sock.close()
|
||||
@@ -0,0 +1,137 @@
|
||||
"""IRC DCC download handler.
|
||||
|
||||
Handles downloading books via IRC DCC protocol.
|
||||
"""
|
||||
|
||||
from pathlib import Path
|
||||
from threading import Event
|
||||
from typing import Callable, Optional
|
||||
|
||||
from shelfmark.core.config import config
|
||||
from shelfmark.core.logger import setup_logger
|
||||
from shelfmark.core.models import DownloadTask
|
||||
from shelfmark.release_sources import DownloadHandler, register_handler
|
||||
|
||||
from .client import IRCClient
|
||||
from .dcc import DCCError, download_dcc
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
|
||||
@register_handler("irc")
|
||||
class IRCDownloadHandler(DownloadHandler):
|
||||
"""Handle IRC DCC downloads."""
|
||||
|
||||
def download(
|
||||
self,
|
||||
task: DownloadTask,
|
||||
cancel_flag: Event,
|
||||
progress_callback: Callable[[float], None],
|
||||
status_callback: Callable[[str, Optional[str]], None],
|
||||
) -> Optional[str]:
|
||||
"""Download a book via IRC DCC. task.task_id contains the IRC request string."""
|
||||
download_request = task.task_id
|
||||
logger.info(f"IRC download: {download_request[:60]}...")
|
||||
|
||||
# Get IRC settings
|
||||
server = config.get("IRC_SERVER", "")
|
||||
port = config.get("IRC_PORT", 6697)
|
||||
channel = config.get("IRC_CHANNEL", "")
|
||||
nick = config.get("IRC_NICK", "")
|
||||
|
||||
if not server or not channel or not nick:
|
||||
logger.warning("IRC not fully configured")
|
||||
status_callback("failed", "IRC not configured")
|
||||
return None
|
||||
|
||||
client = None
|
||||
|
||||
def check_cancelled() -> bool:
|
||||
"""Check if cancelled and handle cleanup."""
|
||||
if not cancel_flag.is_set():
|
||||
return False
|
||||
if client:
|
||||
client.disconnect()
|
||||
status_callback("cancelled", "Cancelled")
|
||||
return True
|
||||
|
||||
try:
|
||||
# Phase 1: Connect to IRC
|
||||
status_callback("resolving", f"Connecting to {server}")
|
||||
|
||||
if check_cancelled():
|
||||
return None
|
||||
|
||||
client = IRCClient(nick, server, port)
|
||||
client.connect()
|
||||
client.join_channel(channel)
|
||||
|
||||
# Phase 2: Send download request
|
||||
status_callback("resolving", "Requesting file from bot")
|
||||
|
||||
if check_cancelled():
|
||||
return None
|
||||
|
||||
# Send the full request line to the channel
|
||||
client.send_message(f"#{channel}", download_request)
|
||||
|
||||
# Phase 3: Wait for DCC offer
|
||||
status_callback("resolving", "Waiting for bot response")
|
||||
|
||||
offer = client.wait_for_dcc(timeout=120.0, result_type=False)
|
||||
|
||||
if not offer:
|
||||
status_callback("error", "No response from bot")
|
||||
client.disconnect()
|
||||
return None
|
||||
|
||||
if check_cancelled():
|
||||
return None
|
||||
|
||||
# Phase 4: Download via DCC
|
||||
status_callback("downloading", "")
|
||||
|
||||
# Get file extension from offer filename
|
||||
ext = Path(offer.filename).suffix.lstrip('.') or task.format or "epub"
|
||||
|
||||
# Stage to temp directory (lazy import to avoid circular import)
|
||||
from shelfmark.download.orchestrator import get_staging_path
|
||||
staging_path = get_staging_path(task.task_id, ext)
|
||||
|
||||
download_dcc(
|
||||
offer=offer,
|
||||
dest_path=staging_path,
|
||||
progress_callback=progress_callback,
|
||||
cancel_flag=cancel_flag,
|
||||
timeout=60.0,
|
||||
)
|
||||
|
||||
client.disconnect()
|
||||
|
||||
if cancel_flag.is_set():
|
||||
# Clean up partial download
|
||||
staging_path.unlink(missing_ok=True)
|
||||
status_callback("cancelled", "Cancelled")
|
||||
return None
|
||||
|
||||
logger.info(f"Download complete: {staging_path}")
|
||||
return str(staging_path)
|
||||
|
||||
except DCCError as e:
|
||||
logger.error(f"DCC error: {e}")
|
||||
status_callback("error", str(e))
|
||||
if client:
|
||||
client.disconnect()
|
||||
return None
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Download failed: {e}")
|
||||
status_callback("error", f"Download failed: {e}")
|
||||
if client:
|
||||
client.disconnect()
|
||||
return None
|
||||
|
||||
def cancel(self, task_id: str) -> bool:
|
||||
"""Cancel an in-progress download (cleanup if cancel_flag fails)."""
|
||||
logger.debug(f"Cancel requested for IRC task: {task_id}")
|
||||
return True
|
||||
@@ -0,0 +1,188 @@
|
||||
"""Search results file parser.
|
||||
|
||||
Parses the text files sent via DCC that contain search results.
|
||||
"""
|
||||
|
||||
import re
|
||||
import zipfile
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
from shelfmark.core.config import config
|
||||
from shelfmark.core.logger import setup_logger
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
# All recognized formats for parsing IRC result lines.
|
||||
# This comprehensive list is used to identify file extensions in results.
|
||||
# User's configured formats are used separately for filtering.
|
||||
# Note: IRC source currently only supports ebooks, but audiobook formats
|
||||
# are included for future-proofing and format detection consistency.
|
||||
ALL_RECOGNIZED_FORMATS = {
|
||||
# Ebook formats
|
||||
'epub', 'mobi', 'azw3', 'azw', 'pdf', 'doc', 'docx',
|
||||
'html', 'htm', 'rtf', 'txt', 'lit', 'fb2', 'djvu',
|
||||
'cbr', 'cbz', 'cdr', 'jpg', 'rar', 'zip',
|
||||
# Audiobook formats
|
||||
'm4b', 'mp3', 'm4a', 'flac', 'ogg', 'wma', 'aac', 'wav', 'opus'
|
||||
}
|
||||
|
||||
|
||||
def _get_supported_formats() -> set[str]:
|
||||
"""Get user's configured supported formats from settings."""
|
||||
formats = config.get("SUPPORTED_FORMATS", ["epub", "mobi", "azw3", "fb2", "djvu", "cbz", "cbr"])
|
||||
if isinstance(formats, str):
|
||||
return {fmt.strip().lower() for fmt in formats.split(",") if fmt.strip()}
|
||||
return {fmt.lower() for fmt in formats}
|
||||
|
||||
# Regex to parse result lines
|
||||
# Format: !Server Author - Title.format ::INFO:: size
|
||||
RESULT_LINE_REGEX = re.compile(
|
||||
r'^!(\S+)\s+' # !ServerName
|
||||
r'(.+?)\s+-\s+' # Author Name -
|
||||
r'(.+?)\.(\w+)' # Title.format
|
||||
r'(?:\s+::INFO::\s*(.+?))?' # Optional ::INFO:: metadata
|
||||
r'(?:\s+::HASH::\s*(\S+))?' # Optional ::HASH::
|
||||
r'\s*$'
|
||||
)
|
||||
|
||||
# Simpler fallback pattern
|
||||
SIMPLE_RESULT_REGEX = re.compile(
|
||||
r'^!(\S+)\s+(.+)$' # !Server everything_else
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class SearchResult:
|
||||
"""Parsed search result entry."""
|
||||
server: str # Bot name (without !)
|
||||
author: str # Author name
|
||||
title: str # Book title
|
||||
format: str # File format (epub, mobi, etc)
|
||||
size: Optional[str] # Human-readable size
|
||||
full_line: str # Original line for download request
|
||||
|
||||
@property
|
||||
def download_request(self) -> str:
|
||||
"""The string to send to IRC to request this book."""
|
||||
return self.full_line.strip()
|
||||
|
||||
@property
|
||||
def display_name(self) -> str:
|
||||
"""Human-readable display name."""
|
||||
return f"{self.author} - {self.title}"
|
||||
|
||||
|
||||
def parse_result_line(line: str) -> Optional[SearchResult]:
|
||||
"""Parse a single search result line. Returns None if unparseable."""
|
||||
line = line.strip()
|
||||
|
||||
# Must start with !
|
||||
if not line.startswith('!'):
|
||||
return None
|
||||
|
||||
# Try detailed pattern first
|
||||
match = RESULT_LINE_REGEX.match(line)
|
||||
if match:
|
||||
server, author, title, fmt, size, _ = match.groups()
|
||||
return SearchResult(
|
||||
server=server,
|
||||
author=author.strip(),
|
||||
title=title.strip(),
|
||||
format=fmt.lower(),
|
||||
size=size.strip() if size else None,
|
||||
full_line=line,
|
||||
)
|
||||
|
||||
# Fallback: simpler parsing
|
||||
match = SIMPLE_RESULT_REGEX.match(line)
|
||||
if match:
|
||||
server, rest = match.groups()
|
||||
|
||||
# Try to extract format from the line
|
||||
fmt = None
|
||||
for known_fmt in ALL_RECOGNIZED_FORMATS:
|
||||
if f'.{known_fmt}' in rest.lower():
|
||||
fmt = known_fmt
|
||||
break
|
||||
|
||||
# Try to split author - title
|
||||
if ' - ' in rest:
|
||||
parts = rest.split(' - ', 1)
|
||||
author = parts[0].strip()
|
||||
title_part = parts[1].strip() if len(parts) > 1 else rest
|
||||
else:
|
||||
author = "Unknown"
|
||||
title_part = rest
|
||||
|
||||
# Extract size if present
|
||||
size = None
|
||||
if '::INFO::' in title_part:
|
||||
title_part, info = title_part.split('::INFO::', 1)
|
||||
size = info.split('::')[0].strip()
|
||||
|
||||
# Clean up title (remove extension)
|
||||
title = title_part
|
||||
for known_fmt in ALL_RECOGNIZED_FORMATS:
|
||||
title = re.sub(rf'\.{known_fmt}\b', '', title, flags=re.IGNORECASE)
|
||||
|
||||
return SearchResult(
|
||||
server=server,
|
||||
author=author,
|
||||
title=title.strip(),
|
||||
format=fmt or 'unknown',
|
||||
size=size,
|
||||
full_line=line,
|
||||
)
|
||||
|
||||
logger.debug(f"Could not parse line: {line[:80]}...")
|
||||
return None
|
||||
|
||||
|
||||
def parse_results_file(content: str) -> list[SearchResult]:
|
||||
"""Parse a search results file into SearchResult objects."""
|
||||
results = []
|
||||
supported = _get_supported_formats()
|
||||
|
||||
for line in content.splitlines():
|
||||
result = parse_result_line(line)
|
||||
if result:
|
||||
# Filter to user's configured formats
|
||||
if result.format in supported or result.format == 'unknown':
|
||||
results.append(result)
|
||||
|
||||
logger.info(f"Parsed {len(results)} results from search file")
|
||||
return results
|
||||
|
||||
|
||||
def extract_results_from_zip(zip_path: Path) -> str:
|
||||
"""Extract and return text content from a search results ZIP."""
|
||||
with zipfile.ZipFile(zip_path, 'r') as zf:
|
||||
# Should contain exactly one text file
|
||||
names = zf.namelist()
|
||||
if not names:
|
||||
raise ValueError("Empty ZIP file")
|
||||
|
||||
# Find the text file
|
||||
txt_file = None
|
||||
for name in names:
|
||||
if name.endswith('.txt'):
|
||||
txt_file = name
|
||||
break
|
||||
|
||||
if not txt_file:
|
||||
# Use first file
|
||||
txt_file = names[0]
|
||||
|
||||
content = zf.read(txt_file)
|
||||
|
||||
# Try different encodings
|
||||
for encoding in ['utf-8', 'latin-1', 'cp1252']:
|
||||
try:
|
||||
return content.decode(encoding)
|
||||
except UnicodeDecodeError:
|
||||
continue
|
||||
|
||||
# Last resort
|
||||
return content.decode('utf-8', errors='replace')
|
||||
@@ -0,0 +1,119 @@
|
||||
"""IRC settings registration.
|
||||
|
||||
Registers IRC settings for the settings UI.
|
||||
"""
|
||||
|
||||
from shelfmark.core.settings_registry import (
|
||||
ActionButton,
|
||||
HeadingField,
|
||||
NumberField,
|
||||
SelectField,
|
||||
TextField,
|
||||
register_settings,
|
||||
)
|
||||
|
||||
|
||||
def _clear_irc_cache():
|
||||
"""Clear all cached IRC search results."""
|
||||
from shelfmark.release_sources.irc.cache import clear_cache, get_cache_stats
|
||||
|
||||
stats = get_cache_stats()
|
||||
count = clear_cache()
|
||||
return {
|
||||
"success": True,
|
||||
"message": f"Cleared {count} cached searches ({stats['total_releases']} releases)",
|
||||
}
|
||||
|
||||
|
||||
@register_settings(
|
||||
name="irc",
|
||||
display_name="IRC",
|
||||
icon="download",
|
||||
order=56,
|
||||
)
|
||||
def irc_settings():
|
||||
"""Define IRC source settings."""
|
||||
return [
|
||||
HeadingField(
|
||||
key="heading",
|
||||
title="IRC",
|
||||
description=(
|
||||
"Search and download books from IRC ebook channels. "
|
||||
"This source connects via IRC and uses DCC for file transfers. "
|
||||
"Configure the connection details below to enable IRC search. "
|
||||
"Note: DCC requires direct TCP connections to arbitrary ports, "
|
||||
"which may not work behind strict firewalls or NAT."
|
||||
),
|
||||
),
|
||||
|
||||
TextField(
|
||||
key="IRC_SERVER",
|
||||
label="Server",
|
||||
placeholder="e.g. irc.example.net",
|
||||
description="IRC server hostname",
|
||||
required=True,
|
||||
env_supported=True,
|
||||
),
|
||||
|
||||
NumberField(
|
||||
key="IRC_PORT",
|
||||
label="Port",
|
||||
default=6697,
|
||||
description="IRC server port (usually 6697 for TLS, 6667 for plain)",
|
||||
env_supported=True,
|
||||
),
|
||||
|
||||
TextField(
|
||||
key="IRC_CHANNEL",
|
||||
label="Channel",
|
||||
placeholder="e.g. ebooks",
|
||||
description="Channel name without the # prefix",
|
||||
required=True,
|
||||
env_supported=True,
|
||||
),
|
||||
|
||||
TextField(
|
||||
key="IRC_NICK",
|
||||
label="Nickname",
|
||||
placeholder="e.g. myusername",
|
||||
description="Your IRC nickname (required). Must be unique on the IRC network.",
|
||||
required=True,
|
||||
env_supported=True,
|
||||
),
|
||||
|
||||
TextField(
|
||||
key="IRC_SEARCH_BOT",
|
||||
label="Search bot",
|
||||
placeholder="e.g. search",
|
||||
description="The search bot to query for results",
|
||||
env_supported=True,
|
||||
),
|
||||
|
||||
HeadingField(
|
||||
key="cache_heading",
|
||||
title="Search Cache",
|
||||
description=(
|
||||
"IRC search results are cached to reduce load on IRC servers. "
|
||||
"Use the Refresh button in the release modal to force a new search."
|
||||
),
|
||||
),
|
||||
|
||||
SelectField(
|
||||
key="IRC_CACHE_TTL",
|
||||
label="Cache Duration",
|
||||
description="How long to keep cached search results before they expire.",
|
||||
options=[
|
||||
{"value": "2592000", "label": "30 days"},
|
||||
{"value": "0", "label": "Forever (until manually cleared)"},
|
||||
],
|
||||
default="2592000", # 30 days
|
||||
),
|
||||
|
||||
ActionButton(
|
||||
key="clear_irc_cache",
|
||||
label="Clear Cache",
|
||||
description="Remove all cached IRC search results.",
|
||||
style="danger",
|
||||
callback=_clear_irc_cache,
|
||||
),
|
||||
]
|
||||
@@ -0,0 +1,347 @@
|
||||
"""IRC release source plugin.
|
||||
|
||||
Searches IRC ebook channels for book releases.
|
||||
"""
|
||||
|
||||
import tempfile
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import List, Optional
|
||||
|
||||
from shelfmark.api.websocket import ws_manager
|
||||
from shelfmark.core.config import config
|
||||
from shelfmark.core.logger import setup_logger
|
||||
from shelfmark.metadata_providers import BookMetadata
|
||||
from shelfmark.release_sources import (
|
||||
ColumnColorHint,
|
||||
ColumnRenderType,
|
||||
ColumnSchema,
|
||||
LeadingCellConfig,
|
||||
LeadingCellType,
|
||||
Release,
|
||||
ReleaseColumnConfig,
|
||||
ReleaseProtocol,
|
||||
ReleaseSource,
|
||||
SourceActionButton,
|
||||
register_source,
|
||||
)
|
||||
|
||||
from .client import IRCClient
|
||||
from .dcc import DCCError, download_dcc
|
||||
from .parser import SearchResult, extract_results_from_zip, parse_results_file
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
|
||||
def _emit_status(message: str, phase: str = 'searching') -> None:
|
||||
"""Emit search status to frontend via WebSocket."""
|
||||
ws_manager.broadcast_search_status(
|
||||
source='irc',
|
||||
provider='',
|
||||
book_id='',
|
||||
message=message,
|
||||
phase=phase,
|
||||
)
|
||||
|
||||
# Rate limiting to avoid server throttling
|
||||
MIN_SEARCH_INTERVAL = 15.0
|
||||
_last_search_time: float = 0
|
||||
|
||||
|
||||
def _enforce_rate_limit() -> None:
|
||||
"""Ensure minimum time between searches."""
|
||||
global _last_search_time
|
||||
|
||||
elapsed = time.time() - _last_search_time
|
||||
if elapsed < MIN_SEARCH_INTERVAL:
|
||||
wait_time = MIN_SEARCH_INTERVAL - elapsed
|
||||
logger.info(f"Rate limiting: waiting {wait_time:.1f}s")
|
||||
time.sleep(wait_time)
|
||||
|
||||
_last_search_time = time.time()
|
||||
|
||||
|
||||
@register_source("irc")
|
||||
class IRCReleaseSource(ReleaseSource):
|
||||
"""Search IRC channels for book releases."""
|
||||
|
||||
name = "irc"
|
||||
display_name = "IRC"
|
||||
supported_content_types = ["ebook"] # IRC only supports ebooks
|
||||
can_be_default = False # Exclude from default source options (requires deliberate selection)
|
||||
|
||||
def __init__(self):
|
||||
# Track online servers from most recent search
|
||||
self._online_servers: Optional[set[str]] = None
|
||||
|
||||
@classmethod
|
||||
def is_available(cls) -> bool:
|
||||
"""Check if IRC is configured (server, channel, and nick are set)."""
|
||||
server = config.get("IRC_SERVER", "")
|
||||
channel = config.get("IRC_CHANNEL", "")
|
||||
nick = config.get("IRC_NICK", "")
|
||||
return bool(server and channel and nick)
|
||||
|
||||
def get_column_config(self) -> ReleaseColumnConfig:
|
||||
"""Configure UI columns for IRC results."""
|
||||
return ReleaseColumnConfig(
|
||||
columns=[
|
||||
ColumnSchema(
|
||||
key="extra.server",
|
||||
label="Server",
|
||||
render_type=ColumnRenderType.TEXT,
|
||||
width="100px",
|
||||
sortable=True,
|
||||
),
|
||||
ColumnSchema(
|
||||
key="format",
|
||||
label="Format",
|
||||
render_type=ColumnRenderType.BADGE,
|
||||
color_hint=ColumnColorHint(type="map", value="format"),
|
||||
width="70px",
|
||||
uppercase=True,
|
||||
sortable=True,
|
||||
),
|
||||
ColumnSchema(
|
||||
key="size",
|
||||
label="Size",
|
||||
render_type=ColumnRenderType.TEXT,
|
||||
width="70px",
|
||||
sortable=True,
|
||||
sort_key="size_bytes",
|
||||
),
|
||||
],
|
||||
grid_template="minmax(0,2fr) 100px 70px 70px",
|
||||
leading_cell=LeadingCellConfig(type=LeadingCellType.NONE),
|
||||
online_servers=list(self._online_servers) if self._online_servers else None,
|
||||
cache_ttl_seconds=1800, # 30 minutes - IRC searches are slow, cache longer
|
||||
supported_filters=["format"], # IRC has no language metadata
|
||||
action_button=SourceActionButton(label="Refresh search"),
|
||||
)
|
||||
|
||||
def search(
|
||||
self,
|
||||
book: BookMetadata,
|
||||
expand_search: bool = False,
|
||||
languages: Optional[List[str]] = None,
|
||||
content_type: str = "ebook"
|
||||
) -> List[Release]:
|
||||
"""Search IRC for books matching metadata.
|
||||
|
||||
The expand_search parameter is repurposed for IRC as a "refresh" flag.
|
||||
When True, it bypasses the cache and forces a fresh search.
|
||||
"""
|
||||
from .cache import get_cached_results, cache_results
|
||||
|
||||
if not self.is_available():
|
||||
logger.debug("IRC source is disabled, skipping search")
|
||||
return []
|
||||
|
||||
# Check cache first (unless expand_search/refresh is requested)
|
||||
if not expand_search:
|
||||
cached = get_cached_results(book.provider, book.provider_id)
|
||||
if cached:
|
||||
_emit_status("Using cached results", phase='complete')
|
||||
self._online_servers = set(cached.get("online_servers", []))
|
||||
return cached["releases"]
|
||||
|
||||
# Build search query
|
||||
query = self._build_query(book)
|
||||
if not query:
|
||||
logger.warning("No search query could be built")
|
||||
return []
|
||||
|
||||
logger.info(f"IRC search: {query}")
|
||||
|
||||
# Enforce rate limit
|
||||
_enforce_rate_limit()
|
||||
|
||||
# Get IRC settings
|
||||
server = config.get("IRC_SERVER", "")
|
||||
port = config.get("IRC_PORT", 6697)
|
||||
channel = config.get("IRC_CHANNEL", "")
|
||||
nick = config.get("IRC_NICK", "")
|
||||
search_bot = config.get("IRC_SEARCH_BOT", "")
|
||||
|
||||
client = None
|
||||
try:
|
||||
# Connect to IRC
|
||||
_emit_status(f"Connecting to {server}...", phase='connecting')
|
||||
client = IRCClient(nick, server, port)
|
||||
client.connect()
|
||||
|
||||
_emit_status(f"Joining #{channel}...", phase='connecting')
|
||||
client.join_channel(channel)
|
||||
|
||||
# Capture online servers (elevated users in channel)
|
||||
self._online_servers = client.online_servers
|
||||
|
||||
# Send search request
|
||||
search_msg = f"@{search_bot} {query}" if search_bot else query
|
||||
client.send_message(f"#{channel}", search_msg)
|
||||
|
||||
# Wait for results DCC - this is the long wait
|
||||
_emit_status(f"Connected to #{channel} - Waiting for results...", phase='searching')
|
||||
offer = client.wait_for_dcc(timeout=60.0, result_type=True)
|
||||
if not offer:
|
||||
logger.info("No search results received")
|
||||
_emit_status("No results found", phase='complete')
|
||||
client.disconnect()
|
||||
# Cache empty result to avoid repeated failed searches
|
||||
cache_results(
|
||||
book.provider,
|
||||
book.provider_id,
|
||||
book.title,
|
||||
[],
|
||||
list(self._online_servers) if self._online_servers else None
|
||||
)
|
||||
return []
|
||||
|
||||
# Download results file
|
||||
_emit_status(f"Connected to #{channel} - Downloading results...", phase='downloading')
|
||||
with tempfile.TemporaryDirectory() as tmpdir:
|
||||
result_path = Path(tmpdir) / offer.filename
|
||||
download_dcc(offer, result_path, timeout=30.0)
|
||||
|
||||
# Parse results
|
||||
if result_path.suffix.lower() == '.zip':
|
||||
content = extract_results_from_zip(result_path)
|
||||
else:
|
||||
content = result_path.read_text(errors='replace')
|
||||
|
||||
client.disconnect()
|
||||
|
||||
# Convert to Release objects
|
||||
results = parse_results_file(content)
|
||||
releases = self._convert_to_releases(results)
|
||||
|
||||
# Cache results
|
||||
cache_results(
|
||||
book.provider,
|
||||
book.provider_id,
|
||||
book.title,
|
||||
releases,
|
||||
list(self._online_servers) if self._online_servers else None
|
||||
)
|
||||
|
||||
return releases
|
||||
|
||||
except DCCError as e:
|
||||
logger.error(f"DCC error during search: {e}")
|
||||
_emit_status(f"DCC error: {e}", phase='error')
|
||||
if client:
|
||||
client.disconnect()
|
||||
return []
|
||||
except Exception as e:
|
||||
logger.error(f"IRC search failed: {e}")
|
||||
_emit_status(f"Search failed: {e}", phase='error')
|
||||
if client:
|
||||
client.disconnect()
|
||||
return []
|
||||
|
||||
def _build_query(self, book: BookMetadata) -> str:
|
||||
"""Build search query from book metadata."""
|
||||
parts = []
|
||||
|
||||
if book.title:
|
||||
parts.append(book.title)
|
||||
|
||||
if book.authors:
|
||||
# Use first author
|
||||
author = book.authors[0] if isinstance(book.authors, list) else book.authors
|
||||
parts.append(author)
|
||||
|
||||
return ' '.join(parts)
|
||||
|
||||
# Format priority for sorting (lower = higher priority)
|
||||
FORMAT_PRIORITY = {
|
||||
'epub': 0,
|
||||
'mobi': 1,
|
||||
'azw3': 2,
|
||||
'azw': 3,
|
||||
'fb2': 4,
|
||||
'djvu': 5,
|
||||
'pdf': 6,
|
||||
'cbr': 7,
|
||||
'cbz': 8,
|
||||
'doc': 9,
|
||||
'docx': 10,
|
||||
'rtf': 11,
|
||||
'txt': 12,
|
||||
'html': 13,
|
||||
'htm': 14,
|
||||
'rar': 15,
|
||||
'zip': 16,
|
||||
}
|
||||
|
||||
def _convert_to_releases(self, results: List[SearchResult]) -> List[Release]:
|
||||
"""Convert parsed results to Release objects, sorted by online/format/server."""
|
||||
releases = []
|
||||
online_servers = self._online_servers if self._online_servers else set()
|
||||
|
||||
for result in results:
|
||||
release = Release(
|
||||
source="irc",
|
||||
source_id=result.download_request, # Full line for download
|
||||
title=result.title,
|
||||
format=result.format,
|
||||
size=result.size,
|
||||
size_bytes=self._parse_size(result.size) if result.size else None,
|
||||
protocol=ReleaseProtocol.DCC,
|
||||
indexer=f"IRC:{result.server}",
|
||||
extra={
|
||||
"server": result.server,
|
||||
"author": result.author,
|
||||
"full_line": result.full_line,
|
||||
},
|
||||
)
|
||||
releases.append(release)
|
||||
|
||||
# Tiered sort: online first, then by format priority, then by server name
|
||||
def sort_key(release: Release) -> tuple:
|
||||
server = release.extra.get("server", "")
|
||||
is_online = server in online_servers
|
||||
fmt = release.format.lower() if release.format else ""
|
||||
format_priority = self.FORMAT_PRIORITY.get(fmt, 99)
|
||||
return (
|
||||
0 if is_online else 1, # Online first
|
||||
format_priority, # Then by format
|
||||
server.lower(), # Then alphabetically by server
|
||||
)
|
||||
|
||||
releases.sort(key=sort_key)
|
||||
|
||||
return releases
|
||||
|
||||
@staticmethod
|
||||
def _parse_size(size_str: str) -> Optional[int]:
|
||||
"""Parse human-readable size (e.g., '1.2MB', '500K') to bytes."""
|
||||
if not size_str:
|
||||
return None
|
||||
|
||||
size_str = size_str.strip().upper()
|
||||
|
||||
# Map suffixes to multipliers (check longer suffixes first)
|
||||
multipliers = [
|
||||
('GB', 1024 * 1024 * 1024),
|
||||
('MB', 1024 * 1024),
|
||||
('KB', 1024),
|
||||
('G', 1024 * 1024 * 1024),
|
||||
('M', 1024 * 1024),
|
||||
('K', 1024),
|
||||
('B', 1),
|
||||
]
|
||||
|
||||
for suffix, mult in multipliers:
|
||||
if size_str.endswith(suffix):
|
||||
try:
|
||||
num = float(size_str[:-len(suffix)].strip())
|
||||
return int(num * mult)
|
||||
except ValueError:
|
||||
return None
|
||||
|
||||
# Try parsing as plain number (bytes)
|
||||
try:
|
||||
return int(float(size_str))
|
||||
except ValueError:
|
||||
return None
|
||||
@@ -0,0 +1,26 @@
|
||||
"""
|
||||
Prowlarr release source plugin.
|
||||
|
||||
This plugin integrates with Prowlarr to search for book releases
|
||||
across multiple indexers (torrent and usenet).
|
||||
|
||||
Includes:
|
||||
- ProwlarrSource: Search integration with Prowlarr
|
||||
- ProwlarrHandler: Download handling via external clients
|
||||
- Download clients: qBittorrent (torrents), NZBGet (usenet)
|
||||
"""
|
||||
|
||||
# Import submodules to trigger decorator registration
|
||||
from shelfmark.release_sources.prowlarr import source # noqa: F401
|
||||
from shelfmark.release_sources.prowlarr import handler # noqa: F401
|
||||
from shelfmark.release_sources.prowlarr import settings # noqa: F401
|
||||
|
||||
# Import clients to trigger client registration
|
||||
# This is in a try/except to handle optional dependencies gracefully
|
||||
try:
|
||||
from shelfmark.release_sources.prowlarr import clients # noqa: F401
|
||||
except ImportError as e:
|
||||
# Log but don't fail - clients require optional dependencies
|
||||
import logging
|
||||
|
||||
logging.getLogger(__name__).debug(f"Prowlarr clients not loaded: {e}")
|
||||
@@ -0,0 +1,148 @@
|
||||
"""Prowlarr API client for connection testing, indexer listing, and search."""
|
||||
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
from urllib.parse import urljoin
|
||||
|
||||
import requests
|
||||
|
||||
from shelfmark.core.logger import setup_logger
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
|
||||
class ProwlarrClient:
|
||||
"""Client for interacting with the Prowlarr API."""
|
||||
|
||||
def __init__(self, url: str, api_key: str, timeout: int = 30):
|
||||
self.base_url = url.rstrip("/")
|
||||
self.api_key = api_key
|
||||
self.timeout = timeout
|
||||
self._session = requests.Session()
|
||||
self._session.headers.update({
|
||||
"X-Api-Key": api_key,
|
||||
"Accept": "application/json",
|
||||
})
|
||||
|
||||
def _request(
|
||||
self,
|
||||
method: str,
|
||||
endpoint: str,
|
||||
params: Optional[Dict[str, Any]] = None,
|
||||
json_data: Optional[Dict[str, Any]] = None,
|
||||
) -> Any:
|
||||
"""Make an API request to Prowlarr. Returns parsed JSON response."""
|
||||
url = urljoin(self.base_url, endpoint)
|
||||
logger.debug(f"Prowlarr API: {method} {url}")
|
||||
|
||||
try:
|
||||
response = self._session.request(
|
||||
method=method,
|
||||
url=url,
|
||||
params=params,
|
||||
json=json_data,
|
||||
timeout=self.timeout,
|
||||
)
|
||||
|
||||
if not response.ok:
|
||||
try:
|
||||
error_body = response.text[:500]
|
||||
logger.error(f"Prowlarr API error response: {error_body}")
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
response.raise_for_status()
|
||||
return response.json()
|
||||
|
||||
except requests.exceptions.JSONDecodeError as e:
|
||||
logger.error(f"Invalid JSON response from Prowlarr: {e}")
|
||||
raise ValueError(f"Invalid JSON response: {e}")
|
||||
except requests.exceptions.HTTPError as e:
|
||||
logger.error(f"Prowlarr API HTTP error: {e.response.status_code} {e.response.reason}")
|
||||
raise
|
||||
except requests.exceptions.RequestException as e:
|
||||
logger.error(f"Prowlarr API request failed: {e}")
|
||||
raise
|
||||
|
||||
def test_connection(self) -> Tuple[bool, str]:
|
||||
"""Test connection to Prowlarr. Returns (success, message)."""
|
||||
logger.info(f"Testing Prowlarr connection to: {self.base_url}")
|
||||
try:
|
||||
data = self._request("GET", "/api/v1/system/status")
|
||||
version = data.get("version", "unknown")
|
||||
logger.info(f"Prowlarr connection successful: version {version}")
|
||||
return True, f"Connected to Prowlarr {version}"
|
||||
except requests.exceptions.ConnectionError:
|
||||
return False, "Could not connect to Prowlarr. Check the URL."
|
||||
except requests.exceptions.HTTPError as e:
|
||||
status = e.response.status_code if e.response is not None else "unknown"
|
||||
if e.response is not None and e.response.status_code == 401:
|
||||
return False, "Invalid API key"
|
||||
return False, f"HTTP error {status}"
|
||||
except Exception as e:
|
||||
return False, f"Connection failed: {str(e)}"
|
||||
|
||||
def get_indexers(self) -> List[Dict[str, Any]]:
|
||||
"""Get all configured indexers."""
|
||||
try:
|
||||
indexers = self._request("GET", "/api/v1/indexer")
|
||||
return indexers
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to get indexers: {e}")
|
||||
return []
|
||||
|
||||
def get_enabled_indexers(self) -> List[Dict[str, Any]]:
|
||||
"""Get enabled indexers with book capability info."""
|
||||
indexers = self.get_indexers()
|
||||
result = []
|
||||
|
||||
for idx in indexers:
|
||||
if not idx.get("enable", False):
|
||||
continue
|
||||
|
||||
# Check for book categories (7000-7999 range)
|
||||
categories = idx.get("capabilities", {}).get("categories", [])
|
||||
has_books = self._has_book_categories(categories)
|
||||
|
||||
result.append({
|
||||
"id": idx.get("id"),
|
||||
"name": idx.get("name"),
|
||||
"protocol": idx.get("protocol"),
|
||||
"has_books": has_books,
|
||||
})
|
||||
|
||||
return result
|
||||
|
||||
def _has_book_categories(self, categories: List[Dict[str, Any]]) -> bool:
|
||||
"""Check if any category or subcategory is in the book range (7000-7999)."""
|
||||
for cat in categories:
|
||||
cat_id = cat.get("id", 0)
|
||||
if 7000 <= cat_id <= 7999:
|
||||
return True
|
||||
for subcat in cat.get("subCategories", []):
|
||||
if 7000 <= subcat.get("id", 0) <= 7999:
|
||||
return True
|
||||
return False
|
||||
|
||||
def search(
|
||||
self,
|
||||
query: str,
|
||||
indexer_ids: Optional[List[int]] = None,
|
||||
categories: Optional[List[int]] = None,
|
||||
limit: int = 100,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Search for releases via Prowlarr."""
|
||||
if not query:
|
||||
return []
|
||||
|
||||
params: Dict[str, Any] = {"query": query, "limit": limit}
|
||||
if indexer_ids:
|
||||
params["indexerIds"] = indexer_ids
|
||||
if categories:
|
||||
params["categories"] = categories
|
||||
|
||||
try:
|
||||
results = self._request("GET", "/api/v1/search", params=params)
|
||||
return results if isinstance(results, list) else []
|
||||
except Exception as e:
|
||||
logger.error(f"Prowlarr search failed: {e}")
|
||||
return []
|
||||
@@ -0,0 +1,113 @@
|
||||
"""
|
||||
Prowlarr release cache.
|
||||
|
||||
Stores search results so the handler can look up releases by source_id.
|
||||
This keeps all Prowlarr-specific data within the plugin.
|
||||
"""
|
||||
|
||||
import time
|
||||
from threading import Lock
|
||||
from typing import Dict, Optional
|
||||
|
||||
from shelfmark.core.logger import setup_logger
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
# Cache TTL in seconds (1 hour - releases should be downloaded within this time)
|
||||
RELEASE_CACHE_TTL = 3600
|
||||
|
||||
# Internal cache storage: source_id -> (release_dict, timestamp)
|
||||
_cache: Dict[str, tuple] = {}
|
||||
_cache_lock = Lock()
|
||||
|
||||
|
||||
def cache_release(source_id: str, release_data: dict) -> None:
|
||||
"""
|
||||
Cache a release by its source_id.
|
||||
|
||||
Args:
|
||||
source_id: The unique identifier for this release (GUID)
|
||||
release_data: The full Prowlarr API result dict
|
||||
"""
|
||||
with _cache_lock:
|
||||
_cache[source_id] = (release_data, time.time())
|
||||
|
||||
|
||||
def get_release(source_id: str) -> Optional[dict]:
|
||||
"""
|
||||
Get a cached release by source_id.
|
||||
|
||||
Args:
|
||||
source_id: The unique identifier for the release
|
||||
|
||||
Returns:
|
||||
The cached release dict, or None if not found or expired
|
||||
"""
|
||||
with _cache_lock:
|
||||
if source_id not in _cache:
|
||||
logger.debug(f"Prowlarr release not in cache: {source_id}")
|
||||
return None
|
||||
|
||||
release_data, cached_at = _cache[source_id]
|
||||
age = time.time() - cached_at
|
||||
|
||||
if age > RELEASE_CACHE_TTL:
|
||||
# Expired - remove from cache
|
||||
del _cache[source_id]
|
||||
logger.debug(f"Prowlarr release expired: {source_id}")
|
||||
return None
|
||||
|
||||
return release_data
|
||||
|
||||
|
||||
def remove_release(source_id: str) -> None:
|
||||
"""
|
||||
Remove a release from the cache (e.g., after successful download).
|
||||
|
||||
Args:
|
||||
source_id: The unique identifier for the release
|
||||
"""
|
||||
with _cache_lock:
|
||||
if source_id in _cache:
|
||||
del _cache[source_id]
|
||||
logger.debug(f"Removed Prowlarr release from cache: {source_id}")
|
||||
|
||||
|
||||
def cleanup_expired() -> int:
|
||||
"""
|
||||
Remove all expired entries from the cache.
|
||||
|
||||
Returns:
|
||||
Number of entries removed
|
||||
"""
|
||||
current_time = time.time()
|
||||
removed = 0
|
||||
|
||||
with _cache_lock:
|
||||
expired_ids = [
|
||||
source_id
|
||||
for source_id, (_, cached_at) in _cache.items()
|
||||
if current_time - cached_at > RELEASE_CACHE_TTL
|
||||
]
|
||||
for source_id in expired_ids:
|
||||
del _cache[source_id]
|
||||
removed += 1
|
||||
|
||||
if removed:
|
||||
logger.debug(f"Cleaned up {removed} expired Prowlarr cache entries")
|
||||
|
||||
return removed
|
||||
|
||||
|
||||
def get_cache_stats() -> dict:
|
||||
"""
|
||||
Get cache statistics for debugging.
|
||||
|
||||
Returns:
|
||||
Dict with cache stats
|
||||
"""
|
||||
with _cache_lock:
|
||||
return {
|
||||
"size": len(_cache),
|
||||
"entries": list(_cache.keys()),
|
||||
}
|
||||
@@ -0,0 +1,300 @@
|
||||
"""
|
||||
Download client infrastructure for Prowlarr integration.
|
||||
|
||||
This module provides:
|
||||
- DownloadState: Enum of valid download states
|
||||
- DownloadStatus: Status dataclass for external download progress
|
||||
- DownloadClient: Abstract base class for download clients
|
||||
- Client registry and factory functions
|
||||
|
||||
Clients register themselves via the @register_client decorator.
|
||||
"""
|
||||
|
||||
import logging
|
||||
from abc import ABC, abstractmethod
|
||||
from dataclasses import dataclass
|
||||
from enum import Enum
|
||||
from typing import Dict, List, Optional, Tuple, Type, Union
|
||||
|
||||
_logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class DownloadState(Enum):
|
||||
"""Valid states for a download."""
|
||||
|
||||
DOWNLOADING = "downloading"
|
||||
COMPLETE = "complete"
|
||||
ERROR = "error"
|
||||
SEEDING = "seeding"
|
||||
PAUSED = "paused"
|
||||
QUEUED = "queued"
|
||||
CHECKING = "checking"
|
||||
PROCESSING = "processing"
|
||||
UNKNOWN = "unknown"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class DownloadStatus:
|
||||
"""Status of an external download (immutable)."""
|
||||
|
||||
progress: float # 0-100
|
||||
state: Union[DownloadState, str] # Prefer DownloadState enum; strings auto-normalized
|
||||
message: Optional[str] # Status message
|
||||
complete: bool # True when download finished
|
||||
file_path: Optional[str] # Path in client's download dir (when complete)
|
||||
download_speed: Optional[int] = None # Bytes per second
|
||||
eta: Optional[int] = None # Seconds remaining
|
||||
|
||||
@classmethod
|
||||
def error(cls, message: str) -> "DownloadStatus":
|
||||
"""Create an error status."""
|
||||
return cls(
|
||||
progress=0,
|
||||
state=DownloadState.ERROR,
|
||||
message=message,
|
||||
complete=False,
|
||||
file_path=None,
|
||||
)
|
||||
|
||||
def __post_init__(self):
|
||||
"""Validate and normalize state."""
|
||||
# Normalize string states to enum
|
||||
if isinstance(self.state, str):
|
||||
try:
|
||||
normalized_state = DownloadState(self.state)
|
||||
object.__setattr__(self, 'state', normalized_state)
|
||||
except ValueError:
|
||||
# Unknown state string - keep as-is for backwards compatibility
|
||||
_logger.warning(f"Unknown download state '{self.state}', keeping as string")
|
||||
|
||||
# Validate progress is in range
|
||||
if not 0 <= self.progress <= 100:
|
||||
_logger.debug(f"Progress {self.progress} out of range, clamping to [0, 100]")
|
||||
object.__setattr__(self, 'progress', max(0, min(100, self.progress)))
|
||||
|
||||
@property
|
||||
def state_value(self) -> str:
|
||||
"""Get the state as a string value (for JSON serialization)."""
|
||||
if isinstance(self.state, DownloadState):
|
||||
return self.state.value
|
||||
return self.state
|
||||
|
||||
|
||||
class DownloadClient(ABC):
|
||||
"""
|
||||
Base class for external download clients.
|
||||
|
||||
Subclasses implement protocol-specific download management:
|
||||
- Torrent clients: qBittorrent, Transmission, Deluge
|
||||
- Usenet clients: NZBGet, SABnzbd
|
||||
|
||||
Subclasses must define:
|
||||
- protocol: "torrent" or "usenet"
|
||||
- name: Unique client identifier (e.g., "qbittorrent", "nzbget")
|
||||
"""
|
||||
|
||||
# Class attributes that subclasses must define
|
||||
protocol: str
|
||||
name: str
|
||||
|
||||
def __init_subclass__(cls, **kwargs):
|
||||
"""Validate that subclasses define required class attributes."""
|
||||
super().__init_subclass__(**kwargs)
|
||||
|
||||
# Skip validation for abstract subclasses
|
||||
if ABC in cls.__bases__:
|
||||
return
|
||||
|
||||
# Validate protocol attribute
|
||||
if not hasattr(cls, 'protocol') or not cls.protocol:
|
||||
raise TypeError(f"{cls.__name__} must define 'protocol' class attribute")
|
||||
if cls.protocol not in ('torrent', 'usenet'):
|
||||
raise TypeError(
|
||||
f"{cls.__name__}.protocol must be 'torrent' or 'usenet', got '{cls.protocol}'"
|
||||
)
|
||||
|
||||
# Validate name attribute
|
||||
if not hasattr(cls, 'name') or not cls.name:
|
||||
raise TypeError(f"{cls.__name__} must define 'name' class attribute")
|
||||
|
||||
@staticmethod
|
||||
@abstractmethod
|
||||
def is_configured() -> bool:
|
||||
"""
|
||||
Check if this client is configured.
|
||||
|
||||
Returns:
|
||||
True if required settings (URL, etc.) are present.
|
||||
"""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def test_connection(self) -> Tuple[bool, str]:
|
||||
"""
|
||||
Test connectivity to the client.
|
||||
|
||||
Returns:
|
||||
Tuple of (success, message).
|
||||
"""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def add_download(self, url: str, name: str, category: str = "cwabd") -> str:
|
||||
"""
|
||||
Add a download to the client.
|
||||
|
||||
Args:
|
||||
url: Download URL (magnet link, .torrent URL, or NZB URL)
|
||||
name: Display name for the download
|
||||
category: Category/label for organization
|
||||
|
||||
Returns:
|
||||
Client-specific download ID (hash for torrents, ID for NZBGet).
|
||||
|
||||
Raises:
|
||||
Exception: If adding fails.
|
||||
"""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def get_status(self, download_id: str) -> DownloadStatus:
|
||||
"""
|
||||
Get status of a download.
|
||||
|
||||
Args:
|
||||
download_id: The ID returned by add_download()
|
||||
|
||||
Returns:
|
||||
Current download status.
|
||||
"""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def remove(self, download_id: str, delete_files: bool = False) -> bool:
|
||||
"""
|
||||
Remove a download from the client.
|
||||
|
||||
Args:
|
||||
download_id: The ID returned by add_download()
|
||||
delete_files: Whether to also delete downloaded files
|
||||
|
||||
Returns:
|
||||
True if removal succeeded.
|
||||
"""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def get_download_path(self, download_id: str) -> Optional[str]:
|
||||
"""
|
||||
Get the path where files were downloaded.
|
||||
|
||||
Args:
|
||||
download_id: The ID returned by add_download()
|
||||
|
||||
Returns:
|
||||
File or directory path, or None if not available.
|
||||
"""
|
||||
pass
|
||||
|
||||
def find_existing(self, url: str) -> Optional[Tuple[str, DownloadStatus]]:
|
||||
"""
|
||||
Check if a download for this URL already exists in the client.
|
||||
|
||||
This is useful for detecting already-completed downloads so we can
|
||||
skip re-downloading and just copy the existing file.
|
||||
|
||||
Args:
|
||||
url: Download URL (magnet link, .torrent URL, or NZB URL)
|
||||
|
||||
Returns:
|
||||
Tuple of (download_id, status) if found, None if not found.
|
||||
Default implementation returns None.
|
||||
"""
|
||||
return None
|
||||
|
||||
|
||||
# Client registry: protocol -> list of client classes
|
||||
_CLIENTS: Dict[str, List[Type[DownloadClient]]] = {}
|
||||
|
||||
|
||||
def register_client(protocol: str):
|
||||
"""
|
||||
Decorator to register a download client for a protocol.
|
||||
|
||||
Multiple clients can be registered for the same protocol.
|
||||
The `is_configured()` method determines which one is active.
|
||||
|
||||
Args:
|
||||
protocol: The protocol this client handles ("torrent" or "usenet")
|
||||
|
||||
Example:
|
||||
@register_client("torrent")
|
||||
class QBittorrentClient(DownloadClient):
|
||||
...
|
||||
"""
|
||||
|
||||
def decorator(cls: Type[DownloadClient]) -> Type[DownloadClient]:
|
||||
if protocol not in _CLIENTS:
|
||||
_CLIENTS[protocol] = []
|
||||
_CLIENTS[protocol].append(cls)
|
||||
return cls
|
||||
|
||||
return decorator
|
||||
|
||||
|
||||
def get_client(protocol: str) -> Optional[DownloadClient]:
|
||||
"""
|
||||
Get a configured client instance for the given protocol.
|
||||
|
||||
Iterates through all registered clients for the protocol and
|
||||
returns the first one that is configured.
|
||||
|
||||
Args:
|
||||
protocol: "torrent" or "usenet"
|
||||
|
||||
Returns:
|
||||
Configured client instance, or None if not available/configured.
|
||||
"""
|
||||
if protocol not in _CLIENTS:
|
||||
return None
|
||||
|
||||
for client_cls in _CLIENTS[protocol]:
|
||||
if client_cls.is_configured():
|
||||
return client_cls()
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def list_configured_clients() -> List[str]:
|
||||
"""
|
||||
List protocols that have configured clients.
|
||||
|
||||
Returns:
|
||||
List of protocol names (e.g., ["torrent", "usenet"]).
|
||||
"""
|
||||
result = []
|
||||
for protocol, client_classes in _CLIENTS.items():
|
||||
for cls in client_classes:
|
||||
if cls.is_configured():
|
||||
result.append(protocol)
|
||||
break
|
||||
return result
|
||||
|
||||
|
||||
def get_all_clients() -> Dict[str, List[Type[DownloadClient]]]:
|
||||
"""
|
||||
Get all registered client classes.
|
||||
|
||||
Returns:
|
||||
Dict of protocol -> list of client classes.
|
||||
"""
|
||||
return dict(_CLIENTS)
|
||||
|
||||
|
||||
# Import client implementations to trigger registration
|
||||
# These imports are at the bottom to avoid circular imports
|
||||
from shelfmark.release_sources.prowlarr.clients import qbittorrent # noqa: F401, E402
|
||||
from shelfmark.release_sources.prowlarr.clients import nzbget # noqa: F401, E402
|
||||
from shelfmark.release_sources.prowlarr.clients import sabnzbd # noqa: F401, E402
|
||||
from shelfmark.release_sources.prowlarr.clients import transmission # noqa: F401, E402
|
||||
from shelfmark.release_sources.prowlarr.clients import deluge # noqa: F401, E402
|
||||
@@ -0,0 +1,308 @@
|
||||
"""
|
||||
Deluge download client for Prowlarr integration.
|
||||
|
||||
Uses the deluge-client library to communicate with Deluge's RPC daemon.
|
||||
Note: Deluge uses a custom binary RPC protocol over TCP (default port 58846,
|
||||
configurable via DELUGE_PORT), which requires the daemon to have
|
||||
"Allow Remote Connections" enabled.
|
||||
"""
|
||||
|
||||
import base64
|
||||
from typing import Any, Optional, Tuple
|
||||
|
||||
from shelfmark.core.config import config
|
||||
from shelfmark.core.logger import setup_logger
|
||||
from shelfmark.release_sources.prowlarr.clients import (
|
||||
DownloadClient,
|
||||
DownloadStatus,
|
||||
register_client,
|
||||
)
|
||||
from shelfmark.release_sources.prowlarr.clients.torrent_utils import (
|
||||
extract_torrent_info,
|
||||
)
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
|
||||
def _decode(value: Any) -> Any:
|
||||
"""Decode bytes to string if needed (Deluge returns bytes for strings)."""
|
||||
return value.decode('utf-8') if isinstance(value, bytes) else value
|
||||
|
||||
|
||||
@register_client("torrent")
|
||||
class DelugeClient(DownloadClient):
|
||||
"""Deluge download client using deluge-client RPC library."""
|
||||
|
||||
protocol = "torrent"
|
||||
name = "deluge"
|
||||
|
||||
def __init__(self):
|
||||
"""Initialize Deluge client with settings from config."""
|
||||
from deluge_client import DelugeRPCClient
|
||||
|
||||
host = config.get("DELUGE_HOST", "localhost")
|
||||
password = config.get("DELUGE_PASSWORD", "")
|
||||
|
||||
if not host:
|
||||
raise ValueError("DELUGE_HOST is required")
|
||||
if not password:
|
||||
raise ValueError("DELUGE_PASSWORD is required")
|
||||
|
||||
port = int(config.get("DELUGE_PORT", "58846"))
|
||||
username = config.get("DELUGE_USERNAME", "")
|
||||
|
||||
self._client = DelugeRPCClient(
|
||||
host=host,
|
||||
port=port,
|
||||
username=username,
|
||||
password=password,
|
||||
)
|
||||
self._connected = False
|
||||
self._category = config.get("DELUGE_CATEGORY", "cwabd")
|
||||
|
||||
def _ensure_connected(self):
|
||||
"""Ensure we're connected to the Deluge daemon."""
|
||||
if not self._connected:
|
||||
logger.debug("Connecting to Deluge daemon...")
|
||||
try:
|
||||
self._client.connect()
|
||||
self._connected = True
|
||||
logger.debug("Connected to Deluge daemon")
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to connect to Deluge daemon: {type(e).__name__}: {e}")
|
||||
raise
|
||||
|
||||
@staticmethod
|
||||
def is_configured() -> bool:
|
||||
"""Check if Deluge is configured and selected as the torrent client."""
|
||||
client = config.get("PROWLARR_TORRENT_CLIENT", "")
|
||||
host = config.get("DELUGE_HOST", "")
|
||||
password = config.get("DELUGE_PASSWORD", "")
|
||||
return client == "deluge" and bool(host) and bool(password)
|
||||
|
||||
def test_connection(self) -> Tuple[bool, str]:
|
||||
"""Test connection to Deluge."""
|
||||
try:
|
||||
self._ensure_connected()
|
||||
# Get daemon info
|
||||
version = self._client.call('daemon.info')
|
||||
return True, f"Connected to Deluge {version}"
|
||||
except Exception as e:
|
||||
self._connected = False
|
||||
return False, f"Connection failed: {str(e)}"
|
||||
|
||||
def add_download(self, url: str, name: str, category: str = None) -> str:
|
||||
"""
|
||||
Add torrent by URL (magnet or .torrent).
|
||||
|
||||
Args:
|
||||
url: Magnet link or .torrent URL
|
||||
name: Display name for the torrent
|
||||
category: Category for organization (uses configured default if not specified)
|
||||
|
||||
Returns:
|
||||
Torrent hash (info_hash).
|
||||
|
||||
Raises:
|
||||
Exception: If adding fails.
|
||||
"""
|
||||
try:
|
||||
self._ensure_connected()
|
||||
|
||||
category = category or self._category
|
||||
|
||||
torrent_info = extract_torrent_info(url)
|
||||
if not torrent_info.is_magnet and not torrent_info.torrent_data:
|
||||
raise Exception("Failed to fetch torrent file")
|
||||
|
||||
options = {}
|
||||
|
||||
if torrent_info.is_magnet:
|
||||
# Use magnet URL if available, otherwise original URL
|
||||
magnet_url = torrent_info.magnet_url or url
|
||||
torrent_id = self._client.call(
|
||||
'core.add_torrent_magnet',
|
||||
magnet_url,
|
||||
options,
|
||||
)
|
||||
else:
|
||||
filedump = base64.b64encode(torrent_info.torrent_data).decode('ascii')
|
||||
torrent_id = self._client.call(
|
||||
'core.add_torrent_file',
|
||||
f"{name}.torrent",
|
||||
filedump,
|
||||
options,
|
||||
)
|
||||
|
||||
if torrent_id:
|
||||
torrent_id = _decode(torrent_id)
|
||||
logger.info(f"Added torrent to Deluge: {torrent_id}")
|
||||
return torrent_id.lower()
|
||||
|
||||
raise Exception("Deluge returned no torrent ID")
|
||||
|
||||
except Exception as e:
|
||||
self._connected = False
|
||||
logger.error(f"Deluge add failed: {e}")
|
||||
raise
|
||||
|
||||
def get_status(self, download_id: str) -> DownloadStatus:
|
||||
"""
|
||||
Get torrent status by hash.
|
||||
|
||||
Args:
|
||||
download_id: Torrent info_hash
|
||||
|
||||
Returns:
|
||||
Current download status.
|
||||
"""
|
||||
try:
|
||||
self._ensure_connected()
|
||||
|
||||
# Get torrent status
|
||||
status = self._client.call(
|
||||
'core.get_torrent_status',
|
||||
download_id,
|
||||
['state', 'progress', 'download_payload_rate', 'eta', 'save_path', 'name'],
|
||||
)
|
||||
|
||||
if not status:
|
||||
return DownloadStatus.error("Torrent not found")
|
||||
|
||||
# Deluge states: Downloading, Seeding, Paused, Checking, Queued, Error, Moving
|
||||
state_map = {
|
||||
'Downloading': ('downloading', None),
|
||||
'Seeding': ('seeding', 'Seeding'),
|
||||
'Paused': ('paused', 'Paused'),
|
||||
'Checking': ('checking', 'Checking files'),
|
||||
'Queued': ('queued', 'Queued'),
|
||||
'Error': ('error', 'Error'),
|
||||
'Moving': ('processing', 'Moving files'),
|
||||
'Allocating': ('downloading', 'Allocating space'),
|
||||
}
|
||||
|
||||
deluge_state = _decode(status.get(b'state', b'Unknown'))
|
||||
state, message = state_map.get(deluge_state, ('unknown', deluge_state))
|
||||
progress = status.get(b'progress', 0)
|
||||
complete = progress >= 100
|
||||
|
||||
if complete:
|
||||
message = "Complete"
|
||||
|
||||
eta = status.get(b'eta')
|
||||
if eta and eta > 604800:
|
||||
eta = None
|
||||
|
||||
file_path = None
|
||||
if complete:
|
||||
save_path = _decode(status.get(b'save_path', b''))
|
||||
name = _decode(status.get(b'name', b''))
|
||||
if save_path and name:
|
||||
file_path = f"{save_path}/{name}"
|
||||
|
||||
return DownloadStatus(
|
||||
progress=progress,
|
||||
state="complete" if complete else state,
|
||||
message=message,
|
||||
complete=complete,
|
||||
file_path=file_path,
|
||||
download_speed=status.get(b'download_payload_rate'),
|
||||
eta=eta,
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
self._connected = False
|
||||
error_type = type(e).__name__
|
||||
logger.error(f"Deluge get_status failed ({error_type}): {e}")
|
||||
return DownloadStatus.error(f"{error_type}: {e}")
|
||||
|
||||
def remove(self, download_id: str, delete_files: bool = False) -> bool:
|
||||
"""
|
||||
Remove a torrent from Deluge.
|
||||
|
||||
Args:
|
||||
download_id: Torrent info_hash
|
||||
delete_files: Whether to also delete files
|
||||
|
||||
Returns:
|
||||
True if successful.
|
||||
"""
|
||||
try:
|
||||
self._ensure_connected()
|
||||
|
||||
result = self._client.call(
|
||||
'core.remove_torrent',
|
||||
download_id,
|
||||
delete_files,
|
||||
)
|
||||
|
||||
if result:
|
||||
logger.info(
|
||||
f"Removed torrent from Deluge: {download_id}"
|
||||
+ (" (with files)" if delete_files else "")
|
||||
)
|
||||
return True
|
||||
return False
|
||||
|
||||
except Exception as e:
|
||||
self._connected = False
|
||||
error_type = type(e).__name__
|
||||
logger.error(f"Deluge remove failed ({error_type}): {e}")
|
||||
return False
|
||||
|
||||
def get_download_path(self, download_id: str) -> Optional[str]:
|
||||
"""
|
||||
Get the path where torrent files are located.
|
||||
|
||||
Args:
|
||||
download_id: Torrent info_hash
|
||||
|
||||
Returns:
|
||||
Content path (file or directory), or None.
|
||||
"""
|
||||
try:
|
||||
self._ensure_connected()
|
||||
|
||||
status = self._client.call(
|
||||
'core.get_torrent_status',
|
||||
download_id,
|
||||
['save_path', 'name'],
|
||||
)
|
||||
|
||||
if status:
|
||||
save_path = _decode(status.get(b'save_path', b''))
|
||||
name = _decode(status.get(b'name', b''))
|
||||
if save_path and name:
|
||||
return f"{save_path}/{name}"
|
||||
return None
|
||||
|
||||
except Exception as e:
|
||||
self._connected = False
|
||||
error_type = type(e).__name__
|
||||
logger.debug(f"Deluge get_download_path failed ({error_type}): {e}")
|
||||
return None
|
||||
|
||||
def find_existing(self, url: str) -> Optional[Tuple[str, DownloadStatus]]:
|
||||
"""Check if a torrent for this URL already exists in Deluge."""
|
||||
try:
|
||||
self._ensure_connected()
|
||||
|
||||
torrent_info = extract_torrent_info(url)
|
||||
if not torrent_info.info_hash:
|
||||
return None
|
||||
|
||||
status = self._client.call(
|
||||
'core.get_torrent_status',
|
||||
torrent_info.info_hash,
|
||||
['state'],
|
||||
)
|
||||
|
||||
if status:
|
||||
full_status = self.get_status(torrent_info.info_hash)
|
||||
return (torrent_info.info_hash, full_status)
|
||||
|
||||
return None
|
||||
except Exception as e:
|
||||
self._connected = False
|
||||
logger.debug(f"Error checking for existing torrent: {e}")
|
||||
return None
|
||||
@@ -0,0 +1,291 @@
|
||||
"""
|
||||
NZBGet download client for Prowlarr integration.
|
||||
|
||||
Uses NZBGet's JSON-RPC API directly via requests (no external dependency).
|
||||
"""
|
||||
|
||||
import json
|
||||
from typing import Any, Optional, Tuple
|
||||
|
||||
import requests
|
||||
|
||||
from shelfmark.core.config import config
|
||||
from shelfmark.core.logger import setup_logger
|
||||
from shelfmark.release_sources.prowlarr.clients import (
|
||||
DownloadClient,
|
||||
DownloadStatus,
|
||||
register_client,
|
||||
)
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
|
||||
@register_client("usenet")
|
||||
class NZBGetClient(DownloadClient):
|
||||
"""NZBGet download client using JSON-RPC API."""
|
||||
|
||||
protocol = "usenet"
|
||||
name = "nzbget"
|
||||
|
||||
def __init__(self):
|
||||
"""Initialize NZBGet client with settings from config."""
|
||||
url = config.get("NZBGET_URL", "")
|
||||
if not url:
|
||||
raise ValueError("NZBGET_URL is required")
|
||||
|
||||
self.url = url.rstrip("/")
|
||||
self.username = config.get("NZBGET_USERNAME", "nzbget")
|
||||
self.password = config.get("NZBGET_PASSWORD", "")
|
||||
self._category = config.get("NZBGET_CATEGORY", "Books")
|
||||
|
||||
@staticmethod
|
||||
def is_configured() -> bool:
|
||||
"""Check if NZBGet is configured and selected as the usenet client."""
|
||||
client = config.get("PROWLARR_USENET_CLIENT", "")
|
||||
url = config.get("NZBGET_URL", "")
|
||||
return client == "nzbget" and bool(url)
|
||||
|
||||
def _rpc_call(self, method: str, params: list = None) -> Any:
|
||||
"""
|
||||
Make a JSON-RPC call to NZBGet.
|
||||
|
||||
Args:
|
||||
method: RPC method name
|
||||
params: Method parameters
|
||||
|
||||
Returns:
|
||||
Result from NZBGet.
|
||||
|
||||
Raises:
|
||||
Exception: If RPC call fails.
|
||||
"""
|
||||
rpc_url = f"{self.url}/jsonrpc"
|
||||
|
||||
payload = json.dumps({
|
||||
"jsonrpc": "2.0",
|
||||
"id": 1,
|
||||
"method": method,
|
||||
"params": params or [],
|
||||
}, separators=(',', ':'))
|
||||
|
||||
response = requests.post(
|
||||
rpc_url,
|
||||
data=payload,
|
||||
headers={"Content-Type": "application/json"},
|
||||
auth=(self.username, self.password),
|
||||
timeout=30,
|
||||
)
|
||||
response.raise_for_status()
|
||||
|
||||
result = response.json()
|
||||
if "error" in result and result["error"]:
|
||||
raise Exception(result["error"].get("message", "RPC error"))
|
||||
|
||||
return result.get("result")
|
||||
|
||||
def test_connection(self) -> Tuple[bool, str]:
|
||||
"""Test connection to NZBGet."""
|
||||
try:
|
||||
status = self._rpc_call("status")
|
||||
version = status.get("Version", "unknown")
|
||||
return True, f"Connected to NZBGet {version}"
|
||||
except requests.exceptions.ConnectionError:
|
||||
return False, "Could not connect to NZBGet"
|
||||
except requests.exceptions.Timeout:
|
||||
return False, "Connection timed out"
|
||||
except Exception as e:
|
||||
return False, f"Connection failed: {str(e)}"
|
||||
|
||||
def add_download(self, url: str, name: str, category: str = None) -> str:
|
||||
"""
|
||||
Add NZB by URL.
|
||||
|
||||
Fetches the NZB content from the URL (e.g., Prowlarr proxy) and sends
|
||||
it base64-encoded to NZBGet, since NZBGet may not handle redirects well.
|
||||
|
||||
Args:
|
||||
url: NZB URL (can be Prowlarr proxy URL)
|
||||
name: Display name for the download
|
||||
category: Category for organization (uses configured default if not specified)
|
||||
|
||||
Returns:
|
||||
NZBGet download ID (NZBID).
|
||||
|
||||
Raises:
|
||||
Exception: If adding fails.
|
||||
"""
|
||||
import base64
|
||||
|
||||
# Use configured category if not explicitly provided
|
||||
category = category or self._category
|
||||
|
||||
try:
|
||||
# Fetch NZB content from the URL (handles Prowlarr proxy redirects)
|
||||
logger.debug(f"Fetching NZB from: {url}")
|
||||
response = requests.get(url, timeout=30)
|
||||
response.raise_for_status()
|
||||
nzb_content = base64.b64encode(response.content).decode('ascii')
|
||||
|
||||
# Ensure filename has .nzb extension
|
||||
nzb_filename = name if name.endswith('.nzb') else f"{name}.nzb"
|
||||
|
||||
# NZBGet append method parameters (all 10 required):
|
||||
# NZBFilename, Content, Category, Priority, AddToTop, AddPaused,
|
||||
# DupeKey, DupeScore, DupeMode, PPParameters
|
||||
nzb_id = self._rpc_call(
|
||||
"append",
|
||||
[
|
||||
nzb_filename, # NZBFilename
|
||||
nzb_content, # Content (base64-encoded NZB)
|
||||
category, # Category
|
||||
0, # Priority (0 = normal)
|
||||
False, # AddToTop
|
||||
False, # AddPaused
|
||||
"", # DupeKey
|
||||
0, # DupeScore
|
||||
"SCORE", # DupeMode
|
||||
[], # PPParameters (empty array)
|
||||
],
|
||||
)
|
||||
|
||||
if nzb_id and nzb_id > 0:
|
||||
logger.info(f"Added NZB to NZBGet: {nzb_id}")
|
||||
return str(nzb_id)
|
||||
|
||||
raise Exception("NZBGet returned invalid ID")
|
||||
except requests.RequestException as e:
|
||||
logger.error(f"Failed to fetch NZB from URL: {e}")
|
||||
raise Exception(f"Failed to fetch NZB: {e}")
|
||||
except Exception as e:
|
||||
logger.error(f"NZBGet add failed: {e}")
|
||||
raise
|
||||
|
||||
def get_status(self, download_id: str) -> DownloadStatus:
|
||||
"""
|
||||
Get NZB status by ID.
|
||||
|
||||
Args:
|
||||
download_id: NZBGet NZBID
|
||||
|
||||
Returns:
|
||||
Current download status.
|
||||
"""
|
||||
try:
|
||||
nzb_id = int(download_id)
|
||||
|
||||
# Check active downloads (queue)
|
||||
groups = self._rpc_call("listgroups", [0])
|
||||
|
||||
for group in groups:
|
||||
if group.get("NZBID") == nzb_id:
|
||||
# Calculate progress
|
||||
# NZBGet uses Hi/Lo for 64-bit values on 32-bit systems
|
||||
file_size = (group.get("FileSizeHi", 0) << 32) + group.get(
|
||||
"FileSizeLo", 0
|
||||
)
|
||||
remaining = (group.get("RemainingSizeHi", 0) << 32) + group.get(
|
||||
"RemainingSizeLo", 0
|
||||
)
|
||||
|
||||
progress = (
|
||||
((file_size - remaining) / file_size * 100)
|
||||
if file_size > 0
|
||||
else 0
|
||||
)
|
||||
status = group.get("Status", "")
|
||||
|
||||
# Map NZBGet status to our states
|
||||
if "DOWNLOADING" in status:
|
||||
state = "downloading"
|
||||
elif "PAUSED" in status:
|
||||
state = "paused"
|
||||
elif "QUEUED" in status:
|
||||
state = "queued"
|
||||
elif "POST-PROCESSING" in status or "UNPACKING" in status:
|
||||
state = "processing"
|
||||
else:
|
||||
state = "unknown"
|
||||
|
||||
return DownloadStatus(
|
||||
progress=progress,
|
||||
state=state,
|
||||
message=status.replace("-", " ").title(),
|
||||
complete=False,
|
||||
file_path=None,
|
||||
download_speed=group.get("DownloadRate"),
|
||||
eta=(
|
||||
group.get("RemainingSec")
|
||||
if group.get("RemainingSec", 0) > 0
|
||||
else None
|
||||
),
|
||||
)
|
||||
|
||||
# Check history for completed downloads
|
||||
history = self._rpc_call("history", [False])
|
||||
|
||||
for item in history:
|
||||
if item.get("NZBID") == nzb_id:
|
||||
status = item.get("Status", "")
|
||||
dest_dir = item.get("DestDir", "")
|
||||
|
||||
if "SUCCESS" in status:
|
||||
return DownloadStatus(
|
||||
progress=100,
|
||||
state="complete",
|
||||
message="Complete",
|
||||
complete=True,
|
||||
file_path=dest_dir,
|
||||
)
|
||||
else:
|
||||
return DownloadStatus(
|
||||
progress=100,
|
||||
state="error",
|
||||
message=f"Download failed: {status}",
|
||||
complete=True,
|
||||
file_path=None,
|
||||
)
|
||||
|
||||
# Not found in queue or history
|
||||
return DownloadStatus.error("Download not found")
|
||||
except Exception as e:
|
||||
error_type = type(e).__name__
|
||||
logger.error(f"NZBGet get_status failed ({error_type}): {e}")
|
||||
return DownloadStatus.error(f"{error_type}: {e}")
|
||||
|
||||
def remove(self, download_id: str, delete_files: bool = False) -> bool:
|
||||
"""
|
||||
Remove a download from NZBGet.
|
||||
|
||||
Args:
|
||||
download_id: NZBGet NZBID
|
||||
delete_files: Whether to permanently delete (vs move to history)
|
||||
|
||||
Returns:
|
||||
True if successful.
|
||||
"""
|
||||
try:
|
||||
nzb_id = int(download_id)
|
||||
# editqueue params: Command (str), Param (str), IDs (int[])
|
||||
# GroupFinalDelete = permanent removal, GroupDelete = move to history
|
||||
command = "GroupFinalDelete" if delete_files else "GroupDelete"
|
||||
result = self._rpc_call("editqueue", [command, "", [nzb_id]])
|
||||
if result:
|
||||
logger.info(f"Removed NZB from NZBGet: {download_id}")
|
||||
return bool(result)
|
||||
except Exception as e:
|
||||
error_type = type(e).__name__
|
||||
logger.error(f"NZBGet remove failed ({error_type}): {e}")
|
||||
return False
|
||||
|
||||
def get_download_path(self, download_id: str) -> Optional[str]:
|
||||
"""
|
||||
Get the path where NZB files are located.
|
||||
|
||||
Args:
|
||||
download_id: NZBGet NZBID
|
||||
|
||||
Returns:
|
||||
Destination directory, or None.
|
||||
"""
|
||||
status = self.get_status(download_id)
|
||||
return status.file_path
|
||||
@@ -0,0 +1,308 @@
|
||||
"""qBittorrent download client for Prowlarr integration."""
|
||||
|
||||
import time
|
||||
from types import SimpleNamespace
|
||||
from typing import List, Optional, Tuple
|
||||
|
||||
from shelfmark.core.config import config
|
||||
from shelfmark.core.logger import setup_logger
|
||||
from shelfmark.release_sources.prowlarr.clients import (
|
||||
DownloadClient,
|
||||
DownloadStatus,
|
||||
register_client,
|
||||
)
|
||||
from shelfmark.release_sources.prowlarr.clients.torrent_utils import (
|
||||
extract_torrent_info,
|
||||
)
|
||||
|
||||
logger = setup_logger(__name__)
|
||||
|
||||
|
||||
def _hashes_match(hash1: str, hash2: str) -> bool:
|
||||
"""Compare hashes, handling Amarr's 40-char zero-padded hashes vs 32-char ed2k hashes."""
|
||||
h1, h2 = hash1.lower(), hash2.lower()
|
||||
if h1 == h2:
|
||||
return True
|
||||
if len(h1) == 40 and len(h2) == 32 and h1.endswith("00000000"):
|
||||
return h1[:32] == h2
|
||||
if len(h2) == 40 and len(h1) == 32 and h2.endswith("00000000"):
|
||||
return h2[:32] == h1
|
||||
return False
|
||||
|
||||
|
||||
@register_client("torrent")
|
||||
class QBittorrentClient(DownloadClient):
|
||||
"""qBittorrent download client."""
|
||||
|
||||
protocol = "torrent"
|
||||
name = "qbittorrent"
|
||||
|
||||
def __init__(self):
|
||||
"""Initialize qBittorrent client with settings from config."""
|
||||
# Lazy import to avoid dependency issues if not using torrents
|
||||
from qbittorrentapi import Client
|
||||
|
||||
url = config.get("QBITTORRENT_URL", "")
|
||||
if not url:
|
||||
raise ValueError("QBITTORRENT_URL is required")
|
||||
|
||||
self._base_url = url.rstrip("/")
|
||||
self._client = Client(
|
||||
host=url,
|
||||
username=config.get("QBITTORRENT_USERNAME", ""),
|
||||
password=config.get("QBITTORRENT_PASSWORD", ""),
|
||||
)
|
||||
self._category = config.get("QBITTORRENT_CATEGORY", "cwabd")
|
||||
|
||||
def _get_torrents_info(self, torrent_hash: Optional[str] = None) -> List:
|
||||
"""Get torrent info using GET (per API spec for read operations)."""
|
||||
import requests
|
||||
|
||||
try:
|
||||
# Ensure session is authenticated before using it directly
|
||||
self._client.auth_log_in()
|
||||
|
||||
params = {"hashes": torrent_hash} if torrent_hash else {}
|
||||
response = self._client._session.get(
|
||||
f"{self._base_url}/api/v2/torrents/info",
|
||||
params=params,
|
||||
timeout=10,
|
||||
)
|
||||
response.raise_for_status()
|
||||
torrents = response.json()
|
||||
return [SimpleNamespace(**t) for t in torrents]
|
||||
except requests.exceptions.HTTPError as e:
|
||||
if e.response is not None and e.response.status_code == 403:
|
||||
logger.warning("qBittorrent auth failed - check credentials")
|
||||
else:
|
||||
logger.warning(f"qBittorrent API error: {e}")
|
||||
return []
|
||||
except requests.exceptions.ConnectionError:
|
||||
logger.warning(f"Cannot connect to qBittorrent at {self._base_url}")
|
||||
return []
|
||||
except Exception as e:
|
||||
logger.debug(f"Failed to get torrents info: {e}")
|
||||
return []
|
||||
|
||||
@staticmethod
|
||||
def is_configured() -> bool:
|
||||
"""Check if qBittorrent is configured and selected as the torrent client."""
|
||||
client = config.get("PROWLARR_TORRENT_CLIENT", "")
|
||||
url = config.get("QBITTORRENT_URL", "")
|
||||
return client == "qbittorrent" and bool(url)
|
||||
|
||||
def test_connection(self) -> Tuple[bool, str]:
|
||||
"""Test connection to qBittorrent."""
|
||||
try:
|
||||
self._client.auth_log_in()
|
||||
api_version = self._client.app.web_api_version
|
||||
return True, f"Connected to qBittorrent (API v{api_version})"
|
||||
except Exception as e:
|
||||
return False, f"Connection failed: {str(e)}"
|
||||
|
||||
def add_download(self, url: str, name: str, category: str = None) -> str:
|
||||
"""
|
||||
Add torrent by URL (magnet or .torrent).
|
||||
|
||||
Args:
|
||||
url: Magnet link or .torrent URL
|
||||
name: Display name for the torrent
|
||||
category: Category for organization (uses configured default if not specified)
|
||||
|
||||
Returns:
|
||||
Torrent hash (info_hash).
|
||||
|
||||
Raises:
|
||||
Exception: If adding fails.
|
||||
"""
|
||||
try:
|
||||
# Use configured category if not explicitly provided
|
||||
category = category or self._category
|
||||
|
||||
# Ensure category exists (may already exist, which is fine)
|
||||
try:
|
||||
self._client.torrents_create_category(name=category)
|
||||
except Exception as e:
|
||||
# Conflict409Error means category exists - that's expected
|
||||
# Log other errors but continue since download may still work
|
||||
if "Conflict" not in type(e).__name__ and "409" not in str(e):
|
||||
logger.debug(f"Could not create category '{category}': {type(e).__name__}: {e}")
|
||||
|
||||
torrent_info = extract_torrent_info(url)
|
||||
expected_hash = torrent_info.info_hash
|
||||
torrent_data = torrent_info.torrent_data
|
||||
|
||||
# Add the torrent - use file content if we have it, otherwise URL
|
||||
if torrent_data:
|
||||
result = self._client.torrents_add(
|
||||
torrent_files=torrent_data,
|
||||
category=category,
|
||||
rename=name,
|
||||
)
|
||||
else:
|
||||
# Use magnet URL if available, otherwise original URL
|
||||
add_url = torrent_info.magnet_url or url
|
||||
result = self._client.torrents_add(
|
||||
urls=add_url,
|
||||
category=category,
|
||||
rename=name,
|
||||
)
|
||||
|
||||
logger.debug(f"qBittorrent add result: {result}")
|
||||
|
||||
if result == "Ok.":
|
||||
if not expected_hash:
|
||||
raise Exception("Could not determine torrent hash from URL")
|
||||
|
||||
# Wait for torrent to appear in client
|
||||
for _ in range(10):
|
||||
torrents = self._get_torrents_info(expected_hash)
|
||||
for t in torrents:
|
||||
if _hashes_match(t.hash, expected_hash):
|
||||
logger.info(f"Added torrent: {t.hash}")
|
||||
return t.hash.lower()
|
||||
time.sleep(0.5)
|
||||
|
||||
# Client said Ok, trust it
|
||||
logger.warning(f"Torrent not yet visible, returning expected hash")
|
||||
return expected_hash
|
||||
|
||||
raise Exception(f"Failed to add torrent: {result}")
|
||||
except Exception as e:
|
||||
logger.error(f"qBittorrent add failed: {e}")
|
||||
raise
|
||||
|
||||
def get_status(self, download_id: str) -> DownloadStatus:
|
||||
"""
|
||||
Get torrent status by hash.
|
||||
|
||||
Args:
|
||||
download_id: Torrent info_hash
|
||||
|
||||
Returns:
|
||||
Current download status.
|
||||
"""
|
||||
try:
|
||||
torrents = self._get_torrents_info(download_id)
|
||||
torrent = next((t for t in torrents if _hashes_match(t.hash, download_id)), None)
|
||||
if not torrent:
|
||||
return DownloadStatus.error("Torrent not found")
|
||||
|
||||
# Map qBittorrent states to our states and user-friendly messages
|
||||
state_info = {
|
||||
"downloading": ("downloading", None), # None = use default progress message
|
||||
"stalledDL": ("downloading", "Stalled"),
|
||||
"metaDL": ("downloading", "Fetching metadata"),
|
||||
"forcedDL": ("downloading", None),
|
||||
"allocating": ("downloading", "Allocating space"),
|
||||
"uploading": ("seeding", "Seeding"),
|
||||
"stalledUP": ("seeding", "Seeding (stalled)"),
|
||||
"forcedUP": ("seeding", "Seeding"),
|
||||
"pausedDL": ("paused", "Paused"),
|
||||
"pausedUP": ("paused", "Paused"),
|
||||
"queuedDL": ("queued", "Queued"),
|
||||
"queuedUP": ("queued", "Queued"),
|
||||
"checkingDL": ("checking", "Checking files"),
|
||||
"checkingUP": ("checking", "Checking files"),
|
||||
"checkingResumeData": ("checking", "Checking resume data"),
|
||||
"moving": ("processing", "Moving files"),
|
||||
"error": ("error", "Error"),
|
||||
"missingFiles": ("error", "Missing files"),
|
||||
"unknown": ("unknown", "Unknown state"),
|
||||
}
|
||||
|
||||
state, message = state_info.get(torrent.state, ("unknown", torrent.state))
|
||||
complete = torrent.progress >= 1.0
|
||||
|
||||
# For active downloads without a special message, leave message as None
|
||||
# so the handler can build the progress message
|
||||
if complete:
|
||||
message = "Complete"
|
||||
|
||||
eta = torrent.eta if 0 < torrent.eta < 604800 else None
|
||||
|
||||
# Get file path for completed downloads
|
||||
file_path = None
|
||||
if complete:
|
||||
if getattr(torrent, 'content_path', ''):
|
||||
file_path = torrent.content_path
|
||||
else:
|
||||
# Fallback for Amarr which doesn't populate content_path
|
||||
save_path = getattr(torrent, 'save_path', '')
|
||||
name = getattr(torrent, 'name', '')
|
||||
if save_path and name:
|
||||
file_path = f"{save_path}/{name}"
|
||||
|
||||
return DownloadStatus(
|
||||
progress=torrent.progress * 100,
|
||||
state="complete" if complete else state,
|
||||
message=message,
|
||||
complete=complete,
|
||||
file_path=file_path,
|
||||
download_speed=torrent.dlspeed,
|
||||
eta=eta,
|
||||
)
|
||||
except Exception as e:
|
||||
error_type = type(e).__name__
|
||||
logger.error(f"qBittorrent get_status failed ({error_type}): {e}")
|
||||
return DownloadStatus.error(f"{error_type}: {e}")
|
||||
|
||||
def remove(self, download_id: str, delete_files: bool = False) -> bool:
|
||||
"""
|
||||
Remove a torrent from qBittorrent.
|
||||
|
||||
Args:
|
||||
download_id: Torrent info_hash
|
||||
delete_files: Whether to also delete files
|
||||
|
||||
Returns:
|
||||
True if successful.
|
||||
"""
|
||||
try:
|
||||
self._client.torrents_delete(
|
||||
torrent_hashes=download_id, delete_files=delete_files
|
||||
)
|
||||
logger.info(
|
||||
f"Removed torrent from qBittorrent: {download_id}"
|
||||
+ (" (with files)" if delete_files else "")
|
||||
)
|
||||
return True
|
||||
except Exception as e:
|
||||
error_type = type(e).__name__
|
||||
logger.error(f"qBittorrent remove failed ({error_type}): {e}")
|
||||
return False
|
||||
|
||||
def get_download_path(self, download_id: str) -> Optional[str]:
|
||||
"""Get the path where torrent files are located."""
|
||||
try:
|
||||
torrents = self._get_torrents_info(download_id)
|
||||
torrent = next((t for t in torrents if _hashes_match(t.hash, download_id)), None)
|
||||
if not torrent:
|
||||
return None
|
||||
# Prefer content_path, fall back to save_path/name (for Amarr compatibility)
|
||||
if getattr(torrent, 'content_path', ''):
|
||||
return torrent.content_path
|
||||
save_path = getattr(torrent, 'save_path', '')
|
||||
name = getattr(torrent, 'name', '')
|
||||
return f"{save_path}/{name}" if save_path and name else None
|
||||
except Exception as e:
|
||||
error_type = type(e).__name__
|
||||
logger.debug(f"qBittorrent get_download_path failed ({error_type}): {e}")
|
||||
return None
|
||||
|
||||
def find_existing(self, url: str) -> Optional[Tuple[str, DownloadStatus]]:
|
||||
"""Check if a torrent for this URL already exists in qBittorrent."""
|
||||
try:
|
||||
torrent_info = extract_torrent_info(url)
|
||||
if not torrent_info.info_hash:
|
||||
return None
|
||||
|
||||
torrents = self._get_torrents_info(torrent_info.info_hash)
|
||||
torrent = next((t for t in torrents if _hashes_match(t.hash, torrent_info.info_hash)), None)
|
||||
if torrent:
|
||||
return (torrent.hash.lower(), self.get_status(torrent.hash.lower()))
|
||||
|
||||
return None
|
||||
except Exception as e:
|
||||
logger.debug(f"Error checking for existing torrent: {e}")
|
||||
return None
|
||||