Compare commits

..
Author SHA1 Message Date
Alex 7a4c913809 Clarify Docker digest update group 2026-05-12 12:14:29 +01:00
Alex 230015ad57 Add Dependabot update cooldown 2026-05-12 12:09:27 +01:00
Alex 28bf26414d Run uv Dependabot checks daily 2026-05-12 11:30:29 +01:00
331 changed files with 3474 additions and 33931 deletions
-24
View File
@@ -42,13 +42,7 @@ updates:
open-pull-requests-limit: 5
groups:
docker-base-image-digests:
# Exclude python from the group on purpose. Dependabot's Docker
# pre-release filter is bypassed for *grouped* updates
# (dependabot-core#9496), so a grouped python update proposes pre-release
# tags like python:3.15.0b2 as if they were a normal stable minor bump.
# node + uv stay grouped into a single digest PR.
patterns: ["*"]
exclude-patterns: ["python"]
ignore:
# Node.js: block major-version bumps so dependabot never proposes
# moving from one LTS line to a non-LTS "Current" release (e.g. 24 -> 25).
@@ -56,24 +50,6 @@ updates:
- dependency-name: "node"
update-types: ["version-update:semver-major"]
# Python: block minor/major bumps. Ungrouping python (above) is NOT enough
# to keep pre-releases out — dependabot-core#13815 rewrote the Docker
# pre-release heuristic to catch PEP 440 tags like 3.15.0a2 / 3.5.0b3, but
# the suffixed real tag still slipped through as PR #1169
# (python:3.14.6-slim -> python:3.15.0b3-slim). CPython spells
# pre-releases without a separator, so tag parsing reads 3.15.0b3 as an
# ordinary version that sorts above 3.14.6.
#
# A minor-version ignore blocks it regardless of spelling. Patch bumps
# (3.14.6 -> 3.14.7) and same-tag digest refreshes still land automatically.
# Moving the runtime to a new Python minor is a manual, deliberate change:
# bump the tag here and confirm C-extension wheels (greenlet/gevent) exist
# for it — a source build against a pre-release ABI boots an app that binds
# its port but never serves, which wedges e2e for the full 6h job limit.
- dependency-name: "python"
update-types:
["version-update:semver-major", "version-update:semver-minor"]
# GitHub Actions
- package-ecosystem: "github-actions"
directory: "/"
@@ -67,10 +67,10 @@ jobs:
run: echo "date=$(date +'%Y-%m-%d')" >> $GITHUB_OUTPUT
- name: Checkout repository
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Log in to the Container registry
uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0
uses: docker/login-action@4907a6ddec9925e35a0a9e82d7399ccc52663121 # v4.1.0
with:
registry: ${{ env.REGISTRY }}
username: ${{ github.actor }}
@@ -78,13 +78,7 @@ jobs:
- name: Extract metadata for ${{ matrix.target }} image
id: meta
uses: docker/metadata-action@dc802804100637a589fabce1cb79ff13a1411302 # v6.2.0
env:
# Annotate both the per-platform manifests and the multi-arch image
# index. The index level is what manifest-list consumers (Renovate's
# minimumReleaseAge soak check, provenance/SBOM tooling) read for the
# standard org.opencontainers.image.* annotations, including `created`.
DOCKER_METADATA_ANNOTATIONS_LEVELS: index,manifest
uses: docker/metadata-action@030e881283bb7a6894de51c315a6bfe6a94e05cf # v6.0.0
with:
images: ${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}${{ matrix.image_name_suffix }}
tags: |
@@ -96,11 +90,11 @@ jobs:
type=ref,event=tag
- name: Set up Docker Buildx
uses: docker/setup-buildx-action@37fe631027851001ddb9b187196cc803df7f5f0e # v4.3.0
uses: docker/setup-buildx-action@4d04d5d9486b7bd6fa91e7baf45bbb4f8b9deedd # v4.0.0
- name: Build and push ${{ matrix.target }} Docker image
id: push
uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7.3.0
uses: docker/build-push-action@bcafcacb16a39f128d818304e6c9c0c18556b85f # v7.1.0
with:
platforms: linux/amd64,linux/arm64
context: .
@@ -111,11 +105,10 @@ jobs:
RELEASE_VERSION=${{ github.ref_name }}
tags: ${{ steps.meta.outputs.tags }}
labels: ${{ steps.meta.outputs.labels }}
annotations: ${{ steps.meta.outputs.annotations }}
- name: Generate artifact attestation for ${{ matrix.target }} image
if: github.event_name != 'pull_request'
uses: actions/attest-build-provenance@4d101475d8b20a2381f78447822ac1eab6504dd8 # v4.2.2
uses: actions/attest-build-provenance@a2bbfa25375fe432b6a289bc6b6cd05ecd0c4c32 # v4.1.0
with:
subject-name: ${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}${{ matrix.image_name_suffix }}
subject-digest: ${{ steps.push.outputs.digest }}
@@ -134,14 +127,14 @@ jobs:
LEGACY_NAME: calibre-web-automated-book-downloader
steps:
- name: Log in to registry
uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0
uses: docker/login-action@4907a6ddec9925e35a0a9e82d7399ccc52663121 # v4.1.0
with:
registry: ${{ env.REGISTRY }}
username: ${{ github.actor }}
password: ${{ secrets.GITHUB_TOKEN }}
- name: Set up Docker Buildx
uses: docker/setup-buildx-action@37fe631027851001ddb9b187196cc803df7f5f0e # v4.3.0
uses: docker/setup-buildx-action@4d04d5d9486b7bd6fa91e7baf45bbb4f8b9deedd # v4.0.0
- name: Create legacy aliases
run: |
+15 -15
View File
@@ -13,10 +13,10 @@ jobs:
runs-on: ubuntu-latest
steps:
- name: Checkout
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Install uv and Python
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0
with:
version: "0.11.3"
python-version: "3.14"
@@ -39,10 +39,10 @@ jobs:
runs-on: ubuntu-latest
steps:
- name: Checkout
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Install uv and Python
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0
with:
version: "0.11.3"
python-version: "3.14"
@@ -59,10 +59,10 @@ jobs:
runs-on: ubuntu-latest
steps:
- name: Checkout
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Install uv and Python
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0
with:
version: "0.11.3"
python-version: "3.14"
@@ -78,13 +78,13 @@ jobs:
runs-on: ubuntu-latest
steps:
- name: Checkout
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Set up Docker Buildx
uses: docker/setup-buildx-action@37fe631027851001ddb9b187196cc803df7f5f0e # v4.3.0
uses: docker/setup-buildx-action@4d04d5d9486b7bd6fa91e7baf45bbb4f8b9deedd # v4.0.0
- name: Build shelfmark-lite image
uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7.3.0
uses: docker/build-push-action@bcafcacb16a39f128d818304e6c9c0c18556b85f # v7.1.0
with:
context: .
target: shelfmark-lite
@@ -99,10 +99,10 @@ jobs:
runs-on: ubuntu-latest
steps:
- name: Checkout
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Set up Node
uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
with:
node-version: 24
cache: "npm"
@@ -122,10 +122,10 @@ jobs:
runs-on: ubuntu-latest
steps:
- name: Checkout
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Set up Node
uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
with:
node-version: 24
cache: "npm"
@@ -142,10 +142,10 @@ jobs:
runs-on: ubuntu-latest
steps:
- name: Checkout
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Set up Node
uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
with:
node-version: 24
cache: "npm"
+4 -4
View File
@@ -22,17 +22,17 @@ jobs:
language: [python, javascript-typescript]
steps:
- name: Checkout
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Initialize CodeQL
uses: github/codeql-action/init@db488ddef3bf6cb639b32c2e9a7c0a7ea8271d28 # v3
uses: github/codeql-action/init@68bde559dea0fdcac2102bfdf6230c5f70eb485e # v3
with:
languages: ${{ matrix.language }}
- name: Autobuild
uses: github/codeql-action/autobuild@db488ddef3bf6cb639b32c2e9a7c0a7ea8271d28 # v3
uses: github/codeql-action/autobuild@68bde559dea0fdcac2102bfdf6230c5f70eb485e # v3
- name: Perform CodeQL Analysis
uses: github/codeql-action/analyze@db488ddef3bf6cb639b32c2e9a7c0a7ea8271d28 # v3
uses: github/codeql-action/analyze@68bde559dea0fdcac2102bfdf6230c5f70eb485e # v3
with:
category: "/language:${{ matrix.language }}"
-135
View File
@@ -1,135 +0,0 @@
name: E2E Platform
# Hermetic end-to-end matrix: boots the app under test against mock
# Anna's Archive / Cloudflare / bypasser / DNS / proxy / Tor / real torrent
# clients and runs the cluster suite under each config profile.
#
# On a PR that touches relevant code, this runs a fast core subset *and* the heavy
# `full` profile (real Chrome solving Cloudflare + DoH + real qBittorrent). The
# `e2e-required` job aggregates them into ONE status check — make that check a
# required status check in branch protection to block merges on any e2e failure
# (see tests/e2e/platform/README.md "Gating PRs").
on:
pull_request:
schedule:
- cron: "0 4 * * *" # nightly full matrix
workflow_dispatch:
concurrency:
group: e2e-platform-${{ github.ref }}
cancel-in-progress: true
jobs:
# Detect whether anything that affects the e2e platform changed. This lets the
# required check always report (never stuck "pending") while only spending CI on
# PRs that can actually break the e2e stack.
changes:
runs-on: ubuntu-latest
outputs:
relevant: ${{ steps.filter.outputs.relevant }}
steps:
- uses: actions/checkout@v7
- uses: dorny/paths-filter@v4.0.3
id: filter
with:
filters: |
relevant:
- 'shelfmark/**'
- 'entrypoint.sh'
- 'tor.sh'
- 'Dockerfile'
- 'tests/e2e/platform/**'
- '.github/workflows/e2e-platform.yml'
select-profiles:
needs: changes
if: needs.changes.outputs.relevant == 'true' || github.event_name != 'pull_request'
runs-on: ubuntu-latest
outputs:
profiles: ${{ steps.pick.outputs.profiles }}
steps:
- id: pick
run: |
if [ "${{ github.event_name }}" = "pull_request" ]; then
echo 'profiles=["baseline","bypasser-external","dns-blocked"]' >> "$GITHUB_OUTPUT"
else
echo 'profiles=["baseline","bypasser-external","bypasser-disabled","dns-manual","dns-blocked","dns-doh","proxy-http","proxy-socks","tor","client-transmission","client-deluge","client-qbittorrent-delayed"]' >> "$GITHUB_OUTPUT"
fi
e2e:
needs: select-profiles
runs-on: ubuntu-latest
# A wedged app under test must not burn GitHub's 6h max job limit. A healthy
# profile run finishes in ~3-5 min; anything past 25 is hung, not slow.
timeout-minutes: 25
strategy:
fail-fast: false
matrix:
profile: ${{ fromJSON(needs.select-profiles.outputs.profiles) }}
name: e2e (${{ matrix.profile }})
steps:
- name: Checkout
uses: actions/checkout@v7
- name: Install uv and Python
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
with:
python-version: "3.14"
enable-cache: true
- name: Sync dependencies
run: make install-python-dev
- name: Run e2e platform (${{ matrix.profile }})
run: tests/e2e/platform/run-e2e.sh env/${{ matrix.profile }}.env
- name: Dump shelfmark logs on failure
if: failure()
run: cat tests/e2e/platform/.state/shelfmark.${{ matrix.profile }}.log || true
# Heavy "everything real" job: real Chrome internal bypasser solving Cloudflare +
# DoH + real qBittorrent webseed download. Runs on relevant PRs and nightly.
e2e-full:
needs: changes
if: needs.changes.outputs.relevant == 'true' || github.event_name != 'pull_request'
runs-on: ubuntu-latest
# Real Chrome + qBittorrent is the slowest profile; still nowhere near 40 min.
timeout-minutes: 40
name: e2e (full — real Chrome + qBittorrent)
steps:
- name: Checkout
uses: actions/checkout@v7
- name: Install uv and Python
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
with:
python-version: "3.14"
enable-cache: true
- name: Sync dependencies
run: make install-python-dev
- name: Run full pipeline
run: tests/e2e/platform/run-e2e.sh env/full.env
- name: Dump logs on failure
if: failure()
run: |
cat tests/e2e/platform/.state/shelfmark.full.log || true
docker logs e2e-qbittorrent || true
# Single aggregated gate. ALWAYS runs (so a required check never hangs "pending"
# on unrelated PRs) and FAILS if any e2e job failed/was cancelled. Make THIS the
# required status check in branch protection.
e2e-required:
name: e2e required
needs: [e2e, e2e-full]
if: always()
runs-on: ubuntu-latest
steps:
- name: Aggregate e2e results
run: |
matrix='${{ needs.e2e.result }}'
full='${{ needs.e2e-full.result }}'
echo "e2e matrix=$matrix, e2e-full=$full"
# success or skipped (unrelated PR) is OK; failure/cancelled blocks.
for r in "$matrix" "$full"; do
if [ "$r" = "failure" ] || [ "$r" = "cancelled" ]; then
echo "::error::An e2e platform job did not pass — blocking."
exit 1
fi
done
echo "All e2e platform jobs passed (or were skipped as not relevant)."
-5
View File
@@ -166,10 +166,6 @@ ENV/
env.bak/
venv.bak/
# ...but the e2e platform test profiles live in an env/ dir and must be tracked
!tests/e2e/platform/env/
!tests/e2e/platform/env/*.env
# Spyder project settings
.spyderproject
.spyproject
@@ -236,7 +232,6 @@ pyrightconfig.json
*.local.*
AGENTS.md
.claude/
CLAUDE.md
.nvmrc
.playwright-mcp/
frontend-dist/
+21 -70
View File
@@ -4,7 +4,7 @@ ARG BUILDPLATFORM
ARG BUILDARCH
# Frontend build stage.
FROM --platform=$BUILDPLATFORM node:24-alpine@sha256:d32cdf619f63fe0471182d08996dd516c6275bb5fd31ae06e55a570bd9e1ad43 AS frontend-builder
FROM --platform=$BUILDPLATFORM node:24-alpine@sha256:d1b3b4da11eefd5941e7f0b9cf17783fc99d9c6fc34884a665f40a06dbdfc94f AS frontend-builder
# Helpful debug output to see what platforms BuildKit thinks it's using
RUN echo "BUILDPLATFORM=$BUILDPLATFORM BUILDARCH=$BUILDARCH TARGETPLATFORM=$TARGETPLATFORM TARGETARCH=$TARGETARCH"
@@ -24,14 +24,10 @@ COPY src/frontend/ ./
# Build the frontend
RUN npm run build
# uv is a build-time tool only, so it is mounted into the RUNs that need it rather
# than copied into the image. A COPY here would land ~24 MB in a `base` layer that
# every published image inherits, and a later `rm` cannot take it back out again --
# a RUN adds a layer, it does not rewrite the one underneath.
FROM ghcr.io/astral-sh/uv:0.12.5@sha256:e85be844203885286c60ffad8a858d48afb6c5a5c237ca0e67f12e74b8f174b1 AS uv
# Use python-slim as the base image
FROM python:3.14.7-slim@sha256:cae66f2ef0ec51a9891263eeee7f987dacf0a9879e8aa9353d5606e0530619a5 AS base
FROM python:3.14-slim@sha256:1697e8e8d39bf168e177ac6b5fdab6df86d81cfc24dae17dfb96cfc3ef76b4dd AS base
COPY --from=ghcr.io/astral-sh/uv:0.11.3@sha256:90bbb3c16635e9627f49eec6539f956d70746c409209041800a0280b93152823 /uv /uvx /bin/
# Add build argument for version
ARG BUILD_VERSION
@@ -63,11 +59,6 @@ ENV FLASK_PORT=8084
# Configure locale, timezone, and perform initial cleanup in a single layer
RUN apt-get update && \
apt-get install -y --no-install-recommends \
# For building C-extensions (cffi, gevent, etc.)
gcc \
g++ \
libffi-dev \
python3-dev \
# For locale
locales tzdata \
# For healthcheck
@@ -81,12 +72,7 @@ RUN apt-get update && \
# --- Tor support (activated via USING_TOR=true) ---
tor \
supervisor \
iptables \
# --- WireGuard support (activated via USING_WIREGUARD=true) ---
wireguard-tools \
iproute2 \
procps \
ca-certificates && \
iptables && \
# Configure iptables alternatives for tor.sh compatibility
update-alternatives --set iptables /usr/sbin/iptables-legacy && \
update-alternatives --set ip6tables /usr/sbin/ip6tables-legacy && \
@@ -115,7 +101,6 @@ WORKDIR /app
# Install core Python dependencies first for better layer caching
COPY pyproject.toml uv.lock ./
RUN --mount=type=cache,target=/root/.cache/uv \
--mount=from=uv,source=/uv,target=/usr/local/bin/uv \
uv sync --locked --no-default-groups
# Runtime dependencies are installed into /app/.venv during the build. Remove the
@@ -146,20 +131,15 @@ RUN mkdir -p \
ln -s /tmp/shelfmark/seleniumbase/archived_files /app/archived_files && \
chown -R 1000:1000 /config /books /home/shelfmark /tmp/shelfmark /var/log/shelfmark && \
chmod -R a+rX /app && \
chmod +x /app/entrypoint.sh /app/tor.sh /app/wireguard.sh /app/genDebug.sh
chmod +x /app/entrypoint.sh /app/tor.sh /app/genDebug.sh
# Expose the application port
EXPOSE ${FLASK_PORT}
# Add healthcheck for container status
# Uses /api/health which doesn't require authentication.
# curl needs -f so an HTTP error status fails the probe instead of passing it:
# plain `curl -s` exits 0 on a 500, which reported a broken app as healthy.
# timeout stays well under interval so a hung probe cannot occupy a whole cycle.
# --start-interval matches the daemon default (5s), made explicit so startup
# probing does not depend on that default staying put.
HEALTHCHECK --interval=30s --timeout=10s --start-period=90s --start-interval=5s --retries=3 \
CMD curl -fsS http://localhost:${FLASK_PORT}/api/health > /dev/null || exit 1
# Uses /api/health which doesn't require authentication
HEALTHCHECK --interval=60s --timeout=60s --start-period=60s --retries=3 \
CMD curl -s http://localhost:${FLASK_PORT}/api/health > /dev/null || exit 1
# Use dumb-init as the entrypoint to handle signals properly
ENTRYPOINT ["/usr/bin/dumb-init", "--"]
@@ -167,39 +147,21 @@ ENTRYPOINT ["/usr/bin/dumb-init", "--"]
FROM base AS shelfmark
# --- Chromium (PINNED to 149.0.7827.196) ---
# Debian's chromium 150.0.7871.46-1~deb13u1 security update (trixie-security,
# 2026-07-05) no longer opens the DevTools remote-debugging TCP port at all
# (no listener, no DevToolsActivePort file, even with a custom --user-data-dir;
# the RemoteDebuggingAllowed policy does not restore it). The SeleniumBase
# Pure-CDP driver connects through that port (/json/version), so with 150 every
# internal bypass dies with "Pure CDP browser startup failed" and all
# CF-gated downloads fail. Install the last working version from
# snapshot.debian.org until the bypasser can talk to Chromium >= 150 (e.g.
# pipe-based DevTools / UC mode) or seleniumbase ships a fix.
# Chrome 144+ requires --enable-unsafe-swiftshader for WebGL in Docker.
# This flag is set in internal_bypasser.py _get_browser_args()
ARG CHROMIUM_VERSION=149.0.7827.196-1~deb13u1
ARG CHROMIUM_SNAPSHOT=20260704T000000Z
RUN echo "deb [check-valid-until=no] https://snapshot.debian.org/archive/debian-security/${CHROMIUM_SNAPSHOT}/ trixie-security main" \
> /etc/apt/sources.list.d/chromium-pin-snapshot.list && \
apt-get update -o Acquire::Retries=5 && \
apt-get install -y --no-install-recommends -o Acquire::Retries=5 \
RUN apt-get update && \
apt-get install -y --no-install-recommends \
# For dumb display
xvfb \
# For screen recording
ffmpeg \
chromium=${CHROMIUM_VERSION} \
chromium-common=${CHROMIUM_VERSION} \
# --- Chromium (unpinned - uses latest from Debian repos) ---
# Chrome 144+ requires --enable-unsafe-swiftshader for WebGL in Docker.
# This flag is set in internal_bypasser.py _get_browser_args()
chromium \
chromium-common \
# For tkinter (pyautogui)
python3-tk \
# For RAR extraction
unrar-free && \
# Keep apt from "upgrading" chromium past the pin inside derived images
printf 'Package: chromium chromium-common\nPin: version %s\nPin-Priority: 1001\n' "${CHROMIUM_VERSION}" \
> /etc/apt/preferences.d/chromium-pin && \
rm /etc/apt/sources.list.d/chromium-pin-snapshot.list && \
# Create symlink so rarfile library can find unrar
ln -sf /usr/bin/unrar-free /usr/bin/unrar && \
# Cleanup APT cache
@@ -209,24 +171,10 @@ RUN echo "deb [check-valid-until=no] https://snapshot.debian.org/archive/debian-
# Install the browser automation stack used by the full image
RUN --mount=type=cache,target=/root/.cache/uv \
--mount=from=uv,source=/uv,target=/usr/local/bin/uv \
uv sync --locked --no-default-groups --extra browser
# Deterministically resolve the Xlib namespace collision.
# pyautogui/mouseinfo pull the stale `python3-xlib` (0.15, 2014), while the
# `--extra browser` set pulls `python-xlib` (0.33). Both packages install into
# the same top-level `Xlib/` namespace, so whichever lands last wins. When the
# 2014 build wins, `Xlib.X` is missing `FamilyServerInterpreted`, which the
# SeleniumBase Pure-CDP driver requires at browser startup -> every bypass fails
# with "module 'Xlib.X' has no attribute 'FamilyServerInterpreted'" and no
# Cloudflare/DDoS-Guard protected download can complete. Drop the stale package
# and force python-xlib 0.33 to own the namespace. pyautogui runs fine against
# 0.33 (superset API).
RUN --mount=type=cache,target=/root/.cache/uv \
--mount=from=uv,source=/uv,target=/usr/local/bin/uv \
uv pip uninstall --python /app/.venv/bin/python python3-xlib && \
uv pip install --python /app/.venv/bin/python --reinstall python-xlib==0.33 && \
/app/.venv/bin/python -c "import Xlib.X; assert hasattr(Xlib.X, 'FamilyServerInterpreted'), 'Xlib.X.FamilyServerInterpreted missing after fix'; print('Xlib namespace OK:', Xlib.__version__)"
# uv is only needed while building the image.
RUN rm -f /usr/bin/uv /usr/bin/uvx
# Keep SeleniumBase's bundled driver cache writable for the fixed non-root user.
RUN SELENIUMBASE_DRIVERS_DIR=$(/app/.venv/bin/python -c "import pathlib, seleniumbase; print(pathlib.Path(seleniumbase.__file__).resolve().parent / 'drivers')") && \
@@ -244,4 +192,7 @@ FROM base AS shelfmark-lite
ENV USING_EXTERNAL_BYPASSER=true
# uv is only needed while building the image.
RUN rm -f /usr/bin/uv /usr/bin/uvx
CMD ["/app/entrypoint.sh"]
+1 -29
View File
@@ -1,4 +1,4 @@
.PHONY: help install install-ci install-python-dev dev build preview frontend-typecheck frontend-lint frontend-format frontend-format-fix frontend-checks frontend-test clean up down docker-build refresh restart build-serve python-lint python-lint-fix python-format python-format-fix python-typecheck python-dead-code python-checks python-test python-test-cov e2e-platform e2e-platform-profile e2e-platform-matrix e2e-platform-full e2e-platform-build checks fix
.PHONY: help install install-ci install-python-dev dev build preview frontend-typecheck frontend-lint frontend-format frontend-format-fix frontend-checks frontend-test clean up down docker-build refresh restart build-serve python-lint python-lint-fix python-format python-format-fix python-typecheck python-dead-code python-checks python-test python-test-cov checks fix
# Frontend directory
FRONTEND_DIR := src/frontend
@@ -38,10 +38,6 @@ help:
@echo " python-checks - Run all Python static analysis checks"
@echo " python-test - Run unit tests"
@echo " python-test-cov - Run unit tests with coverage report"
@echo " e2e-platform - Run e2e docker platform (baseline profile)"
@echo " e2e-platform-profile PROFILE=<name> - Run e2e platform for one profile"
@echo " e2e-platform-matrix - Run e2e platform across all config profiles"
@echo " e2e-platform-full - Run heavy 'full' profile (real Chrome bypasser + DoH + real qBittorrent)"
@echo " clean - Remove node_modules and build artifacts"
@echo ""
@echo "Backend (Docker):"
@@ -131,30 +127,6 @@ python-test-cov:
@echo "Running tests with coverage..."
uv run pytest tests/ -x --tb=short -m "not integration and not e2e" --cov --cov-report=term-missing
# E2E docker platform: hermetic stack (mock AA/Cloudflare/bypasser/DNS/proxy/Tor)
# exercised across config profiles. See tests/e2e/platform/README.md.
E2E_PLATFORM_DIR := tests/e2e/platform
e2e-platform:
@echo "Running e2e platform (baseline profile)..."
cd $(E2E_PLATFORM_DIR) && ./run-e2e.sh env/baseline.env
e2e-platform-profile:
@echo "Running e2e platform (profile=$(PROFILE))..."
cd $(E2E_PLATFORM_DIR) && ./run-e2e.sh env/$(PROFILE).env
e2e-platform-matrix:
@echo "Running e2e platform matrix (all profiles)..."
cd $(E2E_PLATFORM_DIR) && ./run-matrix.sh
e2e-platform-build:
@echo "Pre-building e2e platform images once (reused across profiles)..."
cd $(E2E_PLATFORM_DIR) && ./build-images.sh
e2e-platform-full:
@echo "Running e2e platform FULL profile (real Chrome bypasser + DoH + real qBittorrent)..."
cd $(E2E_PLATFORM_DIR) && ./run-e2e.sh env/full.env
# Frontend linting
frontend-lint:
@echo "Running Oxlint..."
-43
View File
@@ -1,43 +0,0 @@
# Routes all traffic through a WireGuard tunnel - requires root startup.
#
# Mount your wg-quick config at /config/wg0.conf (read-only is fine). All
# non-LAN egress is forced through the tunnel by an iptables kill-switch, so if
# the tunnel drops, external traffic fails closed. LAN ranges (WebUI + internal
# download clients like Prowlarr / qBittorrent) stay reachable off-tunnel.
services:
shelfmark-wireguard:
image: ghcr.io/calibrain/shelfmark:latest
environment:
FLASK_PORT: 8084
# Quoted so it is passed as the literal string "true": entrypoint.sh compares
# $USING_WIREGUARD against "true", and some Compose implementations stringify
# a bare YAML boolean as "True", which would silently NOT enable WireGuard.
USING_WIREGUARD: "true"
# Path to the mounted wg-quick config (default shown).
WIREGUARD_CONFIG: /config/wg0.conf
# CIDRs kept OFF the tunnel so the WebUI and internal clients stay reachable.
LAN_NETWORK: 127.0.0.0/8,10.0.0.0/8,172.16.0.0/12,192.168.0.0/16
PUID: 1000
PGID: 1000
cap_add:
- NET_ADMIN
- NET_RAW
# WireGuard needs the module/kernel routing; NET_ADMIN covers wg-quick.
sysctls:
- net.ipv4.conf.all.src_valid_mark=1
# Disable IPv6 in the container so the kill-switch can guarantee no IPv6
# leak path on kernels/containers without a usable ip6tables. wireguard.sh
# fails closed if IPv6 is neither kill-switched nor disabled. If your host
# DOES have a working ip6tables you may omit these (an ip6tables kill-switch
# is installed instead); or set WIREGUARD_ALLOW_IPV6_LEAK=true only if the
# container genuinely has no IPv6 connectivity.
- net.ipv6.conf.all.disable_ipv6=1
- net.ipv6.conf.default.disable_ipv6=1
ports:
- 8084:8084
restart: unless-stopped
volumes:
- /path/to/books:/books # Default destination for book downloads
- /path/to/config:/config # App configuration (put wg0.conf here)
# Required for torrent / usenet - path must match your download client's volume exactly
# - /path/to/downloads:/path/to/downloads
+53 -60
View File
@@ -1,79 +1,72 @@
[
{ "language": "English", "code": "en", "aliases": ["eng"] },
{ "language": "Chinese", "code": "zh", "aliases": ["chi", "zho"] },
{ "language": "Russian", "code": "ru", "aliases": ["rus"] },
{ "language": "Spanish", "code": "es", "aliases": ["spa"] },
{ "language": "French", "code": "fr", "aliases": ["fra", "fre"] },
{ "language": "German", "code": "de", "aliases": ["deu", "ger"] },
{ "language": "Italian", "code": "it", "aliases": ["ita"] },
{ "language": "Portuguese", "code": "pt", "aliases": ["por"] },
{ "language": "Polish", "code": "pl", "aliases": ["pol"] },
{ "language": "Bulgarian", "code": "bg", "aliases": ["bul"] },
{ "language": "Dutch", "code": "nl", "aliases": ["dut", "nld"] },
{ "language": "Japanese", "code": "ja", "aliases": ["jap", "jpn"] },
{ "language": "Arabic", "code": "ar", "aliases": ["ara"] },
{ "language": "Hebrew", "code": "he", "aliases": ["heb"] },
{ "language": "Turkish", "code": "tr", "aliases": ["tur"] },
{ "language": "Hungarian", "code": "hu", "aliases": ["hun"] },
{ "language": "Latin", "code": "la", "aliases": ["lat"] },
{ "language": "Czech", "code": "cs", "aliases": ["ces", "cze"] },
{ "language": "Korean", "code": "ko", "aliases": ["kor"] },
{ "language": "Ukrainian", "code": "uk", "aliases": ["ukr"] },
{ "language": "Indonesian", "code": "id", "aliases": ["ind"] },
{ "language": "Romanian", "code": "ro", "aliases": ["rom", "ron"] },
{ "language": "Swedish", "code": "sv", "aliases": ["swe"] },
{ "language": "Greek", "code": "el", "aliases": ["ell", "gre"] },
{ "language": "Lithuanian", "code": "lt", "aliases": ["lit"] },
{ "language": "Bangla", "code": "bn", "aliases": ["ben", "bengali"] },
{ "language": "Traditional Chinese", "code": "zh-Hant", "aliases": ["zh‑Hant"] },
{ "language": "Afrikaans", "code": "af", "aliases": ["afr"] },
{ "language": "Catalan", "code": "ca", "aliases": ["cat"] },
{ "language": "Danish", "code": "da", "aliases": ["dan"] },
{ "language": "Thai", "code": "th", "aliases": ["tha"] },
{ "language": "Hindi", "code": "hi", "aliases": ["hin"] },
{ "language": "Irish", "code": "ga", "aliases": ["gle"] },
{ "language": "Latvian", "code": "lv", "aliases": ["lav"] },
{ "language": "English", "code": "en" },
{ "language": "Chinese", "code": "zh" },
{ "language": "Russian", "code": "ru" },
{ "language": "Spanish", "code": "es" },
{ "language": "French", "code": "fr" },
{ "language": "German", "code": "de" },
{ "language": "Italian", "code": "it" },
{ "language": "Portuguese", "code": "pt" },
{ "language": "Polish", "code": "pl" },
{ "language": "Bulgarian", "code": "bg" },
{ "language": "Dutch", "code": "nl" },
{ "language": "Japanese", "code": "ja" },
{ "language": "Arabic", "code": "ar" },
{ "language": "Hebrew", "code": "he" },
{ "language": "Turkish", "code": "tr" },
{ "language": "Hungarian", "code": "hu" },
{ "language": "Latin", "code": "la" },
{ "language": "Czech", "code": "cs" },
{ "language": "Korean", "code": "ko" },
{ "language": "Ukrainian", "code": "uk" },
{ "language": "Indonesian", "code": "id" },
{ "language": "Romanian", "code": "ro" },
{ "language": "Swedish", "code": "sv" },
{ "language": "Greek", "code": "el" },
{ "language": "Lithuanian", "code": "lt" },
{ "language": "Bangla", "code": "bn" },
{ "language": "Traditional Chinese", "code": "zh‑Hant" },
{ "language": "Afrikaans", "code": "af" },
{ "language": "Catalan", "code": "ca" },
{ "language": "Danish", "code": "da" },
{ "language": "Thai", "code": "th" },
{ "language": "Hindi", "code": "hi" },
{ "language": "Irish", "code": "ga" },
{ "language": "Latvian", "code": "lv" },
{ "language": "Tibetan", "code": "bo" },
{ "language": "Kannada", "code": "kn", "aliases": ["kan"] },
{ "language": "Serbian", "code": "sr", "aliases": ["srp"] },
{ "language": "Persian", "code": "fa", "aliases": ["farsi", "fas", "per"] },
{ "language": "Croatian", "code": "hr", "aliases": ["hrv"] },
{ "language": "Kannada", "code": "kn" },
{ "language": "Serbian", "code": "sr" },
{ "language": "Persian", "code": "fa" },
{ "language": "Croatian", "code": "hr" },
{ "language": "Slovak", "code": "sk" },
{ "language": "Javanese", "code": "jv", "aliases": ["jav"] },
{ "language": "Vietnamese", "code": "vi", "aliases": ["vie"] },
{ "language": "Urdu", "code": "ur", "aliases": ["urd"] },
{ "language": "Finnish", "code": "fi", "aliases": ["fin"] },
{ "language": "Norwegian", "code": "no", "aliases": ["nor"] },
{ "language": "Javanese", "code": "jv" },
{ "language": "Vietnamese", "code": "vi" },
{ "language": "Urdu", "code": "ur" },
{ "language": "Finnish", "code": "fi" },
{ "language": "Norwegian", "code": "no" },
{ "language": "Kinyarwanda", "code": "rw" },
{ "language": "Tamil", "code": "ta", "aliases": ["tam"] },
{ "language": "Tamil", "code": "ta" },
{ "language": "Belarusian", "code": "be" },
{ "language": "Kazakh", "code": "kk" },
{ "language": "Mongolian", "code": "mn" },
{ "language": "Georgian", "code": "ka" },
{ "language": "Slovenian", "code": "sl", "aliases": ["slv"] },
{ "language": "Slovenian", "code": "sl" },
{ "language": "Esperanto", "code": "eo" },
{ "language": "Galician", "code": "gl" },
{ "language": "Marathi", "code": "mr", "aliases": ["mar"] },
{ "language": "Filipino", "code": "fil", "aliases": ["tagalog", "tgl"] },
{ "language": "Gujarati", "code": "gu", "aliases": ["guj"] },
{ "language": "Malayalam", "code": "ml", "aliases": ["mal"] },
{ "language": "Marathi", "code": "mr" },
{ "language": "Filipino", "code": "fil" },
{ "language": "Gujarati", "code": "gu" },
{ "language": "Malayalam", "code": "ml" },
{ "language": "Kyrgyz", "code": "ky" },
{ "language": "Azerbaijani", "code": "az" },
{ "language": "Quechua", "code": "qu" },
{ "language": "Swahili", "code": "sw" },
{ "language": "Bashkir", "code": "ba" },
{ "language": "Punjabi", "code": "pa", "aliases": ["pan"] },
{ "language": "Malay", "code": "ms", "aliases": ["may", "msa"] },
{ "language": "Telugu", "code": "te", "aliases": ["tel"] },
{ "language": "Punjabi", "code": "pa" },
{ "language": "Malay", "code": "ms" },
{ "language": "Telugu", "code": "te" },
{ "language": "Albanian", "code": "sq" },
{ "language": "Uyghur", "code": "ug" },
{ "language": "Armenian", "code": "hy" },
{ "language": "Shan", "code": "shn" },
{ "language": "Bosnian", "code": "bs", "aliases": ["bos"] },
{ "language": "Burmese", "code": "my", "aliases": ["bur", "mya"] },
{ "language": "Estonian", "code": "et", "aliases": ["est"] },
{ "language": "Icelandic", "code": "is", "aliases": ["ice", "isl"] },
{ "language": "Manx", "code": "gv", "aliases": ["glv"] },
{ "language": "Scottish Gaelic", "code": "gd", "aliases": ["gla"] },
{ "language": "Sanskrit", "code": "sa", "aliases": ["san"] }
{ "language": "Shan", "code": "shn" }
]
-25
View File
@@ -1,25 +0,0 @@
# Local development - WireGuard variant
services:
shelfmark-wireguard-dev:
extends:
file: ./compose/docker-compose.wireguard.yml
service: shelfmark-wireguard
build:
context: .
dockerfile: Dockerfile
target: shelfmark
environment:
# Quoted so they are passed as the literal string "true" (entrypoint.sh and
# the app compare against "true"); a bare YAML boolean can be stringified as
# "True" by some Compose variants, silently disabling the feature.
DEBUG: "true"
USING_WIREGUARD: "true"
WIREGUARD_CONFIG: /config/wg0.conf
volumes:
- ./.local/config:/config
- ./.local/books:/books
- ./.local/log:/var/log/shelfmark
- ./.local/tmp:/tmp/shelfmark
# Place your wg-quick config at ./.local/config/wg0.conf
# Required for torrent / usenet - path must match your download client's volume exactly
# - /path/to/downloads:/path/to/downloads
-4
View File
@@ -91,10 +91,6 @@ Example:
- Shelfmark can see the same files at `/downloads/books/...`
- Add a mapping from Remote Path `/data/torrents` to Local Path `/downloads`
If the files are copied or synced into Shelfmark on a delay, increase **Completed Path Wait (seconds)**
in Settings -> Advanced. The default is 60 seconds; seedbox or remote-sync setups may need a value
longer than the sync interval.
## File Processing Options
### Transfer Method (Torrent / Usenet Only)
-21
View File
@@ -276,27 +276,6 @@ class DownloadHandler(ABC):
pass
```
### Optional: Listing Files Before Download
Some releases bundle several books (a whole-series torrent). Shelfmark inspects a
release before queueing it so the user can review how it will be split into books.
Override `list_files` when your source can enumerate a release's files without
downloading it; the default returns `None`, which the UI reports as "can't inspect":
```python
from shelfmark.download.postprocess.packs import PackFile
def list_files(self, release_data: dict[str, Any]) -> list[PackFile] | None:
"""Return the release's files (release-relative paths + sizes), or None."""
torrent_bytes = ... # e.g. fetch the .torrent, or scrape the indexer's detail page
return extract_file_list_from_torrent(torrent_bytes) # from download.clients.torrent_utils
```
`release_data` is the same payload the frontend sends to `/api/releases/download`
(`source_id`, `download_url`, `content_type`, `series_name`, ...). Built-in examples:
Prowlarr parses the `.torrent` it already fetches (magnet-only releases return
`None`), and AudiobookBay reads the file table off its detail page.
### Download Method Parameters
| Parameter | Type | Description |
+39 -360
View File
@@ -7,7 +7,6 @@ This document lists all configuration options that can be set via environment va
## Table of Contents
- [Bootstrap Configuration](#bootstrap-configuration)
- [Egress / VPN Routing](#egress--vpn-routing)
- [General](#general)
- [Search Mode](#search-mode)
- [Downloads](#downloads)
@@ -23,7 +22,6 @@ This document lists all configuration options that can be set via environment va
- [Hardcover](#metadata-providers-hardcover)
- [Open Library](#metadata-providers-open-library)
- [Google Books](#metadata-providers-google-books)
- [Moly.hu](#metadata-providers-moly.hu)
- [Direct Download](#direct-download)
- [Download Sources](#direct-download-download-sources)
- [Cloudflare Bypass](#direct-download-cloudflare-bypass)
@@ -147,98 +145,6 @@ Show the onboarding wizard on first run. Set to false to skip (useful for epheme
</details>
## Egress / VPN Routing
These startup-only variables are consumed by `entrypoint.sh` / `wireguard.sh` to select and configure the WireGuard transparent-egress kill-switch. `USING_WIREGUARD` and [`USING_TOR`](#using_tor) (documented under Network) are mutually exclusive; both require root startup.
| Variable | Description | Type | Default |
|----------|-------------|------|---------|
| `USING_WIREGUARD` | Route all traffic through a WireGuard VPN tunnel with a fail-closed iptables kill-switch (non-tunnel egress is dropped). Requires root startup and NET_ADMIN (plus NET_RAW). Mutually exclusive with USING_TOR. | boolean | `false` |
| `WIREGUARD_CONFIG` | Path to the mounted wg-quick configuration file. | string (path) | `/config/wg0.conf` |
| `WIREGUARD_INTERFACE` | WireGuard interface name brought up by wg-quick. | string | `wg0` |
| `LAN_NETWORK` | Comma-separated CIDRs kept off the tunnel so the WebUI and internal download clients (Prowlarr, qBittorrent) stay reachable. | string (comma-separated) | `127.0.0.0/8,10.0.0.0/8,172.16.0.0/12,192.168.0.0/16` |
| `WIREGUARD_ENFORCE_DNS` | Pin the container's resolver so DNS cannot silently fall back to an off-tunnel path. The resolver used is WIREGUARD_DNS if set, else the tunnel config's DNS = line. This does NOT force queries through the tunnel: it is designed for a trusted LAN resolver kept reachable off-tunnel via LAN_NETWORK (the query leaves over the LAN; the resolver encrypts upstream while the download still egresses via the tunnel). Special case: when Docker's embedded resolver (nameserver 127.0.0.11) is present, it is PRESERVED so container-name resolution (e.g. prowlarr, qbittorrent) keeps working, and the embedded resolver's upstream must be pinned via the container's compose dns: list. Fails closed (refuses to start) only when no embedded resolver is present AND no resolver is defined, or /etc/resolv.conf is not writable. | boolean | `true` |
| `WIREGUARD_DNS` | Explicit resolver(s) (comma/space separated) to pin when WIREGUARD_ENFORCE_DNS is true and Docker's embedded resolver is NOT in use. Use when the VPN's pushed DNS filters domains you need; point it at a resolver reachable via the tunnel or an allowed LAN resolver. NOTE: when the embedded resolver (127.0.0.11) is present it is preserved and this value cannot repoint its upstream from inside the container — set the container's compose dns: list to the trusted resolver instead. | string (comma-separated) | `unset (uses config DNS = line)` |
| `WIREGUARD_DISABLE_IPV6` | Strip IPv6 Address/AllowedIPs/DNS from the tunnel config before wg-quick (many container kernels lack the ip6tables raw table wg-quick needs) and remove IPv6 as a leak surface. | boolean | `true` |
| `WIREGUARD_ALLOW_IPV6_LEAK` | Escape hatch: continue startup even when an IPv6 kill-switch cannot be installed AND IPv6 cannot be disabled. Only set when the container has no IPv6 connectivity, as IPv6 egress may otherwise bypass the tunnel. | boolean | `false` |
| `WIREGUARD_ALLOW_WEBUI_OFFTUNNEL` | When false (default) the kill-switch is strictly fail-closed: the only off-tunnel egress permitted is loopback, the tunnel device and the LAN allowlist. Set true only if a NON-LAN client (e.g. a public reverse proxy on a different segment) must reach the WebUI; it permits app-server REPLY packets (--sport FLASK_PORT, conntrack REPLY) to leave off-tunnel. Server replies only, never client-initiated egress, so it cannot leak outbound browsing/downloads or the real IP for outbound requests, but it is still an off-tunnel path while the tunnel is down, hence opt-in. LAN WebUI clients never need it (covered by LAN_NETWORK). | boolean | `false` |
| `WIREGUARD_STALE_AFTER` | Seconds since the last WireGuard handshake before the healthcheck bounces the tunnel. | number | `180` |
<details>
<summary>Detailed descriptions</summary>
#### `USING_WIREGUARD`
Route all traffic through a WireGuard VPN tunnel with a fail-closed iptables kill-switch (non-tunnel egress is dropped). Requires root startup and NET_ADMIN (plus NET_RAW). Mutually exclusive with USING_TOR.
- **Type:** boolean
- **Default:** `false`
#### `WIREGUARD_CONFIG`
Path to the mounted wg-quick configuration file.
- **Type:** string (path)
- **Default:** `/config/wg0.conf`
#### `WIREGUARD_INTERFACE`
WireGuard interface name brought up by wg-quick.
- **Type:** string
- **Default:** `wg0`
#### `LAN_NETWORK`
Comma-separated CIDRs kept off the tunnel so the WebUI and internal download clients (Prowlarr, qBittorrent) stay reachable.
- **Type:** string (comma-separated)
- **Default:** `127.0.0.0/8,10.0.0.0/8,172.16.0.0/12,192.168.0.0/16`
#### `WIREGUARD_ENFORCE_DNS`
Pin the container's resolver so DNS cannot silently fall back to an off-tunnel path. The resolver used is WIREGUARD_DNS if set, else the tunnel config's DNS = line. This does NOT force queries through the tunnel: it is designed for a trusted LAN resolver kept reachable off-tunnel via LAN_NETWORK (the query leaves over the LAN; the resolver encrypts upstream while the download still egresses via the tunnel). Special case: when Docker's embedded resolver (nameserver 127.0.0.11) is present, it is PRESERVED so container-name resolution (e.g. prowlarr, qbittorrent) keeps working, and the embedded resolver's upstream must be pinned via the container's compose dns: list. Fails closed (refuses to start) only when no embedded resolver is present AND no resolver is defined, or /etc/resolv.conf is not writable.
- **Type:** boolean
- **Default:** `true`
#### `WIREGUARD_DNS`
Explicit resolver(s) (comma/space separated) to pin when WIREGUARD_ENFORCE_DNS is true and Docker's embedded resolver is NOT in use. Use when the VPN's pushed DNS filters domains you need; point it at a resolver reachable via the tunnel or an allowed LAN resolver. NOTE: when the embedded resolver (127.0.0.11) is present it is preserved and this value cannot repoint its upstream from inside the container — set the container's compose dns: list to the trusted resolver instead.
- **Type:** string (comma-separated)
- **Default:** `unset (uses config DNS = line)`
#### `WIREGUARD_DISABLE_IPV6`
Strip IPv6 Address/AllowedIPs/DNS from the tunnel config before wg-quick (many container kernels lack the ip6tables raw table wg-quick needs) and remove IPv6 as a leak surface.
- **Type:** boolean
- **Default:** `true`
#### `WIREGUARD_ALLOW_IPV6_LEAK`
Escape hatch: continue startup even when an IPv6 kill-switch cannot be installed AND IPv6 cannot be disabled. Only set when the container has no IPv6 connectivity, as IPv6 egress may otherwise bypass the tunnel.
- **Type:** boolean
- **Default:** `false`
#### `WIREGUARD_ALLOW_WEBUI_OFFTUNNEL`
When false (default) the kill-switch is strictly fail-closed: the only off-tunnel egress permitted is loopback, the tunnel device and the LAN allowlist. Set true only if a NON-LAN client (e.g. a public reverse proxy on a different segment) must reach the WebUI; it permits app-server REPLY packets (--sport FLASK_PORT, conntrack REPLY) to leave off-tunnel. Server replies only, never client-initiated egress, so it cannot leak outbound browsing/downloads or the real IP for outbound requests, but it is still an off-tunnel path while the tunnel is down, hence opt-in. LAN WebUI clients never need it (covered by LAN_NETWORK).
- **Type:** boolean
- **Default:** `false`
#### `WIREGUARD_STALE_AFTER`
Seconds since the last WireGuard handshake before the healthcheck bounces the tunnel.
- **Type:** number
- **Default:** `180`
</details>
## General
| Variable | Description | Type | Default |
@@ -247,7 +153,7 @@ Seconds since the last WireGuard handshake before the healthcheck bounces the tu
| `CALIBRE_WEB_URL` | Adds a navigation button to your book library (Calibre-Web Automated, Grimmory, etc). | string | _none_ |
| `AUDIOBOOK_LIBRARY_URL` | Adds a separate navigation button for your audiobook library (Audiobookshelf, Plex, etc). When both URLs are set, icons are shown instead of text. | string | _none_ |
| `SUPPORTED_FORMATS` | Book formats to include in search results. ZIP/RAR archives are extracted automatically and book files are used if found. | string (comma-separated) | `epub,mobi,azw3,fb2,djvu,cbz,cbr` |
| `SUPPORTED_AUDIOBOOK_FORMATS` | Audiobook formats to include in search results. ZIP/RAR archives are extracted automatically and audiobook files are used if found. | string (comma-separated) | `m4b,mp3,m4a,mp4,flac,ogg,wma,aac,wav,opus,zip,rar` |
| `SUPPORTED_AUDIOBOOK_FORMATS` | Audiobook formats to include in search results. ZIP/RAR archives are extracted automatically and audiobook files are used if found. | string (comma-separated) | `m4b,mp3` |
| `BOOK_LANGUAGE` | Default language filter for searches. | string (comma-separated) | `en` |
<details>
@@ -296,7 +202,16 @@ Book formats to include in search results. ZIP/RAR archives are extracted automa
Audiobook formats to include in search results. ZIP/RAR archives are extracted automatically and audiobook files are used if found.
- **Type:** string (comma-separated)
- **Default:** `m4b,mp3,m4a,mp4,flac,ogg,wma,aac,wav,opus,zip,rar`
- **Default:** `m4b,mp3`
#### `BOOK_LANGUAGE`
**Default Book Languages**
Default language filter for searches.
- **Type:** string (comma-separated)
- **Default:** `en`
</details>
@@ -305,11 +220,9 @@ Audiobook formats to include in search results. ZIP/RAR archives are extracted a
| Variable | Description | Type | Default |
|----------|-------------|------|---------|
| `SEARCH_MODE` | How you want to search for and download books. | string (choice) | `universal` |
| `BOOK_LANGUAGE` | Default language filter for searches. Users can override this for their own account. | string (comma-separated) | `en` |
| `AA_DEFAULT_SORT` | Default sort order for search results. | string (choice) | `relevance` |
| `SHOW_RELEASE_SOURCE_LINKS` | Show clickable release-source links in release and details modals. Metadata provider links stay enabled. | boolean | `true` |
| `SHOW_COMBINED_SELECTOR` | Show the option to search for and download both a book and audiobook together. | boolean | `true` |
| `FORCE_COMBINED_SEARCH` | Force combined search whenever it's available. Locks the combined toggle on. | boolean | `false` |
| `METADATA_PROVIDER` | Choose which metadata provider to use for book searches. | string (choice) | `openlibrary` |
| `METADATA_PROVIDER_AUDIOBOOK` | Metadata provider for audiobook searches. Uses the book provider if not set. | string (choice) | _empty string_ |
| `METADATA_PROVIDER_COMBINED` | Metadata provider for combined mode searches. Uses the book provider if not set. | string (choice) | _empty string_ |
@@ -329,15 +242,6 @@ How you want to search for and download books.
- **Default:** `universal`
- **Options:** `direct` (Direct), `universal` (Universal)
#### `BOOK_LANGUAGE`
**Default Book Languages**
Default language filter for searches. Users can override this for their own account.
- **Type:** string (comma-separated)
- **Default:** `en`
#### `AA_DEFAULT_SORT`
**Default Sort Order**
@@ -366,15 +270,6 @@ Show the option to search for and download both a book and audiobook together.
- **Type:** boolean
- **Default:** `true`
#### `FORCE_COMBINED_SEARCH`
**Always Use Combined Search**
Force combined search whenever it's available. Locks the combined toggle on.
- **Type:** boolean
- **Default:** `false`
#### `METADATA_PROVIDER`
**Book Metadata Provider**
@@ -434,8 +329,8 @@ The release source tab to open by default in the release modal for audiobooks. U
| `BOOKS_OUTPUT_MODE` | Choose where completed book files are sent. | string (choice) | `folder` |
| `INGEST_DIR` | Directory where downloaded files are saved. Use {User} for per-user folders (e.g. /books/{User}). | string | `/books` |
| `FILE_ORGANIZATION` | Choose how downloaded book files are named and organized. | string (choice) | `rename` |
| `TEMPLATE_RENAME` | Variables: {Author}, {Title}, {Year}, {Language}, {User}, {OriginalName} (source filename without extension). Universal adds: {Series}, {SeriesPosition}, {Subtitle}, {PrimaryTitle}. Use arbitrary prefix/suffix: {Vol. SeriesPosition - } outputs 'Vol. 2 - ' when set, nothing when empty. Rename templates are filename-only (no '/' or '\'); use Organize for folders. Applies to single-file downloads. | string | `{Author} - {Title} ({Year})` |
| `TEMPLATE_ORGANIZE` | Use / to create folders. Variables: {Author}, {Title}, {Year}, {Language}, {User}, {OriginalName} (source filename without extension). Universal adds: {Series}, {SeriesPosition}, {Subtitle}, {PrimaryTitle}. Use arbitrary prefix/suffix: {Vol. SeriesPosition - } outputs 'Vol. 2 - ' when set, nothing when empty. | string | `{Author}/{Title} ({Year})` |
| `TEMPLATE_RENAME` | Variables: {Author}, {Title}, {Year}, {User}, {OriginalName} (source filename without extension). Universal adds: {Series}, {SeriesPosition}, {Subtitle}, {PrimaryTitle}. Use arbitrary prefix/suffix: {Vol. SeriesPosition - } outputs 'Vol. 2 - ' when set, nothing when empty. Rename templates are filename-only (no '/' or '\'); use Organize for folders. Applies to single-file downloads. | string | `{Author} - {Title} ({Year})` |
| `TEMPLATE_ORGANIZE` | Use / to create folders. Variables: {Author}, {Title}, {Year}, {User}, {OriginalName} (source filename without extension). Universal adds: {Series}, {SeriesPosition}, {Subtitle}, {PrimaryTitle}. Use arbitrary prefix/suffix: {Vol. SeriesPosition - } outputs 'Vol. 2 - ' when set, nothing when empty. | string | `{Author}/{Title} ({Year})` |
| `HARDLINK_TORRENTS` | Create hardlinks instead of copying. Preserves seeding but archives won't be extracted. Don't use if destination is a library ingest folder. | boolean | `false` |
| `BOOKLORE_HOST` | Base URL of your Grimmory instance | string | _none_ |
| `BOOKLORE_USERNAME` | Grimmory account username | string | _none_ |
@@ -456,8 +351,8 @@ The release source tab to open by default in the release modal for audiobooks. U
| `EMAIL_ALLOW_UNVERIFIED_TLS` | Disable TLS certificate verification (not recommended). | boolean | `false` |
| `DESTINATION_AUDIOBOOK` | Directory where downloaded audiobook files are saved. Leave empty to use the Books destination. | string | _none_ |
| `FILE_ORGANIZATION_AUDIOBOOK` | Choose how downloaded audiobook files are named and organized. | string (choice) | `rename` |
| `TEMPLATE_AUDIOBOOK_RENAME` | Variables: {Author}, {Title}, {Year}, {Language}, {User}, {OriginalName} (source filename without extension), {Series}, {SeriesPosition}, {Subtitle}, {PrimaryTitle}, {PartNumber}. Use arbitrary prefix/suffix: {Vol. SeriesPosition - } outputs 'Vol. 2 - ' when set, nothing when empty. Rename templates are filename-only (no '/' or '\'); use Organize for folders. Applies to single-file downloads. | string | `{Author} - {Title}` |
| `TEMPLATE_AUDIOBOOK_ORGANIZE` | Use / to create folders. Variables: {Author}, {Title}, {Year}, {Language}, {User}, {OriginalName} (source filename without extension), {Series}, {SeriesPosition}, {Subtitle}, {PrimaryTitle}, {PartNumber}. Use arbitrary prefix/suffix: {Vol. SeriesPosition - } outputs 'Vol. 2 - ' when set, nothing when empty. | string | `{Author}/{Title}/{Title}` |
| `TEMPLATE_AUDIOBOOK_RENAME` | Variables: {Author}, {Title}, {Year}, {User}, {OriginalName} (source filename without extension), {Series}, {SeriesPosition}, {Subtitle}, {PrimaryTitle}, {PartNumber}. Use arbitrary prefix/suffix: {Vol. SeriesPosition - } outputs 'Vol. 2 - ' when set, nothing when empty. Rename templates are filename-only (no '/' or '\'); use Organize for folders. Applies to single-file downloads. | string | `{Author} - {Title}` |
| `TEMPLATE_AUDIOBOOK_ORGANIZE` | Use / to create folders. Variables: {Author}, {Title}, {Year}, {User}, {OriginalName} (source filename without extension), {Series}, {SeriesPosition}, {Subtitle}, {PrimaryTitle}, {PartNumber}. Use arbitrary prefix/suffix: {Vol. SeriesPosition - } outputs 'Vol. 2 - ' when set, nothing when empty. | string | `{Author}/{Title}/{Title}` |
| `HARDLINK_TORRENTS_AUDIOBOOK` | Create hardlinks instead of copying. Preserves seeding but archives won't be extracted. Don't use if destination is a library ingest folder. | boolean | `true` |
| `AUTO_OPEN_DOWNLOADS_SIDEBAR` | Automatically open the downloads sidebar when a new download is queued. | boolean | `false` |
| `DOWNLOAD_TO_BROWSER_CONTENT_TYPES` | Automatically download completed files to your browser for the selected content types. | string (comma-separated) | _empty list_ |
@@ -501,7 +396,7 @@ Choose how downloaded book files are named and organized.
**Naming Template**
Variables: {Author}, {Title}, {Year}, {Language}, {User}, {OriginalName} (source filename without extension). Universal adds: {Series}, {SeriesPosition}, {Subtitle}, {PrimaryTitle}. Use arbitrary prefix/suffix: {Vol. SeriesPosition - } outputs 'Vol. 2 - ' when set, nothing when empty. Rename templates are filename-only (no '/' or '\'); use Organize for folders. Applies to single-file downloads.
Variables: {Author}, {Title}, {Year}, {User}, {OriginalName} (source filename without extension). Universal adds: {Series}, {SeriesPosition}, {Subtitle}, {PrimaryTitle}. Use arbitrary prefix/suffix: {Vol. SeriesPosition - } outputs 'Vol. 2 - ' when set, nothing when empty. Rename templates are filename-only (no '/' or '\'); use Organize for folders. Applies to single-file downloads.
- **Type:** string
- **Default:** `{Author} - {Title} ({Year})`
@@ -510,7 +405,7 @@ Variables: {Author}, {Title}, {Year}, {Language}, {User}, {OriginalName} (source
**Path Template**
Use / to create folders. Variables: {Author}, {Title}, {Year}, {Language}, {User}, {OriginalName} (source filename without extension). Universal adds: {Series}, {SeriesPosition}, {Subtitle}, {PrimaryTitle}. Use arbitrary prefix/suffix: {Vol. SeriesPosition - } outputs 'Vol. 2 - ' when set, nothing when empty.
Use / to create folders. Variables: {Author}, {Title}, {Year}, {User}, {OriginalName} (source filename without extension). Universal adds: {Series}, {SeriesPosition}, {Subtitle}, {PrimaryTitle}. Use arbitrary prefix/suffix: {Vol. SeriesPosition - } outputs 'Vol. 2 - ' when set, nothing when empty.
- **Type:** string
- **Default:** `{Author}/{Title} ({Year})`
@@ -705,13 +600,13 @@ Choose how downloaded audiobook files are named and organized.
- **Type:** string (choice)
- **Default:** `rename`
- **Options:** `none` (None), `rename` (Rename Only), `organize` (Rename and Organize), `rename_and_group` (Rename and Group)
- **Options:** `none` (None), `rename` (Rename Only), `organize` (Rename and Organize)
#### `TEMPLATE_AUDIOBOOK_RENAME`
**Naming Template**
Variables: {Author}, {Title}, {Year}, {Language}, {User}, {OriginalName} (source filename without extension), {Series}, {SeriesPosition}, {Subtitle}, {PrimaryTitle}, {PartNumber}. Use arbitrary prefix/suffix: {Vol. SeriesPosition - } outputs 'Vol. 2 - ' when set, nothing when empty. Rename templates are filename-only (no '/' or '\'); use Organize for folders. Applies to single-file downloads.
Variables: {Author}, {Title}, {Year}, {User}, {OriginalName} (source filename without extension), {Series}, {SeriesPosition}, {Subtitle}, {PrimaryTitle}, {PartNumber}. Use arbitrary prefix/suffix: {Vol. SeriesPosition - } outputs 'Vol. 2 - ' when set, nothing when empty. Rename templates are filename-only (no '/' or '\'); use Organize for folders. Applies to single-file downloads.
- **Type:** string
- **Default:** `{Author} - {Title}`
@@ -720,7 +615,7 @@ Variables: {Author}, {Title}, {Year}, {Language}, {User}, {OriginalName} (source
**Path Template**
Use / to create folders. Variables: {Author}, {Title}, {Year}, {Language}, {User}, {OriginalName} (source filename without extension), {Series}, {SeriesPosition}, {Subtitle}, {PrimaryTitle}, {PartNumber}. Use arbitrary prefix/suffix: {Vol. SeriesPosition - } outputs 'Vol. 2 - ' when set, nothing when empty.
Use / to create folders. Variables: {Author}, {Title}, {Year}, {User}, {OriginalName} (source filename without extension), {Series}, {SeriesPosition}, {Subtitle}, {PrimaryTitle}, {PartNumber}. Use arbitrary prefix/suffix: {Vol. SeriesPosition - } outputs 'Vol. 2 - ' when set, nothing when empty.
- **Type:** string
- **Default:** `{Author}/{Title}/{Title}`
@@ -750,7 +645,6 @@ Automatically open the downloads sidebar when a new download is queued.
Automatically download completed files to your browser for the selected content types.
- **Type:** string (comma-separated)
- **Default:** _empty list_
#### `MAX_CONCURRENT_DOWNLOADS`
@@ -1049,13 +943,11 @@ Comma-separated hosts to bypass proxy (e.g., localhost,127.0.0.1,10.*,*.local)
|----------|-------------|------|---------|
| `URL_BASE` | Optional URL path prefix. Use a path like /shelfmark (no hostname). Leave blank for root. | string | _none_ |
| `DEBUG` | Enable verbose logging to console and file. Not recommended for normal use. | boolean | `false` |
| `LOG_LEVEL` | Lowest severity written to the console and log file. Ignored while Debug Mode is on, which forces Debug. | string (choice) | `INFO` |
| `MAIN_LOOP_SLEEP_TIME` | How often the download queue is checked for new items. | number | `5` |
| `DOWNLOAD_PROGRESS_UPDATE_INTERVAL` | How often download progress is broadcast to the UI. | number | `1` |
| `CUSTOM_SCRIPT` | Path to a script to run after each successful download. Must be executable. | string | _none_ |
| `CUSTOM_SCRIPT_PATH_MODE` | Pass the path to the custom script as an absolute path or relative to the destination folder. | string (choice) | `absolute` |
| `CUSTOM_SCRIPT_JSON_PAYLOAD` | Send a JSON payload to the script via stdin. Useful for multi-file imports (audiobooks) or richer metadata without relying on path parsing. | boolean | `false` |
| `DOWNLOAD_CLIENT_COMPLETED_PATH_TIMEOUT` | How long to wait after a torrent or usenet client reports completion for the completed file path to become visible to Shelfmark. Increase this for seedbox or remote-sync workflows. | number | `60` |
| `COVERS_CACHE_ENABLED` | Cache book covers on the server for faster loading. | boolean | `true` |
| `COVERS_CACHE_TTL` | How long to keep cached covers. Set to 0 to keep forever (recommended for static artwork). | number | `0` |
| `COVERS_CACHE_MAX_SIZE_MB` | Maximum disk space for cached covers. Oldest images are removed when limit is reached. | number | `500` |
@@ -1086,17 +978,6 @@ Enable verbose logging to console and file. Not recommended for normal use.
- **Default:** `false`
- **Requires restart:** Yes
#### `LOG_LEVEL`
**Log Level**
Lowest severity written to the console and log file. Ignored while Debug Mode is on, which forces Debug.
- **Type:** string (choice)
- **Default:** `INFO`
- **Requires restart:** Yes
- **Options:** `DEBUG` (Debug), `INFO` (Info), `WARNING` (Warning), `ERROR` (Error), `CRITICAL` (Critical)
#### `MAIN_LOOP_SLEEP_TIME`
**Queue Check Interval (seconds)**
@@ -1147,16 +1028,6 @@ Send a JSON payload to the script via stdin. Useful for multi-file imports (audi
- **Type:** boolean
- **Default:** `false`
#### `DOWNLOAD_CLIENT_COMPLETED_PATH_TIMEOUT`
**Completed Path Wait (seconds)**
How long to wait after a torrent or usenet client reports completion for the completed file path to become visible to Shelfmark. Increase this for seedbox or remote-sync workflows.
- **Type:** number
- **Default:** `60`
- **Constraints:** min: 0, max: 3600
#### `COVERS_CACHE_ENABLED`
**Enable Cover Cache**
@@ -1225,9 +1096,7 @@ How long to cache individual book details. Default: 600 (10 minutes). Max: 60480
| `PROWLARR_URL` | Base URL of your Prowlarr instance | string | _none_ |
| `PROWLARR_API_KEY` | Found in Prowlarr: Settings > General > API Key | string (secret) | _none_ |
| `PROWLARR_INDEXERS` | Select which indexers to search. 📚 = has book categories. Leave empty to search all. | string (comma-separated) | _empty list_ |
| `PROWLARR_INDEXER_TIMEOUT` | How long to wait for a single indexer to answer a search. Indexers behind FlareSolverr can need 90 seconds or more while a cold Cloudflare challenge is solved; raise this if searches come back empty and the Prowlarr log shows the search still running. | number | `90` |
| `PROWLARR_AUTO_EXPAND` | Automatically retry search without category filtering if no results are found | boolean | `false` |
| `PROWLARR_COLLAPSE_DUPLICATES` | Collapse a release that several indexer entries returned down to a single row, keeping the entry with the best Prowlarr priority. Turn this off to see every entry that carried it, which is what makes results from filter-specific entries (freeleech and the like) visible. | boolean | `true` |
| `PROWLARR_USE_SEED_PREFERENCES` | Apply per-indexer seed time and ratio preferences from Prowlarr when sending torrents to the download client | boolean | `false` |
<details>
@@ -1271,16 +1140,6 @@ Select which indexers to search. 📚 = has book categories. Leave empty to sear
- **Type:** string (comma-separated)
- **Default:** _empty list_
#### `PROWLARR_INDEXER_TIMEOUT`
**Indexer Search Timeout (seconds)**
How long to wait for a single indexer to answer a search. Indexers behind FlareSolverr can need 90 seconds or more while a cold Cloudflare challenge is solved; raise this if searches come back empty and the Prowlarr log shows the search still running.
- **Type:** number
- **Default:** `90`
- **Constraints:** min: 5, max: 300
#### `PROWLARR_AUTO_EXPAND`
**Auto-expand search on no results**
@@ -1290,15 +1149,6 @@ Automatically retry search without category filtering if no results are found
- **Type:** boolean
- **Default:** `false`
#### `PROWLARR_COLLAPSE_DUPLICATES`
**Show one row per release**
Collapse a release that several indexer entries returned down to a single row, keeping the entry with the best Prowlarr priority. Turn this off to see every entry that carried it, which is what makes results from filter-specific entries (freeleech and the like) visible.
- **Type:** boolean
- **Default:** `true`
#### `PROWLARR_USE_SEED_PREFERENCES`
**Use Prowlarr seed preferences**
@@ -1315,11 +1165,8 @@ Apply per-indexer seed time and ratio preferences from Prowlarr when sending tor
| Variable | Description | Type | Default |
|----------|-------------|------|---------|
| `NEWZNAB_ENABLED` | Enable searching for books via a Newznab-compatible indexer | boolean | `false` |
| `NEWZNAB_INDEXERS` | Named Newznab connections. Each row accepts `name`, `url`, and `api_key`. | JSON array | `[]` |
| `NEWZNAB_URL` | Legacy single-indexer URL, used when `NEWZNAB_INDEXERS` is empty | string | _none_ |
| `NEWZNAB_API_KEY` | Legacy single-indexer API key | string (secret) | _none_ |
| `NEWZNAB_EBOOK_CATEGORIES` | Newznab category IDs searched for ebooks. Most indexers use the standard 7000, but some use custom IDs. Leave empty to use 7000. | string (comma-separated) | `7000` |
| `NEWZNAB_AUDIOBOOK_CATEGORIES` | Newznab category IDs searched for audiobooks. Most indexers use the standard 3030, but some use custom IDs. Leave empty to use 3030. | string (comma-separated) | `3030` |
| `NEWZNAB_URL` | Base URL of your Newznab indexer or aggregator | string | _none_ |
| `NEWZNAB_API_KEY` | Your Newznab API key (leave blank if not required) | string (secret) | _none_ |
| `NEWZNAB_AUTO_EXPAND` | Automatically retry search without category filtering if no results are found | boolean | `false` |
<details>
@@ -1334,58 +1181,25 @@ Enable searching for books via a Newznab-compatible indexer
- **Type:** boolean
- **Default:** `false`
#### `NEWZNAB_INDEXERS`
**Named Indexers**
Configure multiple named Newznab-compatible indexers. The name is shown beside each search result. For environment-based configuration, provide a JSON array:
```json
[
{"name":"NZBGeek","url":"https://api.nzbgeek.info","api_key":"..."},
{"name":"DrunkenSlug","url":"https://drunkenslug.com","api_key":"..."}
]
```
- **Type:** JSON array
- **Default:** `[]`
#### `NEWZNAB_URL`
**Legacy Newznab URL**
**Newznab URL**
Single-indexer fallback used only when `NEWZNAB_INDEXERS` is empty.
Base URL of your Newznab indexer or aggregator
- **Type:** string
- **Default:** _none_
- **Required:** Yes
#### `NEWZNAB_API_KEY`
**Legacy API Key**
**API Key**
API key for the legacy Newznab URL.
Your Newznab API key (leave blank if not required)
- **Type:** string (secret)
- **Default:** _none_
#### `NEWZNAB_EBOOK_CATEGORIES`
**Ebook Categories**
Newznab category IDs searched for ebooks. Most indexers use the standard 7000, but some use custom IDs. Leave empty to use 7000.
- **Type:** string (comma-separated)
- **Default:** `7000`
#### `NEWZNAB_AUDIOBOOK_CATEGORIES`
**Audiobook Categories**
Newznab category IDs searched for audiobooks. Most indexers use the standard 3030, but some use custom IDs. Leave empty to use 3030.
- **Type:** string (comma-separated)
- **Default:** `3030`
#### `NEWZNAB_AUTO_EXPAND`
**Auto-expand search on no results**
@@ -1467,11 +1281,9 @@ Delay between requests in seconds to avoid rate limiting (0-10).
| `IRC_SERVER` | IRC server hostname | string | _none_ |
| `IRC_PORT` | IRC server port (usually 6697 for TLS, 6667 for plain) | number | `6697` |
| `IRC_USE_TLS` | Enable TLS/SSL encryption for the IRC connection. Disable for servers that don't support TLS. | boolean | `true` |
| `IRC_CHANNEL` | Channel name without the # prefix. Used for all searches unless a separate audiobook channel is configured below. | string | _none_ |
| `IRC_CHANNEL` | Channel name without the # prefix | string | _none_ |
| `IRC_NICK` | Your IRC nickname (required). Must be unique on the IRC network. | string | _none_ |
| `IRC_SEARCH_BOT` | The search bot to address queries to (required). Searches are sent as "@<bot> <query>". | string | _none_ |
| `IRC_AUDIOBOOK_CHANNEL` | Optional. Channel name (without the # prefix) for networks that index audiobooks separately, such as Undernet's bookz. Leave blank (the usual setting) to search the main channel above for audiobooks too. | string | _none_ |
| `IRC_AUDIOBOOK_SEARCH_BOT` | Optional. Search bot for the audiobook channel. Leave blank to reuse the main search bot above. Only used when an audiobook channel is set. | string | _none_ |
| `IRC_SEARCH_BOT` | The search bot to query for results | string | _none_ |
| `IRC_CACHE_TTL` | How long to keep cached search results before they expire. | string (choice) | `2592000` |
<details>
@@ -1509,7 +1321,7 @@ Enable TLS/SSL encryption for the IRC connection. Disable for servers that don't
**Channel**
Channel name without the # prefix. Used for all searches unless a separate audiobook channel is configured below.
Channel name without the # prefix
- **Type:** string
- **Default:** _none_
@@ -1529,26 +1341,7 @@ Your IRC nickname (required). Must be unique on the IRC network.
**Search bot**
The search bot to address queries to (required). Searches are sent as "@<bot> <query>".
- **Type:** string
- **Default:** _none_
- **Required:** Yes
#### `IRC_AUDIOBOOK_CHANNEL`
**Audiobook channel**
Optional. Channel name (without the # prefix) for networks that index audiobooks separately, such as Undernet's bookz. Leave blank (the usual setting) to search the main channel above for audiobooks too.
- **Type:** string
- **Default:** _none_
#### `IRC_AUDIOBOOK_SEARCH_BOT`
**Audiobook search bot**
Optional. Search bot for the audiobook channel. Leave blank to reuse the main search bot above. Only used when an audiobook channel is set.
The search bot to query for results
- **Type:** string
- **Default:** _none_
@@ -1570,12 +1363,9 @@ How long to keep cached search results before they expire.
| Variable | Description | Type | Default |
|----------|-------------|------|---------|
| `PROWLARR_TORRENT_CLIENT` | Choose which torrent client to use | string (choice) | _empty string_ |
| `ALLDEBRID_API_KEY` | AllDebrid API Key (apiv4) from your AllDebrid account settings | string (secret) | _none_ |
| `REALDEBRID_API_KEY` | Real-Debrid API Key (Secret Token) from your Real-Debrid account settings | string (secret) | _none_ |
| `QBITTORRENT_URL` | Web UI URL of your qBittorrent instance | string | _none_ |
| `QBITTORRENT_USERNAME` | qBittorrent Web UI username | string | _none_ |
| `QBITTORRENT_PASSWORD` | qBittorrent Web UI password | string (secret) | _none_ |
| `QBITTORRENT_API_KEY` | Found in qBittorrent: Options > Web UI > API Key (qBittorrent 5.2.0+). Used instead of the username and password when set. | string (secret) | _none_ |
| `QBITTORRENT_CATEGORY` | Category to assign to book downloads in qBittorrent | string | `books` |
| `QBITTORRENT_CATEGORY_AUDIOBOOK` | Category for audiobook downloads. Leave empty to use the book category. | string | _empty string_ |
| `QBITTORRENT_DOWNLOAD_DIR` | Server-side directory where torrents are downloaded (optional, uses qBittorrent default if not specified) | string | _none_ |
@@ -1595,11 +1385,9 @@ How long to keep cached search results before they expire.
| `RTORRENT_URL` | XML-RPC URL of your rTorrent instance | string | _none_ |
| `RTORRENT_USERNAME` | HTTP Basic auth username (if authentication enabled) | string | _none_ |
| `RTORRENT_PASSWORD` | HTTP Basic auth password | string (secret) | _none_ |
| `RTORRENT_LABEL` | Label to assign to ebook downloads in rTorrent | string | `cwabd` |
| `RTORRENT_AUDIOBOOK_LABEL` | Label to assign to audiobook downloads in rTorrent (falls back to Book Label if not set) | string | _none_ |
| `RTORRENT_LABEL` | Label to assign to book downloads in rTorrent | string | `cwabd` |
| `RTORRENT_DOWNLOAD_DIR` | Server-side directory where torrents are downloaded (optional, uses rTorrent default if not specified) | string | _none_ |
| `PROWLARR_TORRENT_ACTION` | Choose whether to keep, remove, or move the torrent to another category or label after import | string (choice) | `keep` |
| `PROWLARR_TORRENT_POST_IMPORT_CATEGORY` | Category or label to assign after a successful import | string | _empty string_ |
| `PROWLARR_TORRENT_ACTION` | Remove deletes the torrent from your client immediately after import (stops seeding, files are kept); Keep leaves it in the client to continue seeding | string (choice) | `keep` |
| `PROWLARR_USENET_CLIENT` | Choose which usenet client to use | string (choice) | _empty string_ |
| `NZBGET_URL` | URL of your NZBGet instance | string | _none_ |
| `NZBGET_USERNAME` | NZBGet control username | string | `nzbget` |
@@ -1623,25 +1411,7 @@ Choose which torrent client to use
- **Type:** string (choice)
- **Default:** _empty string_
- **Options:** `""` (None), `alldebrid` (AllDebrid), `qbittorrent` (qBittorrent), `realdebrid` (Real-Debrid), `transmission` (Transmission), `deluge` (Deluge), `rtorrent` (rTorrent)
#### `ALLDEBRID_API_KEY`
**API Key**
AllDebrid API Key (apiv4) from your AllDebrid account settings
- **Type:** string (secret)
- **Default:** _none_
#### `REALDEBRID_API_KEY`
**API Key**
Real-Debrid API Key (Secret Token) from your Real-Debrid account settings
- **Type:** string (secret)
- **Default:** _none_
- **Options:** `""` (None), `qbittorrent` (qBittorrent), `transmission` (Transmission), `deluge` (Deluge), `rtorrent` (rTorrent)
#### `QBITTORRENT_URL`
@@ -1670,15 +1440,6 @@ qBittorrent Web UI password
- **Type:** string (secret)
- **Default:** _none_
#### `QBITTORRENT_API_KEY`
**API Key**
Found in qBittorrent: Options > Web UI > API Key (qBittorrent 5.2.0+). Used instead of the username and password when set.
- **Type:** string (secret)
- **Default:** _none_
#### `QBITTORRENT_CATEGORY`
**Book Category**
@@ -1854,20 +1615,11 @@ HTTP Basic auth password
**Book Label**
Label to assign to ebook downloads in rTorrent
Label to assign to book downloads in rTorrent
- **Type:** string
- **Default:** `cwabd`
#### `RTORRENT_AUDIOBOOK_LABEL`
**Audiobook Label**
Label to assign to audiobook downloads in rTorrent (falls back to Book Label if not set)
- **Type:** string
- **Default:** _none_
#### `RTORRENT_DOWNLOAD_DIR`
**Download Directory**
@@ -1881,20 +1633,11 @@ Server-side directory where torrents are downloaded (optional, uses rTorrent def
**Torrent Completion Action**
Choose whether to keep, remove, or move the torrent to another category or label after import
Remove deletes the torrent from your client immediately after import (stops seeding, files are kept); Keep leaves it in the client to continue seeding
- **Type:** string (choice)
- **Default:** `keep`
- **Options:** `keep` (Keep), `remove` (Remove), `change_category` (Change Category)
#### `PROWLARR_TORRENT_POST_IMPORT_CATEGORY`
**Post-Import Category**
Category or label to assign after a successful import
- **Type:** string
- **Default:** _empty string_
- **Options:** `keep` (Keep), `remove` (Remove)
#### `PROWLARR_USENET_CLIENT`
@@ -2006,7 +1749,7 @@ Move deletes the job from your usenet client after import; Copy keeps it in the
| Variable | Description | Type | Default |
|----------|-------------|------|---------|
| `HARDCOVER_ENABLED` | Enable Hardcover as a metadata provider for book searches | boolean | `false` |
| `HARDCOVER_API_KEY` | Get your API key from hardcover.app/account/api (starts with hc_pat_) | string (secret) | _none_ |
| `HARDCOVER_API_KEY` | Get your API key from hardcover.app/account/api | string (secret) | _none_ |
| `HARDCOVER_DEFAULT_SORT` | Default sort order for Hardcover search results. | string (choice) | `relevance` |
| `HARDCOVER_EXCLUDE_COMPILATIONS` | Filter out compilations, anthologies, and omnibus editions from search results | boolean | `false` |
| `HARDCOVER_EXCLUDE_UNRELEASED` | Filter out books with a release year in the future | boolean | `false` |
@@ -2028,7 +1771,7 @@ Enable Hardcover as a metadata provider for book searches
**API Key**
Get your API key from hardcover.app/account/api (starts with hc_pat_)
Get your API key from hardcover.app/account/api
- **Type:** string (secret)
- **Default:** _none_
@@ -2146,26 +1889,6 @@ Default sort order for Google Books search results.
</details>
### Metadata Providers: Moly.hu
| Variable | Description | Type | Default |
|----------|-------------|------|---------|
| `MOLY_ENABLED` | Enable Moly.hu as a metadata provider for book searches | boolean | `false` |
<details>
<summary>Detailed descriptions</summary>
#### `MOLY_ENABLED`
**Enable Moly.hu**
Enable Moly.hu as a metadata provider for book searches
- **Type:** boolean
- **Default:** `false`
</details>
## Direct Download
### Direct Download: Download Sources
@@ -2173,13 +1896,11 @@ Enable Moly.hu as a metadata provider for book searches
| Variable | Description | Type | Default |
|----------|-------------|------|---------|
| `DIRECT_DOWNLOAD_ENABLED` | Show Direct Download in release-source lists and allow Direct mode searches. Add your own mirror URLs in the Mirrors tab before using it. | boolean | `false` |
| `DIRECT_DOWNLOAD_LANGUAGE_FROM_PATH` | When language metadata is missing or unknown, parse the distant path (file path shown in search results) for language tags like [BD FR] or [En]. Also enables local language filtering so lgli files without AA language metadata are not excluded before the distant path can be checked. | boolean | `false` |
| `AA_DONATOR_KEY` | Enables fast download access on AA. Get this from your donator account page. | string (secret) | _none_ |
| `FAST_SOURCES_DISPLAY` | Always tried first, no waiting or bypass required. | JSON array | _see UI for defaults_ |
| `SOURCE_PRIORITY` | Fallback sources, may have waiting. Requires bypasser. Drag to reorder. | JSON array | _see UI for defaults_ |
| `MAX_RETRY` | Maximum retry attempts for failed downloads. | number | `10` |
| `DEFAULT_SLEEP` | Wait time between download retry attempts. | number | `5` |
| `RELEASE_SEARCH_TIMEOUT` | How long one release search may run before it gives up and reports why. A first search on a cold start pays for a browser solve, so leave room for one. If you use a reverse proxy, its read timeout should be at least this high or it will cut the search off with a 504 first. | number | `300` |
| `AA_CONTENT_TYPE_ROUTING` | Override destination based on content type metadata. | boolean | `false` |
| `AA_CONTENT_TYPE_DIR_FICTION` | Fiction Books | string | _none_ |
| `AA_CONTENT_TYPE_DIR_NON_FICTION` | Non-Fiction Books | string | _none_ |
@@ -2202,15 +1923,6 @@ Show Direct Download in release-source lists and allow Direct mode searches. Add
- **Type:** boolean
- **Default:** `false`
#### `DIRECT_DOWNLOAD_LANGUAGE_FROM_PATH`
**Detect Language From Distant Path**
When language metadata is missing or unknown, parse the distant path (file path shown in search results) for language tags like [BD FR] or [En]. Also enables local language filtering so lgli files without AA language metadata are not excluded before the distant path can be checked.
- **Type:** boolean
- **Default:** `false`
#### `AA_DONATOR_KEY`
**Account Donator Key**
@@ -2258,16 +1970,6 @@ Wait time between download retry attempts.
- **Default:** `5`
- **Constraints:** min: 1, max: 60
#### `RELEASE_SEARCH_TIMEOUT`
**Release Search Timeout (seconds)**
How long one release search may run before it gives up and reports why. A first search on a cold start pays for a browser solve, so leave room for one. If you use a reverse proxy, its read timeout should be at least this high or it will cut the search off with a 504 first.
- **Type:** number
- **Default:** `300`
- **Constraints:** min: 30, max: 1800
#### `AA_CONTENT_TYPE_ROUTING`
**Enable Content-Type Routing**
@@ -2344,8 +2046,6 @@ Override destination based on content type metadata.
| `EXT_BYPASSER_URL` | URL of the external bypasser service (e.g., FlareSolverr). | string | `http://flaresolverr:8191` |
| `EXT_BYPASSER_PATH` | API path for the external bypasser. | string | `/v1` |
| `EXT_BYPASSER_TIMEOUT` | Timeout for external bypasser requests in milliseconds. | number | `60000` |
| `BYPASS_PAGE_SOURCE_TIMEOUT` | How long to wait for a solved page to produce its content before the bypass is retried. Raise it if solves succeed but searches still fail. | number | `20` |
| `BYPASS_BROWSER_IDLE_TIMEOUT` | How long the bypass helper process may sit unused before it is shut down. Higher keeps more searches fast, lower frees memory sooner. | number | `180` |
<details>
<summary>Detailed descriptions</summary>
@@ -2401,27 +2101,6 @@ Timeout for external bypasser requests in milliseconds.
- **Requires restart:** Yes
- **Constraints:** min: 10000, max: 300000
#### `BYPASS_PAGE_SOURCE_TIMEOUT`
**Page Read Timeout (seconds)**
How long to wait for a solved page to produce its content before the bypass is retried. Raise it if solves succeed but searches still fail.
- **Type:** number
- **Default:** `20`
- **Constraints:** min: 1, max: 120
#### `BYPASS_BROWSER_IDLE_TIMEOUT`
**Bypasser Idle Timeout (seconds)**
How long the bypass helper process may sit unused before it is shut down. Higher keeps more searches fast, lower frees memory sooner.
- **Type:** number
- **Default:** `180`
- **Requires restart:** Yes
- **Constraints:** min: 30, max: 3600
</details>
### Direct Download: Mirrors
+3 -17
View File
@@ -12,7 +12,7 @@ With a subpath (`URL_BASE=/shelfmark/`):
https://<your-shelfmark-domain>/shelfmark/api/auth/oidc/callback
```
The callback URL is constructed from the incoming request, so your reverse proxy must forward `X-Forwarded-Proto` and `X-Forwarded-Host` correctly, including the external port when it is not the protocol default. PKCE (S256) is used automatically.
The callback URL is constructed from the incoming request, so your reverse proxy must forward `X-Forwarded-Proto` and `X-Forwarded-Host` correctly. PKCE (S256) is used automatically.
## Settings
@@ -30,19 +30,7 @@ Configure in **Settings → Security → Authentication Method → OIDC**.
| Auto-Provision Users | Create accounts on first login | `true` |
| Login Button Label | Custom text for the sign-in button | — |
Use **Test Connection** to verify discovery, client configuration, and the provider's token signing keys (JWKS) before attempting login.
> **Authentik users:** make sure your provider has a **Signing Key** selected (e.g. the default self-signed certificate). Without one, Authentik serves an empty JWKS document and every login fails with an OIDC callback error, even though the discovery document looks healthy.
## Account Linking
On login, Shelfmark matches the OIDC identity to a user account in this order:
1. **OIDC subject** — a user who has logged in through this provider before.
2. **Email** — a local account with the same (unique) email address. This only happens when the provider also asserts `email_verified: true` for the address; an unverified email would let anyone claim a local account by registering its address at the IdP.
3. Otherwise, a new account is created when **Auto-Provision Users** is enabled (username conflicts get a numeric suffix), or the login is rejected with "Account not found" when it is disabled.
If the `email_verified` claim is missing or `false`, email linking is silently skipped — a common surprise when the address was never verified at the identity provider (e.g. Keycloak's **Email verified** toggle on the user, or Authentik accounts created without email verification). Make sure the `email` scope is requested and the address is marked verified in your IdP.
Use **Test Connection** to verify discovery and client configuration before attempting login.
## Environment Variables
@@ -58,8 +46,6 @@ If `DISABLE_LOCAL_AUTH` and `OIDC_AUTO_REDIRECT` are both enabled, users are red
## Troubleshooting
- **No token signing keys (empty JWKS)** — The provider's JWKS endpoint returned no keys, so ID tokens can't be verified. In Authentik this happens when the provider has no **Signing Key** selected; pick one (e.g. the default self-signed certificate) and try again.
- **Issuer validation failed** — The issuer in the token doesn't match the discovery document. Check your provider's external URL / issuer configuration.
- **Callback URL mismatch** — Reverse proxy isn't forwarding `X-Forwarded-Proto` or `X-Forwarded-Host`, so the constructed callback URL doesn't match what's registered in the provider.
- **Account not found** — Auto-provision is disabled and the user hasn't been pre-created by an admin. If you pre-created the account with a matching email, see [Account Linking](#account-linking): the provider must send `email_verified: true` for linking to happen.
- **Login created a duplicate account instead of linking to my local one** — Email linking requires a verified email; see [Account Linking](#account-linking). With `DEBUG=true`, the log notes when linking is skipped because the address isn't verified.
- **Account not found** — Auto-provision is disabled and the user hasn't been pre-created by an admin.
+5 -7
View File
@@ -23,11 +23,10 @@ server {
location / {
proxy_pass http://shelfmark:8084;
proxy_http_version 1.1;
proxy_set_header Host $http_host;
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $scheme;
proxy_set_header X-Forwarded-Host $http_host;
proxy_set_header Upgrade $http_upgrade;
proxy_set_header Connection $connection_upgrade;
}
@@ -57,11 +56,11 @@ All Shelfmark paths (UI, API, assets, Socket.IO) are served under the base path.
location /shelfmark/ {
proxy_pass http://shelfmark:8084/shelfmark/;
proxy_http_version 1.1;
proxy_set_header Host $http_host;
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $scheme;
proxy_set_header X-Forwarded-Host $http_host;
proxy_set_header X-Forwarded-Host $host;
proxy_set_header Upgrade $http_upgrade;
proxy_set_header Connection $connection_upgrade;
proxy_read_timeout 86400;
@@ -137,11 +136,11 @@ location /shelfmark/ {
proxy_pass http://shelfmark:8084/shelfmark/;
proxy_http_version 1.1;
proxy_set_header Host $http_host;
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $scheme;
proxy_set_header X-Forwarded-Host $http_host;
proxy_set_header X-Forwarded-Host $host;
proxy_set_header Upgrade $http_upgrade;
proxy_set_header Connection $connection_upgrade;
proxy_read_timeout 86400;
@@ -159,7 +158,6 @@ If login, settings saves, or downloads appear to fail in the browser but the act
- Do not force `Connection: upgrade` on every request. That can break normal `POST` and `PUT` responses while the backend still processes them.
- If your proxy UI does not support conditional websocket headers, remove the forced websocket headers entirely and let Shelfmark fall back to polling.
- Keep the standard forwarded headers: `Host`, `X-Forwarded-For`, `X-Forwarded-Proto`, and `X-Forwarded-Host` when using a subpath or OIDC.
- Preserve the original port in `Host` and `X-Forwarded-Host` by using `$http_host` rather than `$host` when Shelfmark is exposed on a custom port.
This is especially relevant for Nginx Proxy Manager or custom advanced config snippets that add websocket headers globally.
+1 -8
View File
@@ -19,7 +19,7 @@ http://your-server:8084/?q=harry+potter
| `lang` | Filter by language (ISO 639-1 code) | `/?lang=en` |
| `format` | Filter by file format | `/?format=epub` |
| `content` | Filter by content type | `/?content=fiction` |
| `content_type` | Select media type (`ebook`, `audiobook`, or `combined`) in Universal mode only | `/?q=dune&content_type=audiobook` |
| `content_type` | Select media type (`ebook` or `audiobook`) in Universal mode only | `/?q=dune&content_type=audiobook` |
| `sort` | Sort order for results | `/?sort=newest` |
## Multiple Values
@@ -63,11 +63,6 @@ Some parameters support multiple values by repeating the parameter:
/?q=dune&content_type=audiobook
```
**Universal search forcing combined (ebook + audiobook):**
```
/?q=dune&content_type=combined
```
## Search Mode Behavior
### Direct Mode
@@ -79,8 +74,6 @@ When Search Mode is set to Direct, all parameters are used to filter results fro
`q`, `sort`, and `content_type` are used. Other parameters (author, title, format, etc.) are silently ignored since metadata providers have their own search capabilities.
`content_type=combined` forces combined mode (search ebook and audiobook providers together), overriding the last-used preference. It is silently ignored if combined mode is unavailable (e.g. the combined selector is disabled in settings, or either content type is blocked by request policy).
## Notes
- URL parameters are read once on page load
+1 -10
View File
@@ -30,7 +30,7 @@ Requires mounting your Calibre-Web `app.db` to `/auth/app.db`.
Admins can configure per-user settings by editing a user in the user management panel. Non-admin users can also edit their own settings through **My Account** (accessible from the user menu). Admins control which sections are visible in My Account via the **Visible Self-Settings Sections** option.
There are four categories of per-user settings:
There are three categories of per-user settings:
### Delivery Preferences
@@ -42,15 +42,6 @@ Override where a user's downloads are sent. Options depend on the global output
- **BookLore library/path** — Per-user BookLore target (when using BookLore output mode)
- **Email recipient** — Per-user email address (when using Email output mode)
### Search Preferences
Override how a user searches, on top of the global search defaults:
- **Search mode** — Direct or Universal for this user
- **Default book languages** — The languages a user's searches fall back to when they don't pick one themselves. Useful for a shared instance where readers want different languages.
- **Metadata providers** — Book, audiobook, and combined-mode provider for this user
- **Default release sources** — The release tab opened first for books and audiobooks
### Notifications
Users can configure personal notification routes, separate from the global notification settings. Each route targets a URL (e.g. an Apprise-compatible endpoint) and can be scoped to specific event types or all events.
+3 -45
View File
@@ -81,13 +81,6 @@ if is_truthy "$ENABLE_LOGGING_VALUE"; then
fi
fi
# Egress modes are mutually exclusive. Check this BEFORE starting either one so
# we never run tor.sh and then abort, leaving a half-configured network stack.
if [ "$USING_TOR" = "true" ] && [ "$USING_WIREGUARD" = "true" ]; then
echo "USING_TOR and USING_WIREGUARD are mutually exclusive; enable only one egress mode." >&2
exit 1
fi
if [ "$USING_TOR" = "true" ]; then
if [ "$RUN_AS_NON_ROOT" = "true" ]; then
echo "USING_TOR=true requires the container to start as root." >&2
@@ -97,15 +90,6 @@ if [ "$USING_TOR" = "true" ]; then
./tor.sh
fi
if [ "$USING_WIREGUARD" = "true" ]; then
if [ "$RUN_AS_NON_ROOT" = "true" ]; then
echo "USING_WIREGUARD=true requires the container to start as root." >&2
echo "Non-root mode skips the privileged network setup WireGuard depends on." >&2
exit 1
fi
./wireguard.sh
fi
if [ "$FILE_LOGGING_ENABLED" = "true" ]; then
start_file_logging "$LOG_FILE"
fi
@@ -251,22 +235,13 @@ test_write() {
return 1
fi
# This is a probe: a failure here is expected (e.g. a fresh root-owned bind
# mount) and is recovered by the caller via change_ownership + re-probe. Hide
# the shell's "Permission denied"/"Read-only file system" stderr so a handled
# probe miss doesn't masquerade as a real boot failure in the logs.
if ! run_as_target_user sh -c 'echo 0123456789_TEST 2>/dev/null > "$1"' _ "$test_file"; then
if ! run_as_target_user sh -c 'echo 0123456789_TEST > "$1"' _ "$test_file"; then
echo "Failed to write test file in $folder as $USERNAME"
return 1
fi
FILE_CONTENT=$(cat "$test_file" 2>/dev/null || echo "")
# A folder can be writable but not deletable (e.g. a Synology share without
# "Delete subfolders and files"). That is not a boot failure - the app writes
# files in place on such shares - so don't let a failed cleanup print an
# alarming error or fail the probe.
run_as_target_user rm -f "$test_file" 2>/dev/null || \
echo "Note: could not remove test file in $folder (folder is writable but not deletable)"
rm -f "$test_file"
[ "$FILE_CONTENT" = "0123456789_TEST" ]
result=$?
if [ $result -eq 0 ]; then
@@ -473,29 +448,12 @@ else
if [ $config_ok -ne 0 ]; then
fail_unwritable_config_dir "$CONFIG_PATH"
fi
# The ingest/destination library (default /books) is user data and may be a
# bind mount owned by another uid; downloads fail with "Destination not
# writable" if the runtime user can't write there. Fix the top-level dir only
# (root mode) so we don't recursively chown a potentially huge library.
make_writable "${INGEST_DIR:-/books}" root
fi
# Always run Gunicorn (even when DEBUG=true) to ensure Socket.IO WebSocket
# upgrades work reliably on customer machines.
# Map app LOG_LEVEL (often DEBUG/INFO/...) to gunicorn's --log-level (lowercase).
# Gunicorn rejects anything outside its own list, so normalize and fall back to
# info rather than letting a typo stop the container from booting.
if [ "$DEBUG" = "true" ]; then
gunicorn_loglevel=debug
else
gunicorn_loglevel=$(echo "${LOG_LEVEL:-info}" | tr '[:upper:]' '[:lower:]')
[ "$gunicorn_loglevel" = "warn" ] && gunicorn_loglevel=warning
case "$gunicorn_loglevel" in
debug|info|warning|error|critical) ;;
*) gunicorn_loglevel=info ;;
esac
fi
gunicorn_loglevel=$([ "$DEBUG" = "true" ] && echo debug || echo "${LOG_LEVEL:-info}" | tr '[:upper:]' '[:lower:]')
command="${GUNICORN_BIN} --log-level ${gunicorn_loglevel} --access-logfile - --error-logfile - --worker-class geventwebsocket.gunicorn.workers.GeventWebSocketWorker --workers 1 -t 300 -b ${FLASK_HOST:-0.0.0.0}:${FLASK_PORT:-8084} shelfmark.main:app"
# If DEBUG and not using an external bypass
+5 -8
View File
@@ -19,31 +19,28 @@ dependencies = [
"psutil",
"emoji",
"rarfile",
"qbittorrent-api>=2026.8.1",
"qbittorrent-api",
"transmission-rpc",
"authlib>=1.7.2,<1.8",
"apprise>=1.13.0",
# HTTP/2 client for RFC 8484 DoH: quad9 rejects HTTP/1.1 outright (505), which
# requests cannot speak. See shelfmark/download/doh_wireformat.py.
"httpx[http2]>=0.28.1",
"apprise>=1.10.0",
]
[project.optional-dependencies]
browser = [
"pyvirtualdisplay",
"pyautogui",
"seleniumbase==4.53.5",
"seleniumbase==4.48.4",
"python-xlib",
]
[dependency-groups]
dev = [
"basedpyright>=1.39.10",
"basedpyright>=1.39.3",
"prek",
"pytest",
"pytest-cov",
"pytest-xdist>=3.8.0",
"ruff==0.16.5",
"ruff==0.15.12",
"vulture>=2.14",
]
+4 -78
View File
@@ -2,9 +2,6 @@
<img src="src/frontend/public/logo.png" alt="Shelfmark" width="200">
> [!NOTE]
> Shelfmark is feature stable and maintained on a best-effort basis. Bug fixes, security updates, and small quality-of-life improvements are still shipped, and pull requests are reviewed — including new features. There is no roadmap for new features for now.
Shelfmark is a self-hosted web interface for searching and requesting books and audiobooks across multiple sources. Bring your own sources, metadata providers, and download clients to build a single hub for your digital library. Supports multiple users with a built-in request system, so you can share your instance with others and let them browse and request books on their own.
Works great alongside the following library tools, with support for automatic imports:
@@ -44,7 +41,6 @@ Works great alongside the following library tools, with support for automatic im
### Prerequisites
- Docker & Docker Compose
- At least 2 GB of RAM available to the container when using the standard image — see [Memory Requirements](#memory-requirements)
### Installation
@@ -95,30 +91,6 @@ volumes:
- Aggregates releases from multiple configured sources
- Full audiobook support
### Hardcover API Key
Hardcover powers metadata search in Universal mode. Create a token at
[hardcover.app/account/api](https://hardcover.app/account/api) — current keys start with `hc_pat_`
and are far shorter than the JWTs Hardcover issued before August 2026.
Tick these seven scopes on the token screen:
| Scope | Used for |
|-------|----------|
| `read:catalog` | Metadata search, plus book, edition, author and series lookups |
| `read:library` | Your reading status and shelf counts |
| `read:lists` | Your lists and the books on them |
| `read:me:content` | Test Connection and the "Connected as" label |
| `read:users` | Usernames shown alongside lists |
| `write:library` | Setting a book's reading status from Shelfmark |
| `write:lists` | Adding and removing books from lists, including auto-remove on download |
The two `write:` scopes matter only if you set reading status from Shelfmark or leave
**Auto-Remove from List on Download** enabled (it is on by default) — without them those actions
fail silently. Everything else Hardcover offers (journal, goals, reviews, prompts, notifications,
account) can stay unticked. The `all` scope works too, but it grants full account access including
deletion, so prefer the list above.
### Environment Variables
Environment variables work for initial setup and Docker deployments. They serve as defaults that can be overridden in the web interface.
@@ -131,24 +103,13 @@ Environment variables work for initial setup and Docker deployments. They serve
| `PUID` / `PGID` | Runtime user/group for the default root-startup flow (also supports legacy `UID`/`GID`) | `1000` / `1000` |
| `SEARCH_MODE` | `direct` or `universal` | `universal` |
| `USING_TOR` | Enable Tor routing (requires root startup) | `false` |
| `USING_WIREGUARD` | Enable WireGuard VPN egress with kill-switch (requires root startup) | `false` |
| `WIREGUARD_CONFIG` | Path to the mounted wg-quick config | `/config/wg0.conf` |
| `WIREGUARD_INTERFACE` | WireGuard interface name | `wg0` |
| `LAN_NETWORK` | Comma-separated CIDRs kept off the tunnel so the WebUI / internal clients stay reachable | `127.0.0.0/8,10.0.0.0/8,172.16.0.0/12,192.168.0.0/16` |
| `WIREGUARD_ENFORCE_DNS` | Pin the resolver (via `WIREGUARD_DNS`, else the config's `DNS =`) so DNS can't silently fall back to an off-tunnel path. Designed for a trusted LAN resolver kept reachable via `LAN_NETWORK` (query leaves over the LAN; download still egresses via the tunnel) — it does **not** force queries through the tunnel. Docker's embedded resolver (`127.0.0.11`) is preserved when present so container-name resolution keeps working; pin its upstream via the container's `dns:` list. Fails closed if no resolver is available or `/etc/resolv.conf` is not writable. | `true` |
| `WIREGUARD_DNS` | Explicit resolver(s) to pin (comma/space separated). Use when the VPN's pushed DNS filters domains you need; point at a resolver reachable via the tunnel or an allowed LAN resolver. | _(unset; uses config `DNS =`)_ |
| `WIREGUARD_DISABLE_IPV6` | Strip IPv6 from the tunnel config (many container kernels lack the ip6tables `raw` table wg-quick needs) and remove IPv6 as a leak surface. | `true` |
| `WIREGUARD_ALLOW_IPV6_LEAK` | Escape hatch: continue even when an IPv6 kill-switch can't be installed AND IPv6 can't be disabled. Only set if the container has no IPv6 connectivity. | `false` |
| `WIREGUARD_ALLOW_WEBUI_OFFTUNNEL` | Opt-in off-tunnel WebUI reachability. Default (`false`) keeps the kill-switch strictly fail-closed: the only off-tunnel egress is loopback, the tunnel device and the LAN allowlist. Set `true` only if a **non-LAN** client (e.g. a public reverse proxy on another segment) must reach the WebUI; it permits app-server **replies** (`--sport FLASK_PORT`, conntrack REPLY) off-tunnel — server replies only, never client-initiated egress. LAN clients never need it (covered by `LAN_NETWORK`). | `false` |
| `WIREGUARD_STALE_AFTER` | Seconds since the last handshake before the healthcheck bounces the tunnel. | `180` |
See the full [Environment Variables Reference](docs/environment-variables.md) for all available options.
Some of the additional options available in Settings:
- **Prowlarr** - Configure indexers and download clients to download books and audiobooks
- **Additional audiobook sources** - Configure additional sources for audiobook discovery
- **Direct Download mirrors** - Supply your own Anna's Archive mirror URLs; Auto mode tries them in the order listed. The `annas-archive.is` domain does not currently work as a source — use `annas-archive.gl` instead (checked August 2026; mirror availability changes)
- **IRC** - Add details for IRC book sources and download directly from the UI. Most networks serve audiobooks from the same channel as ebooks (on `irc.irchighway.net` that's `#ebooks`, while `#bookz` is effectively inactive), so leave the separate audiobook channel blank unless your network actually indexes one. IRC audiobooks usually arrive as ZIP/RAR archives — keep those enabled under Supported Audiobook Formats or the releases are filtered out of results
- **IRC** - Add details for IRC book sources and download directly from the UI
- **Library Link** - Add a link to your Calibre-Web or Grimmory instance in the UI header
- **File processing** - Customiseable download paths, file renaming and directory creation with template-based renaming
- **Network Settings** - Custom proxy support (SOCKS5 + HTTP/S) and configurable DNS
@@ -164,17 +125,6 @@ docker compose up -d
The full-featured image with all network capabilities included.
#### Memory Requirements
The standard image ships a real Chromium browser, which it launches to solve Cloudflare challenges for Direct Download. Chromium needs room to run:
- **2 GB of RAM available to the container** is a safe minimum; 1 GB or less is where problems usually start
- Only relevant if you use Direct Download. Prowlarr, IRC and audiobook sources don't start the browser
When the container is starved of memory, Chromium fails to start and every Direct Download fails with unrelated-looking errors — repeated `403 detected; switching to bypasser` followed by `No download URL found`, and downloads that never complete. If you're seeing that, check the container's memory limit and the host's free memory before suspecting your ISP or DNS.
If you can't spare the memory, use the [Lite](#lite) image with an external resolver (e.g. FlareSolverr) running elsewhere.
#### Tor Routing
Optional Tor support for network privacy:
```bash
@@ -188,31 +138,12 @@ docker compose -f docker-compose.tor.yml up -d
- Timezone is auto-detected from Tor exit node
- Custom DNS/proxy settings are ignored when Tor is active
#### WireGuard VPN Routing
Optional WireGuard support to route all external egress through a VPN tunnel with a fail-closed kill-switch:
```bash
curl -O https://raw.githubusercontent.com/calibrain/shelfmark/main/compose/docker-compose.wireguard.yml
# place your wg-quick config where the compose mounts /config, as wg0.conf
docker compose -f docker-compose.wireguard.yml up -d
```
**Notes:**
- Requires root startup
- Requires `NET_ADMIN` and `NET_RAW` capabilities
- Mount a standard wg-quick config at `WIREGUARD_CONFIG` (default `/config/wg0.conf`)
- All non-LAN egress is forced through the tunnel; if the tunnel drops, external traffic **fails closed** while LAN ranges (WebUI, Prowlarr, qBittorrent) stay reachable
- IPv4 and IPv6 both fail closed. On kernels without a usable `ip6tables`, disable IPv6 for the container (`sysctls: net.ipv6.conf.all.disable_ipv6=1`, as in the compose example) or the container refuses to start rather than risk an IPv6 leak
- A supervised healthcheck bounces the tunnel if the handshake goes stale, and refreshes the endpoint allow rules so a roaming/rotated peer endpoint can reconnect
- Mutually exclusive with `USING_TOR`
- **DNS trust:** `WIREGUARD_DNS` must be a resolver you trust on a trusted network segment. When it is a LAN resolver (kept reachable off-tunnel by `LAN_NETWORK`), the query to that resolver leaves as plaintext UDP/53 on the LAN — the resolver is responsible for encrypting upstream. Two resolver paths exist: (1) when Docker's embedded resolver (`127.0.0.11`) is present it is **preserved** so container names (Prowlarr, qBittorrent) resolve — you MUST pin its upstream to a trusted resolver via the container's compose `dns:` list, since `WIREGUARD_DNS` cannot repoint the embedded resolver from inside the container; (2) otherwise `WIREGUARD_DNS`/the config `DNS =` line is written to `/etc/resolv.conf`. Setting `WIREGUARD_ENFORCE_DNS=false` is a **foot-gun**: with no embedded resolver present the container then uses its inherited resolver, which forwards to the Docker daemon's upstream **off-tunnel**, leaking your DNS. Leave enforcement on unless you have pinned the resolver another way.
### Lite
A lighter image without the built-in browser automation. Ideal for:
- **External services** - Already running FlareSolverr or similar for other applications
- **Alternative sources** - Using Prowlarr, IRC, or other configured sources
- **Audiobooks** - Using Shelfmark primarily for audiobooks
- **Constrained hosts** - No bundled browser, so it runs comfortably below the standard image's [memory requirements](#memory-requirements)
```bash
curl -O https://raw.githubusercontent.com/calibrain/shelfmark/main/compose/docker-compose.lite.yml
@@ -263,11 +194,9 @@ These are non-goals, not missing features.
## Contributing
Shelfmark's core feature set is complete.
Shelfmark's core feature set is complete. Development focuses on stability, bug fixes, quality-of-life improvements, and refining the search experience. Contributions in these areas are welcome, please file issues or submit pull requests on GitHub.
Pull requests are welcome and all of them get reviewed, new features included. If you want a feature, the fastest path is to send a PR for it rather than to file a request.
Feature requests that fall outside the project scope (library integration, automation, collection management) will be closed, and PRs implementing them won't be merged. If you're unsure whether something fits, open a discussion first.
Feature requests that fall outside the project scope (library integration, automation, collection management) will be closed. If you're unsure whether something fits, open a discussion first.
## Health Monitoring
@@ -287,10 +216,7 @@ Logs are available via:
- `docker logs <container-name>`
- `/var/log/shelfmark/` inside the container (when `ENABLE_LOGGING=true`)
Log level is configurable under Settings → Advanced or via the `LOG_LEVEL` environment
variable (`DEBUG`, `INFO`, `WARNING`, `ERROR`, `CRITICAL`; case-insensitive, defaults to
`INFO`). The environment variable wins over the setting, and `DEBUG=true` forces `DEBUG`
regardless of either. Changes take effect on restart.
Log level is configurable via Settings or `LOG_LEVEL` environment variable.
## Development
-111
View File
@@ -238,113 +238,6 @@ def _generate_bootstrap_env_docs() -> list[str]:
return lines
def _generate_egress_env_docs() -> list[str]:
"""Generate documentation for VPN/Tor egress environment variables.
These are startup-only variables consumed by entrypoint.sh / wireguard.sh
(before and outside the settings registry) to select and configure the
transparent-egress kill-switch. `USING_TOR` has a registry-backed entry
under Network and is cross-referenced rather than repeated here so the two
mutually exclusive egress modes are discoverable side by side without
emitting a duplicate `#### USING_TOR` anchor.
"""
egress_vars = [
{
"name": "USING_WIREGUARD",
"description": "Route all traffic through a WireGuard VPN tunnel with a fail-closed iptables kill-switch (non-tunnel egress is dropped). Requires root startup and NET_ADMIN (plus NET_RAW). Mutually exclusive with USING_TOR.",
"type": "boolean",
"default": "false",
},
{
"name": "WIREGUARD_CONFIG",
"description": "Path to the mounted wg-quick configuration file.",
"type": "string (path)",
"default": "/config/wg0.conf",
},
{
"name": "WIREGUARD_INTERFACE",
"description": "WireGuard interface name brought up by wg-quick.",
"type": "string",
"default": "wg0",
},
{
"name": "LAN_NETWORK",
"description": "Comma-separated CIDRs kept off the tunnel so the WebUI and internal download clients (Prowlarr, qBittorrent) stay reachable.",
"type": "string (comma-separated)",
"default": "127.0.0.0/8,10.0.0.0/8,172.16.0.0/12,192.168.0.0/16",
},
{
"name": "WIREGUARD_ENFORCE_DNS",
"description": "Pin the container's resolver so DNS cannot silently fall back to an off-tunnel path. The resolver used is WIREGUARD_DNS if set, else the tunnel config's DNS = line. This does NOT force queries through the tunnel: it is designed for a trusted LAN resolver kept reachable off-tunnel via LAN_NETWORK (the query leaves over the LAN; the resolver encrypts upstream while the download still egresses via the tunnel). Special case: when Docker's embedded resolver (nameserver 127.0.0.11) is present, it is PRESERVED so container-name resolution (e.g. prowlarr, qbittorrent) keeps working, and the embedded resolver's upstream must be pinned via the container's compose dns: list. Fails closed (refuses to start) only when no embedded resolver is present AND no resolver is defined, or /etc/resolv.conf is not writable.",
"type": "boolean",
"default": "true",
},
{
"name": "WIREGUARD_DNS",
"description": "Explicit resolver(s) (comma/space separated) to pin when WIREGUARD_ENFORCE_DNS is true and Docker's embedded resolver is NOT in use. Use when the VPN's pushed DNS filters domains you need; point it at a resolver reachable via the tunnel or an allowed LAN resolver. NOTE: when the embedded resolver (127.0.0.11) is present it is preserved and this value cannot repoint its upstream from inside the container — set the container's compose dns: list to the trusted resolver instead.",
"type": "string (comma-separated)",
"default": "unset (uses config DNS = line)",
},
{
"name": "WIREGUARD_DISABLE_IPV6",
"description": "Strip IPv6 Address/AllowedIPs/DNS from the tunnel config before wg-quick (many container kernels lack the ip6tables raw table wg-quick needs) and remove IPv6 as a leak surface.",
"type": "boolean",
"default": "true",
},
{
"name": "WIREGUARD_ALLOW_IPV6_LEAK",
"description": "Escape hatch: continue startup even when an IPv6 kill-switch cannot be installed AND IPv6 cannot be disabled. Only set when the container has no IPv6 connectivity, as IPv6 egress may otherwise bypass the tunnel.",
"type": "boolean",
"default": "false",
},
{
"name": "WIREGUARD_ALLOW_WEBUI_OFFTUNNEL",
"description": "When false (default) the kill-switch is strictly fail-closed: the only off-tunnel egress permitted is loopback, the tunnel device and the LAN allowlist. Set true only if a NON-LAN client (e.g. a public reverse proxy on a different segment) must reach the WebUI; it permits app-server REPLY packets (--sport FLASK_PORT, conntrack REPLY) to leave off-tunnel. Server replies only, never client-initiated egress, so it cannot leak outbound browsing/downloads or the real IP for outbound requests, but it is still an off-tunnel path while the tunnel is down, hence opt-in. LAN WebUI clients never need it (covered by LAN_NETWORK).",
"type": "boolean",
"default": "false",
},
{
"name": "WIREGUARD_STALE_AFTER",
"description": "Seconds since the last WireGuard handshake before the healthcheck bounces the tunnel.",
"type": "number",
"default": "180",
},
]
lines = [
"## Egress / VPN Routing",
"",
"These startup-only variables are consumed by `entrypoint.sh` / `wireguard.sh` to select and configure the WireGuard transparent-egress kill-switch. `USING_WIREGUARD` and [`USING_TOR`](#using_tor) (documented under Network) are mutually exclusive; both require root startup.",
"",
"| Variable | Description | Type | Default |",
"|----------|-------------|------|---------|",
]
lines.extend(
f"| `{var['name']}` | {var['description']} | {var['type']} | `{var['default']}` |"
for var in egress_vars
)
lines.append("")
lines.append("<details>")
lines.append("<summary>Detailed descriptions</summary>")
lines.append("")
for var in egress_vars:
lines.append(f"#### `{var['name']}`")
lines.append("")
lines.append(var["description"])
lines.append("")
lines.append(f"- **Type:** {var['type']}")
lines.append(f"- **Default:** `{var['default']}`")
lines.append("")
lines.append("</details>")
lines.append("")
return lines
def generate_env_docs() -> str:
"""Generate markdown documentation for all environment variables."""
# Import settings modules to ensure all settings are registered
@@ -389,7 +282,6 @@ def generate_env_docs() -> str:
# Generate TOC
toc_entries = [
"- [Bootstrap Configuration](#bootstrap-configuration)",
"- [Egress / VPN Routing](#egress--vpn-routing)",
]
# Ungrouped tabs first
@@ -415,9 +307,6 @@ def generate_env_docs() -> str:
# Add bootstrap environment variables documentation
lines.extend(_generate_bootstrap_env_docs())
# Add egress / VPN routing (startup-only, shell-driven) documentation
lines.extend(_generate_egress_env_docs())
# Generate documentation for ungrouped tabs
for tab in grouped_tabs.get(None, []):
lines.extend(_generate_tab_docs(tab))
-11
View File
@@ -3,14 +3,3 @@
class BypassCancelledError(Exception):
"""Raised when a bypass operation is cancelled."""
class ChallengeNotSolvedError(Exception):
"""Raised when a bypasser ran but the site still answered with a challenge.
Distinct from a bypasser that is broken or unreachable, which is what every
"the bypass failed" message used to say. A solver can do its job perfectly and
still be handed something it cannot clear - DDoS-Guard's manual CAPTCHA page is
the case from #1292 - and telling the user to go check that FlareSolverr is
reachable sends them to fix a service that is working.
"""
-52
View File
@@ -1,52 +0,0 @@
"""Challenge-page detection shared by the bypassers and the HTTP retry path.
Kept out of `internal_bypasser` so the HTTP layer can recognise an interstitial
without importing SeleniumBase: that module is imported lazily precisely because its
browser dependencies are optional, and external-bypasser setups run without them.
"""
# Matched against lowercased text, so every entry must be lowercase.
CLOUDFLARE_INDICATORS = [
"just a moment",
"verify you are human",
"verifying you are human",
"cloudflare.com/products/turnstile",
]
DDOS_GUARD_INDICATORS = [
"ddos-guard",
"ddos guard",
"checking your browser before accessing",
"complete the manual check to continue",
"could not verify your browser automatically",
]
# Markers that exist only in raw markup: the bypassers scan rendered innerText, where
# a script src or a <title> never appears. The title match is scoped to the tag on
# purpose - hosts word the rest of that sentence differently, and matching "checking
# your browser" as free text would trip on any page that merely discusses a challenge.
_RAW_HTML_MARKERS = (
"<title>checking your browser",
"/cdn-cgi/challenge-platform",
"/.well-known/ddos-guard/",
)
# An interstitial is a few KB of markup. Past that it is a real page that happens to
# mention a marker - a protected site links its own DDoS-Guard endpoints on every page.
MAX_CHALLENGE_HTML_CHARS = 64 * 1024
def challenge_marker(html: str) -> str | None:
"""Return the marker proving `html` is an unsolved challenge page, or None.
Only meaningful for a response that already carries a challenge status: the
markers appear on protected sites' real pages too, so the status is what
separates "blocked" from "served".
"""
if not html or len(html) > MAX_CHALLENGE_HTML_CHARS:
return None
lowered = html.lower()
for marker in (*_RAW_HTML_MARKERS, *DDOS_GUARD_INDICATORS, *CLOUDFLARE_INDICATORS):
if marker in lowered:
return marker
return None
-281
View File
@@ -1,281 +0,0 @@
"""Clearance cookies won by a bypass, shared by every bypasser implementation.
Kept in its own module rather than inside a bypasser because both of them feed it and
both read from it. The internal bypasser cannot host it: it imports seleniumbase at
module scope, which is exactly the dependency an external-bypasser deployment is
entitled not to have installed.
"""
import threading
import time
from collections.abc import Mapping
from typing import Any
from urllib.parse import urlparse
from shelfmark.core.logger import setup_logger
logger = setup_logger(__name__)
# Cookie storage - shared with requests library for Cloudflare bypass
# Nested mapping of domain to cookie name to cookie metadata.
_cf_cookies: dict[str, dict] = {}
_cf_cookies_lock = threading.Lock()
# User-Agent storage - Cloudflare ties cf_clearance to the UA that solved the challenge
_cf_user_agents: dict[str, str] = {}
# Protection cookie names we care about (Cloudflare and DDoS-Guard)
CF_COOKIE_NAMES = {"cf_clearance", "__cf_bm", "cf_chl_2", "cf_chl_prog"}
DDG_COOKIE_NAMES = {
"__ddg1_",
"__ddg2_",
"__ddg5_",
"__ddg8_",
"__ddg9_",
"__ddg10_",
"__ddgid_",
"__ddgmark_",
"ddg_last_challenge",
}
# DDoS-Guard cookies that describe *one* check rather than granting clearance, and so
# must never be replayed on a later request. Observed live on Anna's Archive:
#
# __ddg9_ the client IP address
# __ddg10_ the unix timestamp the check was issued
# __ddg8_ an opaque token issued with them, same ~40 minute expiry
#
# Clearance itself lives in __ddg1_/__ddg2_/__ddgid_ (roughly a year) and __ddg5_.
# Replaying the trio is actively harmful: once the timestamp ages out - or the egress
# IP changes, which happens routinely behind a VPN - the values no longer describe the
# caller, DDoS-Guard re-arms its check and answers every request with a ?check=1
# redirect. That is the redirect loop, and it is self-inflicted. Dropping them simply
# lets DDoS-Guard issue a fresh set, exactly as it does for a browser.
DDG_EPHEMERAL_COOKIE_NAMES = {
"__ddg8_",
"__ddg9_",
"__ddg10_",
"ddg_last_challenge",
}
def _get_base_domain(domain: str) -> str:
"""Extract base domain from hostname (e.g., 'www.example.com' -> 'example.com')."""
return ".".join(domain.split(".")[-2:]) if "." in domain else domain
def _get_full_cookie_domains() -> set[str]:
"""Return mirror domains that need full-session cookie extraction."""
from shelfmark.core.mirrors import get_zlib_cookie_domains
return {_get_base_domain(domain) for domain in get_zlib_cookie_domains()}
def _replay_per_check_cookies() -> bool:
"""Whether the per-check trio is kept rather than dropped (see env.py)."""
from shelfmark.config import env
return env.DDG_REPLAY_PER_CHECK_COOKIES
def _should_extract_cookie(name: str, *, extract_all: bool) -> bool:
"""Determine if a cookie should be extracted based on its name."""
# Checked before extract_all: a per-check token is wrong to replay for every
# domain, including the full-session ones.
if name in DDG_EPHEMERAL_COOKIE_NAMES and not _replay_per_check_cookies():
return False
if extract_all:
return True
is_cf = name in CF_COOKIE_NAMES or name.startswith("cf_")
is_ddg = name in DDG_COOKIE_NAMES or name.startswith("__ddg")
return is_cf or is_ddg
def _cookie_field(cookie: Any, name: str) -> Any:
"""Read one field from a cookie in either shape we are handed.
The internal bypasser extracts CDP cookie objects; an external bypasser returns
the same fields as JSON objects, so the difference is attribute versus key access.
"""
if isinstance(cookie, Mapping):
return cookie.get(name)
return getattr(cookie, name, None)
def _cookie_expiry(cookie: Any) -> float | None:
"""A cookie's absolute expiry, or None when it is a session cookie.
The two spellings are not interchangeable and both reach this store. CDP and
Playwright cookies carry `expires`; the WebDriver cookie object - what a
Selenium-based solver such as FlareSolverr returns - carries `expiry`. Reading
only one silently turns every cookie from the other into a never-expiring one,
which is exactly how dead clearance ends up replayed forever (see
get_cf_cookies_for_domain).
The value is coerced rather than trusted: it arrives as JSON from a service we
do not control, and a string here used to raise straight out of the store.
"""
for field in ("expires", "expiry"):
raw = _cookie_field(cookie, field)
if raw is None:
continue
try:
expiry = float(raw)
except TypeError, ValueError:
logger.debug("Unreadable cookie expiry %r; treating as a session cookie", raw)
return None
# <= 0 is how both shapes spell "session cookie", not "expired in 1970".
return expiry if expiry > 0 else None
return None
def store_extracted_cookies(
*,
url: str,
cookies: list[Any],
user_agent: str | None = None,
) -> None:
"""Store filtered bypass cookies (and optional UA) for a URL domain."""
parsed = urlparse(url)
domain = parsed.hostname or ""
if not domain:
return
base_domain = _get_base_domain(domain)
extract_all = base_domain in _get_full_cookie_domains()
cookies_found: dict[str, dict[str, Any]] = {}
dropped: list[str] = []
for cookie in cookies:
name = _cookie_field(cookie, "name") or ""
if not _should_extract_cookie(name, extract_all=extract_all):
dropped.append(name)
continue
secure = _cookie_field(cookie, "secure")
cookies_found[name] = {
"value": _cookie_field(cookie, "value") or "",
"domain": _cookie_field(cookie, "domain") or domain,
"path": _cookie_field(cookie, "path") or "/",
"expiry": _cookie_expiry(cookie),
"secure": True if secure is None else bool(secure),
"httpOnly": True,
}
# Names only, never values. Which cookies a solve won, and which of them were held
# back, is the evidence needed to settle what DDoS-Guard actually treats as clearance
# (issue #1276) - and without it a debug log shows a solve succeeding and the next
# request being challenged with nothing in between to explain why.
logger.debug(
"Solve on %s won %s; keeping %s; dropping %s",
base_domain,
sorted({_cookie_field(c, "name") or "" for c in cookies}),
sorted(cookies_found),
sorted(set(dropped)) or "nothing",
)
if not cookies_found:
return
with _cf_cookies_lock:
_cf_cookies[base_domain] = cookies_found
if user_agent:
_cf_user_agents[base_domain] = user_agent
logger.debug("Stored UA for %s: %s...", base_domain, str(user_agent)[:60])
else:
logger.debug("No UA captured for %s", base_domain)
cookie_type = "all" if extract_all else "protection"
logger.debug("Extracted %s %s cookies for %s", len(cookies_found), cookie_type, base_domain)
def _is_cookie_expired(cookie: dict[str, Any]) -> bool:
"""Whether a stored cookie's expiry has passed. Session cookies never expire here."""
expiry = cookie.get("expiry")
if expiry is None:
expiry = cookie.get("expires")
if not expiry or expiry <= 0:
return False
return time.time() > expiry
def get_cf_cookies_for_domain(domain: str) -> dict[str, str]:
"""Get stored cookies for a domain. Returns empty dict if none available."""
if not domain:
return {}
base_domain = _get_base_domain(domain)
with _cf_cookies_lock:
cookies = _cf_cookies.get(base_domain, {})
if not cookies:
return {}
cf_clearance = cookies.get("cf_clearance", {})
if cf_clearance and _is_cookie_expired(cf_clearance):
logger.debug("CF cookies expired for %s", base_domain)
_cf_cookies.pop(base_domain, None)
return {}
# Expiry applies to every cookie, not just Cloudflare's. DDoS-Guard domains
# have no cf_clearance, so the check above never fired for them and dead
# cookies were replayed indefinitely - the server answers those with a
# challenge, which is indistinguishable from having sent nothing at all.
live = {name: c for name, c in cookies.items() if not _is_cookie_expired(c)}
if len(live) != len(cookies):
expired = sorted(set(cookies) - set(live))
logger.debug("Dropping expired cookies for %s: %s", base_domain, expired)
if live:
_cf_cookies[base_domain] = live
else:
_cf_cookies.pop(base_domain, None)
return {name: c["value"] for name, c in live.items()}
def has_valid_cf_cookies(domain: str) -> bool:
"""Check if we have valid Cloudflare cookies for a domain."""
return bool(get_cf_cookies_for_domain(domain))
def get_cf_user_agent_for_domain(domain: str) -> str | None:
"""Get the User-Agent that was used during bypass for a domain."""
if not domain:
return None
with _cf_cookies_lock:
return _cf_user_agents.get(_get_base_domain(domain))
def export_store() -> tuple[dict[str, dict], dict[str, str]]:
"""Snapshot the whole store, for handing to another process.
The internal bypasser's Docker helper solves in a subprocess, so the clearance it
wins has to be serialized back to the parent or the solve is lost with the child.
"""
with _cf_cookies_lock:
return (
{domain: dict(cookies) for domain, cookies in _cf_cookies.items()},
dict(_cf_user_agents),
)
def import_store(cookies: object, user_agents: object) -> None:
"""Merge a snapshot produced by :func:`export_store` into this process's store."""
with _cf_cookies_lock:
if isinstance(cookies, dict):
_cf_cookies.update(cookies)
if isinstance(user_agents, dict):
_cf_user_agents.update(
{str(domain): str(agent) for domain, agent in user_agents.items()}
)
def clear_cf_cookies(domain: str | None = None) -> None:
"""Clear stored Cloudflare cookies and User-Agent. If domain is None, clear all."""
with _cf_cookies_lock:
if domain:
base_domain = _get_base_domain(domain)
_cf_cookies.pop(base_domain, None)
_cf_user_agents.pop(base_domain, None)
else:
_cf_cookies.clear()
_cf_user_agents.clear()
+5 -114
View File
@@ -2,20 +2,17 @@
import random
import time
from typing import TYPE_CHECKING, Any
from typing import TYPE_CHECKING
import requests
from shelfmark.bypass import BypassCancelledError, ChallengeNotSolvedError
from shelfmark.bypass.challenge import challenge_marker
from shelfmark.bypass.cookie_store import store_extracted_cookies
from shelfmark.bypass import BypassCancelledError
from shelfmark.core.config import config
from shelfmark.core.logger import setup_logger
from shelfmark.core.utils import normalize_http_url
from shelfmark.download.network import get_ssl_verify
if TYPE_CHECKING:
from collections.abc import Mapping
from threading import Event
from shelfmark.download import network
@@ -50,55 +47,8 @@ def _coerce_timeout_ms(value: object, default: int) -> int:
return default
def max_duration_seconds() -> float:
"""Upper bound on how long get_bypassed_page() can take for one URL.
MAX_RETRY attempts at the configured read timeout, plus the exponential backoff waited
between them (jitter is < 1s per gap, counted as a full second to stay conservative).
Callers use this to declare a stall-detection grace; see shelfmark.download.activity.
"""
bypasser_timeout = _coerce_timeout_ms(config.get("EXT_BYPASSER_TIMEOUT", 60000), 60000)
read_timeout = min((bypasser_timeout / 1000) + READ_TIMEOUT_BUFFER, MAX_READ_TIMEOUT)
backoff_total = sum(
min(BACKOFF_CAP, BACKOFF_BASE * (2 ** (attempt - 1))) + 1.0
for attempt in range(1, MAX_RETRY)
)
return MAX_RETRY * read_timeout + backoff_total
def _store_solution_clearance(target_url: str, solution: Mapping[str, Any]) -> None:
"""Keep the clearance the solver won, so later requests do not re-solve.
A solve is the expensive part of an external bypass - tens of seconds of real
browser - and FlareSolverr-compatible services hand back the cookies and the
User-Agent that earned it. Dropping them meant every single request paid a 403
plus a full solve, and a file download (which the solver cannot proxy, being
binary) never presented clearance at all.
The UA matters as much as the cookies: Cloudflare ties cf_clearance to the UA
that solved the challenge, so replaying the cookie under our own UA is rejected.
"""
cookies = solution.get("cookies") or []
if not isinstance(cookies, list):
logger.debug("External bypasser returned no usable cookie list for '%s'", target_url)
return
user_agent = solution.get("userAgent")
store_extracted_cookies(
url=target_url,
cookies=cookies,
user_agent=user_agent if isinstance(user_agent, str) else None,
)
def _fetch_via_bypasser(target_url: str) -> str | None:
"""Make a single request to the external bypasser service. Returns HTML or None.
Raises:
ChallengeNotSolvedError: the service answered with a page that is still a
challenge, whatever verdict it reported on itself.
"""
"""Make a single request to the external bypasser service. Returns HTML or None."""
raw_bypasser_url = _coerce_config_str(
config.get("EXT_BYPASSER_URL", "http://flaresolverr:8191"),
"http://flaresolverr:8191",
@@ -150,41 +100,6 @@ def _fetch_via_bypasser(target_url: str) -> str | None:
logger.warning("External bypasser returned empty response for '%s'", target_url)
return None
# "Challenge solved!" is the solver's verdict on its own work, and #1289 showed
# it can be reported alongside a page the caller then rejects. Say what actually
# came back, so a later report does not have to infer it from downstream errors.
marker = challenge_marker(html)
logger.debug(
"External bypasser page for '%s': %d bytes, challenge_marker=%r",
target_url,
len(html),
marker,
)
if marker:
# The solver's verdict is not evidence; the page is. Returning this one as a
# success is what made #1292 unrecoverable: the retry-and-rotate loop that
# could still have saved the search - the next mirror is a different
# DDoS-Guard host, in its own state - was never entered, and the challenge
# page's own __ddg cookies were filed as this host's clearance and replayed
# on every later request.
logger.warning(
"External bypasser reported success but returned a challenge page for "
"'%s' (%d bytes, marker=%r) - the solve did not clear the protection",
target_url,
len(html),
marker,
)
raise ChallengeNotSolvedError(marker)
try:
_store_solution_clearance(target_url, solution)
except AttributeError, KeyError, TypeError, ValueError:
# Storing clearance is an optimisation; the page is the product. The
# solution JSON comes from a service we do not control, so a surprise in
# its cookie shape must not discard HTML that already cost a ~30s solve
# and send the caller round for up to MAX_RETRY more of them.
logger.debug("Could not store bypass clearance for '%s'", target_url, exc_info=True)
except requests.exceptions.Timeout:
logger.warning(
"External bypasser timed out for '%s' (connect: %ss, read: %.0fs)",
@@ -225,33 +140,16 @@ def get_bypassed_page(
selector: network.AAMirrorSelector | None = None,
cancel_flag: Event | None = None,
) -> str | None:
"""Fetch HTML via external bypasser with retries and mirror rotation.
Raises:
ChallengeNotSolvedError: every attempt came back still carrying a challenge.
Reported apart from returning None because the two ask the user for
opposite things: None means go and check the bypasser, this means the
bypasser is fine and the host is the one refusing.
BypassCancelledError: the caller's cancel flag was set.
"""
"""Fetch HTML via external bypasser with retries and mirror rotation."""
from shelfmark.download import network as network_module
sel = selector or network_module.AAMirrorSelector()
unsolved_marker: str | None = None
for attempt in range(1, MAX_RETRY + 1):
_check_cancelled(cancel_flag, "by user")
attempt_url = sel.rewrite(url)
try:
result = _fetch_via_bypasser(attempt_url)
except ChallengeNotSolvedError as e:
# Worth the remaining attempts rather than an immediate give-up: the retry
# rotates onto the next mirror, and that is a different DDoS-Guard host with
# its own idea of whether this caller needs a CAPTCHA.
unsolved_marker = str(e) or unsolved_marker
result = None
result = _fetch_via_bypasser(attempt_url)
if result:
return result
@@ -272,11 +170,4 @@ def get_bypassed_page(
if action in ("mirror", "dns") and new_base:
logger.info("Rotated %s for retry", action)
if unsolved_marker:
msg = (
"The bypasser ran, but the site kept answering with a protection challenge "
f"(marker={unsolved_marker!r}). That is usually a manual CAPTCHA, which no "
"bypasser can answer - the bypasser itself is working. Try again shortly."
)
raise ChallengeNotSolvedError(msg)
return None
File diff suppressed because it is too large Load Diff
+1 -2
View File
@@ -2,7 +2,6 @@
from __future__ import annotations
import hashlib
from typing import Any
from shelfmark.core.config import config
@@ -24,7 +23,7 @@ _BOOKLORE_OPTIONS_CACHE: dict[str, Any] = {
def _get_booklore_cache_key(base_url: str, username: str, password: str) -> str:
return f"{base_url}|{username}|{hashlib.sha256(password.encode()).hexdigest()}"
return f"{base_url}|{username}|{hash(password)}"
def _get_booklore_select_options(
+13 -72
View File
@@ -6,78 +6,34 @@ import shutil
import tempfile
from pathlib import Path
LOG_LEVELS = ("DEBUG", "INFO", "WARNING", "ERROR", "CRITICAL")
def string_to_bool(s: str) -> bool:
"""Convert string to boolean."""
return s.lower() in ["true", "yes", "1", "y"]
def _read_advanced_config(key: str) -> object | None:
"""Read a key from the advanced settings file (import-time safe)."""
config_dir = Path(os.getenv("CONFIG_DIR", "/config"))
config_file = config_dir / "plugins" / "advanced.json"
if config_file.exists():
try:
with config_file.open() as f:
config = json.load(f)
if key in config:
return config[key]
except json.JSONDecodeError, OSError:
pass
return None
def _read_debug_from_config() -> bool:
"""Read DEBUG from env var or config file (import-time safe)."""
env_debug = os.environ.get("DEBUG")
if env_debug is not None:
return string_to_bool(env_debug)
value = _read_advanced_config("DEBUG")
if value is not None:
return bool(value)
# Try to read from config file
config_dir = Path(os.getenv("CONFIG_DIR", "/config"))
config_file = config_dir / "plugins" / "advanced.json"
if config_file.exists():
try:
with config_file.open() as f:
config = json.load(f)
if "DEBUG" in config:
return bool(config["DEBUG"])
except json.JSONDecodeError, OSError:
pass
return False
def normalize_log_level(raw: str | None) -> str:
"""Normalize a log level name, falling back to INFO when unrecognized."""
if raw is None:
return "INFO"
normalized = raw.strip().upper()
# "WARN" is a logging alias, but gunicorn only accepts "warning".
if normalized == "WARN":
normalized = "WARNING"
if normalized not in LOG_LEVELS:
return "INFO"
return normalized
def _read_log_level_from_config(debug: bool) -> str:
"""Resolve the app log level from DEBUG, env var, or config file.
DEBUG wins when enabled, mirroring how entrypoint.sh picks gunicorn's level.
Otherwise LOG_LEVEL is read from the env var, then the settings file, and
falls back to INFO when unset or unrecognized.
"""
if debug:
return "DEBUG"
raw = os.environ.get("LOG_LEVEL")
if raw is None:
value = _read_advanced_config("LOG_LEVEL")
raw = value if isinstance(value, str) else None
return normalize_log_level(raw)
def _is_sqlite_file(path: Path) -> bool:
"""Check if a file is a valid SQLite database by reading magic bytes."""
try:
@@ -145,7 +101,7 @@ INGEST_DIR = Path(os.getenv("INGEST_DIR", "/books"))
# =============================================================================
DEBUG = _read_debug_from_config()
LOG_LEVEL = _read_log_level_from_config(DEBUG)
LOG_LEVEL = "DEBUG" if DEBUG else "INFO"
ENABLE_LOGGING = string_to_bool(os.getenv("ENABLE_LOGGING", "true"))
@@ -203,21 +159,6 @@ ONBOARDING = string_to_bool(os.getenv("ONBOARDING", "true"))
_DEBUG_SKIP_SOURCES_RAW = os.getenv("DEBUG_SKIP_SOURCES", "").strip().lower()
DEBUG_SKIP_SOURCES = {s.strip() for s in _DEBUG_SKIP_SOURCES_RAW.split(",") if s.strip()}
# Debug: keep DDoS-Guard's __ddg8_/__ddg9_/__ddg10_ in the clearance store instead of
# dropping them after a solve.
#
# Which of DDoS-Guard's cookies actually *are* clearance is not settled. The store treats
# the trio as describing one check (client IP, timestamp, token) and drops them, on the
# reasoning that replaying a stale IP/timestamp is what re-arms the ?check=1 loop - see
# shelfmark.bypass.cookie_store. Field reports on issue #1276 point the other way: every
# request after a successful solve was challenged again, which is only consistent with
# what the store keeps not being sufficient clearance on its own.
#
# Deliberately env-only and off by default: this is a knob for reproducing the question
# against a live host, not a setting to offer users. Set it to true, solve once, and watch
# whether the next search still logs "Redirect loop detected".
DDG_REPLAY_PER_CHECK_COOKIES = string_to_bool(os.getenv("DDG_REPLAY_PER_CHECK_COOKIES", "false"))
# =============================================================================
# Legacy migration support - will be removed in future version
+1 -56
View File
@@ -7,7 +7,7 @@ from pathlib import Path
from typing import TYPE_CHECKING, Any, Protocol
if TYPE_CHECKING:
from collections.abc import Callable, Sequence
from collections.abc import Callable
from os import PathLike
_DEPRECATED_SETTINGS_RESTRICTION_KEYS = (
@@ -16,13 +16,6 @@ _DEPRECATED_SETTINGS_RESTRICTION_KEYS = (
"RESTRICT_SETTINGS_TO_ADMIN",
)
# The audiobook format list shipped as the default until the format sets were unified.
# It only covered m4b/mp3, so FLAC/OPUS/OGG/M4A releases were dropped from search results
# and rejected after download - and the wider default alone would never reach existing
# installs, because initialize_default_configs() only writes defaults when the config
# file does not exist yet.
_LEGACY_AUDIOBOOK_FORMATS_DEFAULT = ("m4b", "mp3")
class MigrationLogger(Protocol):
"""Logger surface used by config migration helpers."""
@@ -64,54 +57,6 @@ def _pick_legacy_settings_restriction(config: dict[str, Any]) -> bool | None:
return None
def migrate_audiobook_formats(
*,
load_general_config: Callable[[], dict[str, Any]],
# `object` rather than `None`: the result is discarded, and savers that report
# success (settings_registry.save_config_file returns bool) are not assignable to a
# `-> None` callable.
save_general_config: Callable[[dict[str, Any]], object],
widened_formats: Sequence[str],
logger: MigrationLogger,
) -> None:
"""Widen an untouched audiobook format list to the current, fuller default.
Only a list that still matches the old default exactly is rewritten. Any other value
means someone chose it deliberately, and a migration that "helpfully" re-enabled
formats a user had turned off would be worse than leaving them on the narrow list.
"""
try:
config = load_general_config()
if "SUPPORTED_AUDIOBOOK_FORMATS" not in config:
# Nothing persisted, so the field default already applies.
logger.debug("No persisted audiobook formats - the current default applies")
return
current = config.get("SUPPORTED_AUDIOBOOK_FORMATS")
if not isinstance(current, list):
return
normalized = {str(fmt).strip().lower() for fmt in current if str(fmt).strip()}
if normalized != set(_LEGACY_AUDIOBOOK_FORMATS_DEFAULT):
logger.debug(
"Audiobook formats were customized (%s) - left unchanged", sorted(normalized)
)
return
save_general_config({"SUPPORTED_AUDIOBOOK_FORMATS": list(widened_formats)})
logger.info(
"Widened audiobook formats from the legacy default %s to %s",
list(_LEGACY_AUDIOBOOK_FORMATS_DEFAULT),
list(widened_formats),
)
except FileNotFoundError:
logger.debug("No existing general config file found - nothing to migrate")
except Exception:
logger.exception("Failed to migrate audiobook formats")
def migrate_security_settings(
*,
load_security_config: Callable[[], dict[str, Any]],
+1 -19
View File
@@ -115,7 +115,7 @@ def check_oidc_connection(
response.raise_for_status()
document = response.json()
required_fields = ["issuer", "authorization_endpoint", "token_endpoint", "jwks_uri"]
required_fields = ["issuer", "authorization_endpoint", "token_endpoint"]
missing_fields = [field for field in required_fields if field not in document]
if missing_fields:
return {
@@ -123,24 +123,6 @@ def check_oidc_connection(
"message": f"Discovery document missing fields: {', '.join(missing_fields)}",
}
# Logins verify the ID token against the provider's JWKS, so an empty key
# set (e.g. an Authentik provider with no Signing Key selected) means every
# login will fail even though discovery looks healthy.
jwks_uri = str(document["jwks_uri"])
jwks_response = requests.get(jwks_uri, timeout=10, verify=get_ssl_verify(jwks_uri))
jwks_response.raise_for_status()
jwks_document = jwks_response.json()
jwks_keys = jwks_document.get("keys") if isinstance(jwks_document, dict) else None
if not jwks_keys:
return {
"success": False,
"message": (
"Discovery document is valid, but the provider returned no token "
"signing keys (empty JWKS), so logins will fail. If you use "
"Authentik, select a Signing Key in the provider settings."
),
}
return {"success": True, "message": f"Connected to {document['issuer']}"}
except Exception as exc:
logger.exception("OIDC connection test failed")
+25 -172
View File
@@ -1,5 +1,6 @@
"""Core settings registration and derived configuration values."""
import json
from pathlib import Path
from typing import Any
@@ -14,8 +15,6 @@ from shelfmark.config.download_settings_handlers import (
check_books_destination,
)
from shelfmark.config.email_settings import check_email_connection
from shelfmark.config.migrations import migrate_audiobook_formats
from shelfmark.core.languages import supported_book_languages
from shelfmark.core.logger import setup_logger
from shelfmark.core.settings_registry import (
ActionButton,
@@ -36,10 +35,6 @@ from shelfmark.core.settings_registry import (
register_on_save,
register_settings,
)
from shelfmark.core.utils import ARCHIVE_FORMATS, AUDIOBOOK_FORMATS
_DOWNLOAD_CLIENT_COMPLETED_PATH_TIMEOUT_DEFAULT = 60
_DOWNLOAD_CLIENT_COMPLETED_PATH_TIMEOUT_MAX = 3600
def _on_save_advanced(values: dict[str, Any]) -> dict[str, Any]:
@@ -48,40 +43,6 @@ def _on_save_advanced(values: dict[str, Any]) -> dict[str, Any]:
logger = setup_logger(__name__)
timeout_key = "DOWNLOAD_CLIENT_COMPLETED_PATH_TIMEOUT"
if timeout_key in values:
raw_timeout = values.get(timeout_key)
if isinstance(raw_timeout, bool):
return {
"error": True,
"message": "Completed Path Wait must be a number of seconds",
"values": values,
}
if raw_timeout is None:
return {
"error": True,
"message": "Completed Path Wait must be a number of seconds",
"values": values,
}
try:
timeout_seconds = int(raw_timeout)
except TypeError, ValueError:
return {
"error": True,
"message": "Completed Path Wait must be a number of seconds",
"values": values,
}
if timeout_seconds < 0 or timeout_seconds > _DOWNLOAD_CLIENT_COMPLETED_PATH_TIMEOUT_MAX:
return {
"error": True,
"message": (
"Completed Path Wait must be between 0 and "
f"{_DOWNLOAD_CLIENT_COMPLETED_PATH_TIMEOUT_MAX} seconds"
),
"values": values,
}
values[timeout_key] = timeout_seconds
mappings = values.get("PROWLARR_REMOTE_PATH_MAPPINGS")
if mappings is None:
return {"error": False, "values": values}
@@ -136,20 +97,6 @@ def _on_save_advanced(values: dict[str, Any]) -> dict[str, Any]:
logger = setup_logger(__name__)
def migrate_audiobook_format_settings() -> None:
"""Bring installs created before the audiobook format sets were unified up to date."""
from shelfmark.core.settings_registry import load_config_file, save_config_file
migrate_audiobook_formats(
load_general_config=lambda: load_config_file("general"),
save_general_config=lambda values: save_config_file("general", values),
widened_formats=[*AUDIOBOOK_FORMATS, *ARCHIVE_FORMATS],
logger=logger,
)
_SMTP_PORT_MAX = 65535
_EMAIL_ATTACHMENT_LIMIT_MB_MAX = 600
@@ -159,8 +106,11 @@ for key in ["CONFIG_DIR", "LOG_DIR", "TMP_DIR", "INGEST_DIR", "DEBUG", "DOCKERMO
if hasattr(env, key):
logger.debug(" %s: %s", key, getattr(env, key))
# Selectable book languages, without the resolution aliases clients do not need.
_SUPPORTED_BOOK_LANGUAGE = supported_book_languages()
# Load supported book languages from data file
# Path is relative to the package root, not this file
_DATA_DIR = Path(__file__).resolve().parent.parent.parent / "data"
with (_DATA_DIR / "book-languages.json").open() as file:
_SUPPORTED_BOOK_LANGUAGE = json.load(file)
# Directory settings
BASE_DIR = Path(__file__).resolve().parent.parent.parent
@@ -231,7 +181,11 @@ _FORMAT_OPTIONS = [
]
_AUDIOBOOK_FORMAT_OPTIONS = [
{"value": fmt, "label": fmt.upper()} for fmt in (*AUDIOBOOK_FORMATS, *ARCHIVE_FORMATS)
{"value": "m4b", "label": "M4B"},
{"value": "mp3", "label": "MP3"},
{"value": "m4a", "label": "M4A"},
{"value": "zip", "label": "ZIP"},
{"value": "rar", "label": "RAR"},
]
_DOWNLOAD_TO_BROWSER_CONTENT_TYPE_OPTIONS = [
@@ -428,7 +382,14 @@ def general_settings() -> list[SettingsField]:
label="Supported Audiobook Formats",
description="Audiobook formats to include in search results. ZIP/RAR archives are extracted automatically and audiobook files are used if found.",
options=_AUDIOBOOK_FORMAT_OPTIONS,
default=[*AUDIOBOOK_FORMATS, *ARCHIVE_FORMATS],
default=["m4b", "mp3"],
),
MultiSelectField(
key="BOOK_LANGUAGE",
label="Default Book Languages",
description="Default language filter for searches.",
options=_LANGUAGE_OPTIONS,
default=["en"],
),
]
@@ -467,17 +428,6 @@ def search_mode_settings() -> list[SettingsField]:
default="universal",
user_overridable=True,
),
MultiSelectField(
key="BOOK_LANGUAGE",
label="Default Book Languages",
description=(
"Default language filter for searches. Users can override this for their "
"own account."
),
options=_LANGUAGE_OPTIONS,
default=["en"],
user_overridable=True,
),
SelectField(
key="AA_DEFAULT_SORT",
label="Default Sort Order",
@@ -503,14 +453,6 @@ def search_mode_settings() -> list[SettingsField]:
show_when={"field": "SEARCH_MODE", "value": "universal"},
user_overridable=True,
),
CheckboxField(
key="FORCE_COMBINED_SEARCH",
label="Always Use Combined Search",
description="Force combined search whenever it's available. Locks the combined toggle on.",
default=False,
show_when={"field": "SEARCH_MODE", "value": "universal"},
user_overridable=True,
),
HeadingField(
key="universal_mode_heading",
title="Universal Mode Settings",
@@ -770,10 +712,7 @@ def _on_save_downloads(values: dict[str, Any]) -> dict[str, Any]:
}
# Audiobooks are always folder output.
if effective.get("FILE_ORGANIZATION_AUDIOBOOK", "rename") in {
"rename",
"rename_and_group",
}:
if effective.get("FILE_ORGANIZATION_AUDIOBOOK", "rename") == "rename":
template = effective.get("TEMPLATE_AUDIOBOOK_RENAME", "")
if _contains_path_separators(template):
return {
@@ -1028,7 +967,7 @@ def download_settings() -> list[SettingsField]:
key="TEMPLATE_RENAME",
label="Naming Template",
description=(
"Variables: {Author}, {Title}, {Year}, {Language}, {User}, {OriginalName} "
"Variables: {Author}, {Title}, {Year}, {User}, {OriginalName} "
"(source filename without extension). Universal adds: {Series}, "
"{SeriesPosition}, {Subtitle}, {PrimaryTitle}. Use arbitrary prefix/suffix: "
"{Vol. SeriesPosition - } outputs 'Vol. 2 - ' when set, nothing when empty. "
@@ -1047,7 +986,7 @@ def download_settings() -> list[SettingsField]:
key="TEMPLATE_ORGANIZE",
label="Path Template",
description=(
"Use / to create folders. Variables: {Author}, {Title}, {Year}, {Language}, {User}, "
"Use / to create folders. Variables: {Author}, {Title}, {Year}, {User}, "
"{OriginalName} (source filename without extension). Universal adds: {Series}, "
"{SeriesPosition}, {Subtitle}, {PrimaryTitle}. Use arbitrary prefix/suffix: "
"{Vol. SeriesPosition - } outputs 'Vol. 2 - ' when set, nothing when empty."
@@ -1301,11 +1240,6 @@ def download_settings() -> list[SettingsField]:
"label": "Rename and Organize",
"description": "Create folders and rename files using a template. Recommended for Audiobookshelf. Do not use with ingest folders.",
},
{
"value": "rename_and_group",
"label": "Rename and Group",
"description": "Rename single-file downloads; keep multi-file downloads grouped in their source folder. Do not use with ingest folders.",
},
],
default="rename",
universal_only=True,
@@ -1315,7 +1249,7 @@ def download_settings() -> list[SettingsField]:
key="TEMPLATE_AUDIOBOOK_RENAME",
label="Naming Template",
description=(
"Variables: {Author}, {Title}, {Year}, {Language}, {User}, {OriginalName} "
"Variables: {Author}, {Title}, {Year}, {User}, {OriginalName} "
"(source filename without extension), {Series}, {SeriesPosition}, {Subtitle}, "
"{PrimaryTitle}, {PartNumber}. Use arbitrary prefix/suffix: "
"{Vol. SeriesPosition - } outputs 'Vol. 2 - ' when set, nothing when empty. "
@@ -1324,10 +1258,7 @@ def download_settings() -> list[SettingsField]:
),
default="{Author} - {Title}",
placeholder="{Author} - {Title}{ - Part }{PartNumber}",
show_when={
"field": "FILE_ORGANIZATION_AUDIOBOOK",
"value": ["rename", "rename_and_group"],
},
show_when={"field": "FILE_ORGANIZATION_AUDIOBOOK", "value": "rename"},
universal_only=True,
),
# Organize mode template - folders allowed
@@ -1335,7 +1266,7 @@ def download_settings() -> list[SettingsField]:
key="TEMPLATE_AUDIOBOOK_ORGANIZE",
label="Path Template",
description=(
"Use / to create folders. Variables: {Author}, {Title}, {Year}, {Language}, {User}, "
"Use / to create folders. Variables: {Author}, {Title}, {Year}, {User}, "
"{OriginalName} (source filename without extension), {Series}, {SeriesPosition}, "
"{Subtitle}, {PrimaryTitle}, {PartNumber}. Use arbitrary prefix/suffix: "
"{Vol. SeriesPosition - } outputs 'Vol. 2 - ' when set, nothing when empty."
@@ -1509,17 +1440,6 @@ def download_source_settings() -> list[SettingsField]:
),
default=False,
),
CheckboxField(
key="DIRECT_DOWNLOAD_LANGUAGE_FROM_PATH",
label="Detect Language From Distant Path",
description=(
"When language metadata is missing or unknown, parse the distant path "
"(file path shown in search results) for language tags like [BD FR] or [En]. "
"Also enables local language filtering so lgli files without AA language "
"metadata are not excluded before the distant path can be checked."
),
default=False,
),
PasswordField(
key="AA_DONATOR_KEY",
label="Account Donator Key",
@@ -1560,19 +1480,6 @@ def download_source_settings() -> list[SettingsField]:
min_value=1,
max_value=60,
),
NumberField(
key="RELEASE_SEARCH_TIMEOUT",
label="Release Search Timeout (seconds)",
description=(
"How long one release search may run before it gives up and reports why. "
"A first search on a cold start pays for a browser solve, so leave room "
"for one. If you use a reverse proxy, its read timeout should be at least "
"this high or it will cut the search off with a 504 first."
),
default=300,
min_value=30,
max_value=1800,
),
HeadingField(
key="content_type_routing_heading",
title="Content-Type Routing",
@@ -1683,31 +1590,6 @@ def cloudflare_bypass_settings() -> list[SettingsField]:
requires_restart=True,
show_when={"field": "USING_EXTERNAL_BYPASSER", "value": True},
),
NumberField(
key="BYPASS_PAGE_SOURCE_TIMEOUT",
label="Page Read Timeout (seconds)",
description=(
"How long to wait for a solved page to produce its content before the "
"bypass is retried. Raise it if solves succeed but searches still fail."
),
default=20,
min_value=1,
max_value=120,
show_when={"field": "USING_EXTERNAL_BYPASSER", "value": False},
),
NumberField(
key="BYPASS_BROWSER_IDLE_TIMEOUT",
label="Bypasser Idle Timeout (seconds)",
description=(
"How long the bypass helper process may sit unused before it is shut down. "
"Higher keeps more searches fast, lower frees memory sooner."
),
default=180,
min_value=30,
max_value=3600,
requires_restart=True,
show_when={"field": "USING_EXTERNAL_BYPASSER", "value": False},
),
]
@@ -1826,23 +1708,6 @@ def advanced_settings() -> list[SettingsField]:
default=False,
requires_restart=True,
),
SelectField(
key="LOG_LEVEL",
label="Log Level",
description=(
"Lowest severity written to the console and log file. "
"Ignored while Debug Mode is on, which forces Debug."
),
options=[
{"value": "DEBUG", "label": "Debug", "description": "Everything, very noisy."},
{"value": "INFO", "label": "Info", "description": "Normal activity (default)."},
{"value": "WARNING", "label": "Warning", "description": "Warnings and problems."},
{"value": "ERROR", "label": "Error", "description": "Failures only."},
{"value": "CRITICAL", "label": "Critical", "description": "Fatal errors only."},
],
default="INFO",
requires_restart=True,
),
NumberField(
key="MAIN_LOOP_SLEEP_TIME",
label="Queue Check Interval (seconds)",
@@ -1896,18 +1761,6 @@ def advanced_settings() -> list[SettingsField]:
title="Remote Path Mappings",
description="Map download client paths to paths inside Shelfmark. Needed when volume mounts differ between containers.",
),
NumberField(
key="DOWNLOAD_CLIENT_COMPLETED_PATH_TIMEOUT",
label="Completed Path Wait (seconds)",
description=(
"How long to wait after a torrent or usenet client reports completion "
"for the completed file path to become visible to Shelfmark. Increase "
"this for seedbox or remote-sync workflows."
),
default=_DOWNLOAD_CLIENT_COMPLETED_PATH_TIMEOUT_DEFAULT,
min_value=0,
max_value=_DOWNLOAD_CLIENT_COMPLETED_PATH_TIMEOUT_MAX,
),
TableField(
key="PROWLARR_REMOTE_PATH_MAPPINGS",
label="Path Mappings",
+4 -41
View File
@@ -7,7 +7,6 @@ that talks to /api/admin/users endpoints.
from typing import Any
from shelfmark.core.languages import normalize_language
from shelfmark.core.request_policy import (
get_source_content_type_capabilities,
parse_policy_mode,
@@ -62,7 +61,7 @@ _SELF_SETTINGS_SECTION_OPTIONS = [
{
"value": "search",
"label": "Search Preferences",
"description": "Show personal search mode, language, and provider settings.",
"description": "Show personal search mode and provider settings.",
},
{
"value": "notifications",
@@ -78,13 +77,11 @@ _SEARCH_PREFERENCE_PROVIDER_KEYS = {
"METADATA_PROVIDER_AUDIOBOOK",
"METADATA_PROVIDER_COMBINED",
}
SEARCH_PREFERENCE_VALIDATABLE_KEYS = {
_SEARCH_PREFERENCE_VALIDATABLE_KEYS = {
"SEARCH_MODE",
"BOOK_LANGUAGE",
"DEFAULT_RELEASE_SOURCE",
"DEFAULT_RELEASE_SOURCE_AUDIOBOOK",
"SHOW_COMBINED_SELECTOR",
"FORCE_COMBINED_SEARCH",
*_SEARCH_PREFERENCE_PROVIDER_KEYS,
}
@@ -180,43 +177,14 @@ def _get_request_policy_rule_columns() -> list[dict[str, object]]:
]
def _validate_book_languages(value: Any) -> tuple[Any, str | None]:
"""Validate a per-user default language list against the known languages.
Accepts the list the settings UI sends as well as a comma-separated string, so an
API client can spell the value the way the env var does. Blank entries are skipped
rather than rejected, which makes "" and "en," mean the same as [] and ["en"]. An
empty result is a deliberate override meaning "no default language filter", so it
is kept as-is; ``None`` clears the override further up the chain.
"""
entries = value.split(",") if isinstance(value, str) else value
if not isinstance(entries, (list, tuple)):
return value, "BOOK_LANGUAGE must be a list of language codes"
normalized: list[str] = []
for entry in entries:
if entry is None or (isinstance(entry, str) and not entry.strip()):
continue
code = normalize_language(entry)
if code is None:
return value, f"BOOK_LANGUAGE contains an unsupported language: {entry}"
if code not in normalized:
normalized.append(code)
return normalized, None
def validate_search_preference_value(key: str, value: Any) -> tuple[Any, str | None]:
"""Validate and normalize a search preference value for user overrides."""
if key not in SEARCH_PREFERENCE_VALIDATABLE_KEYS:
if key not in _SEARCH_PREFERENCE_VALIDATABLE_KEYS:
return value, None
if value is None:
return None, None
if key == "BOOK_LANGUAGE":
return _validate_book_languages(value)
normalized_value = str(value).strip()
if key == "SEARCH_MODE":
@@ -255,11 +223,6 @@ def validate_search_preference_value(key: str, value: Any) -> tuple[Any, str | N
return value, None
return bool(value), None
if key == "FORCE_COMBINED_SEARCH":
if isinstance(value, bool):
return value, None
return bool(value), None
return value, None
@@ -329,7 +292,7 @@ def _on_save_users(values: dict[str, object]) -> dict[str, object]:
}
values["REQUEST_POLICY_RULES"] = normalized_rules
for key in SEARCH_PREFERENCE_VALIDATABLE_KEYS:
for key in _SEARCH_PREFERENCE_VALIDATABLE_KEYS:
if key not in values:
continue
normalized_value, validation_error = validate_search_preference_value(key, values[key])
+8 -7
View File
@@ -11,10 +11,7 @@ from shelfmark.config.notifications_settings import (
is_valid_notification_url,
normalize_notification_routes,
)
from shelfmark.config.users_settings import (
SEARCH_PREFERENCE_VALIDATABLE_KEYS,
validate_search_preference_value,
)
from shelfmark.config.users_settings import validate_search_preference_value
from shelfmark.core.config import config as app_config
from shelfmark.core.request_policy import parse_policy_mode, validate_policy_rules
from shelfmark.core.settings_registry import load_config_file
@@ -94,9 +91,13 @@ def validate_user_settings(
if search_validation_error:
errors.append(search_validation_error)
continue
# Every key the search validator recognises keeps its normalized value;
# a hand-maintained subset here silently dropped normalization for the rest.
if key in SEARCH_PREFERENCE_VALIDATABLE_KEYS:
if key in {
"SEARCH_MODE",
"METADATA_PROVIDER",
"METADATA_PROVIDER_AUDIOBOOK",
"DEFAULT_RELEASE_SOURCE",
"DEFAULT_RELEASE_SOURCE_AUDIOBOOK",
}:
valid[key] = normalized_search_value
continue
-1
View File
@@ -39,7 +39,6 @@ def upsert_cwa_user(
email=normalized_email,
role=role,
allow_email_link=True,
sync_username=True,
collision_strategy=collision_strategy,
alias_suffix=_CWA_ALIAS_SUFFIX,
context=context,
+4 -62
View File
@@ -108,7 +108,6 @@ def _build_updates(
auth_source: str,
role: str,
sync_role: bool,
username: str | object,
email: str | None | object,
display_name: str | None | object,
subject_field: str | None,
@@ -117,8 +116,6 @@ def _build_updates(
updates: dict[str, Any] = {"auth_source": auth_source}
if sync_role:
updates["role"] = _normalize_role(role)
if username is not UNSET:
updates["username"] = _normalize_username(username)
if email is not UNSET:
updates["email"] = _normalize_email(email)
if display_name is not UNSET:
@@ -128,17 +125,10 @@ def _build_updates(
return updates
def _next_suffix_username(
user_db: UserDB,
base_username: str,
*,
exclude_user_id: int | None = None,
) -> str:
def _next_suffix_username(user_db: UserDB, base_username: str) -> str:
candidate = base_username
suffix = 1
while existing := user_db.get_user(username=candidate):
if exclude_user_id is not None and int(existing.get("id") or 0) == exclude_user_id:
return candidate
while user_db.get_user(username=candidate):
candidate = f"{base_username}_{suffix}"
suffix += 1
return candidate
@@ -159,7 +149,7 @@ def _find_existing_alias_user(
]
if not candidates:
return None
return min(candidates, key=lambda user: int(user.get("id") or 0), default=None)
return sorted(candidates, key=lambda user: int(user.get("id") or 0))[0]
def _resolve_create_username(
@@ -195,38 +185,6 @@ def _resolve_create_username(
return _next_suffix_username(user_db, alias_base), None, "username_collision_alias"
def _resolve_update_username(
user_db: UserDB,
*,
current_user: dict[str, Any],
requested_username: str,
strategy: CollisionStrategy,
alias_suffix: str,
) -> str:
current_user_id = int(current_user["id"])
existing = user_db.get_user(username=requested_username)
if existing is None or int(existing.get("id") or 0) == current_user_id:
return requested_username
if strategy == "suffix":
return _next_suffix_username(
user_db,
requested_username,
exclude_user_id=current_user_id,
)
if strategy == "alias":
return _next_suffix_username(
user_db,
f"{requested_username}{alias_suffix}",
exclude_user_id=current_user_id,
)
# `takeover` can select an existing row during creation, but once an
# identity is already matched it must never replace a different username
# owner. Preserve the matched row's current collision-free name instead.
return str(current_user["username"])
def upsert_external_user(
user_db: UserDB,
*,
@@ -239,7 +197,6 @@ def upsert_external_user(
subject: str | None = None,
allow_email_link: bool = False,
sync_role: bool = True,
sync_username: bool = False,
allow_create: bool = True,
collision_strategy: CollisionStrategy = "takeover",
alias_suffix: str | None = None,
@@ -272,26 +229,10 @@ def upsert_external_user(
subject=subject,
allow_email_link=allow_email_link,
)
resolved_alias_suffix = alias_suffix or f"__{auth_source}"
update_username: str | object = UNSET
if (
matched is not None
and sync_username
and normalize_auth_source(matched.get("auth_source"), matched.get("oidc_subject"))
== auth_source
):
update_username = _resolve_update_username(
user_db,
current_user=matched,
requested_username=normalized_username,
strategy=collision_strategy,
alias_suffix=resolved_alias_suffix,
)
updates = _build_updates(
auth_source=auth_source,
role=normalized_role,
sync_role=sync_role,
username=update_username,
email=normalized_email if email is not UNSET else UNSET,
display_name=normalized_display_name if display_name is not UNSET else UNSET,
subject_field=subject_field,
@@ -320,6 +261,7 @@ def upsert_external_user(
)
return None, "not_found"
resolved_alias_suffix = alias_suffix or f"__{auth_source}"
create_username, takeover_target, create_reason = _resolve_create_username(
user_db,
auth_source=auth_source,
-138
View File
@@ -1,138 +0,0 @@
"""Canonical language resolution shared by every release source.
Release sources report a language in whatever shape their upstream uses: a
two-letter code, an ISO 639-2 three-letter code in either the bibliographic or
terminological form, or an English name. They all need the same ISO 639-1 code
out the other side, so the aliases live in one place (``data/book-languages.json``)
and adding a language means editing one file.
"""
import json
import threading
import unicodedata
from pathlib import Path
from shelfmark.core.logger import setup_logger
logger = setup_logger(__name__)
LANGUAGE_DATA_PATH = Path(__file__).resolve().parents[1].parent / "data" / "book-languages.json"
# Values a source uses to mean "we could not tell".
LANGUAGE_PLACEHOLDERS = frozenset({"", "-", "--", "unknown", "unk", "n/a", "na", "none", "null"})
_ALIAS_TO_CODE: dict[str, str] | None = None
_CODE_TO_NAME: dict[str, str] | None = None
_LOCK = threading.Lock()
# Separators that stand in for the hyphen in a subtag. The dashes turn up in
# codes copied from web pages -- "zh‑Hant" used U+2011, which renders close
# enough to both a hyphen and an underscore to go unnoticed -- and the
# underscore is the spelling Direct Download accepted before this module existed.
_SUBTAG_SEPARATORS = dict.fromkeys(map(ord, "‐‑‒–—―−﹘﹣-_"), "-")
def _fold(value: str) -> str:
"""Casefold, strip accents, and normalize subtag separators, so 'Español'
and 'espanol', or 'zh-Hant', 'zh‑Hant' and 'zh_Hant', all match."""
decomposed = unicodedata.normalize("NFKD", value).translate(_SUBTAG_SEPARATORS)
stripped = "".join(ch for ch in decomposed if not unicodedata.combining(ch))
return " ".join(stripped.split()).casefold()
def _load() -> tuple[dict[str, str], dict[str, str]]:
global _ALIAS_TO_CODE, _CODE_TO_NAME
if _ALIAS_TO_CODE is not None and _CODE_TO_NAME is not None:
return _ALIAS_TO_CODE, _CODE_TO_NAME
with _LOCK:
if _ALIAS_TO_CODE is not None and _CODE_TO_NAME is not None:
return _ALIAS_TO_CODE, _CODE_TO_NAME
alias_to_code: dict[str, str] = {}
code_to_name: dict[str, str] = {}
try:
raw = json.loads(LANGUAGE_DATA_PATH.read_text(encoding="utf-8"))
except OSError, ValueError:
logger.exception("Failed to load language data from %s", LANGUAGE_DATA_PATH)
raw = []
if not isinstance(raw, list):
logger.warning("Language data at %s is not a list", LANGUAGE_DATA_PATH)
raw = []
for item in raw:
if not isinstance(item, dict):
continue
code = str(item.get("code") or "").strip()
name = str(item.get("language") or "").strip()
if not code:
continue
code_to_name.setdefault(code, name or code)
for candidate in (code, name, *(item.get("aliases") or [])):
folded = _fold(str(candidate))
if folded and folded not in LANGUAGE_PLACEHOLDERS:
alias_to_code.setdefault(folded, code)
_ALIAS_TO_CODE = alias_to_code
_CODE_TO_NAME = code_to_name
return alias_to_code, code_to_name
def normalize_language(value: object) -> str | None:
"""Resolve any known spelling of a language to its ISO 639-1 code.
Accepts a two-letter code, an ISO 639-2 three-letter code in either the
bibliographic or terminological form, or an English name. Returns None for
anything unrecognised or for the placeholders a source uses to say it does
not know, so callers can treat "no language" uniformly.
"""
if value is None:
return None
folded = _fold(str(value))
if not folded or folded in LANGUAGE_PLACEHOLDERS:
return None
alias_to_code, _ = _load()
return alias_to_code.get(folded)
def language_name(code: str | None) -> str | None:
"""Return the English name for a language code, or None if unknown."""
if not code:
return None
_, code_to_name = _load()
return code_to_name.get(str(code).strip())
def language_alias_map() -> dict[str, str]:
"""Every known alias mapped to its code, for callers doing their own matching.
Direct Download scans free-text paths and needs the whole alias set up front
to look for, rather than resolving one candidate at a time.
"""
alias_to_code, _ = _load()
return dict(alias_to_code)
def supported_book_languages() -> list[dict[str, str]]:
"""The selectable languages, as ``{"language": ..., "code": ...}``.
Aliases are an implementation detail of resolution, so they are left out of
what the settings dropdown and the API hand to clients.
"""
_, code_to_name = _load()
return [{"language": name, "code": code} for code, name in code_to_name.items()]
def known_language_codes() -> frozenset[str]:
"""Every ISO 639-1 code the bundled language data defines."""
_, code_to_name = _load()
return frozenset(code_to_name)
-10
View File
@@ -108,9 +108,6 @@ class DownloadTask:
retry_expected_hash: str | None = None # Optional torrent hash used to match client downloads
retry_ratio_limit: float | None = None # Optional post-download seeding ratio
retry_seeding_time_limit_minutes: int | None = None # Optional post-download seeding time limit
retry_source_context: dict[str, Any] = field(
default_factory=dict
) # Source-private context for retry/re-resolution
can_retry_without_staged_source: bool = (
True # Whether the source can restart without a preserved staged file
)
@@ -119,7 +116,6 @@ class DownloadTask:
series_name: str | None = None
series_position: float | None = None # Float for novellas (e.g., 1.5)
subtitle: str | None = None # Book subtitle for naming templates
language: str | None = None # Release language code for the {Language} template variable
# Hardlinking support
original_download_path: str | None = None # Path in download client (for hardlinking)
@@ -136,12 +132,6 @@ class DownloadTask:
default_factory=dict
) # Per-output parameters (e.g. email recipient)
# Multi-book packs: one release holding several books. `book_plan` is the split the
# user approved before download (list of {title, series_position, year, files});
# `multi_book` asks post-processing to split heuristically when no plan exists.
multi_book: bool = False
book_plan: list[dict[str, Any]] | None = None
# User association (multi-user support)
user_id: int | None = None # DB user ID who queued this download
username: str | None = None # Username for {User} template variable
+2 -30
View File
@@ -4,7 +4,6 @@ import re
from pathlib import Path
from typing import TYPE_CHECKING
from shelfmark.core.languages import LANGUAGE_PLACEHOLDERS, normalize_language
from shelfmark.core.logger import setup_logger
if TYPE_CHECKING:
@@ -20,7 +19,6 @@ KNOWN_TOKENS = [
"primarytitle",
"originalname",
"partnumber",
"language",
"subtitle",
"author",
"series",
@@ -68,33 +66,6 @@ def format_series_position(position: str | float | None) -> str:
return str(position)
def normalize_language_code(language: str | None) -> str:
"""Resolve a release language to the single spelling used in a path.
Sources report the same language in different shapes: "en", "eng", "English".
All of them have to collapse to one code, or the editions they identify end
up in separate folders, which is the collision this token exists to prevent.
Placeholder values render empty so `{ (Language)}` disappears entirely
rather than labelling a folder "(unknown)".
A language the bundled data does not know is kept, casefolded, rather than
dropped: it still separates editions, and it cannot collide with a resolved
code precisely because nothing resolves it.
"""
if not language:
return ""
resolved = normalize_language(language)
if resolved is not None:
return resolved
normalized = " ".join(str(language).split()).strip().casefold()
if normalized in LANGUAGE_PLACEHOLDERS:
return ""
return normalized
def derive_primary_title(title: str | None, subtitle: str | None) -> str:
"""Return the title without an explicit subtitle suffix when possible."""
title_value = " ".join(str(title or "").split()).strip()
@@ -120,7 +91,8 @@ PAD_NUMBERS_PATTERN = re.compile(r"\d+")
def natural_sort_key(path: str | Path) -> str:
"""Generate a sort key with padded numbers for natural sorting."""
return PAD_NUMBERS_PATTERN.sub(lambda m: m.group().zfill(9), str(path).lower())
filename = Path(path).name.lower()
return PAD_NUMBERS_PATTERN.sub(lambda m: m.group().zfill(9), filename)
def assign_part_numbers(
-42
View File
@@ -393,41 +393,6 @@ def _plugin_label(plugin: object, fallback_scheme: str) -> str:
return " ".join(parts)
def _apprise_proxy_env() -> dict[str, str]:
"""Build proxy env vars from app config so Apprise respects the proxy setting."""
import os
from shelfmark.core.config import config as _cfg
mode = str(_cfg.get("PROXY_MODE", "") or "").lower()
env: dict[str, str] = {}
if mode == "http":
http = str(_cfg.get("HTTP_PROXY", "") or "").strip()
https = str(_cfg.get("HTTPS_PROXY", "") or "").strip() or http
if http:
env["HTTP_PROXY"] = http
env["http_proxy"] = http
if https:
env["HTTPS_PROXY"] = https
env["https_proxy"] = https
elif mode == "socks5":
socks = str(_cfg.get("SOCKS5_PROXY", "") or "").strip()
if socks:
env["HTTP_PROXY"] = socks
env["http_proxy"] = socks
env["HTTPS_PROXY"] = socks
env["https_proxy"] = socks
no_proxy = str(_cfg.get("NO_PROXY", "") or "").strip()
if no_proxy and env:
env["NO_PROXY"] = no_proxy
env["no_proxy"] = no_proxy
# Don't override if the user already set these in the environment directly
return {k: v for k, v in env.items() if not os.environ.get(k)}
def _dispatch_to_apprise(
urls: Iterable[str],
*,
@@ -435,8 +400,6 @@ def _dispatch_to_apprise(
body: str,
notify_type: object,
) -> dict[str, Any]:
import os
normalized_urls = _normalize_urls(list(urls))
url_schemes = _extract_url_schemes(normalized_urls)
if not normalized_urls:
@@ -445,11 +408,6 @@ def _dispatch_to_apprise(
if apprise is None:
return {"success": False, "message": "Apprise is not installed"}
proxy_env = _apprise_proxy_env()
if proxy_env:
logger.debug("Applying proxy env for Apprise dispatch: %s", list(proxy_env.keys()))
os.environ.update(proxy_env)
valid_urls = 0
invalid_urls = 0
delivered_urls = 0
-33
View File
@@ -33,11 +33,6 @@ logger = setup_logger(__name__)
oauth = OAuth()
_RETURN_TO_SESSION_KEY = "oidc_return_to"
_OIDC_CLIENT_ERRORS = (OAuthError, OSError, RuntimeError, TypeError, ValueError)
_EMPTY_JWKS_MESSAGE = (
"Authentication failed: the identity provider returned no token signing keys "
"(empty JWKS). If you use Authentik, select a Signing Key in the provider "
"settings and try again."
)
class _ClaimsMappingLike(Protocol):
@@ -126,17 +121,6 @@ def _normalize_return_to(raw_return_to: object) -> str | None:
return urlunsplit(("", "", path, parsed.query, parsed.fragment))
def _idp_jwks_has_no_keys(client: Any) -> bool:
"""Return True when the IdP's JWKS document verifiably contains no signing keys."""
try:
jwk_set = client.fetch_jwk_set(force=True)
except (*_OIDC_CLIENT_ERRORS, KeyError):
return False
if not isinstance(jwk_set, Mapping):
return False
return not jwk_set.get("keys")
def _get_pending_return_to(*, clear: bool = False) -> str | None:
"""Read the pending post-login target from the session."""
raw_return_to = (
@@ -290,17 +274,6 @@ def register_oidc_routes(app: Flask, user_db: UserDB) -> None:
return redirect(
_login_error_url(f"OIDC token claim validation failed: {claim_name}")
)
except KeyError, ValueError:
# An IdP serving an empty JWKS document (e.g. an Authentik provider
# with no Signing Key selected) surfaces as KeyError('keys') while
# importing the key set. Test Connection only validates discovery,
# so this is the first place the misconfiguration becomes visible.
if _idp_jwks_has_no_keys(client):
logger.exception(
"OIDC callback failed: the IdP JWKS document contains no signing keys"
)
return redirect(_login_error_url(_EMPTY_JWKS_MESSAGE))
raise
claims = _normalize_claims(token.get("userinfo"))
# If userinfo is missing or claims are too sparse, request it explicitly.
@@ -333,12 +306,6 @@ def register_oidc_routes(app: Flask, user_db: UserDB) -> None:
is_admin = admin_group in groups
allow_email_link = bool(user_info.get("email")) and _is_email_verified(claims)
if user_info.get("email") and not allow_email_link:
logger.debug(
"OIDC email %s is not marked verified by the IdP; skipping "
"email-based account linking",
user_info["email"],
)
user = provision_oidc_user(
user_db,
user_info,
-102
View File
@@ -1,102 +0,0 @@
"""Pre-download release inspection: list a release's files and plan a multi-book split."""
from __future__ import annotations
from typing import TYPE_CHECKING, Any
from flask import jsonify, request
from shelfmark.core.logger import setup_logger
from shelfmark.core.utils import is_audiobook
from shelfmark.download.postprocess.packs import PackFile, PackPlan, plan_pack
from shelfmark.download.postprocess.policy import (
get_supported_audiobook_formats,
get_supported_formats,
)
from shelfmark.release_sources import get_handler
if TYPE_CHECKING:
from collections.abc import Callable
from flask import Flask, Response
logger = setup_logger(__name__)
_INSPECT_ERRORS = (OSError, RuntimeError, ValueError, TypeError, KeyError, AttributeError)
NOT_INSPECTABLE_REASON = "This source cannot list the release's files before downloading"
def _serialize_plan(plan: PackPlan) -> dict[str, Any]:
return {
"is_pack": plan.is_pack,
"ignored": plan.ignored,
"books": [
{
"title": book.title,
"series_position": book.series_position,
"year": book.year,
"files": book.files,
}
for book in plan.books
],
}
def inspect_release(data: dict[str, Any]) -> dict[str, Any]:
"""Build the inspect response for a release payload (same shape as a download)."""
source = str(data["source"])
handler = get_handler(source)
try:
files: list[PackFile] | None = handler.list_files(data)
except _INSPECT_ERRORS as exc:
logger.warning(
"Could not list files for %s release %s: %s", source, data.get("source_id"), exc
)
return {"inspected": False, "reason": str(exc), "files": [], "plan": None}
if files is None:
return {"inspected": False, "reason": NOT_INSPECTABLE_REASON, "files": [], "plan": None}
content_type = data.get("content_type")
supported = (
get_supported_audiobook_formats()
if is_audiobook(content_type if isinstance(content_type, str) else None)
else get_supported_formats()
)
series_name = data.get("series_name")
author_name = data.get("author")
plan = plan_pack(
files,
supported_extensions=set(supported),
series_name=series_name if isinstance(series_name, str) else None,
author_name=author_name if isinstance(author_name, str) else None,
)
return {
"inspected": True,
"reason": None,
"files": [{"path": f.path, "size": f.size} for f in files],
"plan": _serialize_plan(plan),
}
def register_release_inspect_routes(
app: Flask,
login_required: Callable[..., Any],
) -> None:
"""Register POST /api/releases/inspect."""
@app.route("/api/releases/inspect", methods=["POST"])
@login_required
def api_inspect_release() -> Response | tuple[Response, int]:
data = request.get_json(silent=True)
if not isinstance(data, dict):
return jsonify({"error": "No data provided"}), 400
if not data.get("source_id"):
return jsonify({"error": "source_id is required"}), 400
if not data.get("source"):
return jsonify({"error": "source is required"}), 400
try:
get_handler(str(data["source"]))
except ValueError as exc:
return jsonify({"error": str(exc)}), 400
return jsonify(inspect_release(data))
-134
View File
@@ -1,134 +0,0 @@
"""A wall-clock budget for one release search, enforced through the existing cancel flag.
`/api/releases` is synchronous: the browser waits on it while the search runs. Nothing
bounded that wait, and the bypasser's own worst case is minutes long
(`internal_bypasser.max_duration_seconds()`), so a search that ran into an unsolvable
protection challenge outlived every reverse proxy in front of it. The user then saw
"Server unavailable (504)" - a gateway timeout that says nothing about what went wrong
and points the blame at their proxy config. See issue #1276.
The budget is expressed as the cancel flag the download path already understands: an
Event armed by a timer. `html_get_page`, the bypassers and the helper subprocess all poll
it, so an expired budget stops a solve already in flight rather than only refusing the
next one. When it trips, the search fails with a message that names the real cause.
Scoped to a context variable so it applies to the request that set it and to nothing else
- a queued download must keep its own, much longer, budget.
"""
from __future__ import annotations
import threading
import time
from contextlib import contextmanager
from contextvars import ContextVar
from typing import TYPE_CHECKING
from shelfmark.core.logger import setup_logger
if TYPE_CHECKING:
from collections.abc import Iterator
logger = setup_logger(__name__)
# What one search may spend. A first search on a cold start legitimately pays for a
# browser solve - jfmlima measured 60-120s for a successful one on Anna's Archive - so
# this cannot be as tight as a proxy's default read timeout without breaking working
# setups. It is instead well below the ~840s the bypass path could previously reach,
# which is what turned a failing challenge into a gateway timeout.
DEFAULT_SEARCH_BUDGET_SECONDS = 300.0
_MIN_SEARCH_BUDGET_SECONDS = 30.0
_MAX_SEARCH_BUDGET_SECONDS = 1800.0
# Raised to the caller when the budget runs out, so the API can say so plainly.
SEARCH_DEADLINE_MESSAGE = (
"The release search ran out of time (%.0fs). Anna's Archive is behind a protection "
"challenge the bypasser could not solve in that window. Raise the release search "
"timeout if your setup is simply slow."
)
class SearchDeadline:
"""A budget with an Event that trips when it expires."""
def __init__(self, budget_seconds: float) -> None:
self.budget_seconds = budget_seconds
self.expires_at = time.monotonic() + budget_seconds
# A plain threading.Event on purpose: this is handed on as a cancel flag, and
# that is the type the download path, the CDP worker thread and the bypass helper
# already poll.
self.event = threading.Event()
self._timer = threading.Timer(budget_seconds, self.event.set)
self._timer.daemon = True
def start(self) -> None:
self._timer.start()
def cancel(self) -> None:
self._timer.cancel()
@property
def remaining(self) -> float:
return max(0.0, self.expires_at - time.monotonic())
@property
def expired(self) -> bool:
return self.event.is_set() or self.remaining <= 0
_current: ContextVar[SearchDeadline | None] = ContextVar("search_deadline", default=None)
def budget_seconds() -> float:
"""The configured budget for one release search."""
from shelfmark.core.config import config as app_config
raw = app_config.get("RELEASE_SEARCH_TIMEOUT", DEFAULT_SEARCH_BUDGET_SECONDS)
if isinstance(raw, bool) or not isinstance(raw, int | float | str):
return DEFAULT_SEARCH_BUDGET_SECONDS
try:
value = float(raw)
except TypeError, ValueError:
return DEFAULT_SEARCH_BUDGET_SECONDS
if value <= 0:
return DEFAULT_SEARCH_BUDGET_SECONDS
return min(max(value, _MIN_SEARCH_BUDGET_SECONDS), _MAX_SEARCH_BUDGET_SECONDS)
@contextmanager
def search_deadline(budget: float | None = None) -> Iterator[SearchDeadline]:
"""Apply a budget to everything the calling context does."""
deadline = SearchDeadline(budget if budget is not None else budget_seconds())
token = _current.set(deadline)
deadline.start()
logger.debug("Release search budget: %.0fs", deadline.budget_seconds)
try:
yield deadline
finally:
deadline.cancel()
_current.reset(token)
def current() -> SearchDeadline | None:
"""The budget in force, or None outside a search."""
return _current.get()
def expired() -> bool:
"""Whether the budget in force has run out. False when there is no budget."""
deadline = _current.get()
return deadline is not None and deadline.expired
def cancel_event() -> threading.Event | None:
"""The Event that trips when the budget runs out, for use as a cancel flag."""
deadline = _current.get()
return deadline.event if deadline is not None else None
def deadline_message() -> str:
"""The failure to report when the budget has run out."""
deadline = _current.get()
budget = deadline.budget_seconds if deadline else DEFAULT_SEARCH_BUDGET_SECONDS
return SEARCH_DEADLINE_MESSAGE % budget
+28 -103
View File
@@ -7,7 +7,6 @@ from dataclasses import dataclass
from typing import TYPE_CHECKING
from shelfmark.core.config import config
from shelfmark.core.logger import setup_logger
from shelfmark.metadata_providers import (
BookMetadata,
build_localized_search_titles,
@@ -17,8 +16,6 @@ from shelfmark.metadata_providers import (
if TYPE_CHECKING:
from shelfmark.core.models import SearchFilters
logger = setup_logger(__name__)
MANUAL_QUERY_MAX_LEN = 256
@@ -55,110 +52,44 @@ class ReleaseSearchPlan:
return self.title_variants[0].query if self.title_variants else ""
def _to_language_codes(values: Iterable[object], *, source: str) -> list[str] | None:
"""Resolve any spelling of a language to the ISO code the sources expect.
Anna's Archive matches `lang=` against ISO codes: `lang=english` is not a loose
spelling of `lang=en`, it is a facet value AA does not have, and it filters every
search down to nothing. Only the *per-user* override was normalised
(config.users_settings.validate), so a global BOOK_LANGUAGE=english - the spelling
the old docs used - reached the query verbatim and silently emptied every search
with no error anywhere. See issue #1276.
An entry that resolves to nothing is dropped with a warning rather than passed
through: searching unfiltered and saying so beats reporting "no results" for a book
the source is full of.
"""
from shelfmark.core.languages import normalize_language
codes: list[str] = []
unresolved: list[str] = []
for value in values:
text = str(value).strip() if value is not None else ""
if not text:
continue
if text.lower() == "all":
# An explicit "search every language", not a language.
return None
code = normalize_language(text)
if code is None:
unresolved.append(text)
continue
if code not in codes:
codes.append(code)
if unresolved:
logger.warning(
"Ignoring unrecognised language(s) in %s: %s. Use an ISO code such as 'en', "
"a three-letter code, or an English name like 'English'.",
source,
", ".join(unresolved),
)
return codes or None
def _normalize_languages(languages: list[str] | None, user_id: int | None) -> list[str] | None:
def _normalize_languages(languages: list[str] | None) -> list[str] | None:
if not languages:
default = config.get("BOOK_LANGUAGE", None, user_id=user_id)
default = getattr(config, "BOOK_LANGUAGE", None)
if isinstance(default, str):
default_values: list[object] = [default]
elif isinstance(default, Iterable) and not isinstance(default, (bytes, bytearray, dict)):
default_values = list(default)
else:
return None
return _to_language_codes(default_values, source="BOOK_LANGUAGE")
return [str(lang).strip() for lang in default_values if str(lang).strip()]
return _to_language_codes(languages, source="the search request")
normalized: list[str] = []
for lang in languages:
if not lang:
continue
s = str(lang).strip()
if not s:
continue
normalized.append(s)
if any(lang.lower() == "all" for lang in normalized):
return None
return normalized or None
def first_author(value: str) -> str:
"""The first name in a possibly comma-joined author string.
Both ends of the app hand us every contributor in one string. The frontend joins
`authors` with ", " for display (`bookTransformers.ts`) and that display string comes
straight back as the `author` request parameter, while several providers set
`search_author` from the same joined text. Searching a release source for
"Blindness Jose Saramago, Giovanni Pontiero, ..." - the author plus two translators -
matches nothing, and the user is told the book has no releases at all.
A "Last, First" author collapses to the surname, which is still a usable search term
and is what the authors[] fallback has always done with the same input. See #1252.
"""
first, _, _ = value.partition(",")
return first.strip()
def pick_search_author(book: BookMetadata) -> str:
"""The one author a release query should carry, from whichever field holds one.
Every release source that builds its own query wants exactly this, so it lives here
rather than being re-derived per source - the two branches below drifted apart once
already (#1252) and the IRC source carried a third copy of the same preference.
#1290 fixed the same report by merging the two branches and trimming whichever one
won; this keeps that outcome ("Blindness Jose Saramago" from either field, measured
there at 0 releases before and 49 after) and adds the empty-narrowing fallback, so a
credit list that merely starts with a blank entry does not fall out to title-only.
"""
# Narrowing can come back empty - the joined string starts with a comma because the
# first contributor was blank, and `authors.join(', ')` does not drop the empty entry.
# Falling through to authors[] then still finds a usable name; returning "" would
# search by title alone and lose the author we were holding all along.
def _pick_search_author(book: BookMetadata) -> str:
if book.search_author:
narrowed = first_author(book.search_author)
if narrowed:
return narrowed
return book.search_author
# A bare string here would otherwise be iterated one character at a time; the IRC
# source guarded against exactly that before it shared this helper.
authors = book.authors if isinstance(book.authors, list) else [book.authors or ""]
for author in authors:
narrowed = first_author(author or "")
if narrowed:
return narrowed
if not book.authors:
return ""
return ""
first = book.authors[0]
if "," in first:
first = first.split(",")[0].strip()
return first
def _pick_search_title(book: BookMetadata) -> str:
@@ -171,21 +102,15 @@ def build_release_search_plan(
manual_query: str | None = None,
indexers: list[str] | None = None,
source_filters: SearchFilters | None = None,
user_id: int | None = None,
) -> ReleaseSearchPlan:
"""Build normalized search variants shared across release sources.
``user_id`` picks up that user's default languages when the caller does not
filter explicitly, so a search started without a language filter uses the
reader's own default rather than the instance-wide one.
"""
resolved_languages = _normalize_languages(languages, user_id)
"""Build normalized search variants shared across release sources."""
resolved_languages = _normalize_languages(languages)
resolved_manual_query = None
if manual_query:
resolved_manual_query = manual_query.strip()[:MANUAL_QUERY_MAX_LEN] or None
author = pick_search_author(book)
author = _pick_search_author(book)
base_title = _pick_search_title(book)
if resolved_manual_query:
-4
View File
@@ -9,7 +9,6 @@ from typing import TYPE_CHECKING, Any
from werkzeug.utils import secure_filename
from shelfmark.config.env import normalize_log_level
from shelfmark.core.logger import setup_logger
from shelfmark.core.request_helpers import coerce_bool, normalize_optional_text
@@ -899,9 +898,6 @@ def _get_env_value_for_field(field: FieldBase) -> tuple[bool, object | None]:
"WELIB_MIRROR_URLS",
} and isinstance(parsed, list):
parsed = _normalize_mirror_env_urls(parsed)
if field.key == "LOG_LEVEL" and isinstance(parsed, str):
# LOG_LEVEL is commonly set lowercase; the field options are uppercase.
parsed = normalize_log_level(parsed)
return True, parsed
if field.key == "AA_MIRROR_URLS":
-2
View File
@@ -344,7 +344,6 @@ class UserDB:
_ALLOWED_UPDATE_COLUMNS: ClassVar[frozenset[str]] = frozenset(
{
"username",
"email",
"display_name",
"password_hash",
@@ -354,7 +353,6 @@ class UserDB:
}
)
_USER_UPDATE_STATEMENTS: ClassVar[dict[str, str]] = {
"username": "UPDATE users SET username = ? WHERE id = ?",
"email": "UPDATE users SET email = ? WHERE id = ?",
"display_name": "UPDATE users SET display_name = ? WHERE id = ?",
"password_hash": "UPDATE users SET password_hash = ? WHERE id = ?",
-27
View File
@@ -52,13 +52,6 @@ def normalize_http_url(
if scheme:
normalized = f"{scheme}://{normalized}"
# Strip query string and fragment — mirrors are used as base URLs for
# constructing search requests; params/fragments on the configured URL
# produce malformed URLs when paths are appended (issue #999).
parsed = urlparse(normalized)
if parsed.query or parsed.fragment:
normalized = parsed._replace(query="", fragment="").geturl()
if strip_trailing_slash:
normalized = normalized.rstrip("/")
@@ -115,26 +108,6 @@ def is_audiobook(content_type: str | None) -> bool:
return bool(content_type and "audiobook" in content_type.lower())
# Every audio format an audiobook can legitimately arrive in, and the single source of
# truth for that list. The settings UI, release-source parsing, archive extraction and
# post-download scanning all derive from it, so a format added here becomes selectable,
# searchable AND downloadable at once. These used to be four hand-maintained copies that
# had drifted apart: the settings UI only offered m4b/mp3/m4a, which meant a FLAC
# audiobook could never be enabled, was silently dropped from every search result, and
# was rejected after download as "format not supported".
#
# "mp4" is here because some trackers (MyAnonamouse in particular) ship AAC audiobooks
# as per-chapter .mp4 files - the same ISO-BMFF container as .m4a/.m4b, just with the
# generic extension. Without it those releases downloaded fine and then failed
# post-processing with "No book files found in download".
AUDIOBOOK_FORMATS = ("m4b", "mp3", "m4a", "mp4", "flac", "ogg", "wma", "aac", "wav", "opus")
# Multi-file audiobooks are almost always distributed as an archive. These are containers
# rather than formats: they are what a *release* looks like, and the formats above are
# what comes out of one after extraction.
ARCHIVE_FORMATS = ("zip", "rar")
CONTENT_TYPES = [
"book (fiction)",
"book (non-fiction)",
-81
View File
@@ -1,81 +0,0 @@
"""Stall-detection grace signalling for long single-shot download operations.
The orchestrator cancels a download after `STALL_TIMEOUT` seconds without activity, where
"activity" means a *changed* status event or a *changed* progress value. That de-duplication
is deliberate - a keep-alive that repeats the same payload on a timer proves nothing about
whether the operation is still making progress, so letting it refresh the stall clock would
make a genuinely wedged download immortal.
Operations that legitimately take longer than `STALL_TIMEOUT` but cannot report incremental
progress therefore declare an explicit upper bound up front instead:
request_activity_grace(status_callback, my_worst_case_seconds)
try:
...one long blocking call...
finally:
release_activity_grace(status_callback)
The grace is a single absolute deadline. It is never extended, so the operation still dies
if it overruns its own declared budget - just at *its* bound rather than at a global 300s.
The signal rides on the existing `status_callback` channel using a sentinel status, which
avoids threading a new parameter through every handler, post-processor and output module.
`shelfmark.download.orchestrator`'s per-task `status_callback` closure intercepts the
sentinel and never forwards it to `update_download_status`.
Adopters should be operations that yield to the gevent hub while blocking (`requests`,
patched `subprocess`). An operation that blocks the hub outright - `shutil.copy2`, sqlite -
will still be killed by the gunicorn worker timeout regardless of any grace, and must go
through `shelfmark.download.fs.run_blocking_io` first.
Current adopters: `shelfmark.download.http.html_get_page` (protection bypass).
Candidates: `download.clients.base_handler._wait_for_completed_path`, archive extraction in
`download.postprocess.scan`, large-file copies in `download.outputs.folder`, email/BookLore
uploads, and the Anna's Archive countdown in `release_sources.direct_download` (which today
refreshes the stall clock on every tick of a loop that proves nothing about the remote).
"""
from collections.abc import Callable
# Not a QueueStatus value, so `update_download_status` would reject it anyway; the
# orchestrator's status_callback intercepts it before that point.
ACTIVITY_GRACE_STATUS = "__activity_grace__"
StatusCallback = Callable[[str, str | None], None]
# A status_callback is caller-supplied and may raise; a failed liveness hint must never
# break the operation it was protecting. Mirrors http._STATUS_CALLBACK_ERRORS.
_CALLBACK_ERRORS = (AttributeError, KeyError, OSError, RuntimeError, TypeError, ValueError)
def request_activity_grace(status_callback: StatusCallback | None, seconds: float) -> None:
"""Ask the orchestrator to suppress stall detection for up to `seconds` from now."""
_emit(status_callback, seconds)
def release_activity_grace(status_callback: StatusCallback | None) -> None:
"""Drop any outstanding grace and count now as activity."""
_emit(status_callback, 0)
def parse_activity_grace(status: str, message: str | None) -> float | None:
"""Return the requested grace in seconds, or None if this is not a grace event.
Never raises: a malformed sentinel is treated as "not a grace event" so a bad emitter
cannot take down the status pipeline.
"""
if status != ACTIVITY_GRACE_STATUS:
return None
try:
return max(float(message or 0), 0.0)
except TypeError, ValueError:
return 0.0
def _emit(status_callback: StatusCallback | None, seconds: float) -> None:
if status_callback is None:
return
try:
status_callback(ACTIVITY_GRACE_STATUS, str(float(seconds)))
except _CALLBACK_ERRORS:
return
+1 -2
View File
@@ -7,7 +7,6 @@ from pathlib import Path
from typing import TYPE_CHECKING, cast
from shelfmark.core.logger import setup_logger
from shelfmark.core.utils import AUDIOBOOK_FORMATS
from shelfmark.core.utils import is_audiobook as check_audiobook
from shelfmark.download.fs import atomic_move
from shelfmark.download.postprocess.policy import (
@@ -99,7 +98,7 @@ ALL_EBOOK_EXTENSIONS = {
}
# All known audio extensions (superset of what user might enable for audiobooks)
ALL_AUDIO_EXTENSIONS = {f".{fmt}" for fmt in AUDIOBOOK_FORMATS}
ALL_AUDIO_EXTENSIONS = {".m4b", ".mp3", ".m4a", ".aac", ".flac", ".ogg", ".wma", ".wav", ".opus"}
def _filter_files(
+2 -18
View File
@@ -325,19 +325,6 @@ class DownloadClient(ABC):
"""
def set_category(self, download_id: str, category: str) -> bool:
"""Update a download's category or label when supported by the client.
Args:
download_id: The client-specific download ID.
category: Category or label to assign.
Returns:
True if the category was updated, otherwise False.
"""
return False
@abstractmethod
def get_download_path(self, download_id: str) -> str | None:
"""Get the path where files were downloaded.
@@ -372,13 +359,10 @@ class DownloadClient(ABC):
# Client registry: protocol -> list of client classes
_CLIENTS: dict[str, list[type[DownloadClient]]] = {}
ClientType = TypeVar("ClientType", bound=DownloadClient)
_BUILTIN_CLIENT_MODULES = (
"shelfmark.download.clients.alldebrid",
"shelfmark.download.clients.deluge",
"shelfmark.download.clients.nzbget",
"shelfmark.download.clients.qbittorrent",
"shelfmark.download.clients.realdebrid",
"shelfmark.download.clients.rtorrent",
"shelfmark.download.clients.sabnzbd",
"shelfmark.download.clients.transmission",
@@ -399,7 +383,7 @@ def _ensure_builtin_clients_registered() -> None:
def register_client(
protocol: str,
) -> Callable[[type[ClientType]], type[ClientType]]:
) -> Callable[[type[DownloadClient]], type[DownloadClient]]:
"""Register a download client for a protocol.
Multiple clients can be registered for the same protocol.
@@ -415,7 +399,7 @@ def register_client(
"""
def decorator(cls: type[ClientType]) -> type[ClientType]:
def decorator(cls: type[DownloadClient]) -> type[DownloadClient]:
if protocol not in _CLIENTS:
_CLIENTS[protocol] = []
_CLIENTS[protocol].append(cls)
-718
View File
@@ -1,718 +0,0 @@
"""AllDebrid debrid service client for Shelfmark.
Routes magnet links through the AllDebrid API (v4/v4.1) to download
torrent content via AllDebrid's CDN infrastructure.
"""
from __future__ import annotations
import shutil
import threading
import time
from dataclasses import dataclass, field
from pathlib import Path
from typing import Any, ClassVar, NoReturn
from urllib.parse import quote
import requests
from shelfmark.config.env import TMP_DIR
from shelfmark.core.config import config
from shelfmark.core.logger import setup_logger
from shelfmark.download.clients import (
DownloadClient,
DownloadState,
DownloadStatus,
register_client,
)
from shelfmark.download.clients._coercion import config_text
from shelfmark.download.clients.torrent_utils import (
DebridMagnet,
DebridUpload,
resolve_debrid_upload,
)
from shelfmark.download.http import download_url
from shelfmark.download.network import get_ssl_verify
logger = setup_logger(__name__)
_API_BASE = "https://api.alldebrid.com/v4"
_AGENT = "shelfmark"
_ALLDEBRID_CLIENT_ERRORS = (
AttributeError,
OSError,
requests.exceptions.RequestException,
RuntimeError,
TypeError,
ValueError,
)
# AllDebrid magnet status codes (from API v4.1 documentation).
_STATUS_DOWNLOADING = frozenset({0, 1, 2, 3})
_STATUS_READY = 4
# Timeouts and retry limits for API calls.
_API_TIMEOUT = 30
_STATUS_TIMEOUT = 15
_DELAYED_POLL_INTERVAL = 5
_DELAYED_POLL_MAX_ATTEMPTS = 12
# File extensions recognised as book or audiobook content.
_BOOK_EXTENSIONS = (
".aac",
".azw",
".azw3",
".cbr",
".cbz",
".djvu",
".doc",
".docx",
".epub",
".fb2",
".flac",
".lit",
".m4a",
".m4b",
".mobi",
".mp3",
".mp4",
".ogg",
".opus",
".pdf",
".rtf",
".txt",
".wma",
)
def _flatten_magnet_files(
entries: list[dict[str, Any]],
prefix: str = "",
) -> list[dict[str, Any]]:
"""Flatten AllDebrid's nested file tree into a list of file dicts.
AllDebrid returns files with ``"n"`` (name), ``"s"`` (size),
``"l"`` (link), and ``"e"`` (children) keys. Directories use
``"e"`` to nest their contents.
Returns:
List of ``{"filename": ..., "size": ..., "link": ...}`` dicts.
"""
flat: list[dict[str, Any]] = []
for entry in entries:
name = entry.get("n", "")
if "e" in entry:
flat.extend(
_flatten_magnet_files(entry["e"], prefix=f"{prefix}{name}/"),
)
elif entry.get("l"):
flat.append(
{
"filename": f"{prefix}{name}",
"size": entry.get("s", 0),
"link": entry["l"],
}
)
return flat
def _raise_runtime_error(message: str) -> NoReturn:
raise RuntimeError(message)
@dataclass
class _DownloadState:
"""Internal mutable state for an in-progress AllDebrid download."""
magnet_id: str
name: str
target_dir: Path
phase: str = "uploading"
error_message: str | None = None
progress: float = 0.0
download_thread: threading.Thread | None = None
lock: threading.Lock = field(default_factory=threading.Lock)
@register_client("torrent")
class AllDebridClient(DownloadClient):
"""AllDebrid debrid service client.
Downloads torrent content by uploading magnet links to AllDebrid,
waiting for the torrent to complete on their servers, then fetching
the resulting files via direct HTTP download from AllDebrid's CDN.
API documentation: https://docs.alldebrid.com/
"""
protocol = "torrent"
name = "alldebrid"
_downloads: ClassVar[dict[str, _DownloadState]] = {}
_downloads_lock = threading.Lock()
def __init__(self) -> None:
self._api_key = config_text(config.get("ALLDEBRID_API_KEY", ""))
def _auth_headers(self) -> dict[str, str]:
"""Return Authorization header dict for API requests."""
return {"Authorization": f"Bearer {self._api_key}"}
# ------------------------------------------------------------------
# DownloadClient interface
# ------------------------------------------------------------------
@staticmethod
def is_configured() -> bool:
"""Return True when AllDebrid is selected and an API key exists."""
client = config_text(config.get("PROWLARR_TORRENT_CLIENT", ""))
api_key = config_text(config.get("ALLDEBRID_API_KEY", ""))
return client == "alldebrid" and bool(api_key)
def test_connection(self) -> tuple[bool, str]:
"""Validate the API key and check Premium subscription status."""
if not self._api_key:
return False, "AllDebrid API Key is required"
try:
url = f"{_API_BASE}/user"
resp = requests.get(
url,
headers=self._auth_headers(),
timeout=_STATUS_TIMEOUT,
verify=get_ssl_verify(url),
)
resp.raise_for_status()
data = resp.json()
if data.get("status") != "success":
err = data.get("error", {}).get("message", "API error")
return False, f"AllDebrid error: {err}"
user = data.get("data", {}).get("user", {})
username = user.get("username", "Unknown")
if not user.get("isPremium", False):
return (
False,
f"AllDebrid user '{username}' does not have a Premium subscription",
)
except _ALLDEBRID_CLIENT_ERRORS as e:
return False, f"Connection failed: {e}"
else:
return True, f"Connected to AllDebrid as '{username}' (Premium)"
def add_download(
self,
url: str,
name: str,
category: str | None = None,
expected_hash: str | None = None,
**kwargs: object,
) -> str:
"""Send a torrent to AllDebrid and return the magnet ID.
Accepts a magnet link, a .torrent URL, or an indexer proxy URL; anything
that is not already a magnet is resolved first, since an HTTP URL posted
as a magnet is rejected rather than downloaded (#1250).
"""
if not self._api_key:
msg = "AllDebrid API key is not configured"
raise RuntimeError(msg)
try:
upload = resolve_debrid_upload(url, expected_hash=expected_hash)
info = self._send_torrent(upload)
magnet_id = str(info.get("id", ""))
if not magnet_id:
msg = "No magnet ID returned from AllDebrid"
_raise_runtime_error(msg)
target_dir = TMP_DIR / f"alldebrid_{magnet_id}"
target_dir.mkdir(parents=True, exist_ok=True)
state = _DownloadState(
magnet_id=magnet_id,
name=name,
target_dir=target_dir,
phase="waiting_ad",
)
with self._downloads_lock:
self._downloads[magnet_id] = state
logger.info(
"Added torrent to AllDebrid: ID %s (%s)",
magnet_id,
name,
)
except Exception:
logger.exception("Failed to add torrent to AllDebrid")
raise
else:
return magnet_id
def _send_torrent(self, upload: DebridUpload) -> dict[str, Any]:
"""Hand the torrent to AllDebrid, as a magnet or as a file upload.
Both endpoints answer with the same envelope and the same per-entry
error shape, differing only in which key holds the entries.
"""
if isinstance(upload, DebridMagnet):
api_url = f"{_API_BASE}/magnet/upload"
entries_key = "magnets"
resp = requests.post(
api_url,
headers=self._auth_headers(),
data={"magnets[]": upload.magnet_url},
timeout=_API_TIMEOUT,
verify=get_ssl_verify(api_url),
)
else:
api_url = f"{_API_BASE}/magnet/upload/file"
entries_key = "files"
resp = requests.post(
api_url,
headers=self._auth_headers(),
files={
"files[]": (
"release.torrent",
upload.torrent_data,
"application/x-bittorrent",
)
},
timeout=_API_TIMEOUT,
verify=get_ssl_verify(api_url),
)
resp.raise_for_status()
data = resp.json()
if data.get("status") != "success":
code = data.get("error", {}).get("code", "UNKNOWN")
msg = f"AllDebrid upload failed: {code}"
_raise_runtime_error(msg)
entries = data.get("data", {}).get(entries_key, [])
if not entries:
msg = "AllDebrid accepted the upload but returned no torrent"
_raise_runtime_error(msg)
info = entries[0]
if info.get("error"):
code = info["error"].get("code", "UNKNOWN")
msg = f"AllDebrid rejected the torrent: {code}"
_raise_runtime_error(msg)
return info
def get_status(self, download_id: str) -> DownloadStatus:
"""Poll AllDebrid for magnet status and drive the download."""
state = self._ensure_state(download_id)
# Return cached terminal / in-flight states immediately.
with state.lock:
if state.phase == "error":
return DownloadStatus.error(
state.error_message or "AllDebrid error",
)
if state.phase == "complete":
return DownloadStatus(
progress=100.0,
state=DownloadState.COMPLETE,
message="Complete",
complete=True,
file_path=str(state.target_dir),
)
if state.phase == "downloading_http":
return DownloadStatus(
progress=state.progress,
state=DownloadState.DOWNLOADING,
message="Downloading files via HTTP...",
complete=False,
file_path=None,
)
# Ask AllDebrid for the current magnet status.
try:
status_url = f"{_API_BASE.replace('/v4', '/v4.1')}/magnet/status"
resp = requests.post(
status_url,
headers=self._auth_headers(),
data={"id": download_id},
timeout=_STATUS_TIMEOUT,
verify=get_ssl_verify(status_url),
)
resp.raise_for_status()
data = resp.json()
if data.get("status") != "success":
err = data.get("error", {}).get("message", "Status failed")
return DownloadStatus.error(
f"AllDebrid status error: {err}",
)
mag = self._extract_magnet_info(data)
return self._handle_magnet_status(mag, state)
except Exception as e:
logger.exception(
"Error checking AllDebrid status for %s",
download_id,
)
return DownloadStatus.error(str(e))
def remove(
self,
download_id: str,
*,
delete_files: bool = False,
) -> bool:
"""Delete the magnet from AllDebrid and clean up local files."""
try:
url = f"{_API_BASE}/magnet/delete"
requests.post(
url,
headers=self._auth_headers(),
data={"id": download_id},
timeout=_STATUS_TIMEOUT,
verify=get_ssl_verify(url),
)
except _ALLDEBRID_CLIENT_ERRORS as e:
logger.warning("Failed to delete magnet from AllDebrid: %s", e)
with self._downloads_lock:
state = self._downloads.pop(download_id, None)
if state and state.target_dir.exists():
shutil.rmtree(state.target_dir, ignore_errors=True)
return True
def get_download_path(self, download_id: str) -> str | None:
"""Return the local directory containing downloaded files."""
with self._downloads_lock:
state = self._downloads.get(download_id)
if state and state.phase == "complete":
return str(state.target_dir)
target_dir = TMP_DIR / f"alldebrid_{download_id}"
if target_dir.exists():
return str(target_dir)
return None
# ------------------------------------------------------------------
# Internal helpers
# ------------------------------------------------------------------
def _ensure_state(self, download_id: str) -> _DownloadState:
"""Get or create download state for the given magnet ID."""
with self._downloads_lock:
state = self._downloads.get(download_id)
if state:
return state
target_dir = TMP_DIR / f"alldebrid_{download_id}"
state = _DownloadState(
magnet_id=download_id,
name=f"Download {download_id}",
target_dir=target_dir,
phase="waiting_ad",
)
with self._downloads_lock:
self._downloads[download_id] = state
return state
@staticmethod
def _extract_magnet_info(data: dict[str, Any]) -> dict[str, Any]:
"""Extract magnet info dict from a status API response."""
mag_data = data.get("data", {}).get("magnets", {})
if isinstance(mag_data, list) and mag_data:
return mag_data[0]
if isinstance(mag_data, dict):
return mag_data
return {}
def _handle_magnet_status(
self,
mag: dict[str, Any],
state: _DownloadState,
) -> DownloadStatus:
"""Map AllDebrid magnet status to a DownloadStatus."""
status_code = mag.get("statusCode")
if status_code in _STATUS_DOWNLOADING:
size = mag.get("size", 0)
downloaded = mag.get("downloaded", 0)
pct = (downloaded / size * 100.0) if size > 0 else 0.0
return DownloadStatus(
progress=pct * 0.5,
state=DownloadState.DOWNLOADING,
message=(f"AllDebrid downloading torrent ({mag.get('filename', state.name)})"),
complete=False,
file_path=None,
download_speed=mag.get("downloadSpeed", 0),
)
if status_code == _STATUS_READY or mag.get("ready", False):
self._maybe_start_download_thread(state)
return DownloadStatus(
progress=50.0,
state=DownloadState.DOWNLOADING,
message="AllDebrid ready, retrieving files...",
complete=False,
file_path=None,
)
# Terminal error from AllDebrid.
error_txt = mag.get("error", {}).get("message") or f"AllDebrid status code {status_code}"
with state.lock:
state.phase = "error"
state.error_message = error_txt
return DownloadStatus.error(error_txt)
def _maybe_start_download_thread(self, state: _DownloadState) -> None:
"""Spawn a background thread to unlock and download files."""
with state.lock:
already_running = state.phase in (
"unlocking",
"downloading_http",
"complete",
)
thread_alive = state.download_thread is not None and state.download_thread.is_alive()
if already_running or thread_alive:
return
state.phase = "unlocking"
t = threading.Thread(
target=self._process_and_download,
args=(state,),
daemon=True,
)
state.download_thread = t
t.start()
# ------------------------------------------------------------------
# Link unlocking
# ------------------------------------------------------------------
def _unlock_file_link(self, link: str) -> str:
"""Resolve an AllDebrid file link to a direct CDN download URL.
AllDebrid's ``/v4/magnet/files`` endpoint returns virtual links
(``alldebrid.com/f/...``) that must be converted to direct CDN
URLs via ``/v4/link/unlock``.
Strategy:
1. If the link is already a CDN URL (``/dl/``), return it.
2. ``POST /v4/link/unlock`` with Bearer auth (primary).
3. ``GET /v4/link/unlock`` with query parameters (fallback).
4. Append ``apikey=`` to ``alldebrid.com/f/`` links
(last-resort fallback for ghost-cached torrents).
"""
# 1. Already a direct CDN link.
if "/dl/" in link:
return link
headers = self._auth_headers()
unlock_url = f"{_API_BASE}/link/unlock"
err_msg = "Unknown unlock error"
# 2. POST unlock (primary method).
try:
resp = requests.post(
unlock_url,
headers=headers,
data={"link": link},
timeout=_API_TIMEOUT,
verify=get_ssl_verify(unlock_url),
)
if resp.status_code == 200:
body = resp.json()
if body.get("status") == "success":
direct = self._resolve_unlock_data(
body.get("data", {}),
headers,
)
if direct:
return direct
err_msg = body.get("error", {}).get(
"message",
"Unlock failed",
)
except _ALLDEBRID_CLIENT_ERRORS as e:
logger.debug("POST unlock exception: %s", e)
# 3. GET unlock fallback with URL-encoded link.
try:
encoded = quote(link, safe="")
get_url = (
f"{_API_BASE}/link/unlock?agent={_AGENT}&apikey={self._api_key}&link={encoded}"
)
resp = requests.get(
get_url,
headers=headers,
timeout=_API_TIMEOUT,
verify=get_ssl_verify(get_url),
)
if resp.status_code == 200:
body = resp.json()
if body.get("status") == "success":
direct = body.get("data", {}).get("link")
if direct:
return direct
err_msg = body.get("error", {}).get("message", err_msg)
except _ALLDEBRID_CLIENT_ERRORS as e:
logger.debug("GET unlock exception: %s", e)
# 4. Last-resort: append apikey to alldebrid.com/f/ links.
if "alldebrid.com/f/" in link:
logger.info(
"Using apikey fallback for AllDebrid file link: %s",
link,
)
if "apikey=" not in link:
sep = "&" if "?" in link else "?"
return f"{link}{sep}apikey={self._api_key}"
return link
logger.error(
"AllDebrid unlock failed for '%s': %s",
link,
err_msg,
)
msg = f"AllDebrid unlock failed: {err_msg}"
raise RuntimeError(msg)
def _resolve_unlock_data(
self,
data: dict[str, Any],
headers: dict[str, str],
) -> str | None:
"""Extract the direct link from unlock response data.
Handles the *delayed link* flow where AllDebrid returns a
``delayed`` ID instead of an immediate download link.
"""
# Delayed link: poll until the CDN file is ready.
if "delayed" in data:
delayed_id = data["delayed"]
logger.info(
"AllDebrid link delayed (ID %s), polling...",
delayed_id,
)
delayed_url = f"{_API_BASE}/link/delayed"
for _ in range(_DELAYED_POLL_MAX_ATTEMPTS):
time.sleep(_DELAYED_POLL_INTERVAL)
try:
resp = requests.post(
delayed_url,
headers=headers,
data={"id": delayed_id},
timeout=_STATUS_TIMEOUT,
verify=get_ssl_verify(delayed_url),
)
if resp.status_code != 200:
continue
body = resp.json()
d = body.get("data", {})
if body.get("status") == "success" and d.get("status") == 2 and d.get("link"):
return d["link"]
except _ALLDEBRID_CLIENT_ERRORS as e:
logger.debug("Delayed poll exception: %s", e)
return data.get("link")
# ------------------------------------------------------------------
# File download pipeline
# ------------------------------------------------------------------
def _process_and_download(self, state: _DownloadState) -> None:
"""Fetch the file list, unlock links, and download via HTTP.
Runs in a background thread spawned by ``_maybe_start_download_thread``.
"""
try:
files = self._fetch_file_list(state.magnet_id)
relevant = [f for f in files if f["filename"].lower().endswith(_BOOK_EXTENSIONS)]
if not relevant:
relevant = files
with state.lock:
state.phase = "downloading_http"
total = len(relevant)
for idx, file_info in enumerate(relevant):
direct_link = self._unlock_file_link(file_info["link"])
rel_path = Path(file_info["filename"])
dest = state.target_dir / rel_path
dest.parent.mkdir(parents=True, exist_ok=True)
logger.info(
"Downloading AllDebrid file %d/%d: %s",
idx + 1,
total,
rel_path,
)
buf = download_url(
direct_link,
referer="https://alldebrid.com/",
)
if not buf:
msg = f"Failed to download from {direct_link}"
_raise_runtime_error(msg)
with dest.open("wb") as fh:
fh.write(buf.getvalue())
with state.lock:
state.progress = 50.0 + (idx + 1) / total * 50.0
with state.lock:
state.phase = "complete"
state.progress = 100.0
logger.info(
"AllDebrid download complete for ID %s at %s",
state.magnet_id,
state.target_dir,
)
except Exception:
logger.exception(
"Error in AllDebrid download for ID %s",
state.magnet_id,
)
with state.lock:
state.phase = "error"
state.error_message = str(
state.error_message or "Download failed",
)
def _fetch_file_list(
self,
magnet_id: str,
) -> list[dict[str, Any]]:
"""Retrieve and flatten the file tree for a magnet."""
url = f"{_API_BASE}/magnet/files"
resp = requests.post(
url,
headers=self._auth_headers(),
data={"id[]": magnet_id},
timeout=_API_TIMEOUT,
verify=get_ssl_verify(url),
)
resp.raise_for_status()
data = resp.json()
if data.get("status") != "success":
msg = f"Failed to list magnet files: {data.get('error')}"
raise RuntimeError(msg)
magnets = data.get("data", {}).get("magnets", [])
if not magnets:
msg = "No magnet files returned"
raise RuntimeError(msg)
files = _flatten_magnet_files(magnets[0].get("files", []))
if not files:
msg = "No files found in torrent"
raise RuntimeError(msg)
return files
+29 -182
View File
@@ -3,7 +3,6 @@
from __future__ import annotations
import errno
import math
import shutil
import time
from abc import ABC, abstractmethod
@@ -56,19 +55,6 @@ SECONDS_PER_HOUR = 3600
# How long to wait for completed files to appear (seconds)
COMPLETED_PATH_RETRY_INTERVAL = 5
COMPLETED_PATH_MAX_ATTEMPTS = 12 # 12 attempts * 5s = 60s grace period
COMPLETED_PATH_TIMEOUT_SETTING = "DOWNLOAD_CLIENT_COMPLETED_PATH_TIMEOUT"
COMPLETED_PATH_TIMEOUT_MAX_SECONDS = 3600
_RETRYABLE_COMPLETED_PATH_ERRNOS = frozenset(
code
for code in (
errno.ENOENT,
getattr(errno, "ESTALE", None),
getattr(errno, "EAGAIN", None),
getattr(errno, "EBUSY", None),
getattr(errno, "ETIMEDOUT", None),
)
if code is not None
)
@dataclass(frozen=True)
@@ -83,39 +69,6 @@ class DownloadRequest:
ratio_limit: float | None = None
@dataclass(frozen=True)
class _CompletedPathResolution:
path: Path | None
error: str | None
retryable: bool
def _coerce_completed_path_timeout_seconds(value: object, default: float) -> float:
if isinstance(value, bool) or value is None:
return default
if isinstance(value, (int, float)):
parsed = float(value)
elif isinstance(value, str):
try:
parsed = float(value.strip())
except ValueError:
return default
else:
return default
if not math.isfinite(parsed) or parsed < 0:
return default
return min(parsed, float(COMPLETED_PATH_TIMEOUT_MAX_SECONDS))
def _is_retryable_completed_path_probe(error: OSError | None) -> bool:
return error is not None and error.errno in _RETRYABLE_COMPLETED_PATH_ERRNOS
def _path_needs_mapping(path: str) -> bool:
return (len(path) >= WINDOWS_DRIVE_PREFIX_LENGTH and path[1] == ":") or "\\" in path
def _diagnose_path_issue(path: str) -> str:
"""Analyze a path and return diagnostic hints for common issues.
@@ -214,23 +167,6 @@ class ExternalClientHandler(DownloadHandler, ABC):
"""Maximum attempts when waiting for completed files."""
return COMPLETED_PATH_MAX_ATTEMPTS
def _completed_path_timeout_seconds(self) -> float:
"""Total time to wait for completed files to appear on disk."""
fallback = self._completed_path_retry_interval() * self._completed_path_max_attempts()
configured = config.get(COMPLETED_PATH_TIMEOUT_SETTING, fallback)
return _coerce_completed_path_timeout_seconds(configured, fallback)
def _refresh_download_request_after_add_failure(
self,
*,
task: DownloadTask,
request: DownloadRequest,
error: Exception,
status_callback: Callable[[str, str | None], None],
) -> DownloadRequest | None:
"""Give source handlers one chance to refresh stale resolved download data."""
return None
def _get_category_for_task(self, client: DownloadClient, task: DownloadTask) -> str | None:
"""Get audiobook category if configured and applicable, else None for default."""
if not is_audiobook(task.content_type):
@@ -284,47 +220,17 @@ class ExternalClientHandler(DownloadHandler, ABC):
)
elif protocol == "torrent":
torrent_action = config.get("PROWLARR_TORRENT_ACTION", "keep")
if torrent_action == "remove":
try:
client.remove(download_id, delete_files=False)
except _CLIENT_CLEANUP_ERRORS as e:
logger.warning(
"Failed to remove torrent %s from %s: %s",
download_id,
getattr(client, "name", "client"),
e,
)
if config.get("PROWLARR_TORRENT_ACTION", "keep") != "remove":
return
if torrent_action != "change_category":
return
post_import_category = normalize_optional_text(
config.get("PROWLARR_TORRENT_POST_IMPORT_CATEGORY", "")
)
if post_import_category is None:
return
try:
category_updated = client.set_category(download_id, post_import_category)
client.remove(download_id, delete_files=False)
except _CLIENT_CLEANUP_ERRORS as e:
logger.warning(
"Failed to set post-import category for torrent %s in %s: %s",
"Failed to remove torrent %s from %s: %s",
download_id,
getattr(client, "name", "client"),
e,
)
return
if not category_updated:
# Clients that cannot label torrents (debrid services) return False here,
# and the ones that can already log the specific failure themselves.
logger.debug(
"Post-import category not applied to torrent %s in %s",
download_id,
getattr(client, "name", "client"),
)
def _remove_usenet_download(
self,
@@ -481,21 +387,6 @@ class ExternalClientHandler(DownloadHandler, ABC):
log_details: bool,
) -> tuple[Path | None, str | None]:
"""Resolve and validate the completed download path once."""
result = self._resolve_download_path_once_detailed(
client,
download_id,
log_details=log_details,
)
return result.path, result.error
def _resolve_download_path_once_detailed(
self,
client: DownloadClient,
download_id: str,
*,
log_details: bool,
) -> _CompletedPathResolution:
"""Resolve and validate a completed path, including retryability."""
try:
raw_path = client.get_download_path(download_id)
except Exception as e:
@@ -511,7 +402,7 @@ class ExternalClientHandler(DownloadHandler, ABC):
logger.debug(
"Failed to resolve download path for %s %s: %s", client.name, download_id, e
)
return _CompletedPathResolution(None, message, retryable=False)
return None, message
if not raw_path:
message = (
@@ -526,7 +417,7 @@ class ExternalClientHandler(DownloadHandler, ABC):
logger.debug(
"Download client returned empty path for %s %s", client.name, download_id
)
return _CompletedPathResolution(None, message, retryable=False)
return None, message
from shelfmark.core.path_mappings import (
get_client_host_identifier,
@@ -566,7 +457,7 @@ class ExternalClientHandler(DownloadHandler, ABC):
logger.error(failure_log, *failure_args)
else:
logger.debug(failure_log, *failure_args)
return _CompletedPathResolution(None, message, retryable=False)
return None, message
remapped_exists, remapped_error = _probe_completed_path(remapped)
@@ -589,7 +480,7 @@ class ExternalClientHandler(DownloadHandler, ABC):
source_path_obj,
remapped,
)
return _CompletedPathResolution(remapped, None, retryable=False)
return remapped, None
message = (
f"Remapped path '{remapped}' does not exist. "
@@ -608,11 +499,7 @@ class ExternalClientHandler(DownloadHandler, ABC):
logger.error(failure_log, *failure_args)
else:
logger.debug(failure_log, *failure_args)
return _CompletedPathResolution(
None,
message,
retryable=_is_retryable_completed_path_probe(remapped_error),
)
return None, message
source_exists, source_error = _probe_completed_path(source_path_obj)
@@ -635,7 +522,7 @@ class ExternalClientHandler(DownloadHandler, ABC):
download_id,
source_path_obj,
)
return _CompletedPathResolution(source_path_obj, None, retryable=False)
return source_path_obj, None
hint = _diagnose_path_issue(raw_path)
if mappings:
@@ -668,12 +555,7 @@ class ExternalClientHandler(DownloadHandler, ABC):
logger.error(failure_log, *failure_args)
else:
logger.debug(failure_log, *failure_args)
return _CompletedPathResolution(
None,
message,
retryable=not _path_needs_mapping(raw_path)
and _is_retryable_completed_path_probe(source_error),
)
return None, message
def _wait_for_completed_path(
self,
@@ -685,37 +567,23 @@ class ExternalClientHandler(DownloadHandler, ABC):
) -> tuple[Path | None, str | None]:
"""Wait briefly for completed files to appear on disk."""
last_error: str | None = None
max_attempts = self._completed_path_max_attempts()
retry_interval = self._completed_path_retry_interval()
timeout_seconds = self._completed_path_timeout_seconds()
if retry_interval <= 0 or timeout_seconds <= 0:
max_attempts = 1
else:
max_attempts = int(math.ceil(timeout_seconds / retry_interval)) + 1
for attempt in range(1, max_attempts + 1):
if cancel_flag and cancel_flag.is_set():
return None, last_error
log_details = attempt == max_attempts
result = self._resolve_download_path_once_detailed(
resolved_path, error = self._resolve_download_path_once(
client,
download_id,
log_details=log_details,
)
if result.path:
return result.path, None
if resolved_path:
return resolved_path, None
last_error = result.error
if not result.retryable:
if not log_details:
logger.error(
"Completed path resolution is not retryable for %s (%s): %s",
client.name,
download_id,
last_error,
)
return None, last_error
last_error = error
if attempt < max_attempts:
status_callback("locating", "Waiting for completed files...")
@@ -832,41 +700,20 @@ class ExternalClientHandler(DownloadHandler, ABC):
status_callback("downloading", "Resuming existing download")
else:
# No existing download - add new
refresh_attempted = False
while True:
status_callback("resolving", f"Sending to {client.name}")
try:
download_id = client.add_download(
url=request.url,
name=request.release_name,
category=category,
expected_hash=request.expected_hash,
seeding_time_limit=request.seeding_time_limit,
ratio_limit=request.ratio_limit,
# rTorrent has no category concept, so its audiobook label
# can only be chosen from the content type (#1235).
content_type=task.content_type,
)
except Exception as e:
if not refresh_attempted:
refresh_attempted = True
refreshed_request = self._refresh_download_request_after_add_failure(
task=task,
request=request,
error=e,
status_callback=status_callback,
)
if (
refreshed_request is not None
and refreshed_request.protocol == request.protocol
):
request = refreshed_request
continue
logger.exception("Failed to add to %s", client.name)
status_callback("error", f"Failed to add to {client.name}: {e}")
return None
break
status_callback("resolving", f"Sending to {client.name}")
try:
download_id = client.add_download(
url=request.url,
name=request.release_name,
category=category,
expected_hash=request.expected_hash,
seeding_time_limit=request.seeding_time_limit,
ratio_limit=request.ratio_limit,
)
except Exception as e:
logger.exception("Failed to add to %s", client.name)
status_callback("error", f"Failed to add to {client.name}: {e}")
return None
logger.info(
"Added to %s: %s for '%s'", client.name, download_id, request.release_name
+3 -18
View File
@@ -227,10 +227,10 @@ class DelugeClient(DownloadClient):
return self._rpc_call("daemon.info")
def _try_set_label(self, torrent_id: str, label: str) -> bool:
def _try_set_label(self, torrent_id: str, label: str) -> None:
"""Best-effort label assignment (requires Deluge Label plugin)."""
if not label:
return False
return
try:
# label.add will error if the plugin is unavailable or the label exists.
@@ -240,9 +240,6 @@ class DelugeClient(DownloadClient):
self._rpc_call("label.set_torrent", torrent_id, label)
except _DELUGE_CLIENT_ERRORS as e:
logger.debug("Could not set Deluge label '%s' for %s: %s", label, torrent_id, e)
return False
else:
return True
@staticmethod
def is_configured() -> bool:
@@ -280,10 +277,7 @@ class DelugeClient(DownloadClient):
torrent_info = extract_torrent_info(url, expected_hash=expected_hash)
if not torrent_info.is_magnet and not torrent_info.torrent_data:
message = "Failed to fetch torrent file"
if torrent_info.fetch_error:
message = f"{message}: {torrent_info.fetch_error}"
_raise_runtime_error(message)
_raise_runtime_error("Failed to fetch torrent file")
options: dict[str, Any] = {}
if self._download_dir:
@@ -425,15 +419,6 @@ class DelugeClient(DownloadClient):
else:
return False
def set_category(self, download_id: str, category: str) -> bool:
"""Assign a label to a torrent using Deluge's Label plugin."""
try:
self._ensure_connected()
return self._try_set_label(download_id, category)
except _DELUGE_CLIENT_ERRORS as e:
self._log_error("set_category", e)
return False
def get_download_path(self, download_id: str) -> str | None:
"""Return the resolved download path for a Deluge torrent."""
try:
+182 -279
View File
@@ -44,11 +44,6 @@ _HASH_LENGTH_40 = 40
_HASH_LENGTH_ED2K = 32
_HTTP_STATUS_FORBIDDEN = HTTPStatus.FORBIDDEN
_HTTP_STATUS_NOT_FOUND = HTTPStatus.NOT_FOUND
_METADATA_DOWNLOAD_STATES = {"forcedMetaDL", "metaDL"}
# How long add_download waits for magnet metadata before falling back to the info
# hash it already knows, rather than holding the download queue on a thin swarm.
_METADATA_WAIT_POLLS = 20
_METADATA_WAIT_INTERVAL_SECONDS = 0.5
_ONE_WEEK_IN_SECONDS = 604800
@@ -99,25 +94,6 @@ def _hashes_match(hash1: str, hash2: str) -> bool:
return False
def _torrent_matches_download_id(torrent: object, download_id: str) -> bool:
"""Match an ID against every identity qBittorrent exposes.
For hybrid torrents, qBittorrent's primary `hash` can change from the v1
hash to the truncated v2 hash after metadata resolution. The full
`infohash_v1` and `infohash_v2` fields preserve the torrent's identities.
"""
identifiers = (
getattr(torrent, "hash", None),
getattr(torrent, "infohash_v1", None),
getattr(torrent, "infohash_v2", None),
)
return any(
isinstance(identifier, str) and identifier and _hashes_match(identifier, download_id)
for identifier in identifiers
)
def _raise_runtime_error(message: str) -> NoReturn:
raise RuntimeError(message)
@@ -190,6 +166,63 @@ def _build_qbittorrent_child_path(base_path: object, child_path: object) -> str
class QBittorrentClient(DownloadClient):
"""qBittorrent download client."""
def _is_torrent_loaded(self, torrent_hash: str) -> tuple[bool, str | None]:
"""Check whether qBittorrent has registered a torrent yet.
Uses `/api/v2/torrents/properties?hash=<hash>`.
Returns:
(loaded, error_message)
Notes:
A false result with no error means "not loaded yet".
"""
url = f"{self._base_url}/api/v2/torrents/properties"
params = {"hash": torrent_hash}
try:
self._client.auth_log_in()
response = self._client._session.get(url, params=params, timeout=10)
# Re-authenticate and retry once on 403
if response.status_code == _HTTP_STATUS_FORBIDDEN:
logger.debug(
"qBittorrent returned 403 for properties; re-authenticating and retrying"
)
self._client.auth_log_in()
response = self._client._session.get(url, params=params, timeout=10)
if response.status_code == _HTTP_STATUS_FORBIDDEN:
return False, "qBittorrent authentication failed (HTTP 403)"
# qBittorrent returns 404/409-ish responses depending on version when missing.
if response.status_code == _HTTP_STATUS_NOT_FOUND:
return False, None
response.raise_for_status()
except requests.exceptions.HTTPError as e:
status = getattr(getattr(e, "response", None), "status_code", None)
if status == _HTTP_STATUS_NOT_FOUND:
return False, None
if status:
return False, f"qBittorrent API request failed (HTTP {status})"
return False, "qBittorrent API request failed"
except requests.exceptions.ConnectionError:
return False, f"Cannot connect to qBittorrent at {self._base_url}"
except requests.exceptions.Timeout:
return False, f"qBittorrent request timed out at {self._base_url}"
except requests.exceptions.InvalidSchema:
return (
False,
"qBittorrent URL is invalid (missing http:// or https://). "
f"Configured: {self._base_url}",
)
except _QBITTORRENT_CLIENT_ERRORS as e:
return False, f"qBittorrent API error: {type(e).__name__}: {e}"
else:
return True, None
protocol = "torrent"
name = "qbittorrent"
@@ -211,7 +244,6 @@ class QBittorrentClient(DownloadClient):
username = config_text(config.get("QBITTORRENT_USERNAME", ""))
password = config_text(config.get("QBITTORRENT_PASSWORD", ""))
self._api_key = config_text(config.get("QBITTORRENT_API_KEY", ""))
# qbittorrent-api accepts either a full URL or host:port; prefer the normalized URL
# for consistency.
@@ -219,43 +251,43 @@ class QBittorrentClient(DownloadClient):
host=self._base_url,
username=username,
password=password,
api_key=self._api_key or None,
VERIFY_WEBUI_CERTIFICATE=get_ssl_verify(self._base_url),
)
self._category = config_text(config.get("QBITTORRENT_CATEGORY", "books"))
self._download_dir = config_text(config.get("QBITTORRENT_DOWNLOAD_DIR", ""))
self._tags = _normalize_tags(config.get("QBITTORRENT_TAG", []))
# download_id -> qBittorrent's current primary hash, for identities that no
# longer match it directly. See _resolve_torrent().
self._primary_hashes: dict[str, str] = {}
@property
def _can_reauthenticate(self) -> bool:
"""Whether a 403 is worth retrying; a bearer token cannot be refreshed like a session."""
return not self._api_key
def _ensure_authenticated(self) -> None:
"""Authenticate the underlying HTTP session before it is used directly.
API keys (qBittorrent 5.2.0+) are sent as a bearer header on every request and
have no login endpoint, so there is no session to establish up front.
"""
if self._api_key:
return
self._client.auth_log_in()
def _request_torrent_info_records(
self, params: dict[str, str]
def _get_torrents_info(
self, torrent_hash: str | None = None
) -> tuple[list[SimpleNamespace], str | None]:
"""Request torrent info records from qBittorrent."""
"""Get torrent info using GET.
Behaviors:
- Retry once on HTTP 403 by re-authenticating.
- Keep "API/auth/connect" errors distinct from "torrent missing".
- If a hash-specific query returns empty, fall back to listing by category
and matching locally.
Returns:
(torrents, error_message)
"""
url = f"{self._base_url}/api/v2/torrents/info"
try:
self._ensure_authenticated()
response = self._client._session.get(url, params=params, timeout=10)
if response.status_code == _HTTP_STATUS_FORBIDDEN and self._can_reauthenticate:
def do_request(params: dict[str, str]) -> requests.Response:
# Ensure session is authenticated before using it directly
self._client.auth_log_in()
return self._client._session.get(url, params=params, timeout=10)
def parse_response(
response: requests.Response,
*,
request_params: dict[str, str],
) -> tuple[list[SimpleNamespace], str | None]:
if response.status_code == _HTTP_STATUS_FORBIDDEN:
logger.debug("qBittorrent returned 403; re-authenticating and retrying")
self._ensure_authenticated()
response = self._client._session.get(url, params=params, timeout=10)
self._client.auth_log_in()
response = self._client._session.get(url, params=request_params, timeout=10)
if response.status_code == _HTTP_STATUS_FORBIDDEN:
logger.warning("qBittorrent authentication failed (HTTP 403)")
@@ -264,6 +296,41 @@ class QBittorrentClient(DownloadClient):
response.raise_for_status()
torrents = response.json()
return [SimpleNamespace(**t) for t in torrents], None
try:
primary_params: dict[str, str] = {}
if torrent_hash:
primary_params["hashes"] = torrent_hash
response = do_request(primary_params)
torrents, error = parse_response(response, request_params=primary_params)
if error:
return [], error
if torrent_hash and not torrents:
# Fallback 1: list by configured category
category_params: dict[str, str] = {}
if self._category:
category_params["category"] = self._category
category_response = do_request(category_params)
category_torrents, category_error = parse_response(
category_response, request_params=category_params
)
if category_error:
return [], category_error
if category_torrents:
return category_torrents, None
# Fallback 2: list everything (handles per-task categories like audiobooks)
all_response = do_request({})
all_torrents, all_error = parse_response(all_response, request_params={})
if all_error:
return [], all_error
return all_torrents, None
except requests.exceptions.HTTPError as e:
status = getattr(getattr(e, "response", None), "status_code", None)
if status:
@@ -288,141 +355,8 @@ class QBittorrentClient(DownloadClient):
except _QBITTORRENT_CLIENT_ERRORS as e:
logger.debug("Failed to get torrents info: %s", e)
return [], f"qBittorrent API error: {type(e).__name__}: {e}"
def _get_torrent_info(self, download_id: str) -> tuple[SimpleNamespace | None, str | None]:
"""Get one torrent by its current qBittorrent hash."""
torrents, error = self._request_torrent_info_records({"hashes": download_id})
if error or not torrents:
return None, error
return (
next(
(
torrent
for torrent in torrents
if isinstance(getattr(torrent, "hash", None), str)
and _hashes_match(torrent.hash, download_id)
),
None,
),
None,
)
def _list_torrents_by_category(
self, category: str | None
) -> tuple[list[SimpleNamespace], str | None]:
"""List torrent records in a category, or all records when unset."""
params = {"category": category} if category else {}
return self._request_torrent_info_records(params)
def _remember_primary_hash(self, download_id: str, torrent: SimpleNamespace) -> None:
"""Note the primary hash a listing scan found, so later lookups skip the scan."""
torrent_hash = getattr(torrent, "hash", None)
if isinstance(torrent_hash, str) and torrent_hash:
self._primary_hashes[download_id.lower()] = torrent_hash.lower()
def _resolve_torrent(
self, download_id: str, category: str | None = None
) -> tuple[SimpleNamespace | None, str | None]:
"""Resolve any known torrent identity to its current qBittorrent record.
A hybrid torrent's primary hash switches from the v1 hash to the truncated v2
hash once metadata resolves, so a download tracked by its v1 hash misses the
`hashes=` lookup and falls through to a full listing. Since `get_status()`
polls every couple of seconds for the life of the download, remember the
primary hash a scan finds and try it first.
"""
cached = self._primary_hashes.get(download_id.lower())
for candidate in (item for item in dict.fromkeys((cached, download_id)) if item):
torrent, error = self._get_torrent_info(candidate)
if error:
return None, error
if torrent:
self._remember_primary_hash(download_id, torrent)
return torrent, None
categories = [candidate for candidate in (category, self._category) if candidate]
for candidate in dict.fromkeys(categories):
torrents, error = self._list_torrents_by_category(candidate)
if error:
return None, error
torrent = next(
(item for item in torrents if _torrent_matches_download_id(item, download_id)),
None,
)
if torrent:
self._remember_primary_hash(download_id, torrent)
return torrent, None
torrents, error = self._list_torrents_by_category(None)
if error:
return None, error
torrent = next(
(item for item in torrents if _torrent_matches_download_id(item, download_id)),
None,
)
if torrent:
self._remember_primary_hash(download_id, torrent)
else:
# The torrent is gone; drop the note so a re-add is not looked up by a
# hash that no longer exists.
self._primary_hashes.pop(download_id.lower(), None)
return torrent, None
def _current_hash(self, download_id: str) -> str:
"""qBittorrent's current primary hash for any identity we know the torrent by.
Falls back to the given ID when the torrent cannot be found, so callers
still address the hash they were handed and surface the client's error.
"""
try:
torrent, error = self._resolve_torrent(download_id)
except _QBITTORRENT_CLIENT_ERRORS as e:
logger.debug("Could not resolve current hash for %s: %s", download_id, e)
return download_id
if error or not torrent:
return download_id
torrent_hash = getattr(torrent, "hash", None)
if isinstance(torrent_hash, str) and torrent_hash:
return torrent_hash
return download_id
def _list_category_hashes(self, category: str | None) -> set[str] | None:
"""Snapshot the hashes qBittorrent currently reports for a category."""
torrents, error = self._list_torrents_by_category(category)
if error:
logger.debug("Could not snapshot qBittorrent torrents: %s", error)
return None
return {str(torrent.hash).lower() for torrent in torrents if getattr(torrent, "hash", None)}
def _discover_added_torrent_hash(
self,
name: str,
category: str | None,
known_hashes: set[str] | None,
) -> str | None:
"""Recover the hash of a torrent that was added without a known info_hash.
A `known_hashes` of None means the pre-add snapshot failed, so only a
torrent matching the requested rename can identify the new arrival.
"""
for _ in range(20):
torrents, error = self._list_torrents_by_category(category)
if error:
logger.debug("qBittorrent hash discovery: %s", error)
else:
new_torrents = [
torrent
for torrent in torrents
if getattr(torrent, "hash", None)
and (known_hashes is None or str(torrent.hash).lower() not in known_hashes)
]
for torrent in new_torrents:
if getattr(torrent, "name", None) == name:
return str(torrent.hash).lower()
if known_hashes is not None and len(new_torrents) == 1:
return str(new_torrents[0].hash).lower()
time.sleep(0.5)
return None
return torrents, None
@staticmethod
def is_configured() -> bool:
@@ -434,7 +368,7 @@ class QBittorrentClient(DownloadClient):
def test_connection(self) -> tuple[bool, str]:
"""Test connection to qBittorrent."""
try:
self._ensure_authenticated()
self._client.auth_log_in()
api_version = self._client.app.web_api_version
except _QBITTORRENT_CLIENT_ERRORS as e:
return False, f"Connection failed: {e!s}"
@@ -491,10 +425,6 @@ class QBittorrentClient(DownloadClient):
expected_hash = torrent_info.info_hash
torrent_data = torrent_info.torrent_data
known_hashes: set[str] | None = None
if not expected_hash:
known_hashes = self._list_category_hashes(category)
# Per-torrent seeding limits from indexer
seeding_time_limit_value = kwargs.get("seeding_time_limit")
seeding_time_limit = coerce_optional_int(seeding_time_limit_value)
@@ -529,47 +459,33 @@ class QBittorrentClient(DownloadClient):
result_text = _normalize_add_result(result)
logger.debug("qBittorrent add result: %s", result_text)
if not expected_hash:
_raise_runtime_error("Could not determine torrent hash from URL")
if _is_explicit_add_failure(result):
_raise_runtime_error(f"Failed to add torrent: {result_text}")
if not expected_hash:
# qBittorrent fetches .torrent URLs itself, so the add can succeed
# even when no hash could be extracted up front. Recover it by
# watching for the new torrent to appear.
expected_hash = self._discover_added_torrent_hash(name, category, known_hashes)
if not expected_hash:
message = "Could not determine torrent hash from URL"
if torrent_info.fetch_error:
message = f"{message} (torrent file fetch failed: {torrent_info.fetch_error})"
_raise_runtime_error(message)
# Prefer qBittorrent's primary torrent ID, which for hybrid torrents
# switches from the v1 hash to the truncated v2 hash once metadata
# resolves. A magnet with few peers can take minutes to fetch metadata,
# and the torrent is worth keeping in the meantime: every lookup goes
# through `_resolve_torrent`, which still matches the v1 hash against
# `infohash_v1` after the primary ID has changed.
for _ in range(_METADATA_WAIT_POLLS):
torrent, error = self._resolve_torrent(expected_hash, category)
# Some qBittorrent-compatible clients return HTTP 200 with an empty body
# instead of qBittorrent's literal "Ok." response. Prefer verifying that
# the torrent becomes visible over trusting the response body alone.
for _ in range(10):
loaded, error = self._is_torrent_loaded(expected_hash)
if error:
logger.debug("qBittorrent add_download: %s", error)
elif torrent and getattr(torrent, "state", None) not in _METADATA_DOWNLOAD_STATES:
torrent_hash = getattr(torrent, "hash", None)
if isinstance(torrent_hash, str) and torrent_hash:
logger.info("Added torrent: %s", torrent_hash)
return torrent_hash.lower()
time.sleep(_METADATA_WAIT_INTERVAL_SECONDS)
if loaded:
logger.info("Added torrent: %s", expected_hash)
return expected_hash.lower()
time.sleep(0.5)
logger.info(
"Added torrent %s; metadata still pending after %.0fs, tracking it by info hash",
expected_hash,
_METADATA_WAIT_POLLS * _METADATA_WAIT_INTERVAL_SECONDS,
logger.warning(
"Torrent add was not confirmed within the visibility grace period (response=%s), returning expected hash",
result_text,
)
except _QBITTORRENT_CLIENT_ERRORS:
logger.exception("qBittorrent add failed")
raise
else:
return expected_hash.lower()
return expected_hash
def get_status(self, download_id: str) -> DownloadStatus:
"""Get torrent status by hash.
@@ -582,9 +498,19 @@ class QBittorrentClient(DownloadClient):
"""
try:
torrent, error = self._resolve_torrent(download_id)
torrents, error = self._get_torrents_info(download_id)
if error:
return DownloadStatus.error(error)
torrent = next(
(
t
for t in torrents
if isinstance(getattr(t, "hash", None), str)
and _hashes_match(t.hash, download_id)
),
None,
)
if not torrent:
return DownloadStatus.error("Torrent not found in qBittorrent")
@@ -666,9 +592,7 @@ class QBittorrentClient(DownloadClient):
"""
try:
torrent_hash = self._current_hash(download_id)
self._client.torrents_delete(torrent_hashes=torrent_hash, delete_files=delete_files)
self._primary_hashes.pop(download_id.lower(), None)
self._client.torrents_delete(torrent_hashes=download_id, delete_files=delete_files)
logger.info(
"Removed torrent from qBittorrent: %s%s",
download_id,
@@ -680,26 +604,6 @@ class QBittorrentClient(DownloadClient):
else:
return True
def set_category(self, download_id: str, category: str) -> bool:
"""Assign a category to a torrent in qBittorrent."""
try:
try:
self._client.torrents_create_category(name=category)
except _QBITTORRENT_CLIENT_ERRORS as e:
if "Conflict" not in type(e).__name__ and "409" not in str(e):
logger.debug("Could not create category '%s': %s", category, e)
self._client.torrents_set_category(
torrent_hashes=self._current_hash(download_id),
category=category,
)
logger.info("Set qBittorrent category for %s to '%s'", download_id, category)
except _QBITTORRENT_CLIENT_ERRORS as e:
self._log_error("set_category", e)
return False
else:
return True
def get_download_path(self, download_id: str) -> str | None:
"""Get the path where torrent files are located.
@@ -712,10 +616,20 @@ class QBittorrentClient(DownloadClient):
- join `save_path` with the torrent's top-level directory
"""
try:
torrent, error = self._resolve_torrent(download_id)
torrents, error = self._get_torrents_info(download_id)
if error:
logger.debug("qBittorrent get_download_path: %s", error)
return None
torrent = next(
(
t
for t in torrents
if isinstance(getattr(t, "hash", None), str)
and _hashes_match(t.hash, download_id)
),
None,
)
if not torrent:
return None
@@ -761,11 +675,11 @@ class QBittorrentClient(DownloadClient):
import os
def get_with_auth(url: str, params: dict[str, str]) -> requests.Response:
self._ensure_authenticated()
self._client.auth_log_in()
resp = self._client._session.get(url, params=params, timeout=10)
if resp.status_code == _HTTP_STATUS_FORBIDDEN and self._can_reauthenticate:
if resp.status_code == _HTTP_STATUS_FORBIDDEN:
logger.debug("qBittorrent returned 403; re-authenticating and retrying")
self._ensure_authenticated()
self._client.auth_log_in()
resp = self._client._session.get(url, params=params, timeout=10)
return resp
@@ -813,33 +727,6 @@ class QBittorrentClient(DownloadClient):
)
return None
def _await_existing_torrent(
self, info_hash: str, category: str | None
) -> tuple[str, DownloadStatus] | None:
"""Report a torrent already in qBittorrent, waiting out magnet metadata first."""
for _ in range(_METADATA_WAIT_POLLS):
torrent, error = self._resolve_torrent(info_hash, category)
if error:
logger.debug("qBittorrent find_existing: %s", error)
return None
if not torrent:
return None
if getattr(torrent, "state", None) not in _METADATA_DOWNLOAD_STATES:
torrent_hash = getattr(torrent, "hash", None)
if isinstance(torrent_hash, str) and torrent_hash:
torrent_hash = torrent_hash.lower()
return (torrent_hash, self.get_status(torrent_hash))
time.sleep(_METADATA_WAIT_INTERVAL_SECONDS)
# Metadata is still pending, but the torrent is here and `add_download` keeps
# one in this state rather than giving up. Report it by info hash so the
# caller joins the download in progress instead of adding a duplicate.
logger.info(
"Existing torrent %s is still fetching metadata; joining it by info hash",
info_hash,
)
return (info_hash.lower(), self.get_status(info_hash))
def find_existing(
self, url: str, category: str | None = None
) -> tuple[str, DownloadStatus] | None:
@@ -849,9 +736,25 @@ class QBittorrentClient(DownloadClient):
if not torrent_info.info_hash:
return None
existing = self._await_existing_torrent(torrent_info.info_hash, category)
torrents, error = self._get_torrents_info(torrent_info.info_hash)
if error:
logger.debug("qBittorrent find_existing: %s", error)
return None
torrent = next(
(
t
for t in torrents
if isinstance(getattr(t, "hash", None), str)
and _hashes_match(t.hash, torrent_info.info_hash)
),
None,
)
if torrent and isinstance(getattr(torrent, "hash", None), str):
torrent_hash = torrent.hash
return (torrent_hash.lower(), self.get_status(torrent_hash.lower()))
except _QBITTORRENT_CLIENT_ERRORS as e:
logger.debug("Error checking for existing torrent: %s", e)
return None
else:
return existing
return None
-545
View File
@@ -1,545 +0,0 @@
"""Real-Debrid debrid service client for Shelfmark.
Routes magnet links through the Real-Debrid REST API (v1.0) to download
torrent content via Real-Debrid's CDN infrastructure.
"""
from __future__ import annotations
import shutil
import threading
from dataclasses import dataclass, field
from pathlib import Path
from typing import Any, ClassVar, NoReturn
import requests
from shelfmark.config.env import TMP_DIR
from shelfmark.core.config import config
from shelfmark.core.logger import setup_logger
from shelfmark.download.clients import (
DownloadClient,
DownloadState,
DownloadStatus,
register_client,
)
from shelfmark.download.clients._coercion import config_text
from shelfmark.download.clients.torrent_utils import (
DebridMagnet,
DebridUpload,
resolve_debrid_upload,
)
from shelfmark.download.http import download_url
from shelfmark.download.network import get_ssl_verify
logger = setup_logger(__name__)
_API_BASE = "https://api.real-debrid.com/rest/1.0"
_REALDEBRID_CLIENT_ERRORS = (
AttributeError,
OSError,
requests.exceptions.RequestException,
RuntimeError,
TypeError,
ValueError,
)
# Real-Debrid torrent status values.
_STATUS_DOWNLOADING = frozenset(
{
"magnet_conversion",
"waiting_files_selection",
"downloading",
"compressing",
"uploading",
}
)
_STATUS_READY = "downloaded"
_STATUS_ERROR = frozenset({"error", "virus", "dead"})
# Timeouts for API calls.
_API_TIMEOUT = 30
_STATUS_TIMEOUT = 15
# File extensions recognised as book or audiobook content.
_BOOK_EXTENSIONS = (
".aac",
".azw",
".azw3",
".cbr",
".cbz",
".djvu",
".doc",
".docx",
".epub",
".fb2",
".flac",
".lit",
".m4a",
".m4b",
".mobi",
".mp3",
".mp4",
".ogg",
".opus",
".pdf",
".rtf",
".txt",
".wma",
)
def _raise_runtime_error(message: str) -> NoReturn:
raise RuntimeError(message)
@dataclass
class _DownloadState:
"""Internal mutable state for an in-progress Real-Debrid download."""
torrent_id: str
name: str
target_dir: Path
phase: str = "uploading"
error_message: str | None = None
progress: float = 0.0
download_thread: threading.Thread | None = None
lock: threading.Lock = field(default_factory=threading.Lock)
@register_client("torrent")
class RealDebridClient(DownloadClient):
"""Real-Debrid debrid service client.
Downloads torrent content by uploading magnet links to Real-Debrid,
selecting all files for download on their servers, then unrestricting
and fetching the resulting files via direct HTTP download from
Real-Debrid's CDN.
API documentation: https://api.real-debrid.com/
"""
protocol = "torrent"
name = "realdebrid"
_downloads: ClassVar[dict[str, _DownloadState]] = {}
_downloads_lock = threading.Lock()
def __init__(self) -> None:
self._api_key = config_text(config.get("REALDEBRID_API_KEY", ""))
def _auth_headers(self) -> dict[str, str]:
"""Return Authorization header dict for API requests."""
return {"Authorization": f"Bearer {self._api_key}"}
# ------------------------------------------------------------------
# DownloadClient interface
# ------------------------------------------------------------------
@staticmethod
def is_configured() -> bool:
"""Return True when Real-Debrid is selected and an API key exists."""
client = config_text(config.get("PROWLARR_TORRENT_CLIENT", ""))
api_key = config_text(config.get("REALDEBRID_API_KEY", ""))
return client == "realdebrid" and bool(api_key)
def test_connection(self) -> tuple[bool, str]:
"""Validate the API key and check Premium subscription status."""
if not self._api_key:
return False, "Real-Debrid API Key is required"
try:
url = f"{_API_BASE}/user"
resp = requests.get(
url,
headers=self._auth_headers(),
timeout=_STATUS_TIMEOUT,
verify=get_ssl_verify(url),
)
resp.raise_for_status()
user = resp.json()
username = user.get("username", "Unknown")
account_type = user.get("type", "free")
if account_type != "premium":
return (
False,
f"Real-Debrid user '{username}' does not have "
f"a Premium subscription (type: {account_type})",
)
except _REALDEBRID_CLIENT_ERRORS as e:
return False, f"Connection failed: {e}"
else:
return True, f"Connected to Real-Debrid as '{username}' (Premium)"
def add_download(
self,
url: str,
name: str,
category: str | None = None,
expected_hash: str | None = None,
**kwargs: object,
) -> str:
"""Send a torrent to Real-Debrid and select all files.
Accepts a magnet link, a .torrent URL, or an indexer proxy URL; anything
that is not already a magnet is resolved first, because Real-Debrid
answers a non-magnet body on addMagnet with a bare 404 (#1250).
"""
if not self._api_key:
msg = "Real-Debrid API key is not configured"
raise RuntimeError(msg)
try:
upload = resolve_debrid_upload(url, expected_hash=expected_hash)
data = self._send_torrent(upload)
torrent_id = str(data.get("id", ""))
if not torrent_id:
msg = "No torrent ID returned from Real-Debrid"
_raise_runtime_error(msg)
# Select all files so Real-Debrid starts downloading the torrent
select_url = f"{_API_BASE}/torrents/selectFiles/{torrent_id}"
sel_resp = requests.post(
select_url,
headers=self._auth_headers(),
data={"files": "all"},
timeout=_API_TIMEOUT,
verify=get_ssl_verify(select_url),
)
sel_resp.raise_for_status()
target_dir = TMP_DIR / f"realdebrid_{torrent_id}"
target_dir.mkdir(parents=True, exist_ok=True)
state = _DownloadState(
torrent_id=torrent_id,
name=name,
target_dir=target_dir,
phase="waiting_rd",
)
with self._downloads_lock:
self._downloads[torrent_id] = state
logger.info(
"Added torrent to Real-Debrid: ID %s (%s)",
torrent_id,
name,
)
except Exception:
logger.exception("Failed to add torrent to Real-Debrid")
raise
else:
return torrent_id
def _send_torrent(self, upload: DebridUpload) -> dict[str, Any]:
"""Hand the torrent to Real-Debrid, as a magnet or as a file upload."""
if isinstance(upload, DebridMagnet):
add_url = f"{_API_BASE}/torrents/addMagnet"
resp = requests.post(
add_url,
headers=self._auth_headers(),
data={"magnet": upload.magnet_url},
timeout=_API_TIMEOUT,
verify=get_ssl_verify(add_url),
)
else:
# addTorrent is a PUT that takes the raw file as the request body,
# not a form field: https://api.real-debrid.com/
add_url = f"{_API_BASE}/torrents/addTorrent"
resp = requests.put(
add_url,
headers={
**self._auth_headers(),
"Content-Type": "application/x-bittorrent",
},
data=upload.torrent_data,
timeout=_API_TIMEOUT,
verify=get_ssl_verify(add_url),
)
resp.raise_for_status()
return resp.json()
def get_status(self, download_id: str) -> DownloadStatus:
"""Poll Real-Debrid for torrent status and drive the download."""
state = self._ensure_state(download_id)
# Return cached terminal / in-flight states immediately.
with state.lock:
if state.phase == "error":
return DownloadStatus.error(
state.error_message or "Real-Debrid error",
)
if state.phase == "complete":
return DownloadStatus(
progress=100.0,
state=DownloadState.COMPLETE,
message="Complete",
complete=True,
file_path=str(state.target_dir),
)
if state.phase == "downloading_http":
return DownloadStatus(
progress=state.progress,
state=DownloadState.DOWNLOADING,
message="Downloading files via HTTP...",
complete=False,
file_path=None,
)
# Query Real-Debrid for torrent info.
try:
info_url = f"{_API_BASE}/torrents/info/{download_id}"
resp = requests.get(
info_url,
headers=self._auth_headers(),
timeout=_STATUS_TIMEOUT,
verify=get_ssl_verify(info_url),
)
resp.raise_for_status()
info = resp.json()
return self._handle_torrent_info(info, state)
except Exception as e:
logger.exception(
"Error checking Real-Debrid status for %s",
download_id,
)
return DownloadStatus.error(str(e))
def remove(
self,
download_id: str,
*,
delete_files: bool = False,
) -> bool:
"""Delete the torrent from Real-Debrid and clean up local files."""
try:
url = f"{_API_BASE}/torrents/delete/{download_id}"
requests.delete(
url,
headers=self._auth_headers(),
timeout=_STATUS_TIMEOUT,
verify=get_ssl_verify(url),
)
except _REALDEBRID_CLIENT_ERRORS as e:
logger.warning("Failed to delete torrent from Real-Debrid: %s", e)
with self._downloads_lock:
state = self._downloads.pop(download_id, None)
if state and state.target_dir.exists():
shutil.rmtree(state.target_dir, ignore_errors=True)
return True
def get_download_path(self, download_id: str) -> str | None:
"""Return the local directory containing downloaded files."""
with self._downloads_lock:
state = self._downloads.get(download_id)
if state and state.phase == "complete":
return str(state.target_dir)
target_dir = TMP_DIR / f"realdebrid_{download_id}"
if target_dir.exists():
return str(target_dir)
return None
# ------------------------------------------------------------------
# Internal helpers
# ------------------------------------------------------------------
def _ensure_state(self, download_id: str) -> _DownloadState:
"""Get or create download state for the given torrent ID."""
with self._downloads_lock:
state = self._downloads.get(download_id)
if state:
return state
target_dir = TMP_DIR / f"realdebrid_{download_id}"
state = _DownloadState(
torrent_id=download_id,
name=f"Download {download_id}",
target_dir=target_dir,
phase="waiting_rd",
)
with self._downloads_lock:
self._downloads[download_id] = state
return state
def _handle_torrent_info(
self,
info: dict[str, Any],
state: _DownloadState,
) -> DownloadStatus:
"""Map Real-Debrid torrent info to a DownloadStatus."""
status = info.get("status", "")
if status in _STATUS_DOWNLOADING:
progress = float(info.get("progress", 0.0))
speed = int(info.get("speed", 0))
filename = info.get("filename", state.name)
return DownloadStatus(
progress=progress * 0.5,
state=DownloadState.DOWNLOADING,
message=f"Real-Debrid downloading torrent ({filename})",
complete=False,
file_path=None,
download_speed=speed,
)
if status == _STATUS_READY:
links = info.get("links", [])
files = info.get("files", [])
self._maybe_start_download_thread(state, links, files)
return DownloadStatus(
progress=50.0,
state=DownloadState.DOWNLOADING,
message="Real-Debrid ready, retrieving files...",
complete=False,
file_path=None,
)
# Terminal error from Real-Debrid.
error_txt = f"Real-Debrid status error: {status}"
with state.lock:
state.phase = "error"
state.error_message = error_txt
return DownloadStatus.error(error_txt)
def _maybe_start_download_thread(
self,
state: _DownloadState,
links: list[str],
files: list[dict[str, Any]],
) -> None:
"""Spawn a background thread to unrestrict and download files."""
with state.lock:
already_running = state.phase in (
"unrestricting",
"downloading_http",
"complete",
)
thread_alive = state.download_thread is not None and state.download_thread.is_alive()
if already_running or thread_alive:
return
state.phase = "unrestricting"
t = threading.Thread(
target=self._process_and_download,
args=(state, links, files),
daemon=True,
)
state.download_thread = t
t.start()
# ------------------------------------------------------------------
# File download pipeline
# ------------------------------------------------------------------
def _process_and_download(
self,
state: _DownloadState,
links: list[str],
files: list[dict[str, Any]],
) -> None:
"""Unrestrict links and download files via HTTP.
Runs in a background thread spawned by ``_maybe_start_download_thread``.
"""
try:
if not links:
msg = "No download links returned by Real-Debrid"
_raise_runtime_error(msg)
# Match selected files with links
selected_files = [f for f in files if f.get("selected") == 1]
# Filter relevant ebook / audiobook files
relevant_indices: list[int] = []
for i, f_info in enumerate(selected_files):
path_str = f_info.get("path", "").lower()
if path_str.endswith(_BOOK_EXTENSIONS):
relevant_indices.append(i)
if not relevant_indices:
relevant_indices = list(range(len(links)))
with state.lock:
state.phase = "downloading_http"
total = len(relevant_indices)
for idx, rel_idx in enumerate(relevant_indices):
if rel_idx >= len(links):
continue
link = links[rel_idx]
# Unrestrict the Real-Debrid link to get direct CDN download URL
unrestrict_url = f"{_API_BASE}/unrestrict/link"
unl_resp = requests.post(
unrestrict_url,
headers=self._auth_headers(),
data={"link": link},
timeout=_API_TIMEOUT,
verify=get_ssl_verify(unrestrict_url),
)
unl_resp.raise_for_status()
unl_data = unl_resp.json()
direct_url = unl_data.get("download")
filename = unl_data.get("filename")
if not direct_url:
msg = f"Failed to unrestrict Real-Debrid link: {link}"
_raise_runtime_error(msg)
# Determine relative file path
if rel_idx < len(selected_files):
rel_path_str = selected_files[rel_idx].get("path", "").lstrip("/")
rel_path = Path(rel_path_str)
else:
rel_path = Path(filename or f"file_{idx + 1}")
dest = state.target_dir / rel_path
dest.parent.mkdir(parents=True, exist_ok=True)
logger.info(
"Downloading Real-Debrid file %d/%d: %s",
idx + 1,
total,
rel_path,
)
buf = download_url(
direct_url,
referer="https://real-debrid.com/",
)
if not buf:
msg = f"Failed to download from {direct_url}"
_raise_runtime_error(msg)
with dest.open("wb") as fh:
fh.write(buf.getvalue())
with state.lock:
state.progress = 50.0 + (idx + 1) / total * 50.0
with state.lock:
state.phase = "complete"
state.progress = 100.0
logger.info(
"Real-Debrid download complete for ID %s at %s",
state.torrent_id,
state.target_dir,
)
except Exception:
logger.exception(
"Error in Real-Debrid download for ID %s",
state.torrent_id,
)
with state.lock:
state.phase = "error"
state.error_message = str(
state.error_message or "Download failed",
)
+7 -102
View File
@@ -4,19 +4,13 @@ Uses xmlrpc to communicate with rTorrent's RPC interface.
"""
import ssl
import time
import xmlrpc.client as stdlib_xmlrpc_client
from typing import Any, NoReturn, Protocol, cast
from urllib.parse import urlparse
from shelfmark.core.config import config
from shelfmark.core.logger import setup_logger
from shelfmark.core.utils import (
get_hardened_xmlrpc_client,
)
from shelfmark.core.utils import (
is_audiobook as check_audiobook,
)
from shelfmark.core.utils import get_hardened_xmlrpc_client
from shelfmark.download.clients import (
DownloadClient,
DownloadStatus,
@@ -52,13 +46,7 @@ class _RTorrentLoadProtocol(Protocol):
def start(self, target: str, url: str, commands: str) -> object: ...
class _RTorrentCustom1Protocol(Protocol):
def set(self, download_id: str, value: str) -> object: ...
class _RTorrentDownloadProtocol(Protocol):
custom1: _RTorrentCustom1Protocol
def multicall2(self, *args: object) -> list[list[Any]]: ...
def delete_tied(self, download_id: str) -> object: ...
@@ -127,7 +115,6 @@ class RTorrentClient(DownloadClient):
self._rpc = _create_rtorrent_server_proxy(self._base_url)
self._download_dir = config_text(config.get("RTORRENT_DOWNLOAD_DIR", ""))
self._label = config_text(config.get("RTORRENT_LABEL", ""))
self._audiobook_label = config_text(config.get("RTORRENT_AUDIOBOOK_LABEL", ""))
@staticmethod
def is_configured() -> bool:
@@ -172,18 +159,9 @@ class RTorrentClient(DownloadClient):
try:
torrent_info = extract_torrent_info(url, expected_hash=expected_hash)
known_hashes: set[str] | None = None
if not (torrent_info.info_hash or expected_hash):
known_hashes = self._list_torrent_hashes()
commands = []
content_type = kwargs.get("content_type")
is_audiobook = check_audiobook(content_type if isinstance(content_type, str) else None)
default_label = (
self._audiobook_label if is_audiobook and self._audiobook_label else self._label
)
label = category or default_label
label = category or self._label
if label:
logger.debug("Setting rTorrent label: %s", label)
commands.append(f"d.custom1.set={label}")
@@ -213,15 +191,7 @@ class RTorrentClient(DownloadClient):
torrent_hash = torrent_info.info_hash or expected_hash
if not torrent_hash:
# rTorrent fetches .torrent URLs itself, so the add can succeed
# even when no hash could be extracted up front. Recover it by
# watching for the new download to appear.
torrent_hash = self._discover_added_torrent_hash(name, label, known_hashes)
if not torrent_hash:
message = "Could not determine torrent hash from URL"
if torrent_info.fetch_error:
message = f"{message} (torrent file fetch failed: {torrent_info.fetch_error})"
_raise_runtime_error(message)
_raise_runtime_error("Could not determine torrent hash from URL")
logger.debug("Added torrent to rTorrent: %s", torrent_hash)
@@ -344,14 +314,12 @@ class RTorrentClient(DownloadClient):
"""
try:
# rtorrent is somehow case sensitive and requires uppercase hashes for look
torrent_hash = download_id.upper()
if delete_files:
self._rpc.d.delete_tied(torrent_hash)
self._rpc.d.erase(torrent_hash)
self._rpc.d.delete_tied(download_id)
self._rpc.d.erase(download_id)
else:
self._rpc.d.stop(torrent_hash)
self._rpc.d.erase(torrent_hash)
self._rpc.d.stop(download_id)
self._rpc.d.erase(download_id)
logger.info(
"Removed torrent from rTorrent: %s%s",
@@ -365,19 +333,6 @@ class RTorrentClient(DownloadClient):
else:
return True
def set_category(self, download_id: str, category: str) -> bool:
"""Assign a label to a torrent using rTorrent's custom1 field."""
try:
# rtorrent is somehow case sensitive and requires uppercase hashes for look
self._rpc.d.custom1.set(download_id.upper(), category)
logger.info("Set rTorrent label for %s to '%s'", download_id, category)
except _RTORRENT_CLIENT_ERRORS as e:
error_type = type(e).__name__
logger.exception("rTorrent set_category failed (%s)", error_type)
return False
else:
return True
def get_download_path(self, download_id: str) -> str | None:
"""Get the path where torrent files are located.
@@ -427,56 +382,6 @@ class RTorrentClient(DownloadClient):
except _RTORRENT_CLIENT_ERRORS:
return "/downloads"
def _list_torrent_hashes(self) -> set[str] | None:
"""Snapshot the hashes rTorrent currently reports."""
try:
all_torrents = self._rpc.d.multicall2("", "", "d.hash=")
except _RTORRENT_CLIENT_ERRORS as e:
logger.debug("Could not snapshot rTorrent downloads: %s", e)
return None
return {str(row[0]).lower() for row in all_torrents if row and row[0]}
def _discover_added_torrent_hash(
self,
name: str,
label: str,
known_hashes: set[str] | None,
) -> str | None:
"""Recover the hash of a torrent that was added without a known info_hash.
rTorrent fetches .torrent URLs itself, so the add can succeed even when
no hash could be extracted up front. A `known_hashes` of None means the
pre-add snapshot failed, so only an exact name match can identify the
new arrival.
"""
for _ in range(20):
try:
all_torrents = self._rpc.d.multicall2("", "", "d.hash=", "d.name=", "d.custom1=")
except _RTORRENT_CLIENT_ERRORS as e:
logger.debug("rTorrent hash discovery: %s", e)
else:
new_torrents = [
row
for row in all_torrents
if row
and row[0]
and (known_hashes is None or str(row[0]).lower() not in known_hashes)
]
# The label set at add time distinguishes concurrent arrivals,
# but rTorrent may not have applied it yet, so it only ever
# narrows a non-empty candidate list.
if label:
labeled = [row for row in new_torrents if len(row) > 2 and row[2] == label]
if labeled:
new_torrents = labeled
for row in new_torrents:
if len(row) > 1 and row[1] == name:
return str(row[0]).lower()
if known_hashes is not None and len(new_torrents) == 1:
return str(new_torrents[0][0]).lower()
time.sleep(0.5)
return None
def _get_torrent_path(self, download_id: str) -> str | None:
"""Get the file path of a torrent by hash.
-9
View File
@@ -248,15 +248,6 @@ class SABnzbdClient(DownloadClient):
if trusted_url and _url_origin(trusted_url) == target_origin:
return True
named_indexers = config.get("NEWZNAB_INDEXERS", [])
if isinstance(named_indexers, list):
for row in named_indexers:
if not isinstance(row, dict):
continue
trusted_url = normalize_http_config_url(row.get("url"))
if trusted_url and _url_origin(trusted_url) == target_origin:
return True
return False
def _get_prowlarr_headers(self, url: str) -> dict:
+3 -102
View File
@@ -159,7 +159,6 @@ def _test_qbittorrent_connection(current_values: dict[str, Any] | None = None) -
raw_url = _resolve_string_setting(current_values, config.get, "QBITTORRENT_URL")
username = _resolve_string_setting(current_values, config.get, "QBITTORRENT_USERNAME")
password = _resolve_string_setting(current_values, config.get, "QBITTORRENT_PASSWORD")
api_key = _resolve_string_setting(current_values, config.get, "QBITTORRENT_API_KEY")
if not raw_url:
return {"success": False, "message": "qBittorrent URL is required"}
@@ -175,7 +174,6 @@ def _test_qbittorrent_connection(current_values: dict[str, Any] | None = None) -
host=url,
username=username,
password=password,
api_key=api_key or None,
VERIFY_WEBUI_CERTIFICATE=get_ssl_verify(url),
)
client.auth_log_in()
@@ -183,18 +181,9 @@ def _test_qbittorrent_connection(current_values: dict[str, Any] | None = None) -
except ImportError:
return {"success": False, "message": "qbittorrent-api package not installed"}
except _QBITTORRENT_SETTINGS_ERRORS as e:
if isinstance(e, _QBittorrentLoginFailed):
# LoginFailed carries no message of its own, so name the rejected credential.
rejected = "API key" if api_key else "username or password"
return {"success": False, "message": f"qBittorrent rejected the {rejected}"}
return {"success": False, "message": f"Connection failed: {e!s}"}
else:
# Both credentials can be set at once, so name the one that actually authenticated.
used = " using the API key" if api_key else ""
return {
"success": True,
"message": f"Connected to qBittorrent (API v{api_version}){used}",
}
return {"success": True, "message": f"Connected to qBittorrent (API v{api_version})"}
def _test_transmission_connection(current_values: dict[str, Any] | None = None) -> dict[str, Any]:
@@ -531,40 +520,6 @@ def _test_sabnzbd_connection(current_values: dict[str, Any] | None = None) -> di
return {"success": True, "message": f"Connected to SABnzbd {version}"}
def _test_alldebrid_connection(current_values: dict[str, Any] | None = None) -> dict[str, Any]:
"""Test the AllDebrid API connection using current form values."""
from shelfmark.core.config import config
from shelfmark.download.clients.alldebrid import AllDebridClient
current_values = current_values or {}
api_key = _resolve_string_setting(current_values, config.get, "ALLDEBRID_API_KEY")
if not api_key:
return {"success": False, "message": "AllDebrid API Key is required"}
client = AllDebridClient()
client._api_key = api_key
success, message = client.test_connection()
return {"success": success, "message": message}
def _test_realdebrid_connection(current_values: dict[str, Any] | None = None) -> dict[str, Any]:
"""Test the Real-Debrid API connection using current form values."""
from shelfmark.core.config import config
from shelfmark.download.clients.realdebrid import RealDebridClient
current_values = current_values or {}
api_key = _resolve_string_setting(current_values, config.get, "REALDEBRID_API_KEY")
if not api_key:
return {"success": False, "message": "Real-Debrid API Key is required"}
client = RealDebridClient()
client._api_key = api_key
success, message = client.test_connection()
return {"success": success, "message": message}
# ==================== Download Clients Tab ====================
@@ -589,45 +544,13 @@ def prowlarr_clients_settings() -> list[SettingsField]:
description="Choose which torrent client to use",
options=[
{"value": "", "label": "None"},
{"value": "alldebrid", "label": "AllDebrid"},
{"value": "qbittorrent", "label": "qBittorrent"},
{"value": "realdebrid", "label": "Real-Debrid"},
{"value": "transmission", "label": "Transmission"},
{"value": "deluge", "label": "Deluge"},
{"value": "rtorrent", "label": "rTorrent"},
],
default="",
),
# --- AllDebrid Settings ---
PasswordField(
key="ALLDEBRID_API_KEY",
label="API Key",
description="AllDebrid API Key (apiv4) from your AllDebrid account settings",
show_when={"field": "PROWLARR_TORRENT_CLIENT", "value": "alldebrid"},
),
ActionButton(
key="test_alldebrid",
label="Test Connection",
description="Verify your AllDebrid configuration",
style="primary",
callback=_test_alldebrid_connection,
show_when={"field": "PROWLARR_TORRENT_CLIENT", "value": "alldebrid"},
),
# --- Real-Debrid Settings ---
PasswordField(
key="REALDEBRID_API_KEY",
label="API Key",
description="Real-Debrid API Key (Secret Token) from your Real-Debrid account settings",
show_when={"field": "PROWLARR_TORRENT_CLIENT", "value": "realdebrid"},
),
ActionButton(
key="test_realdebrid",
label="Test Connection",
description="Verify your Real-Debrid configuration",
style="primary",
callback=_test_realdebrid_connection,
show_when={"field": "PROWLARR_TORRENT_CLIENT", "value": "realdebrid"},
),
# --- qBittorrent Settings ---
TextField(
key="QBITTORRENT_URL",
@@ -649,12 +572,6 @@ def prowlarr_clients_settings() -> list[SettingsField]:
description="qBittorrent Web UI password",
show_when={"field": "PROWLARR_TORRENT_CLIENT", "value": "qbittorrent"},
),
PasswordField(
key="QBITTORRENT_API_KEY",
label="API Key",
description="Found in qBittorrent: Options > Web UI > API Key (qBittorrent 5.2.0+). Used instead of the username and password when set.",
show_when={"field": "PROWLARR_TORRENT_CLIENT", "value": "qbittorrent"},
),
ActionButton(
key="test_qbittorrent",
label="Test Connection",
@@ -831,18 +748,11 @@ def prowlarr_clients_settings() -> list[SettingsField]:
TextField(
key="RTORRENT_LABEL",
label="Book Label",
description="Label to assign to ebook downloads in rTorrent",
description="Label to assign to book downloads in rTorrent",
placeholder="cwabd",
default="cwabd",
show_when={"field": "PROWLARR_TORRENT_CLIENT", "value": "rtorrent"},
),
TextField(
key="RTORRENT_AUDIOBOOK_LABEL",
label="Audiobook Label",
description="Label to assign to audiobook downloads in rTorrent (falls back to Book Label if not set)",
placeholder="audiobooks",
show_when={"field": "PROWLARR_TORRENT_CLIENT", "value": "rtorrent"},
),
TextField(
key="RTORRENT_DOWNLOAD_DIR",
label="Download Directory",
@@ -854,23 +764,14 @@ def prowlarr_clients_settings() -> list[SettingsField]:
SelectField(
key="PROWLARR_TORRENT_ACTION",
label="Torrent Completion Action",
description="Choose whether to keep, remove, or move the torrent to another category or label after import",
description="Remove deletes the torrent from your client immediately after import (stops seeding, files are kept); Keep leaves it in the client to continue seeding",
options=[
{"value": "keep", "label": "Keep"},
{"value": "remove", "label": "Remove"},
{"value": "change_category", "label": "Change Category"},
],
default="keep",
show_when={"field": "PROWLARR_TORRENT_CLIENT", "notEmpty": True},
),
TextField(
key="PROWLARR_TORRENT_POST_IMPORT_CATEGORY",
label="Post-Import Category",
description="Category or label to assign after a successful import",
placeholder="imported",
default="",
show_when={"field": "PROWLARR_TORRENT_ACTION", "value": "change_category"},
),
# --- Usenet Client Selection ---
HeadingField(
key="usenet_heading",
+38 -207
View File
@@ -5,10 +5,8 @@ from __future__ import annotations
import base64
import hashlib
import re
import time
from binascii import Error as BinasciiError
from dataclasses import dataclass
from threading import Lock
from urllib.parse import ParseResult, parse_qs, urljoin, urlparse
import requests
@@ -17,12 +15,10 @@ from shelfmark.core.config import config
from shelfmark.core.logger import setup_logger
from shelfmark.core.utils import normalize_http_url
from shelfmark.download.network import get_ssl_verify
from shelfmark.download.postprocess.packs import PackFile
logger = setup_logger(__name__)
_MAGNET_RESPONSE_MAX_BYTES = 2000
_TORRENT_FETCH_MAX_REDIRECTS = 5
_BASE32_BTMH_TAG_BYTES = 34
_BTIH_INFO_BYTE_HEX = 0x20
_BTIH_PREFIX_BYTE = 0x12
@@ -39,15 +35,6 @@ _TORRENT_FETCH_ERRORS = (
_TORRENT_PARSE_ERRORS = (IndexError, KeyError, TypeError, ValueError)
_TRUSTED_TORRENT_FETCH_URL_CONFIG_KEYS = ("PROWLARR_URL", "NEWZNAB_URL")
# Successful torrent fetches are reused for a short window so one add attempt
# hits the download link only once. Tracker download links (e.g. private
# trackers behind Prowlarr's proxy) can be slow, rate-limited, or single-use,
# and both find_existing() and add_download() resolve the same URL (#1111).
_TORRENT_FETCH_CACHE_TTL_SECONDS = 120.0
_TORRENT_FETCH_CACHE_MAX_ENTRIES = 8
_torrent_fetch_cache_lock = Lock()
_torrent_fetch_cache: dict[str, tuple[float, TorrentInfo]] = {}
type BencodeValue = dict[str | bytes, BencodeValue] | list[BencodeValue] | int | bytes | str
@@ -67,9 +54,6 @@ class TorrentInfo:
magnet_url: str | None = None
"""The actual magnet URL, if available."""
fetch_error: str | None = None
"""Why fetching the .torrent URL failed, or None if it succeeded/was skipped."""
def with_info_hash(self, info_hash: str | None) -> TorrentInfo:
"""Return a copy with the info_hash replaced when provided."""
if info_hash:
@@ -78,65 +62,10 @@ class TorrentInfo:
torrent_data=self.torrent_data,
is_magnet=self.is_magnet,
magnet_url=self.magnet_url,
fetch_error=self.fetch_error,
)
return self
@dataclass
class DebridMagnet:
"""A magnet link, ready to hand to a debrid service as-is."""
magnet_url: str
@dataclass
class DebridTorrentFile:
"""Raw .torrent bytes, for a debrid service's file-upload endpoint."""
torrent_data: bytes
# A debrid service takes one or the other, never an indexer page or a proxy URL.
type DebridUpload = DebridMagnet | DebridTorrentFile
def resolve_debrid_upload(url: str, *, expected_hash: str | None = None) -> DebridUpload:
"""Resolve a release download URL into a magnet link or .torrent bytes.
Prowlarr hands out a proxy URL, with no magnetUrl and no infoHash, for any
indexer that only publishes torrent files - 1337x among them. Posting that
URL to a debrid service as if it were a magnet is what produced a bare 404
from the service instead of a download (#1250).
The torrent file is preferred over a synthesized `urn:btih:` magnet because
it carries the tracker list, which is how the service finds a swarm that is
not already cached. Fetches are shared with the rest of the add path through
the torrent fetch cache, so resolving here costs at most one request.
Raises:
ValueError: The URL resolved to neither form, so there is nothing to send.
"""
if url.startswith("magnet:"):
return DebridMagnet(magnet_url=url)
info = extract_torrent_info(url, expected_hash=expected_hash)
if info.is_magnet and info.magnet_url:
# The download URL redirected to, or returned, a magnet link.
return DebridMagnet(magnet_url=info.magnet_url)
if info.torrent_data:
return DebridTorrentFile(torrent_data=info.torrent_data)
if info.info_hash:
# No file to upload, but the hash alone still identifies the torrent.
return DebridMagnet(magnet_url=f"magnet:?xt=urn:btih:{info.info_hash}")
reason = info.fetch_error or "no magnet link, info hash, or torrent file was available"
msg = f"Could not resolve a torrent to send from {url[:120]} ({reason})"
raise ValueError(msg)
def extract_torrent_info(
url: str,
*,
@@ -166,58 +95,10 @@ def extract_torrent_info(
# Not a magnet - try to fetch and parse the .torrent file
if not fetch_torrent:
return TorrentInfo(info_hash=expected_hash, torrent_data=None, is_magnet=False)
if not _is_trusted_torrent_fetch_url(url):
logger.debug("Skipping torrent prefetch for untrusted URL: %s...", url[:80])
return TorrentInfo(info_hash=expected_hash, torrent_data=None, is_magnet=False)
info = _get_cached_torrent_fetch(url)
if info is None:
info = _fetch_torrent_info(url)
if info.fetch_error is None:
_store_cached_torrent_fetch(url, info)
return info.with_info_hash(info.info_hash or expected_hash)
def _get_cached_torrent_fetch(url: str) -> TorrentInfo | None:
with _torrent_fetch_cache_lock:
entry = _torrent_fetch_cache.get(url)
if entry is None:
return None
fetched_at, info = entry
if time.monotonic() - fetched_at > _TORRENT_FETCH_CACHE_TTL_SECONDS:
del _torrent_fetch_cache[url]
return None
logger.debug("Reusing recently fetched torrent data for: %s...", url[:80])
return info
def _store_cached_torrent_fetch(url: str, info: TorrentInfo) -> None:
with _torrent_fetch_cache_lock:
_torrent_fetch_cache[url] = (time.monotonic(), info)
while len(_torrent_fetch_cache) > _TORRENT_FETCH_CACHE_MAX_ENTRIES:
oldest_url = min(_torrent_fetch_cache, key=lambda key: _torrent_fetch_cache[key][0])
del _torrent_fetch_cache[oldest_url]
def clear_torrent_fetch_cache() -> None:
"""Drop all cached torrent fetches (used by tests)."""
with _torrent_fetch_cache_lock:
_torrent_fetch_cache.clear()
def _fetch_torrent_info(url: str) -> TorrentInfo:
"""Fetch a .torrent URL and parse out the info_hash and raw torrent data.
On failure, the returned TorrentInfo carries the reason in `fetch_error`
so callers can surface it instead of a generic hash error.
"""
# A release source can legitimately hand us a download URL on a different
# origin than the configured Prowlarr/Newznab endpoint (e.g. a direct
# tracker link, or Prowlarr reached through a separate proxy), and a trusted
# Prowlarr download URL commonly redirects to the indexer's own download
# link. We still need to fetch the .torrent to recover the info_hash when
# the source did not provide one, so the prefetch runs regardless of origin
# and follows cross-origin redirects. The Prowlarr API key, however, is
# re-evaluated per hop and only ever sent to a trusted origin so it can
# never leak to an arbitrary indexer/tracker host.
headers: dict[str, str] = {"Accept": "application/x-bittorrent"}
# TODO(shelfmark): Move this source-specific Prowlarr auth handling into a source hook.
api_key = str(config.get("PROWLARR_API_KEY", "") or "").strip()
@@ -233,47 +114,44 @@ def _fetch_torrent_info(url: str) -> TorrentInfo:
try:
logger.debug("Fetching torrent file from: %s...", url[:80])
# Redirects are followed manually: some indexers redirect download URLs
# to magnet links, and each hop must decide anew whether it may see the
# API key.
current_url = url
redirects_remaining = _TORRENT_FETCH_MAX_REDIRECTS
while True:
request_headers = dict(headers)
if not _is_trusted_torrent_fetch_url(current_url):
request_headers.pop("X-Api-Key", None)
# Use allow_redirects=False to handle magnet link redirects manually
# Some indexers redirect download URLs to magnet links
resp = requests.get(
url,
timeout=30,
allow_redirects=False,
headers=headers,
verify=get_ssl_verify(url),
)
resp = requests.get(
current_url,
timeout=30,
allow_redirects=False,
headers=request_headers,
verify=get_ssl_verify(current_url),
)
if resp.status_code not in (301, 302, 303, 307, 308):
break
redirect_url = resolve_url(current_url, resp.headers.get("Location", ""))
# Check if this is a redirect to a magnet link
if resp.status_code in (301, 302, 303, 307, 308):
redirect_url = resolve_url(url, resp.headers.get("Location", ""))
if redirect_url.startswith("magnet:"):
logger.debug("Download URL redirected to magnet link")
info_hash = extract_hash_from_magnet(redirect_url)
if not info_hash and expected_hash:
info_hash = expected_hash
return TorrentInfo(
info_hash=extract_hash_from_magnet(redirect_url),
info_hash=info_hash,
torrent_data=None,
is_magnet=True,
magnet_url=redirect_url,
)
if redirects_remaining <= 0:
logger.warning("Too many redirects fetching torrent file: %s...", url[:80])
return TorrentInfo(
info_hash=None,
torrent_data=None,
is_magnet=False,
fetch_error="too many redirects",
if not _is_trusted_torrent_fetch_url(redirect_url):
logger.debug(
"Skipping torrent prefetch redirect to untrusted URL: %s...",
redirect_url[:80],
)
redirects_remaining -= 1
return TorrentInfo(info_hash=expected_hash, torrent_data=None, is_magnet=False)
# Not a magnet redirect, follow it manually
logger.debug("Following redirect to: %s...", redirect_url[:80])
current_url = redirect_url
resp = requests.get(
redirect_url,
timeout=30,
headers=headers,
verify=get_ssl_verify(redirect_url),
)
resp.raise_for_status()
torrent_data = resp.content
@@ -284,22 +162,25 @@ def _fetch_torrent_info(url: str) -> TorrentInfo:
text_content = torrent_data.decode("utf-8", errors="ignore").strip()
if text_content.startswith("magnet:"):
logger.debug("Download URL returned magnet link as response body")
info_hash = extract_hash_from_magnet(text_content)
if not info_hash and expected_hash:
info_hash = expected_hash
return TorrentInfo(
info_hash=extract_hash_from_magnet(text_content),
info_hash=info_hash,
torrent_data=None,
is_magnet=True,
magnet_url=text_content,
)
info_hash = extract_info_hash_from_torrent(torrent_data)
info_hash = extract_info_hash_from_torrent(torrent_data) or expected_hash
if info_hash:
logger.debug("Extracted hash from torrent file: %s", info_hash)
else:
logger.warning("Could not extract hash from torrent file")
return TorrentInfo(info_hash=info_hash, torrent_data=torrent_data, is_magnet=False)
except _TORRENT_FETCH_ERRORS as e:
logger.warning("Could not fetch torrent file: %s", e)
return TorrentInfo(info_hash=None, torrent_data=None, is_magnet=False, fetch_error=str(e))
logger.debug("Could not fetch torrent file: %s", e)
return TorrentInfo(info_hash=expected_hash, torrent_data=None, is_magnet=False)
def _is_trusted_torrent_fetch_url(url: str) -> bool:
@@ -434,56 +315,6 @@ def extract_info_hash_from_torrent(torrent_data: bytes) -> str | None:
return None
def _decode_torrent_text(value: object) -> str | None:
if isinstance(value, bytes):
return value.decode("utf-8", errors="replace")
if isinstance(value, str):
return value
return None
def extract_file_list_from_torrent(torrent_data: bytes) -> list[PackFile] | None:
"""List the files a .torrent describes, release-relative, without downloading it.
Multi-file torrents nest every path under the torrent name (which becomes the
client's save folder); single-file torrents are just the named file.
"""
try:
decoded, _ = bencode_decode(torrent_data)
except _TORRENT_PARSE_ERRORS as e:
logger.debug("Failed to parse torrent file list: %s", e)
return None
if not isinstance(decoded, dict):
return None
info = decoded.get(b"info")
if not isinstance(info, dict):
return None
name = _decode_torrent_text(info.get(b"name")) or ""
raw_files = info.get(b"files")
if not isinstance(raw_files, list):
length = info.get(b"length")
if not name:
return None
return [PackFile(name, length if isinstance(length, int) else None)]
files: list[PackFile] = []
for entry in raw_files:
if not isinstance(entry, dict):
continue
raw_path = entry.get(b"path")
if not isinstance(raw_path, list):
continue
segments = [seg for seg in (_decode_torrent_text(part) for part in raw_path) if seg]
if not segments:
continue
if name:
segments.insert(0, name)
length = entry.get(b"length")
files.append(PackFile("/".join(segments), length if isinstance(length, int) else None))
return files
def extract_hash_from_magnet(magnet_url: str) -> str | None:
"""Extract info_hash from a magnet URL."""
if not magnet_url.startswith("magnet:"):
+2 -30
View File
@@ -316,12 +316,8 @@ class TransmissionClient(DownloadClient):
state, message = status_map.get(status_value, ("downloading", "Downloading"))
progress = torrent.percent_done * 100
# Only mark complete when seeding or stopped (e.g. if seed limit/ratio is 0)
# and progress is complete. seed pending means files still being moved
complete = progress >= _SEEDING_PROGRESS_PERCENT and status_value in (
"seeding",
"stopped",
)
# Only mark complete when seeding - seed pending means files still being moved
complete = progress >= _SEEDING_PROGRESS_PERCENT and status_value == "seeding"
if complete:
message = "Complete"
@@ -390,30 +386,6 @@ class TransmissionClient(DownloadClient):
else:
return True
def _get_torrent_labels(self, download_id: str) -> list[str]:
"""Return a torrent's current labels, preserving their order."""
torrent = self._client.get_torrent(download_id)
raw_labels = getattr(torrent, "labels", None) or []
return [str(label) for label in raw_labels if str(label)]
def set_category(self, download_id: str, category: str) -> bool:
"""Add the post-import label to a torrent, keeping labels set elsewhere."""
try:
existing_labels = self._get_torrent_labels(download_id)
if category in existing_labels:
logger.debug(
"Transmission torrent %s already has label '%s'", download_id, category
)
return True
self._client.change_torrent(ids=download_id, labels=[*existing_labels, category])
logger.info("Added Transmission label '%s' to %s", category, download_id)
except _TRANSMISSION_CLIENT_ERRORS as e:
self._log_error("set_category", e)
return False
else:
return True
def get_download_path(self, download_id: str) -> str | None:
"""Get the path where torrent files are located.
-158
View File
@@ -1,158 +0,0 @@
"""RFC 8484 DNS wireformat encoding/decoding for DoH providers.
Providers split into two incompatible camps and the difference is not cosmetic:
* **JSON** (Cloudflare, Google) - ``?name=<host>&type=A`` returning a JSON body. A
convention, not a standard, and the only one Shelfmark used to speak.
* **Wireformat** (Quad9, OpenDNS) - RFC 8484 proper: a base64url-encoded DNS message
in ``?dns=``, answered with ``application/dns-message``. Quad9 additionally
*requires HTTP/2* per RFC 8484 section 5.2 and answers HTTP/1.1 with 505.
This module carries the codec only; the transport choice lives in the resolver.
Encoding a query is a handful of bytes, and parsing an answer needs message
compression support (RFC 1035 section 4.1.4) because answer names are almost always
pointers back into the question.
"""
from __future__ import annotations
import base64
import secrets
import struct
# Record types we resolve.
TYPE_A = 1
TYPE_AAAA = 28
_CLASS_IN = 1
_HEADER = struct.Struct(">HHHHHH")
_RR_FIXED = struct.Struct(">HHIH") # type, class, ttl, rdlength
_FLAG_RECURSION_DESIRED = 0x0100
_MAX_LABEL_JUMPS = 64 # cap pointer-following so a malicious answer cannot loop
_MAX_NAME_LENGTH = 255
class WireformatError(ValueError):
"""Raised when a DNS wireformat message cannot be parsed."""
def encode_query(hostname: str, record_type: int) -> bytes:
"""Build a DNS query message for ``hostname``.
The ID is zero because RFC 8484 section 4.1 requires it for cacheability, but the
caller may randomise it when not using a cache.
"""
if not hostname:
msg = "hostname must not be empty"
raise WireformatError(msg)
question = bytearray()
for label in hostname.rstrip(".").split("."):
encoded = label.encode("idna") if not label.isascii() else label.encode("ascii")
if not encoded or len(encoded) > 63:
msg = f"invalid DNS label in {hostname!r}"
raise WireformatError(msg)
question.append(len(encoded))
question.extend(encoded)
question.append(0)
question.extend(struct.pack(">HH", record_type, _CLASS_IN))
header = _HEADER.pack(0, _FLAG_RECURSION_DESIRED, 1, 0, 0, 0)
return header + bytes(question)
def encode_query_param(hostname: str, record_type: int) -> str:
"""Return the base64url ``dns=`` parameter value for a query (padding stripped)."""
return base64.urlsafe_b64encode(encode_query(hostname, record_type)).rstrip(b"=").decode()
def _read_name(message: bytes, offset: int) -> int:
"""Skip over a (possibly compressed) name, returning the offset after it."""
jumps = 0
length = 0
while True:
if offset >= len(message):
msg = "truncated DNS name"
raise WireformatError(msg)
label_len = message[offset]
if label_len == 0:
return offset + 1
if label_len & 0xC0 == 0xC0:
# A pointer ends this name; the rest of the record follows the 2 bytes.
if offset + 1 >= len(message):
msg = "truncated DNS name pointer"
raise WireformatError(msg)
return offset + 2
offset += 1 + label_len
length += 1 + label_len
jumps += 1
if jumps > _MAX_LABEL_JUMPS or length > _MAX_NAME_LENGTH:
msg = "malformed DNS name"
raise WireformatError(msg)
def decode_answer(message: bytes, record_type: int) -> list[str]:
"""Extract the IP addresses of ``record_type`` from a DNS response message.
Returns an empty list for a well-formed response that carries no matching record
(NXDOMAIN, or only CNAMEs), and raises WireformatError for a malformed one - the
caller treats those differently.
"""
if len(message) < _HEADER.size:
msg = "DNS response shorter than its header"
raise WireformatError(msg)
_id, _flags, qdcount, ancount, _ns, _ar = _HEADER.unpack_from(message, 0)
offset = _HEADER.size
for _ in range(qdcount):
offset = _read_name(message, offset)
offset += 4 # QTYPE + QCLASS
results: list[str] = []
for _ in range(ancount):
offset = _read_name(message, offset)
if offset + _RR_FIXED.size > len(message):
msg = "truncated resource record"
raise WireformatError(msg)
rtype, rclass, _ttl, rdlength = _RR_FIXED.unpack_from(message, offset)
offset += _RR_FIXED.size
rdata = message[offset : offset + rdlength]
if len(rdata) != rdlength:
msg = "truncated record data"
raise WireformatError(msg)
offset += rdlength
if rclass != _CLASS_IN or rtype != record_type:
continue
if rtype == TYPE_A and rdlength == 4:
results.append(".".join(str(b) for b in rdata))
elif rtype == TYPE_AAAA and rdlength == 16:
groups = struct.unpack(">8H", rdata)
results.append(_compress_ipv6(groups))
return results
def _compress_ipv6(groups: tuple[int, ...]) -> str:
"""Render an IPv6 address with the longest zero run collapsed to '::'."""
best_start = best_len = -1
run_start = -1
for i, group in enumerate([*list(groups), 1]): # sentinel closes a trailing run
if group == 0 and i < len(groups):
if run_start < 0:
run_start = i
elif run_start >= 0:
if i - run_start > best_len:
best_start, best_len = run_start, i - run_start
run_start = -1
parts = [format(g, "x") for g in groups]
if best_len > 1:
return ":".join(parts[:best_start]) + "::" + ":".join(parts[best_start + best_len :])
return ":".join(parts)
def random_query_id() -> int:
"""A random DNS message ID, for callers that do not want the RFC 8484 zero."""
return secrets.randbelow(0x10000)
+6 -153
View File
@@ -4,7 +4,6 @@ These utilities handle file collisions atomically, avoiding TOCTOU race conditio
when multiple workers may try to write to the same path simultaneously.
"""
import contextlib
import errno
import os
import shutil
@@ -105,57 +104,6 @@ _PUBLISH_VERIFY_RETRY_SECONDS = 0.25
_TEMPFILE_PREFIX = ".shelfmark."
_TEMPFILE_SUFFIX = ".tmp"
# Destinations that accept writes but reject unlink/rename, e.g. a Synology share
# with "Delete subfolders and files" unticked. Publishing a temp file into place
# removes a directory entry, so those paths must be written in place instead.
_DELETE_DENIED_DIRS: set[str] = set()
class _PublishDeniedError(Exception):
"""A fully-written temp file could not be renamed onto its final path."""
def _is_delete_denied_error(error: Exception) -> bool:
return isinstance(error, OSError) and error.errno in {errno.EACCES, errno.EPERM}
def mark_delete_denied(directory: Path) -> None:
"""Record that `directory` rejects deletes so later writes skip the temp file."""
key = str(directory)
if key in _DELETE_DENIED_DIRS:
return
_DELETE_DENIED_DIRS.add(key)
logger.warning(
"Destination %s rejects delete/rename; writing files in place instead of "
"publishing atomically. Grant delete permission to restore atomic writes.",
directory,
)
def clear_delete_denied(directory: Path) -> None:
"""Forget recorded denials for `directory` and anything beneath it.
Subdirectories get marked independently (an `organize` layout publishes into
per-author folders), so clearing only the exact key would leave a fixed
destination writing in place until restart.
"""
if not _DELETE_DENIED_DIRS:
return
key = str(directory)
prefix = f"{key}{os.sep}"
_DELETE_DENIED_DIRS.difference_update(
{marked for marked in _DELETE_DENIED_DIRS if marked == key or marked.startswith(prefix)}
)
def is_delete_denied(directory: Path) -> bool:
"""True if `directory` or one of its ancestors is known to reject deletes."""
if not _DELETE_DENIED_DIRS:
return False
if str(directory) in _DELETE_DENIED_DIRS:
return True
return any(str(parent) in _DELETE_DENIED_DIRS for parent in directory.parents)
def _verify_transfer_size(
dest: Path,
@@ -413,40 +361,10 @@ def _create_temp_path(dest_path: Path) -> Path:
return Path(temp_path)
def _discard_path(path: Path) -> None:
"""Best-effort unlink that tolerates destinations which reject deletes."""
try:
run_blocking_io(path.unlink, missing_ok=True)
except OSError as exc:
logger.warning("Could not remove %s: %s", path, exc)
def _copy_into_claimed(source_path: Path, dest_path: Path, expected_size: int) -> None:
"""Copy content straight into an already-claimed destination path.
Used when the destination rejects rename/unlink: there is no temp file to
publish, so the final name is written in place. This is not atomic - a
watcher can observe a partial file - but it is the only way to deliver on
such a share. `copyfile` (not `copy2`) because metadata copying needs chmod,
which those shares also tend to refuse.
"""
try:
run_blocking_io(shutil.copyfile, str(source_path), str(dest_path))
_verify_transfer_size(dest_path, expected_size, "copy")
except Exception:
with contextlib.suppress(OSError):
run_blocking_io(dest_path.unlink, missing_ok=True)
raise
def _publish_temp_file(temp_path: Path, dest_path: Path) -> bool:
"""Publish a temp file to its final path without overwriting existing files.
Returns True on success, False if the destination already exists.
Raises `_PublishDeniedError` when the rename is refused for lack of delete
permission. The claimed destination is left in place so the caller can write
into it directly instead.
"""
claimed = _claim_destination(dest_path)
if not claimed:
@@ -456,19 +374,7 @@ def _publish_temp_file(temp_path: Path, dest_path: Path) -> bool:
# Publish by renaming the fully-written temp file into place. This gives
# watchers an IN_MOVED_TO-style event on the final path instead of relying
# on hardlink support in the destination filesystem.
try:
run_blocking_io(os.replace, str(temp_path), str(dest_path))
except OSError as e:
if _is_delete_denied_error(e):
log_transfer_permission_context(
"publish_replace",
source=temp_path,
dest=dest_path,
error=e,
)
mark_delete_denied(dest_path.parent)
raise _PublishDeniedError(str(e)) from e
raise
run_blocking_io(os.replace, str(temp_path), str(dest_path))
# Best-effort nudge for watchers that only react to close-write on the
# final filename rather than rename/move events.
@@ -477,8 +383,6 @@ def _publish_temp_file(temp_path: Path, dest_path: Path) -> bool:
run_blocking_io(os.close, fd)
except OSError:
pass
except _PublishDeniedError:
raise
except Exception as e:
if _is_permission_error(e):
log_transfer_permission_context(
@@ -487,23 +391,12 @@ def _publish_temp_file(temp_path: Path, dest_path: Path) -> bool:
dest=dest_path,
error=e,
)
_discard_path(dest_path)
run_blocking_io(dest_path.unlink, missing_ok=True)
raise
else:
return True
def _move_via_copy(source_path: Path, dest_path: Path, max_attempts: int) -> Path:
"""Deliver a move as copy + source unlink.
For destinations that reject rename. The source lives in TMP_DIR (which we
own and can delete), so only the destination-side semantics change.
"""
final_path = atomic_copy(source_path, dest_path, max_attempts=max_attempts)
_discard_path(source_path)
return final_path
def atomic_move(source_path: Path, dest_path: Path, max_attempts: int = 100) -> Path:
"""Move a file with collision detection.
@@ -530,11 +423,6 @@ def atomic_move(source_path: Path, dest_path: Path, max_attempts: int = 100) ->
ext = dest_path.suffix
parent = dest_path.parent
# rename() removes a directory entry, so a destination that refuses deletes
# cannot be moved into. Deliver it as copy + source unlink instead.
if is_delete_denied(parent):
return _move_via_copy(source_path, dest_path, max_attempts)
for attempt in range(max_attempts):
try_path = dest_path if attempt == 0 else parent / f"{base}_{attempt}{ext}"
@@ -561,13 +449,6 @@ def atomic_move(source_path: Path, dest_path: Path, max_attempts: int = 100) ->
run_blocking_io(try_path.unlink, missing_ok=True)
continue
except OSError as e:
if _is_delete_denied_error(e):
# Destination refuses the rename; fall back to copy + unlink source.
mark_delete_denied(parent)
if claimed:
_discard_path(try_path)
return _move_via_copy(source_path, dest_path, max_attempts)
# Cross-filesystem - copy to temp and publish atomically.
if e.errno != errno.EXDEV:
if claimed:
@@ -616,7 +497,7 @@ def atomic_move(source_path: Path, dest_path: Path, max_attempts: int = 100) ->
try:
_verify_published_file(try_path, expected_size, "move")
except Exception:
_discard_path(try_path)
run_blocking_io(try_path.unlink, missing_ok=True)
raise
run_blocking_io(source_path.unlink)
@@ -627,18 +508,9 @@ def atomic_move(source_path: Path, dest_path: Path, max_attempts: int = 100) ->
if temp_path:
run_blocking_io(temp_path.unlink, missing_ok=True)
continue
except _PublishDeniedError:
# Destination is claimed but unrenameable; write into it directly.
_copy_into_claimed(source_path, try_path, expected_size)
if temp_path:
_discard_path(temp_path)
_discard_path(source_path)
if attempt > 0:
logger.info("File collision resolved: %s", try_path.name)
return try_path
except Exception:
if temp_path:
_discard_path(temp_path)
run_blocking_io(temp_path.unlink, missing_ok=True)
raise
else:
return try_path
@@ -757,17 +629,6 @@ def atomic_copy(source_path: Path, dest_path: Path, max_attempts: int = 100) ->
try_path = dest_path if attempt == 0 else parent / f"{base}_{attempt}{ext}"
if run_blocking_io(try_path.exists):
continue
# Known-undeletable destination: skip the temp file entirely, otherwise
# every transfer would strand a `.shelfmark.*.tmp` we cannot clean up.
if is_delete_denied(parent):
if not _claim_destination(try_path):
continue
_copy_into_claimed(source_path, try_path, expected_size)
if attempt > 0:
logger.info("File collision resolved: %s", try_path.name)
return try_path
temp_path: Path | None = None
try:
temp_path = _create_temp_path(try_path)
@@ -819,22 +680,14 @@ def atomic_copy(source_path: Path, dest_path: Path, max_attempts: int = 100) ->
try:
_verify_published_file(try_path, expected_size, "copy")
except Exception:
_discard_path(try_path)
run_blocking_io(try_path.unlink, missing_ok=True)
raise
if attempt > 0:
logger.info("File collision resolved: %s", try_path.name)
except _PublishDeniedError:
# The destination is claimed but unrenameable; write into it directly.
_copy_into_claimed(source_path, try_path, expected_size)
if temp_path:
_discard_path(temp_path)
if attempt > 0:
logger.info("File collision resolved: %s", try_path.name)
return try_path
except Exception:
if temp_path:
_discard_path(temp_path)
run_blocking_io(temp_path.unlink, missing_ok=True)
raise
else:
return try_path
+66 -455
View File
@@ -4,61 +4,44 @@ import random
import time
from http import HTTPStatus
from io import BytesIO
from threading import Event, Thread
from typing import TYPE_CHECKING, NoReturn
from urllib.parse import parse_qsl, urlencode, urljoin, urlparse, urlunparse
from urllib.parse import urljoin, urlparse
import requests
from tqdm import tqdm
from shelfmark.bypass import BypassCancelledError, ChallengeNotSolvedError, cookie_store
from shelfmark.bypass.challenge import challenge_marker
from shelfmark.core import search_deadline
from shelfmark.bypass import BypassCancelledError
from shelfmark.core.config import config as app_config
from shelfmark.core.logger import setup_logger
from shelfmark.core.request_helpers import coerce_bool, normalize_positive_int
from shelfmark.download import network
from shelfmark.download.activity import release_activity_grace, request_activity_grace
from shelfmark.download.network import get_proxies, get_ssl_verify
if TYPE_CHECKING:
from collections.abc import Callable
from threading import Event
from types import ModuleType
logger = setup_logger(__name__)
_RNG = random.SystemRandom()
_MAX_REDIRECTS = 5
# DDoS-Guard's re-check probe. Its 302 to `?check=1` is one hop of a handshake rather
# than a page: the parameter asserts the caller already holds the cookies that hop
# issued.
_DDG_CHECK_PARAM = "check"
# Z-Library answers the first hit with a 503 whose only real payload is a Set-Cookie; echoing
# that cookie back returns the 302 to the real page. Two attempts cover the handshake without
# letting a server that keeps re-issuing cookies hold us in the loop.
_MAX_COOKIE_HANDSHAKE_RETRIES = 2
_HTTP_STATUS_FORBIDDEN = HTTPStatus.FORBIDDEN
_HTTP_STATUS_NOT_FOUND = HTTPStatus.NOT_FOUND
_HTTP_STATUS_RATE_LIMITED = HTTPStatus.TOO_MANY_REQUESTS
_HTTP_STATUS_SERVICE_UNAVAILABLE = HTTPStatus.SERVICE_UNAVAILABLE
_HTTP_STATUS_OK = HTTPStatus.OK
_HTTP_STATUS_RANGE_NOT_SATISFIABLE = HTTPStatus.REQUESTED_RANGE_NOT_SATISFIABLE
_HTTP_STATUS_PARTIAL_CONTENT = HTTPStatus.PARTIAL_CONTENT
_HTTP_STATUS_NON_RETRYABLE = (_HTTP_STATUS_FORBIDDEN, _HTTP_STATUS_NOT_FOUND)
_STATUS_CALLBACK_ERRORS = (AttributeError, KeyError, OSError, RuntimeError, TypeError, ValueError)
# Added on top of the active bypasser's own budget so it reports its real failure before
# stall detection cancels the download.
_BYPASS_GRACE_SLACK_SECONDS = 30.0
_BYPASSER_ERRORS = (
AttributeError,
BypassCancelledError,
ChallengeNotSolvedError,
KeyError,
OSError,
RuntimeError,
TypeError,
ValueError,
network.RateLimitedError,
requests.exceptions.RequestException,
)
@@ -71,18 +54,6 @@ def _raise_too_many_redirects(message: str) -> NoReturn:
raise requests.exceptions.TooManyRedirects(message)
def _new_cookies(response: requests.Response, already_sent: dict[str, str]) -> dict[str, str]:
"""Cookies a response set that we were not already echoing back.
Returning only the *new* ones is what makes the retry terminate: a server that keeps
re-issuing the same cookie yields nothing here, so we stop instead of spinning.
"""
jar = getattr(response, "cookies", None)
if not jar:
return {}
return {name: value for name, value in jar.items() if already_sent.get(name) != value}
def _get_internal_bypasser() -> ModuleType:
"""Lazy import of internal bypasser module."""
global _internal_bypasser
@@ -128,19 +99,6 @@ def _is_cf_bypass_enabled() -> bool:
return coerce_bool(app_config.get("USE_CF_BYPASS", True))
def _bypass_grace_seconds() -> float:
"""How long a bypass may block before stall detection should give up on it.
Each bypasser knows its own retry/timeout budget, so ask the active one rather than
duplicating the arithmetic here. The slack keeps the bypasser's own deadline expiring
first, so the user sees its real error instead of a generic "Download stalled".
"""
bypasser = (
_get_external_bypasser() if _is_using_external_bypasser() else _get_internal_bypasser()
)
return bypasser.max_duration_seconds() + _BYPASS_GRACE_SLACK_SECONDS
def get_bypassed_page(
url: str,
selector: network.AAMirrorSelector | None = None,
@@ -153,13 +111,19 @@ def get_bypassed_page(
def get_cf_cookies_for_domain(domain: str) -> dict[str, str]:
"""Get the clearance cookies won by whichever bypasser solved this domain."""
return cookie_store.get_cf_cookies_for_domain(domain)
"""Get CF cookies - only available with internal bypasser."""
if _is_using_external_bypasser():
logger.debug("External bypasser in use, CF cookies not available for %s", domain)
return {}
return _get_internal_bypasser().get_cf_cookies_for_domain(domain)
def get_cf_user_agent_for_domain(domain: str) -> str | None:
"""Get the User-Agent that solved this domain's challenge, if one is stored."""
return cookie_store.get_cf_user_agent_for_domain(domain)
"""Get CF user agent - only available with internal bypasser."""
if _is_using_external_bypasser():
logger.debug("External bypasser in use, CF user agent not available for %s", domain)
return None
return _get_internal_bypasser().get_cf_user_agent_for_domain(domain)
def _apply_cf_bypass(url: str, headers: dict) -> dict:
@@ -236,93 +200,13 @@ def _is_retryable_error(e: Exception) -> bool:
return status is not None and status in RETRYABLE_CODES
# Statuses that mean the host is gone rather than busy: 410 Gone and 451 Unavailable
# For Legal Reasons are what a seized domain answers with.
_DEAD_MIRROR_CODES = (410, 451)
def _response_challenge_marker(response: requests.Response) -> str | None:
"""The challenge marker in a response body, or None if it carries no challenge.
Content type is checked first so a JSON or octet-stream error body is never
decoded just to be scanned; a missing header is scanned anyway, since an
interstitial served without one is still an interstitial.
"""
content_type = response.headers.get("Content-Type", "")
if content_type and "html" not in content_type.lower():
return None
try:
return challenge_marker(response.text)
except UnicodeDecodeError, ValueError:
return None
def _solvable_url(url: str) -> str:
"""The URL a solver should open, given one we may be mid-handshake on.
The manual AA redirect follower in `html_get_page` walks DDoS-Guard's handshake by
reassigning `current_url`, so by the time a 403, a 503 challenge or a redirect loop
hands that URL to a bypasser it is often the `?check=1` probe rather than the page
we actually wanted. A solver opens it in a fresh browser holding none of the cookies
the probe exists to collect, so DDoS-Guard cannot verify it automatically and answers
with the manual CAPTCHA page that nothing can solve - the failure in #1292, where
FlareSolverr reported "Challenge solved!" over a 4.7 KB DDOS-GUARD interstitial.
Handing over the pre-probe URL instead lets the solver's browser run the whole
handshake itself, which is what a real browser does and what the solver is for.
Scoped to the hosts whose redirects we follow manually: everywhere else `check` is
an ordinary query parameter and none of our business.
"""
if not network.should_rotate_dns_for_url(url):
return url
parsed = urlparse(url)
params = parse_qsl(parsed.query, keep_blank_values=True)
kept = [(key, value) for key, value in params if key != _DDG_CHECK_PARAM]
if len(kept) == len(params):
return url
return urlunparse(parsed._replace(query=urlencode(kept)))
def _fatal_mirror_reason(e: Exception) -> str | None:
"""Return why ``e`` proves the mirror is unusable, or None if it may recover.
Hard evidence only - the name does not resolve, nothing is listening, or the host
says it is gone for good. A timeout, a 5xx or a challenge all mean the mirror is
alive, and rotating off it discards the bypass clearance held for that domain.
"""
status = _get_status_code(e)
if status is not None and status in _DEAD_MIRROR_CODES:
return f"HTTP {status}"
# requests wraps the real cause; a read timeout subclasses ConnectionError for
# some adapters, so exclude timeouts explicitly before inspecting the message.
if isinstance(e, requests.exceptions.Timeout):
return None
if not isinstance(e, requests.exceptions.ConnectionError):
return None
text = str(e).lower()
if "nameresolutionerror" in text or "failed to resolve" in text or "name or service" in text:
return "DNS does not resolve"
if "connection refused" in text or "no route to host" in text:
return "connection refused"
return None
def _try_rotation(
original_url: str,
current_url: str,
selector: network.AAMirrorSelector,
*,
fatal_reason: str | None = None,
original_url: str, current_url: str, selector: network.AAMirrorSelector
) -> str | None:
"""Try mirror/DNS rotation. Returns new URL or None."""
aa_base_url = network.get_aa_base_url()
if aa_base_url and current_url.startswith(aa_base_url):
new_base, action = selector.next_mirror_or_rotate_dns(
fatal=fatal_reason is not None, reason=fatal_reason or ""
)
new_base, action = selector.next_mirror_or_rotate_dns()
if action in ("mirror", "dns") and new_base:
new_url = selector.rewrite(original_url)
logger.info("[%s] switching to: %s", action, new_url)
@@ -354,11 +238,8 @@ def html_get_page(
selector: Mirror selector used for AA mirror and DNS rotation.
cancel_flag: Optional event used to abort retries early.
status_callback: Optional callback for UI status updates.
allow_bypasser_fallback: Whether a challenge may be handed to the bypasser.
If False, a 403 triggers mirror rotation instead, and an AA redirect loop
gives up immediately rather than waiting on a browser solve. Use False for
best-effort fetches whose result is optional (e.g. the download count on
the details modal); search and detail pages pass True.
allow_bypasser_fallback: If False, 403 errors will trigger mirror rotation
instead of switching to the bypasser. Use for search operations.
use_bypasser: Whether to start with the bypasser instead of direct HTTP.
include_response_url: If True, return `(html, final_url)` to expose the
resolved response URL after redirects.
@@ -367,172 +248,59 @@ def html_get_page(
"""
# Normalise before the closures below capture it: they touch selector.last_failure,
# so it must be a concrete selector, not the Optional parameter.
selector = selector or network.AAMirrorSelector()
# A release search runs under a wall-clock budget (see shelfmark.core.search_deadline).
# Adopting it as the cancel flag is what makes the budget bite on a solve already in
# flight: the bypassers and the helper subprocess poll this flag but know nothing about
# deadlines. Only when the caller has no flag of its own - a queued download brings one
# and must keep it, and runs outside any search context anyway.
if cancel_flag is None:
cancel_flag = search_deadline.cancel_event()
def _result(html: str, response_url: str) -> str | tuple[str, str]:
if include_response_url:
return html, response_url
return html
def _fail(reason: str, response_url: str) -> str | tuple[str, str]:
"""Record why the fetch is giving up, then return the empty result.
Every give-up path returns an empty page, which is all the caller used to
see. Stashing the concrete reason on the shared selector lets the caller
surface it (see release_sources.direct_download) rather than reporting the
same generic "network restricted or mirrors blocked" for every cause.
"""
selector.last_failure = reason
return _result("", response_url)
def _run_bypasser(bypass_url: str) -> str | tuple[str, str]:
"""Run the active bypasser for one URL and return its result.
Factored out so the redirect-loop handoff below can invoke it directly. That
call site sits inside the inner redirect `while`, so it cannot reach the
retry-loop branch above with `continue`, and with MAX_RETRY=1 there is no
later attempt for that branch to run on either.
"""
# Every handoff reaches the solver through here, so this is the one place the
# mid-handshake `?check=1` URL has to be unwound. See _solvable_url.
bypass_url = _solvable_url(bypass_url)
# Never start a minutes-long browser solve on a budget that has already run out:
# nothing downstream would get to report the real reason before the caller's
# deadline (or its reverse proxy) cut the request off.
if search_deadline.expired():
logger.info("Release search budget spent; not starting a bypass for %s", bypass_url)
return _fail(search_deadline.deadline_message(), bypass_url)
if status_callback:
status_callback("resolving", "Bypassing protection...")
try:
# A bypass is one long blocking call with no incremental progress, so
# tell the orchestrator up front how long it may legitimately take
# instead of trying to fake activity while it runs. Inside the try so a
# bypasser that fails to load is still reported as a bypasser error.
request_activity_grace(status_callback, _bypass_grace_seconds())
result = get_bypassed_page(bypass_url, selector, cancel_flag)
if result:
return _result(result, bypass_url)
return _fail(
"The protection bypasser returned an empty page — the challenge was "
"not solved. Check that FlareSolverr/the CF bypasser is reachable.",
bypass_url,
)
except network.RateLimitedError as e:
# Not a bypasser malfunction: the host is throttling this IP and a solve
# cannot help. Surface the wait as a plain failure so the search ends cleanly
# instead of looping another minutes-long solve against a 429.
logger.info("Skipping bypass (rate-limited): %s", e)
if status_callback:
try:
status_callback("resolving", "Rate limited, try again shortly")
except _STATUS_CALLBACK_ERRORS:
logger.debug("Rate-limit status callback failed", exc_info=True)
return _fail(str(e), bypass_url)
except ChallengeNotSolvedError as e:
# Not a bypasser malfunction: it ran, and the host answered with something it
# cannot clear - DDoS-Guard's manual CAPTCHA, typically. Must precede the
# generic handler below, whose "the protection bypasser failed" is what sent
# #1292 off to fix a FlareSolverr that was working perfectly.
logger.info("Bypass ran but did not clear the protection: %s", e)
if status_callback:
try:
status_callback("error", str(e))
except _STATUS_CALLBACK_ERRORS:
logger.debug("Unsolved-challenge status callback failed", exc_info=True)
return _fail(str(e), bypass_url)
except _BYPASSER_ERRORS as e:
logger.warning("Bypasser error: %s: %s", type(e).__name__, e)
# Surface the real reason. Without this the caller only sees an empty
# page and the download dies with a generic failure, hiding e.g. a
# FlareSolverr 500 behind a silent wait.
if status_callback and not isinstance(e, BypassCancelledError):
try:
status_callback("error", f"Bypass failed: {type(e).__name__}: {e}")
except _STATUS_CALLBACK_ERRORS:
logger.debug("Bypass error status callback failed", exc_info=True)
if isinstance(e, BypassCancelledError):
# The budget trips the same cancel flag a user's cancel does, so tell them
# apart here - "cancelled" is a confusing thing to read when nobody did.
if search_deadline.expired():
return _fail(search_deadline.deadline_message(), bypass_url)
return _fail("The protection bypass was cancelled.", bypass_url)
return _fail(f"The protection bypasser failed: {type(e).__name__}: {e}", bypass_url)
finally:
release_activity_grace(status_callback)
def _bypass_handoff_allowed() -> bool:
"""Whether a challenge on the current URL may be handed to the bypasser.
allow_bypasser_fallback is honoured for the same reason the 403 path honours it:
callers such as the /dyn/md5/summary fetch behind the details modal pass False
precisely so a best-effort request fails fast instead of holding the UI open for
a minutes-long browser solve.
"""
return allow_bypasser_fallback and _is_cf_bypass_enabled() and not use_bypasser_now
def _purge_clearance(target_url: str) -> None:
"""Drop the host's stored clearance cookies.
Called whenever the protection answered a request that *carried* cookies:
being challenged while presenting them proves they no longer work, so keeping
them only guarantees the same rejection on every later request. Applies to
either bypasser, since both fill the same store.
"""
hostname = urlparse(target_url).hostname or ""
# An empty domain means "clear every host" to the store, so skip the purge
# rather than wipe clearance for sites that are working fine.
if hostname:
cookie_store.clear_cf_cookies(hostname)
def _redirect_loop_handoff(bypass_url: str) -> str | tuple[str, str]:
"""Drop the host's stale clearance cookies, then bypass `bypass_url`.
A `?check=1` loop is how DDoS-Guard answers a clearance cookie that has gone
stale, so the dead cookie has to go before the solve — otherwise it is merged
back over the fresh one on the next request and the loop simply resumes.
"""
_purge_clearance(bypass_url)
return _run_bypasser(bypass_url)
configured_retry = normalize_positive_int(app_config.MAX_RETRY)
retry_limit = (
retry if retry is not None else (configured_retry if configured_retry is not None else 1)
)
selector = selector or network.AAMirrorSelector()
original_url = url
current_url = selector.rewrite(original_url)
use_bypasser_now = use_bypasser
# Survives across attempts so a cookie won once is still presented on later retries.
handshake_cookies: dict[str, str] = {}
handshake_retries = 0
# Last transport error seen, so the exhausted-retries path can name the real
# cause (timeout, connection refused, DNS, ...) instead of a generic message.
last_error: Exception | None = None
for attempt in range(1, retry_limit + 1):
# Check for cancellation before each attempt
if cancel_flag and cancel_flag.is_set():
if search_deadline.expired():
logger.info("Release search budget spent before attempt %s", attempt)
return _fail(search_deadline.deadline_message(), current_url)
logger.info("html_get_page cancelled before attempt %s", attempt)
return _fail("The request was cancelled.", current_url)
return _result("", current_url)
cookies: dict[str, str] = {}
try:
if use_bypasser_now and _is_cf_bypass_enabled():
return _run_bypasser(current_url)
if status_callback:
status_callback("resolving", "Bypassing protection...")
heartbeat_stop = Event()
heartbeat_thread: Thread | None = None
if status_callback:
def _heartbeat() -> None:
# Keep the download "alive" during long bypass operations so the orchestrator
# doesn't flag it as stalled.
if cancel_flag and cancel_flag.is_set():
return
try:
status_callback("resolving", "Bypassing protection...")
except _STATUS_CALLBACK_ERRORS:
return
heartbeat_thread = Thread(
target=_heartbeat, daemon=True, name="BypassHeartbeat"
)
heartbeat_thread.start()
try:
result = get_bypassed_page(current_url, selector, cancel_flag)
return _result(result or "", current_url)
except _BYPASSER_ERRORS as e:
logger.warning("Bypasser error: %s: %s", type(e).__name__, e)
return _result("", current_url)
finally:
heartbeat_stop.set()
if heartbeat_thread:
heartbeat_thread.join(timeout=1)
logger.debug("GET: %s", current_url)
@@ -554,67 +322,12 @@ def html_get_page(
current_url,
proxies=get_proxies(current_url),
timeout=REQUEST_TIMEOUT,
# Handshake cookies win. They were issued by *this* exchange, so by
# definition they are fresher than anything the store holds, and the
# server is waiting to see them echoed back on the very next hop.
# Letting the store overwrite them meant a stored cookie of the same
# name (DDoS-Guard reuses __ddg1_/__ddg2_ for both) was replayed on
# every hop and the freshly issued value never left this process - the
# ?check=1 probe could then never terminate, so every request ended in
# the redirect-loop handoff and paid for a full browser solve.
cookies={**cookies, **handshake_cookies},
cookies=cookies,
headers=headers,
allow_redirects=allow_redirects,
verify=get_ssl_verify(current_url),
)
# Z-Library gates the first hit with a 503 that carries nothing but a
# Set-Cookie; echoing it back yields the 302 to the real page. Without this
# the cookie is dropped and every retry re-runs the same rejected request.
if (
response.status_code == _HTTP_STATUS_SERVICE_UNAVAILABLE
and handshake_retries < _MAX_COOKIE_HANDSHAKE_RETRIES
):
issued = _new_cookies(response, handshake_cookies)
if issued:
handshake_cookies.update(issued)
handshake_retries += 1
logger.debug(
"503 set %s cookie(s); retrying with them: %s",
len(issued),
current_url,
)
continue
# A 503 still serving a challenge is protection, not a busy origin. The
# handshake above has nothing left to echo back, and 503 is in
# RETRYABLE_CODES, so without this the request spends every attempt on
# the same wall: the bypasser is only ever reached from the 403 branch
# and the AA redirect rescues. Gate on the body, not the status, so a
# genuine overloaded-origin 503 keeps its retry path.
if response.status_code == _HTTP_STATUS_SERVICE_UNAVAILABLE:
marker = _response_challenge_marker(response)
if marker and _bypass_handoff_allowed():
if cookies:
# Challenged while presenting clearance means those cookies
# are dead; same reasoning as the 403 branch below.
logger.debug(
"503 challenge with cookies presented; purging: %s", current_url
)
_purge_clearance(current_url)
logger.info(
"503 challenge detected (%s); switching to bypasser: %s",
marker,
current_url,
)
return _run_bypasser(current_url)
if marker:
logger.debug(
"503 challenge (%s) but no bypasser handoff available: %s",
marker,
current_url,
)
if is_aa_url and response.is_redirect:
location = response.headers.get("Location", "")
if not location:
@@ -636,19 +349,13 @@ def html_get_page(
redirect_host,
current_url,
)
return _fail(
f"The configured mirror {current_host} redirected to "
f"{redirect_host}; it may be down or seized. Point MIRROR at "
"a working host or switch to auto mode.",
current_url,
)
return _result("", current_url)
new_url = _try_rotation(original_url, current_url, selector)
if new_url:
current_url = new_url
# Reset per-request state for the new host.
headers = {"User-Agent": DOWNLOAD_HEADERS["User-Agent"]}
handshake_cookies.clear()
is_aa_url = network.should_rotate_dns_for_url(current_url)
allow_redirects = not is_aa_url
redirects_followed = 0
@@ -660,47 +367,12 @@ def html_get_page(
redirect_host,
current_url,
)
return _fail(
"Every Anna's Archive mirror redirected away to a dead host — "
"all configured mirrors are unreachable.",
current_url,
)
return _result("", current_url)
# Same-host redirect (relative or absolute) - follow manually.
# DDoS-Guard gates AA /search behind a cookie probe: the 302 to
# ?check=1 carries Set-Cookie (__ddg*) which must be echoed back on
# the next hop, or the server just re-issues the redirect forever.
issued = _new_cookies(response, handshake_cookies)
if issued:
handshake_cookies.update(issued)
redirects_followed += 1
if redirects_followed > _MAX_REDIRECTS:
# A same-host redirect loop on AA is not a network fault — it is
# how DDoS-Guard presents a handshake that is unsolved, or whose
# clearance cookie has gone stale: /search redirects to
# /search&check=1, which redirects back, indefinitely. Hand it
# straight to the bypasser rather than raising, which would send it
# down the retry path to re-run the whole loop on every attempt
# (10 x 6 = ~60 requests to AA) without ever offering the URL to the
# bypasser. `continue` is no use here either — it would target this
# inner redirect loop rather than the retry branch below.
if _bypass_handoff_allowed():
logger.info(
"Redirect loop detected; switching to bypasser: %s", current_url
)
return _redirect_loop_handoff(current_url)
# No bypasser to hand it to. Every AA mirror shares the challenge,
# so rotating only collects another loop — give up now instead of
# raising and burning the same ~60 requests over the retry budget.
logger.warning(
"Redirect loop and no bypasser available, giving up: %s", current_url
)
return _fail(
"Anna's Archive is behind a protection challenge (endless "
"redirect loop) and no bypasser is enabled to solve it. Enable "
"FlareSolverr/the CF bypasser.",
current_url,
)
_raise_too_many_redirects(f"Too many redirects for {current_url}")
current_url = redirect_url
continue
@@ -710,24 +382,8 @@ def html_get_page(
return _result(response.text, response.url)
except Exception as e:
last_error = e
status = _get_status_code(e)
# The same DDoS-Guard rescue, for the loops the manual AA follower above hands
# back rather than resolving inline — an AA redirect missing its Location
# header. TooManyRedirects carries no status, so the 403 rescue below never
# fires and every retry would re-send the dead cookies. Scoped to the hosts
# whose redirects we follow manually: elsewhere `requests` follows them itself,
# and a loop there is an ordinary misconfiguration that a cookie purge and a
# minutes-long browser solve would be the wrong answer to.
if (
isinstance(e, requests.exceptions.TooManyRedirects)
and network.should_rotate_dns_for_url(current_url)
and _bypass_handoff_allowed()
):
logger.info("Redirect loop detected; switching to bypasser: %s", current_url)
return _redirect_loop_handoff(current_url)
# 403 = Cloudflare/DDoS-Guard protection
if status == _HTTP_STATUS_FORBIDDEN:
# If bypasser fallback is disabled, try mirrors instead
@@ -737,73 +393,38 @@ def html_get_page(
current_url = new_url
continue
logger.warning("403 error, mirrors exhausted: %s", current_url)
return _fail(
"Anna's Archive returned 403 (blocked) and all mirrors are exhausted.",
current_url,
)
return _result("", current_url)
if _is_cf_bypass_enabled() and not use_bypasser_now:
# Before switching to bypasser, check if cookies have become available
# (another concurrent download may have completed bypass and extracted cookies)
parsed = urlparse(current_url)
fresh_cookies = get_cf_cookies_for_domain(parsed.hostname or "")
if fresh_cookies and not cookies and attempt < retry_limit:
# Cookies are now available - retry with cookies before using bypasser.
# Guarded on there being a next attempt: `continue` on the last one
# ends the retry loop and abandons the request without ever offering
# the URL to the bypasser, and MAX_RETRY=1 is the supported setting.
# Same reasoning as the bypasser invocation below.
if fresh_cookies and not cookies:
# Cookies are now available - retry with cookies before using bypasser
logger.debug(
"403 but cookies now available - retrying with cookies: %s",
current_url,
)
continue
if cookies:
# Challenged *while presenting* clearance: those cookies are
# dead. Without this they survive the solve and get merged back
# over the fresh ones, so every later request re-presents a
# known-rejected cookie and is challenged again - the stale
# retry that never ends.
logger.debug("403 with cookies presented; purging: %s", current_url)
_purge_clearance(current_url)
logger.info("403 detected; switching to bypasser: %s", current_url)
# Invoke it here rather than setting use_bypasser_now and continuing.
# The branch that acts on that flag runs at the top of the *next* retry
# attempt, so under the supported MAX_RETRY=1 there is no next attempt
# and the bypasser was never reached — a 403 simply ended the search.
# Same reasoning as the redirect-loop handoffs.
return _run_bypasser(current_url)
if status_callback:
status_callback("resolving", "Bypassing protection...")
use_bypasser_now = True
continue
logger.warning("403 error, giving up: %s", current_url)
return _fail(
"Anna's Archive returned 403 (blocked) and no bypasser is enabled "
"to solve the protection challenge.",
current_url,
)
return _result("", current_url)
# 404 = Not found
if status == _HTTP_STATUS_NOT_FOUND:
logger.warning("404 error: %s", current_url)
return _fail(
f"Anna's Archive returned 404 Not Found for {current_url}.", current_url
)
return _result("", current_url)
# 429 = origin throttling this IP. Arm the per-host backoff so selection and
# the bypasser stop hammering it, then fall through to normal rotation onto a
# mirror that is not (yet) rate-limited.
if status == _HTTP_STATUS_RATE_LIMITED:
network.note_rate_limited(current_url)
# Try mirror/DNS rotation on retryable errors. A failure that proves the
# mirror is unusable also drops it from this process's rotation, so the
# next search does not pay for it again.
fatal_reason = _fatal_mirror_reason(e)
if fatal_reason or _is_retryable_error(e):
new_url = _try_rotation(
original_url, current_url, selector, fatal_reason=fatal_reason
)
# Try mirror/DNS rotation on retryable errors
if _is_retryable_error(e):
new_url = _try_rotation(original_url, current_url, selector)
if new_url:
current_url = new_url
handshake_cookies.clear()
continue
# Retry with backoff
@@ -820,16 +441,7 @@ def html_get_page(
else:
logger.exception("Giving up after %s attempts: %s", retry_limit, current_url)
if last_error is not None:
return _fail(
f"Could not reach Anna's Archive after {retry_limit} attempt(s): "
f"{type(last_error).__name__}: {last_error}",
current_url,
)
return _fail(
"Could not reach Anna's Archive — all mirrors were exhausted without a usable response.",
current_url,
)
return _result("", current_url)
def download_url(
@@ -946,7 +558,6 @@ def download_url(
# Rate limited - skip to next source immediately
# (waiting doesn't help with concurrent downloads hitting the same server)
if status == _HTTP_STATUS_RATE_LIMITED:
network.note_rate_limited(current_url)
logger.info("Rate limited (429) - trying next source")
if status_callback:
status_callback("resolving", "Server busy, trying next")
+57 -429
View File
@@ -3,16 +3,14 @@
import fnmatch
import ipaddress
import socket
import time
import urllib.parse
import urllib.request
from datetime import UTC, datetime, timedelta
from http import HTTPStatus
from socket import AddressFamily, SocketKind
from typing import TYPE_CHECKING, Any, NamedTuple, cast
from typing import TYPE_CHECKING, Any, cast
import dns.resolver
import httpx
import requests
from dns.exception import DNSException
@@ -279,116 +277,6 @@ _current_aa_url_index = 0
_aa_urls: list[str] = [] # Initialized lazily in _initialize_aa_state()
_aa_base_url: str = "" # Current active AA URL
# Mirrors quarantined for this process: domains that are not a working AA mirror at
# all (NXDOMAIN, refused, or a 200 that isn't AA - seized/parked/for-sale domains all
# land here). Kept separate from ordinary failures: a 403 challenge or a 5xx means the
# mirror is alive and rotating away from it only discards the DDoS-Guard clearance we
# hold for it. Deliberately in-memory only, so a restart re-probes everything.
_dead_aa_urls: set[str] = set()
_dead_aa_urls_lock = _RLock()
# Per-host rate-limit backoff. A 429 is the origin throttling *this IP*, not a challenge:
# a DDoS-Guard/Cloudflare solve still renders, so the bypass "succeeds" yet the cleared
# request is rejected again and the throttle is only renewed. The single answer is to
# wait, so a 429 sidelines the host for a growing window - mirror selection and the
# bypasser both skip a cooling-down host until its deadline passes. The wait escalates
# 2 -> 5 -> 10 -> 15 -> 30 minutes each time the host throttles us again *after* we
# already waited a full window out; a host left clear for longer than the top step
# starts the ladder over. Keyed by host so every mirror and source shares one view;
# in-memory only, so a restart starts clean.
_RATE_LIMIT_COOLDOWN_LADDER_SECONDS: tuple[float, ...] = (120.0, 300.0, 600.0, 900.0, 1800.0)
# A host that has been clear this long is treated as a fresh episode: the next 429
# restarts the ladder at 2 minutes rather than resuming the escalation.
_RATE_LIMIT_RESET_AFTER_SECONDS = 1800.0
class _Cooldown(NamedTuple):
"""One host's active rate-limit window and how far up the ladder it has climbed."""
deadline: float # time.monotonic() value at which the wait expires
level: int # index into _RATE_LIMIT_COOLDOWN_LADDER_SECONDS
_host_cooldowns: dict[str, _Cooldown] = {}
_host_cooldowns_lock = _RLock()
class RateLimitedError(Exception):
"""Raised to abandon a request whose host is in a 429 cooldown.
Not a transport failure - nothing is wrong with the network, the origin is
throttling this IP and only time clears it. Callers surface it as a plain failure
rather than retrying or handing the URL to the bypasser.
"""
def _cooldown_key(url: str) -> str:
"""Host a cooldown is keyed by; '' when the URL carries none."""
return (urllib.parse.urlparse(url).hostname or "").lower()
def note_rate_limited(url: str) -> float:
"""Escalate a host's 429 backoff and (re)arm its cooldown; return the wait applied.
The step advances only when a fresh 429 arrives *after* the previous window already
elapsed - i.e. we waited it out and the host throttled us again. A 429 that lands
while the host is still cooling is the same episode: it neither escalates the level
nor shortens the wait. See the ladder note above.
"""
host = _cooldown_key(url)
if not host:
return 0.0
now = time.monotonic()
ladder = _RATE_LIMIT_COOLDOWN_LADDER_SECONDS
with _host_cooldowns_lock:
prev = _host_cooldowns.get(host)
if prev is not None and now < prev.deadline:
# Still inside the current window - same throttling episode, leave it be.
return prev.deadline - now
if prev is None or now - prev.deadline > _RATE_LIMIT_RESET_AFTER_SECONDS:
level = 0
else:
level = min(prev.level + 1, len(ladder) - 1)
wait = ladder[level]
_host_cooldowns[host] = _Cooldown(deadline=now + wait, level=level)
logger.info(
"Rate limited (429): backing off %s for %.0fs (step %d/%d)",
host,
wait,
level + 1,
len(ladder),
)
return wait
def host_cooldown_remaining(url: str) -> float:
"""Seconds left on a host's 429 cooldown; 0.0 when clear or expired.
Leaves an expired record in place: the ladder level it carries is what a later 429
escalates from (or resets, once the clear gap is long enough).
"""
host = _cooldown_key(url)
if not host:
return 0.0
now = time.monotonic()
with _host_cooldowns_lock:
rec = _host_cooldowns.get(host)
if rec is None or rec.deadline <= now:
return 0.0
return rec.deadline - now
def is_host_cooling_down(url: str) -> bool:
"""True while ``url``'s host is inside its 429 cooldown window."""
return host_cooldown_remaining(url) > 0.0
def clear_host_cooldowns() -> None:
"""Forget all rate-limit cooldowns (manual reset / tests)."""
with _host_cooldowns_lock:
_host_cooldowns.clear()
def _ensure_initialized() -> None:
"""Lazy guard so runtime setup happens once and late calls still work."""
@@ -410,24 +298,6 @@ DNS_PROVIDERS = [
("opendns", ["208.67.222.222", "208.67.220.220"], "https://doh.opendns.com/dns-query"),
]
# httpx raises its own hierarchy, which shares no base class with requests', so a
# wireformat failure would escape a requests-only except clause.
_DOH_REQUEST_ERRORS = (OSError, ValueError, requests.RequestException, httpx.HTTPError)
def _first_proxy(proxies: dict[str, str] | None) -> str | None:
"""Pick a single proxy URL from a requests-style mapping, for httpx."""
if not proxies:
return None
return proxies.get("https") or proxies.get("http") or None
# DoH providers that speak RFC 8484 wireformat rather than the (non-standard) JSON API
# Cloudflare and Google popularised. Verified against the live services: both reject a
# ?name=&type= query outright - Quad9 with 505 (it also mandates HTTP/2 per RFC 8484
# section 5.2, which requests cannot speak), OpenDNS with 400 "No valid query received".
_DOH_WIREFORMAT_HOSTS = frozenset({"dns.quad9.net", "doh.opendns.com"})
# Domain patterns that should trigger DNS rotation on failure
DNS_ROTATION_DOMAINS = [
"annas-archive",
@@ -592,16 +462,8 @@ class DoHResolver:
# DNS cache: {(hostname, record_type): (ip_list, timestamp)}
self._cache: dict[tuple[str, str], tuple[list[str], datetime]] = {}
# RFC 8484 providers get a separate transport: they need wireformat, and Quad9
# additionally refuses HTTP/1.1, which requests has no way to upgrade from.
self.use_wireformat = urllib.parse.urlparse(self.base_url).hostname in (
_DOH_WIREFORMAT_HOSTS
)
self._http2_client: Any | None = None
if self.use_wireformat:
self.session.headers.update({"Accept": "application/dns-message"})
elif "google" in self.base_url:
# Different headers based on provider
if "google" in self.base_url:
self.session.headers.update(
{
"Accept": "application/json",
@@ -614,35 +476,6 @@ class DoHResolver:
}
)
def _get_http2_client(self) -> Any:
"""Lazily build the HTTP/2 client used for RFC 8484 providers.
Built on first use so a resolver pointed at a JSON provider never opens an
HTTP/2 connection pool it will not use.
"""
if self._http2_client is None:
self._http2_client = httpx.Client(
http2=True,
timeout=10,
verify=get_ssl_verify(self.base_url),
proxy=_first_proxy(get_proxies(self.base_url)),
)
return self._http2_client
def _resolve_wireformat(self, hostname: str, record_type: str) -> list[str]:
"""Resolve via RFC 8484: base64url query in, DNS message out."""
from shelfmark.download import doh_wireformat
qtype = doh_wireformat.TYPE_AAAA if record_type == "AAAA" else doh_wireformat.TYPE_A
param = doh_wireformat.encode_query_param(hostname, qtype)
response = self._get_http2_client().get(
self.base_url,
params={"dns": param},
headers={"Accept": "application/dns-message"},
)
response.raise_for_status()
return doh_wireformat.decode_answer(response.content, qtype)
def _get_cached(self, hostname: str, record_type: str) -> list[str] | None:
"""Get cached DNS result if still valid."""
key = (hostname, record_type)
@@ -692,37 +525,34 @@ class DoHResolver:
return cached
try:
if self.use_wireformat:
answers = self._resolve_wireformat(hostname, record_type)
else:
params = {"name": hostname, "type": "AAAA" if record_type == "AAAA" else "A"}
params = {"name": hostname, "type": "AAAA" if record_type == "AAAA" else "A"}
response = self.session.get(
self.base_url,
params=params,
proxies=get_proxies(self.base_url),
timeout=10, # Increased from 5s to handle slow network conditions
verify=get_ssl_verify(self.base_url),
)
response.raise_for_status()
response = self.session.get(
self.base_url,
params=params,
proxies=get_proxies(self.base_url),
timeout=10, # Increased from 5s to handle slow network conditions
verify=get_ssl_verify(self.base_url),
)
response.raise_for_status()
data = response.json()
if "Answer" not in data:
logger.warning("DoH resolution failed for %s: %s", hostname, data)
return []
data = response.json()
if "Answer" not in data:
logger.warning("DoH resolution failed for %s: %s", hostname, data)
return []
# Extract IP addresses from the response
answers = [
answer["data"]
for answer in data["Answer"]
if answer.get("type") == (28 if record_type == "AAAA" else 1)
]
# Extract IP addresses from the response
answers = [
answer["data"]
for answer in data["Answer"]
if answer.get("type") == (28 if record_type == "AAAA" else 1)
]
# Cache the result
self._set_cached(hostname, record_type, answers)
# Don't log here - the caller (custom_getaddrinfo) will log the final result
except _DOH_REQUEST_ERRORS as e:
except (OSError, ValueError, requests.RequestException) as e:
logger.warning("DoH resolution failed for %s: %s", hostname, e)
return []
else:
@@ -786,6 +616,8 @@ def create_custom_getaddrinfo(
source: str,
provider_label: str,
res: Sequence[tuple[AddressFamily, SocketKind, int, str, tuple[Any, ...]]],
*,
is_bypass: bool = False,
) -> None:
"""Emit a unified resolver log with the IPs returned.
@@ -793,6 +625,7 @@ def create_custom_getaddrinfo(
source: Description of resolver source
provider_label: Label for the DNS provider
res: Resolution results
is_bypass: If True, log at DEBUG level (for local/IP addresses)
"""
# Skip logging entirely for localhost to reduce noise
@@ -808,7 +641,11 @@ def create_custom_getaddrinfo(
ip = sockaddr[0]
if isinstance(ip, str):
ips.append(ip)
logger.debug("Resolved %s via %s [%s]: %s", host_str, source, provider_label, ips)
msg = f"Resolved {host_str} via {source} [{provider_label}]: {ips}"
if is_bypass:
logger.debug(msg)
else:
logger.info(msg)
# Skip custom resolution for IP addresses, local addresses, or if skip check passes
if (
@@ -818,7 +655,7 @@ def create_custom_getaddrinfo(
):
# Quietly bypass custom resolution for IP/local targets
res = original_getaddrinfo(host, port, family, socket_type, proto, flags)
_log_results("system resolver (bypass)", "system", res)
_log_results("system resolver (bypass)", "system", res, is_bypass=True)
return res
results: list[tuple[AddressFamily, SocketKind, int, str, tuple[Any, ...]]] = []
@@ -1015,99 +852,6 @@ def _init_custom_resolver_internal(servers: list[str]) -> dns.resolver.Resolver:
return custom_resolver
# --- ISP / network DNS interference detection ---------------------------------
# Compare what the (tamperable) system resolver returns for a host against a
# tamper-resistant DoH lookup. Divergent answers are a strong signal the network is
# hijacking or NXDOMAIN-blocking the domain (a common reason AA downloads "work" but
# land on an ISP block page). Used to surface an actionable hint to the user.
_dns_interference_warned: set[str] = set()
_dns_interference_active = False
def _build_detection_doh_resolver() -> DoHResolver | None:
"""Build a throwaway DoH resolver for interference checks (no socket patching).
Honours the DoH provider the user selected (``DNS_PROVIDERS[_current_dns_index]``),
falling back to the first configured provider when none is active. The endpoint is
pinned to the provider's own nameserver IP so resolving the DoH host can't be
redirected by the very DNS layer the check is meant to detect.
"""
if 0 <= _current_dns_index < len(DNS_PROVIDERS):
_name, servers, doh_url = DNS_PROVIDERS[_current_dns_index]
elif DNS_PROVIDERS:
_name, servers, doh_url = DNS_PROVIDERS[0]
else:
return None
server_hostname = urllib.parse.urlparse(doh_url).hostname or ""
if not server_hostname or not servers:
return None
return DoHResolver(doh_url, server_hostname, servers[0])
def detect_dns_interference(hostname: str) -> dict[str, list[str]] | None:
"""Detect network DNS interference by comparing system DNS against DoH.
Returns ``{"system_ips": [...], "doh_ips": [...]}`` when the two resolvers disagree
(no overlapping IPs), otherwise None. No-op for IP literals / local hostnames and
when DoH resolution is unavailable, so it never produces a false positive.
"""
host = (hostname or "").strip().lower()
if not host or _is_ip_address(host) or _is_local_address(host):
return None
resolver = _build_detection_doh_resolver()
if resolver is None:
return None
try:
system_ips = {str(info[4][0]) for info in original_getaddrinfo(host, 443, socket.AF_INET)}
except OSError:
return None
if not system_ips:
return None
doh_ips = {ip for ip in resolver.resolve(host, "A") if ip}
if not doh_ips or (system_ips & doh_ips):
return None
return {"system_ips": sorted(system_ips), "doh_ips": sorted(doh_ips)}
def note_possible_dns_interference(hostname: str) -> bool:
"""Check ``hostname`` for DNS interference, logging an actionable warning once.
Returns True when interference has been detected this session. The check runs at
most once per host to avoid repeated DoH lookups and log spam.
"""
global _dns_interference_active
host = (hostname or "").strip().lower()
if not host or host in _dns_interference_warned:
return _dns_interference_active
_dns_interference_warned.add(host)
result = detect_dns_interference(host)
if not result:
return _dns_interference_active
_dns_interference_active = True
routing_via_doh = _current_dns_index >= 0 and bool(DOH_SERVER)
remedy = (
"Shelfmark is routing this domain through DNS-over-HTTPS to work around it."
if routing_via_doh
else "Enable DNS-over-HTTPS (USE_DOH=true) or set a custom DNS provider to bypass it."
)
logger.warning(
"Possible ISP/network DNS interference for %s: system DNS resolves to %s but DoH "
"resolves to %s. The network appears to be blocking or redirecting this domain. %s",
host,
result["system_ips"],
result["doh_ips"],
remedy,
)
return True
def dns_interference_detected() -> bool:
"""Whether network DNS interference has been detected this session."""
return _dns_interference_active
def init_doh_resolver(doh_server: str = "") -> DoHResolver | None:
"""Initialize DNS over HTTPS resolver."""
server = doh_server or DOH_SERVER
@@ -1178,18 +922,14 @@ def rotate_dns_and_reset_aa() -> bool:
configured_url = _get_configured_aa_url()
if configured_url == "auto":
# Auto mode always resets to the first mirror to restart the cascade. Skip any
# quarantined ones: a new DNS provider cannot revive a parked or seized domain.
with _dead_aa_urls_lock:
restart_urls = [url for url in _aa_urls if url not in _dead_aa_urls] or _aa_urls
if restart_urls:
_aa_base_url = restart_urls[0]
_current_aa_url_index = _aa_urls.index(_aa_base_url)
# Auto mode always resets to the first mirror to restart the cascade
_current_aa_url_index = 0
if _aa_urls:
_aa_base_url = _aa_urls[0]
logger.info("After DNS switch, resetting AA URL to: %s", _aa_base_url)
_save_state(aa_url=_aa_base_url)
else:
_aa_base_url = ""
_current_aa_url_index = 0
logger.info("After DNS switch, AA URL remains unconfigured")
else:
# Keep the user's configured primary mirror (if it exists in the list),
@@ -1359,17 +1099,8 @@ def _initialize_aa_state() -> None:
global _aa_base_url, _current_aa_url_index, _aa_urls
# Build URL list from config
previous_urls = _aa_urls
_aa_urls = _build_aa_urls()
# Drop quarantine decisions only when the mirror list itself changed - they were
# made about a list that no longer applies. This runs on every re-init (settings
# sync, DNS rotation, helper subprocess startup), and clearing unconditionally
# would resurrect a parked mirror mid-session.
if previous_urls != _aa_urls:
with _dead_aa_urls_lock:
_dead_aa_urls.clear()
# Get configured base URL from config
configured_url = _get_configured_aa_url()
@@ -1385,34 +1116,26 @@ def _initialize_aa_state() -> None:
return
if configured_url == "auto":
# Never restore or probe a mirror quarantined this session: re-init happens
# often, and re-electing a parked domain costs a wasted request every time
# (its parking page answers 200, so the probe would happily pick it).
with _dead_aa_urls_lock:
candidates = [url for url in _aa_urls if url not in _dead_aa_urls]
restored = state.get("aa_base_url")
if restored and restored in candidates:
_current_aa_url_index = _aa_urls.index(restored)
_aa_base_url = restored
if state.get("aa_base_url") and state["aa_base_url"] in _aa_urls:
_current_aa_url_index = _aa_urls.index(state["aa_base_url"])
_aa_base_url = state["aa_base_url"]
else:
logger.debug("AA_BASE_URL: auto, checking available urls %s", candidates)
for url in candidates:
logger.debug("AA_BASE_URL: auto, checking available urls %s", _aa_urls)
for i, url in enumerate(_aa_urls):
try:
response = requests.get(
url, proxies=get_proxies(url), timeout=3, verify=get_ssl_verify(url)
)
if response.status_code == HTTPStatus.OK:
_current_aa_url_index = _aa_urls.index(url)
_current_aa_url_index = i
_aa_base_url = url
_save_state(aa_url=_aa_base_url)
break
except (OSError, requests.RequestException) as exc:
logger.debug("Could not reach AA mirror candidate %s: %s", url, exc)
# Also covers the case where every probe failed and the previous base is
# itself quarantined - keeping it would aim the next search at a dead host.
if not _aa_base_url or _aa_base_url == "auto" or _aa_base_url not in candidates:
_aa_base_url = (candidates or _aa_urls)[0]
_current_aa_url_index = _aa_urls.index(_aa_base_url)
if not _aa_base_url or _aa_base_url == "auto":
_aa_base_url = _aa_urls[0]
_current_aa_url_index = 0
elif configured_url not in _aa_urls:
logger.info("AA_BASE_URL set to custom value %s; skipping auto-switch", configured_url)
_aa_base_url = configured_url
@@ -1510,80 +1233,22 @@ def is_aa_auto_mode() -> bool:
def get_available_aa_urls() -> list[str]:
"""Get configured AA URLs (copy), minus any quarantined this process.
Falls back to the full list when every mirror has been quarantined: a wrong
classification must not leave the app with nowhere to search.
"""
"""Get list of configured AA URLs (copy)."""
_ensure_initialized()
with _dead_aa_urls_lock:
alive = [url for url in _aa_urls if url not in _dead_aa_urls]
if not alive and _aa_urls:
logger.warning("All AA mirrors quarantined; retrying the full list")
_dead_aa_urls.clear()
alive = _aa_urls.copy()
# Prefer mirrors that are not serving a 429 cooldown so rotation stops hammering a
# throttled host. When every live mirror is cooling, keep the full live list rather
# than returning nothing: selection must never be left with nowhere to point, and
# the bypasser's fail-fast reports the "all rate-limited" case with a clear error.
breathing = [url for url in alive if not is_host_cooling_down(url)]
return breathing or alive
def _aa_base_for_url(url: str) -> str:
"""Return the configured mirror base that ``url`` belongs to, if any."""
for base in _aa_urls:
if base and url.startswith(base):
return base
return ""
def mark_aa_url_dead(url: str, reason: str) -> bool:
"""Quarantine an AA mirror for the rest of this process.
Only for hard evidence that the host is not a working AA mirror. Transient
failures (403 challenge, 429, 5xx, timeouts) must never come through here -
quarantining a live mirror throws away its bypass clearance.
"""
_ensure_initialized()
base = _aa_base_for_url(url) or url
with _dead_aa_urls_lock:
if base not in _aa_urls or base in _dead_aa_urls:
return False
# Keep at least one mirror in play, even if it is the failing one.
if len([u for u in _aa_urls if u not in _dead_aa_urls]) <= 1:
logger.warning("Not quarantining last remaining AA mirror %s (%s)", base, reason)
return False
_dead_aa_urls.add(base)
logger.warning("Quarantined AA mirror %s for this session: %s", base, reason)
return True
def get_dead_aa_urls() -> set[str]:
"""Return the mirrors quarantined this process (copy)."""
with _dead_aa_urls_lock:
return set(_dead_aa_urls)
def set_aa_url(url: str) -> bool:
"""Set the active AA base URL; returns True if applied."""
_ensure_initialized()
global _aa_base_url, _current_aa_url_index
if url not in _aa_urls:
return False
_current_aa_url_index = _aa_urls.index(url)
_aa_base_url = url
logger.info("Set AA URL to: %s", _aa_base_url)
_save_state(aa_url=_aa_base_url)
return True
return _aa_urls.copy()
def set_aa_url_index(new_index: int) -> bool:
"""Set AA base URL by index in the full configured list; True if applied."""
"""Set AA base URL by index in available list; returns True if applied."""
_ensure_initialized()
global _aa_base_url, _current_aa_url_index
if new_index < 0 or new_index >= len(_aa_urls):
return False
return set_aa_url(_aa_urls[new_index])
_current_aa_url_index = new_index
_aa_base_url = _aa_urls[_current_aa_url_index]
logger.info("Set AA URL to: %s", _aa_base_url)
_save_state(aa_url=_aa_base_url)
return True
class AAMirrorSelector:
@@ -1594,20 +1259,11 @@ class AAMirrorSelector:
def __init__(self) -> None:
"""Initialize mirror state from the current AA configuration."""
# Set by html_get_page at each give-up path so a caller that only sees the
# returned empty page can still report *why* the fetch produced nothing
# (403, 404, redirect loop, bypasser error, mirrors exhausted, ...) instead
# of a blanket "network restricted" guess. None means "no failure recorded".
self.last_failure: str | None = None
self._ensure_fresh_state(reset_attempts=True)
def _ensure_fresh_state(self, *, reset_attempts: bool = False) -> None:
_ensure_initialized()
self.aa_urls = get_available_aa_urls()
# Rotation walks the live mirrors, but rewriting has to recognise every
# configured base: a URL built before a mirror was quarantined still points at
# it, and failing to rewrite would send the retry back to the dead host.
self.all_aa_urls = _aa_urls.copy()
self._index = self._safe_index(get_aa_base_url())
self.current_base = self.aa_urls[self._index] if self.aa_urls else ""
if reset_attempts:
@@ -1620,41 +1276,16 @@ class AAMirrorSelector:
def rewrite(self, url: str) -> str:
"""Replace any known AA base in url with current_base."""
for base in self.all_aa_urls:
for base in self.aa_urls:
if url.startswith(base):
return url.replace(base, self.current_base, 1)
return url
def quarantine_current(self, reason: str) -> bool:
"""Quarantine the mirror this selector is on (hard failures only)."""
if not self.current_base:
return False
dropped = mark_aa_url_dead(self.current_base, reason)
if dropped:
# Rebuild from the surviving mirrors so the dead one is out of the cycle.
self._ensure_fresh_state(reset_attempts=False)
return dropped
def next_mirror_or_rotate_dns(
self, *, allow_dns: bool = True, fatal: bool = False, reason: str = ""
) -> tuple[str | None, str]:
def next_mirror_or_rotate_dns(self, *, allow_dns: bool = True) -> tuple[str | None, str]:
"""Advance to the next mirror or rotate DNS if needed.
``fatal`` marks the current mirror as not-an-AA-mirror (NXDOMAIN, refused, a
200 that isn't AA) and drops it from this process's rotation. Leave it False
for anything the mirror can recover from - a challenge or a 5xx means the host
is alive, and quarantining it would discard its bypass clearance.
Returns (new_base, action) where action is 'mirror', 'dns', or 'exhausted'.
"""
if fatal and self.quarantine_current(reason or "unusable mirror"):
# Quarantining rebuilt the state onto a surviving mirror, so that mirror is
# the next one to try - advancing again here would skip straight past it.
self.attempts_this_dns += 1
if self.current_base and is_aa_auto_mode():
set_aa_url(self.current_base)
return self.current_base, "mirror"
self.attempts_this_dns += 1
max_attempts = len(self.aa_urls) if is_aa_auto_mode() else 1
if self.attempts_this_dns >= max_attempts:
@@ -1667,11 +1298,8 @@ class AAMirrorSelector:
# Mirror is explicitly configured; do not fail over to other mirrors.
return None, "exhausted"
if not self.aa_urls:
return None, "exhausted"
next_index = (self._index + 1) % len(self.aa_urls)
set_aa_url(self.aa_urls[next_index])
set_aa_url_index(next_index)
self._ensure_fresh_state(reset_attempts=False)
return self.current_base, "mirror"
+23 -170
View File
@@ -12,7 +12,7 @@ from concurrent.futures import Future, ThreadPoolExecutor
from email.utils import parseaddr
from pathlib import Path
from threading import Event, Lock
from typing import TYPE_CHECKING, Any
from typing import Any
from shelfmark.core.config import config
from shelfmark.core.logger import setup_logger
@@ -24,7 +24,6 @@ from shelfmark.core.request_helpers import (
)
from shelfmark.core.utils import is_audiobook as check_audiobook
from shelfmark.core.utils import transform_cover_url
from shelfmark.download.activity import parse_activity_grace
from shelfmark.download.fs import run_blocking_io
from shelfmark.download.postprocess.pipeline import is_torrent_source, safe_cleanup_path
from shelfmark.download.postprocess.router import post_process_download
@@ -34,9 +33,6 @@ from shelfmark.release_sources import (
get_source_display_name,
)
if TYPE_CHECKING:
from collections.abc import Iterable
logger = setup_logger(__name__)
_RNG = random.SystemRandom()
@@ -69,16 +65,7 @@ _last_progress_value: dict[str, float] = {}
# De-duplicate status updates (keep-alive updates shouldn't spam clients)
_last_status_event: dict[str, tuple[str, str | None]] = {}
STALL_TIMEOUT = 300 # 5 minutes without progress/status update = stalled
# Absolute deadlines (time.time()) until which stall detection is suppressed for a task.
# Long single-shot operations (protection bypass, etc.) declare their own upper bound via
# `shelfmark.download.activity` instead of faking progress. See set_activity_grace().
_activity_grace: dict[str, float] = {}
# A caller cannot buy immortality: the largest grace any operation may request. Must stay
# above the largest budget any caller can declare (see http._bypass_grace_seconds).
_MAX_ACTIVITY_GRACE_SECONDS = 960.0
COORDINATOR_LOOP_ERROR_RETRY_DELAY = 1.0
# Ceiling for the exponential backoff applied to repeated coordinator loop failures.
_COORDINATOR_LOOP_ERROR_MAX_DELAY = 30.0
_PROGRESS_BROADCAST_START_PERCENT = 1
_PROGRESS_BROADCAST_COMPLETE_PERCENT = 99
_PROGRESS_BROADCAST_MIN_DELTA = 10
@@ -183,19 +170,16 @@ def _build_retry_resolution_fields(
retry_download_url = normalize_optional_text(release_data.get("download_url"))
protocol = normalize_optional_text(release_data.get("protocol"))
source = normalize_optional_text(release_data.get("source"))
retry_source_context: dict[str, Any] = {}
if source is not None:
handler = get_handler(source)
source_retry_fields = handler.build_retry_resolution_fields(release_data)
if "retry_download_url" in source_retry_fields:
retry_download_url = normalize_optional_text(
source_retry_fields.get("retry_download_url")
)
if "retry_download_protocol" in source_retry_fields:
protocol = normalize_optional_text(source_retry_fields.get("retry_download_protocol"))
raw_retry_source_context = source_retry_fields.get("retry_source_context")
if isinstance(raw_retry_source_context, dict):
retry_source_context = dict(raw_retry_source_context)
retry_download_url = (
normalize_optional_text(source_retry_fields.get("retry_download_url"))
or retry_download_url
)
protocol = (
normalize_optional_text(source_retry_fields.get("retry_download_protocol")) or protocol
)
ratio_limit = _optional_number(release_data.get("ratio_limit"))
if ratio_limit is None and config.get("PROWLARR_USE_SEED_PREFERENCES", False):
@@ -218,7 +202,6 @@ def _build_retry_resolution_fields(
),
"retry_ratio_limit": ratio_limit,
"retry_seeding_time_limit_minutes": seeding_time_limit_minutes,
"retry_source_context": retry_source_context,
"can_retry_without_staged_source": True,
}
@@ -264,9 +247,6 @@ def queue_release(
series_name = release_data.get("series_name") or extra.get("series_name")
series_position = release_data.get("series_position") or extra.get("series_position")
subtitle = release_data.get("subtitle") or extra.get("subtitle")
language = release_data.get("language") or extra.get("language")
multi_book = bool(release_data.get("multi_book") or extra.get("multi_book"))
book_plan = _normalize_book_plan(release_data.get("book_plan") or extra.get("book_plan"))
books_output_mode = (
str(config.get("BOOKS_OUTPUT_MODE", "folder", user_id=user_id) or "folder")
@@ -301,9 +281,6 @@ def queue_release(
series_name=series_name,
series_position=series_position,
subtitle=subtitle,
language=language,
multi_book=multi_book or book_plan is not None,
book_plan=book_plan,
search_mode=search_mode,
output_mode=output_mode,
output_args=output_args,
@@ -412,33 +389,6 @@ def can_retry_download_task(
return _has_staged_retry_source(task)
def _normalize_book_plan(value: object) -> list[dict[str, Any]] | None:
"""Keep only well-formed pack books: a title plus a non-empty list of file paths."""
if not isinstance(value, list):
return None
books: list[dict[str, Any]] = []
for entry in value:
if not isinstance(entry, dict):
continue
title = normalize_optional_text(entry.get("title"))
raw_files = entry.get("files")
if title is None or not isinstance(raw_files, list):
continue
files = [f for f in raw_files if isinstance(f, str) and f.strip()]
if not files:
continue
year = entry.get("year")
books.append(
{
"title": title,
"series_position": _optional_number(entry.get("series_position")),
"year": year if isinstance(year, int) and not isinstance(year, bool) else None,
"files": files,
}
)
return books or None
def serialize_task_for_retry(task: DownloadTask) -> dict[str, Any]:
"""Serialize the task state needed for restart-safe retries."""
raw_search_mode = getattr(task, "search_mode", None)
@@ -450,7 +400,6 @@ def serialize_task_for_retry(task: DownloadTask) -> dict[str, Any]:
search_mode = normalized_search_mode or None
raw_output_args = getattr(task, "output_args", None)
raw_retry_source_context = getattr(task, "retry_source_context", None)
return {
"task_id": getattr(task, "task_id", None),
@@ -466,10 +415,7 @@ def serialize_task_for_retry(task: DownloadTask) -> dict[str, Any]:
"series_name": getattr(task, "series_name", None),
"series_position": getattr(task, "series_position", None),
"subtitle": getattr(task, "subtitle", None),
"language": getattr(task, "language", None),
"search_mode": search_mode,
"multi_book": bool(getattr(task, "multi_book", False)),
"book_plan": _normalize_book_plan(getattr(task, "book_plan", None)),
"output_mode": getattr(task, "output_mode", None),
"output_args": dict(raw_output_args) if isinstance(raw_output_args, dict) else {},
"user_id": getattr(task, "user_id", None),
@@ -482,9 +428,6 @@ def serialize_task_for_retry(task: DownloadTask) -> dict[str, Any]:
"retry_expected_hash": getattr(task, "retry_expected_hash", None),
"retry_ratio_limit": getattr(task, "retry_ratio_limit", None),
"retry_seeding_time_limit_minutes": getattr(task, "retry_seeding_time_limit_minutes", None),
"retry_source_context": (
dict(raw_retry_source_context) if isinstance(raw_retry_source_context, dict) else {}
),
"can_retry_without_staged_source": bool(
getattr(task, "can_retry_without_staged_source", True)
),
@@ -510,7 +453,6 @@ def _restore_task_from_retry_payload(payload: object) -> DownloadTask | None:
search_mode = None
output_args = payload.get("output_args")
retry_source_context = payload.get("retry_source_context")
return DownloadTask(
task_id=task_id,
@@ -526,10 +468,7 @@ def _restore_task_from_retry_payload(payload: object) -> DownloadTask | None:
series_name=normalize_optional_text(payload.get("series_name")),
series_position=_optional_number(payload.get("series_position")),
subtitle=normalize_optional_text(payload.get("subtitle")),
language=normalize_optional_text(payload.get("language")),
search_mode=search_mode,
multi_book=bool(payload.get("multi_book", False)),
book_plan=_normalize_book_plan(payload.get("book_plan")),
output_mode=normalize_optional_text(payload.get("output_mode")),
output_args=dict(output_args) if isinstance(output_args, dict) else {},
user_id=normalize_positive_int(payload.get("user_id")),
@@ -544,9 +483,6 @@ def _restore_task_from_retry_payload(payload: object) -> DownloadTask | None:
retry_seeding_time_limit_minutes=_optional_positive_int(
payload.get("retry_seeding_time_limit_minutes")
),
retry_source_context=(
dict(retry_source_context) if isinstance(retry_source_context, dict) else {}
),
can_retry_without_staged_source=bool(payload.get("can_retry_without_staged_source", True)),
)
@@ -697,17 +633,6 @@ def _download_task(task_id: str, cancel_flag: Event) -> str | None:
update_download_progress(task_id, progress)
def status_callback(status: str, message: str | None = None) -> None:
# Liveness hint from a long single-shot operation, not a user-visible status.
# Handled here so it never reaches update_download_status (which dedupes status
# transitions on purpose). See shelfmark.download.activity.
grace = parse_activity_grace(status, message)
if grace is not None:
if grace > 0:
set_activity_grace(task_id, grace)
else:
clear_activity_grace(task_id)
return
status_key = status.lower()
if status_key == "error":
_capture_task_error(
@@ -945,7 +870,6 @@ def _cleanup_progress_tracking(task_id: str) -> None:
_last_activity.pop(task_id, None)
_last_progress_value.pop(task_id, None)
_last_status_event.pop(task_id, None)
_activity_grace.pop(task_id, None)
def _finalize_download_failure(task_id: str) -> None:
@@ -993,66 +917,6 @@ def _process_single_download(task_id: str, cancel_flag: Event) -> None:
ws_manager.broadcast_status_update(queue_status())
def set_activity_grace(book_id: str, seconds: float) -> None:
"""Suppress stall detection for `book_id` for up to `seconds` from now.
For long single-shot operations that cannot report incremental progress (protection
bypass being the motivating case). The grace is a single absolute deadline computed
once, so it cannot be extended into immortality by a keep-alive that carries no real
liveness information - an operation that hangs forever is still cancelled once its
declared budget expires.
Deliberately touches neither the queue nor the WebSocket: this is a liveness
assertion, not a user-visible status transition.
"""
grace = _config_float(seconds, 0.0)
grace = min(max(grace, 0.0), _MAX_ACTIVITY_GRACE_SECONDS)
with _progress_lock:
_activity_grace[book_id] = time.time() + grace
def clear_activity_grace(book_id: str) -> None:
"""Drop any activity grace for `book_id` and count this moment as activity.
Resetting `_last_activity` means a nested or abandoned grace degrades to a fresh
full STALL_TIMEOUT window rather than an immediate stall.
"""
with _progress_lock:
_activity_grace.pop(book_id, None)
_last_activity[book_id] = time.time()
def _find_stalled_tasks(task_ids: Iterable[str], now: float) -> list[str]:
"""Return the task ids with no activity inside STALL_TIMEOUT and no active grace.
Holds `_progress_lock` for dict reads only - never call into `book_queue` from here,
see _cancel_stalled_task().
"""
stalled: list[str] = []
with _progress_lock:
for task_id in task_ids:
last_active = _last_activity.get(task_id, now)
deadline = max(last_active + STALL_TIMEOUT, _activity_grace.get(task_id, 0.0))
if now > deadline:
stalled.append(task_id)
return stalled
def _cancel_stalled_task(task_id: str) -> None:
"""Cancel a stalled download.
Must be called WITHOUT `_progress_lock` held. `book_queue.cancel_download` runs the
terminal-status hooks, which reach a sqlite write that gevent does not patch; holding
the progress lock across that blocks the hub and every other download worker.
"""
logger.warning("Download stalled for %s, cancelling", task_id)
book_queue.cancel_download(task_id)
book_queue.update_status_message(
task_id,
f"Download stalled (no activity for {STALL_TIMEOUT}s)",
)
def concurrent_download_loop() -> None:
"""Run the main concurrent download coordinator."""
max_workers = normalize_positive_int(config.MAX_CONCURRENT_DOWNLOADS) or 1
@@ -1062,7 +926,6 @@ def concurrent_download_loop() -> None:
with ThreadPoolExecutor(max_workers=max_workers, thread_name_prefix="Download") as executor:
active_futures: dict[Future, tuple[str, Event]] = {} # Track active download futures
stalled_tasks: set[str] = set() # Track tasks already cancelled due to stall
consecutive_errors = 0
while True:
try:
@@ -1113,14 +976,19 @@ def concurrent_download_loop() -> None:
# Check for stalled downloads (no activity in STALL_TIMEOUT seconds)
current_time = time.time()
candidates = [
task_id
for _future, (task_id, _cancel_flag) in list(active_futures.items())
if task_id not in stalled_tasks
]
for task_id in _find_stalled_tasks(candidates, current_time):
_cancel_stalled_task(task_id)
stalled_tasks.add(task_id)
with _progress_lock:
for _future, (task_id, _cancel_flag) in list(active_futures.items()):
if task_id in stalled_tasks:
continue
last_active = _last_activity.get(task_id, current_time)
if current_time - last_active > STALL_TIMEOUT:
logger.warning("Download stalled for %s, cancelling", task_id)
book_queue.cancel_download(task_id)
book_queue.update_status_message(
task_id,
f"Download stalled (no activity for {STALL_TIMEOUT}s)",
)
stalled_tasks.add(task_id)
# Start new downloads if we have capacity
while len(active_futures) < max_workers:
@@ -1143,24 +1011,9 @@ def concurrent_download_loop() -> None:
# Brief sleep to prevent busy waiting
time.sleep(main_loop_sleep_time)
consecutive_errors = 0
# This loop is the only thing driving the download queue; if it exits, nothing
# is ever picked up again and the app looks healthy while doing nothing (#823,
# #1166). A narrow exception list let gevent's LoopExit and friends through, so
# catch everything short of BaseException - GreenletExit and gevent.Timeout must
# still propagate, and the tests' loop-stopping sentinels derive from
# BaseException for exactly this reason.
except Exception as e: # noqa: BLE001 - coordinator loop must never die
consecutive_errors += 1
except (AttributeError, KeyError, OSError, RuntimeError, TypeError, ValueError) as e:
logger.error_trace("Download coordinator loop error: %s", e)
# Back off when the failure is persistent so we don't spin at 1Hz forever,
# but keep the first delay unchanged for a normal transient blip.
time.sleep(
min(
COORDINATOR_LOOP_ERROR_RETRY_DELAY * 2 ** min(consecutive_errors - 1, 5),
_COORDINATOR_LOOP_ERROR_MAX_DELAY,
)
)
time.sleep(COORDINATOR_LOOP_ERROR_RETRY_DELAY)
# Download coordinator thread (started explicitly via start())
+1 -11
View File
@@ -105,7 +105,6 @@ def process_folder_output(
maybe_run_custom_script,
prepare_output_files,
record_step,
resolve_book_groups,
transfer_book_files,
)
@@ -206,7 +205,6 @@ def process_folder_output(
is_torrent=is_torrent,
preserve_source=preserve_source,
organization_mode=plan.organization_mode,
source_root=source_path,
)
if error:
@@ -261,15 +259,7 @@ def process_folder_output(
prepared.cleanup_paths,
)
pack_groups = resolve_book_groups(
task, prepared.files, organization_mode=plan.organization_mode
)
if pack_groups is not None:
message = f"Complete ({len(pack_groups)} books, {len(final_paths)} files)"
elif len(final_paths) == 1:
message = "Complete"
else:
message = f"Complete ({len(final_paths)} files)"
message = "Complete" if len(final_paths) == 1 else f"Complete ({len(final_paths)} files)"
status_callback("complete", message)
return str(final_paths[0])
@@ -229,7 +229,6 @@ def _build_custom_script_payload(
"series_name": context.task.series_name,
"series_position": context.task.series_position,
"subtitle": context.task.subtitle,
"language": context.task.language,
"original_download_path": context.task.original_download_path,
},
"output": {
+4 -33
View File
@@ -2,7 +2,7 @@
from __future__ import annotations
import contextlib
import uuid
from typing import TYPE_CHECKING
from shelfmark.core.logger import setup_logger
@@ -12,11 +12,7 @@ from shelfmark.core.utils import (
from shelfmark.core.utils import (
is_audiobook as check_audiobook,
)
from shelfmark.download.fs import (
clear_delete_denied,
mark_delete_denied,
run_blocking_io,
)
from shelfmark.download.fs import run_blocking_io
from shelfmark.download.permissions_debug import log_path_permission_context
from shelfmark.release_sources import get_source
@@ -28,8 +24,6 @@ if TYPE_CHECKING:
logger = setup_logger("shelfmark.download.postprocess.pipeline")
_WRITE_PROBE_NAME = ".shelfmark_write_test.tmp"
def validate_destination(
destination: Path, status_callback: Callable[[str, str | None], None]
@@ -46,20 +40,16 @@ def validate_destination(
status_callback("error", f"Destination is not a directory: {destination}")
return False
created_by_us = False
if not destination_exists:
try:
run_blocking_io(destination.mkdir, parents=True, exist_ok=True)
created_by_us = True
except (OSError, PermissionError) as exc:
log_path_permission_context("destination_create", destination)
logger.warning("Cannot create destination: %s (%s)", destination, exc)
status_callback("error", f"Cannot create destination: {destination} ({exc})")
return False
# Stable name: on shares that refuse deletes the probe file cannot be cleaned
# up, so reusing one name bounds the leftovers at a single hidden file.
test_path = destination / _WRITE_PROBE_NAME
test_path = destination / f".shelfmark_write_test_{uuid.uuid4().hex}.tmp"
try:
test_content = (
@@ -67,33 +57,14 @@ def validate_destination(
"It should've been automatically deleted. Feel free to delete it.\n"
)
run_blocking_io(test_path.write_text, test_content)
run_blocking_io(test_path.unlink, missing_ok=True)
except OSError as exc:
logger.debug("Destination write probe path: %s", test_path)
log_path_permission_context("destination_write_probe", destination)
logger.warning("Destination not writable: %s (%s)", destination, exc)
status_callback("error", f"Destination not writable: {destination} ({exc})")
if created_by_us:
with contextlib.suppress(OSError):
run_blocking_io(destination.rmdir)
return False
try:
run_blocking_io(test_path.unlink, missing_ok=True)
except OSError as exc:
# Writable but not deletable, e.g. a Synology share with "Delete
# subfolders and files" unticked. Not fatal: record it so transfers write
# files in place instead of publishing a temp file via rename.
mark_delete_denied(destination)
logger.warning(
"Destination %s is writable but refuses deletes (%s); leaving probe file %s "
"behind and writing files in place",
destination,
exc,
test_path.name,
)
else:
clear_delete_denied(destination)
return True
-423
View File
@@ -1,423 +0,0 @@
"""Multi-book ("pack") release planning.
A pack is one release that contains several books: a whole-series torrent with one
subfolder per book, or a flat folder of `Series 1.0 - Title.m4b` files. The same
planning rules serve pre-download inspection (the file list comes from the release
source) and post-processing (the file list comes from disk), so what the user
approved in the modal is what gets filed.
"""
from __future__ import annotations
import os
import re
from dataclasses import dataclass
from pathlib import Path, PurePosixPath
from shelfmark.core.utils import AUDIOBOOK_FORMATS
# m4b/m4a hold a whole audiobook in one file; every other audio format (mp3, flac, ...) is
# chaptered - many files make up one book. Ebook formats are always one file per book, so
# only chaptered *audio* matters here. A flat folder is split one-book-per-file only when
# none of its files are chaptered audio: a bare list of `01 - Chapter.mp3` tracks is a
# single chaptered audiobook, not a pack of books.
_SINGLE_FILE_AUDIO_CONTAINERS = frozenset({"m4b", "m4a"})
_CHAPTERED_AUDIO_EXTENSIONS = frozenset(AUDIOBOOK_FORMATS) - _SINGLE_FILE_AUDIO_CONTAINERS
_YEAR_SUFFIX_RE = re.compile(r"\s*\(\s*(?P<year>\d{4})\s*\)\s*$")
_SERIES_MARKER_RE = re.compile(
r"""
^\s*
(?:
\[\s*\#?(?P<bracket>\d+(?:\.\d+)?)\s*\] # [03] / [#3]
| \#(?P<hash>\d+(?:\.\d+)?) # #3
| book\.?\s*(?P<book>\d+(?:\.\d+)?) # Book 3 / Book. 03
| (?P<plain>\d+(?:\.\d+)?)(?=[\s\-:.]) # 03 - / 1.0 - / 3.
)
\s*(?:[-:.]\s*)?
""",
re.IGNORECASE | re.VERBOSE,
)
_SEPARATOR_CHARS = " \t-_:."
# "Gods of Risk 2.5 - Gods of Risk": the title repeated on both sides of the position.
_REPEATED_TITLE_RE = re.compile(
r"^(?P<left>.+?)\s+(?P<position>\d+(?:\.\d+)?)\s*[-:\u2013]\s*(?P<right>.+)$"
)
_SERIES_LABEL_WORDS = r"(?:novella|novellas|short\s+story|short|story|novel)"
# "Uncrowned Cradle, Book 7" / "Reaper Cradle, Volume 10" / "Wintersteel (Cradle, Book 8)":
# an explicit word marks the position at the END of the name. A bare trailing number
# is deliberately not matched — "Title - 02" is a chapter, not a series position.
_TRAILING_MARKER_RE = re.compile(
r"""
[\s,\-:\u2013(]*
(?:book|volume|vol\.?)\s*\#?(?P<position>\d+(?:\.\d+)?)
\s*\)?\s*$
""",
re.IGNORECASE | re.VERBOSE,
)
# AudiobookBay renders a file inside a folder as "<folder> <file>" with no separator,
# so a pack row reads "Author - Title Series, Book 1 Title Series, Book 1".
_GLUED_FOLDER_RE = re.compile(
r"^(?P<prefix>.+?\s[-\u2013]\s)?(?P<core>.+?)\s+(?P=core)$", re.IGNORECASE
)
@dataclass(frozen=True)
class PackFile:
"""One file inside a release, path relative to the release root."""
path: str
size: int | None = None
@dataclass(frozen=True)
class PackBook:
"""One book split out of a pack, files as release-relative paths."""
title: str
series_position: float | None
year: int | None
files: list[str]
@dataclass(frozen=True)
class PackPlan:
books: list[PackBook]
ignored: list[str]
@property
def is_pack(self) -> bool:
return len(self.books) > 1
@dataclass(frozen=True)
class BookGroup:
"""One book's on-disk files, ready for transfer."""
title: str
series_position: float | None
year: int | None
files: list[Path]
def _strip_series_name(name: str, series_name: str | None) -> str:
if not series_name:
return name
prefix = series_name.strip()
if not prefix or not name.lower().startswith(prefix.lower()):
return name
remainder = name[len(prefix) :]
if remainder and remainder[0].isalnum():
return name
return remainder.lstrip(_SEPARATOR_CHARS)
def _strip_series_label(work: str, series_name: str | None) -> str:
"""Drop a leading "An <Series> Novella - " style label that some packs prepend."""
if not series_name:
return work
# "The Expanse" is labelled "An Expanse Novella", so match without the article.
core = re.sub(r"^(?:the|an?)\s+", "", series_name.strip(), flags=re.IGNORECASE)
if not core:
return work
pattern = re.compile(
rf"^(?:an?\s+|the\s+)?{re.escape(core)}\s+{_SERIES_LABEL_WORDS}\s*[-:\u2013]\s*",
re.IGNORECASE,
)
return pattern.sub("", work, count=1)
def _collapse_glued_folder(name: str) -> str:
match = _GLUED_FOLDER_RE.match(name)
if not match:
return name
prefix = match.group("prefix") or ""
core = match.group("core")
# "Author - X X" → "Author - X" (the folder carried the author, the file did not).
return (prefix + core).strip()
def _strip_author_name(name: str, author_name: str | None) -> str:
"""Drop a leading "Author - " (packs are often filed as `Author - Title`)."""
if not author_name:
return name
prefix = author_name.strip()
if not prefix or not name.lower().startswith(prefix.lower()):
return name
remainder = name[len(prefix) :]
stripped = remainder.lstrip(_SEPARATOR_CHARS + "\u2013")
if stripped == remainder: # no separator after the author: part of the title
return name
return stripped
def _strip_trailing_series_name(work: str, series_name: str | None) -> str:
"""Drop a trailing series name left behind by a trailing position marker."""
if not series_name:
return work
suffix = series_name.strip()
if not suffix or not work.lower().endswith(suffix.lower()):
return work
remainder = work[: -len(suffix)]
stripped = remainder.rstrip(_SEPARATOR_CHARS + ",(\u2013")
if not stripped or stripped == remainder:
return work
return stripped
def parse_pack_book_name(
name: str, *, series_name: str | None, author_name: str | None = None
) -> tuple[str, float | None, int | None]:
"""Split a book folder/file-stem name into (title, series position, year).
Strips a leading series name, a leading position marker (`Book 3 - `, `03 - `,
`1.0 - `, `3. `, `[03] `, `#3 `) and a trailing `(YYYY)`. Also understands a
trailing marker (`Title Series, Book 3`, `Title (Series, Volume 3)`), a leading
`Author - `, and AudiobookBay's glued `<folder> <file>` names. Returns the name
unchanged with no position/year when nothing would be left of the title.
"""
work = _collapse_glued_folder(name.strip())
work = _strip_author_name(work, author_name)
work = _strip_series_name(work, series_name)
year: int | None = None
year_match = _YEAR_SUFFIX_RE.search(work)
if year_match:
year = int(year_match.group("year"))
work = work[: year_match.start()]
position: float | None = None
repeated = _REPEATED_TITLE_RE.match(work.strip())
if (
repeated
and repeated.group("left").strip().lower() == repeated.group("right").strip().lower()
):
return repeated.group("right").strip(), float(repeated.group("position")), year
marker = _SERIES_MARKER_RE.match(work)
if marker:
raw = (
marker.group("bracket")
or marker.group("hash")
or marker.group("book")
or marker.group("plain")
)
position = float(raw)
work = work[marker.end() :]
else:
trailing = _TRAILING_MARKER_RE.search(work)
if trailing and trailing.start() > 0:
position = float(trailing.group("position"))
work = _strip_trailing_series_name(work[: trailing.start()], series_name)
work = _strip_series_label(work, series_name)
title = work.strip().strip(_SEPARATOR_CHARS).strip()
if not title:
return name, None, None
return title, position, year
def _book_from_name(
name: str, files: list[str], series_name: str | None, author_name: str | None = None
) -> PackBook:
title, position, year = parse_pack_book_name(
name, series_name=series_name, author_name=author_name
)
return PackBook(title=title, series_position=position, year=year, files=files)
def _common_root_parts(paths: list[PurePosixPath]) -> tuple[str, ...]:
parents = [p.parent.parts for p in paths]
common: list[str] = []
for parts in zip(*parents, strict=False):
if len(set(parts)) != 1:
break
common.append(parts[0])
return tuple(common)
def plan_pack(
files: list[PackFile],
*,
supported_extensions: set[str],
series_name: str | None,
author_name: str | None = None,
root_depth: int | None = None,
) -> PackPlan:
"""Group a release's file list into books.
Files in a subfolder (relative to the common root) group by that subfolder. Files
directly in the root split one-book-per-file only when at least two of them carry
a series position in their names; otherwise they are one book (a chaptered
audiobook, e.g. `01.mp3`, `02.mp3`). `root_depth` fixes how many leading path
components form the root instead of deriving it from the files' common parent.
"""
supported = {ext.lower().lstrip(".") for ext in supported_extensions}
book_files: list[PurePosixPath] = []
ignored: list[str] = []
for pack_file in files:
rel = PurePosixPath(pack_file.path.replace("\\", "/").lstrip("./"))
if rel.suffix.lower().lstrip(".") in supported:
book_files.append(rel)
else:
ignored.append(pack_file.path)
if not book_files:
return PackPlan(books=[], ignored=ignored)
root_parts = (
_common_root_parts(book_files) if root_depth is None else book_files[0].parts[:root_depth]
)
depth = len(root_parts)
root_files: list[PurePosixPath] = []
folders: dict[str, list[str]] = {}
for rel in book_files:
remainder = rel.parts[depth:]
if len(remainder) > 1:
folders.setdefault(remainder[0], []).append(str(rel))
else:
root_files.append(rel)
books: list[PackBook] = []
if root_files:
parsed = [
parse_pack_book_name(f.stem, series_name=series_name, author_name=author_name)
for f in root_files
]
positions = {p[1] for p in parsed if p[1] is not None}
titles = {p[0].strip().lower() for p in parsed if p[0]}
one_book_per_file = all(
rel.suffix.lower().lstrip(".") not in _CHAPTERED_AUDIO_EXTENSIONS for rel in root_files
)
# Split a flat folder into a book per file only with real evidence of distinct
# books: two or more series positions, more than one title, and no chaptered audio
# (a bare list of `01 - Chapter.mp3` tracks is one book, not a pack).
if len(positions) >= 2 and len(titles) >= 2 and one_book_per_file:
books.extend(
PackBook(title=title, series_position=position, year=year, files=[str(f)])
for f, (title, position, year) in zip(root_files, parsed, strict=True)
)
elif len(root_files) == 1:
books.append(
_book_from_name(root_files[0].stem, [str(root_files[0])], series_name, author_name)
)
else:
group_name = root_parts[-1] if root_parts else ""
books.append(
_book_from_name(group_name, [str(f) for f in root_files], series_name, author_name)
)
books.extend(
_book_from_name(folder, paths, series_name, author_name)
for folder, paths in folders.items()
)
return PackPlan(books=books, ignored=ignored)
def _relative_paths(
book_files: list[Path], root: Path | None = None
) -> tuple[Path, dict[Path, str]]:
if root is None:
root = Path(os.path.commonpath([str(f.parent) for f in book_files]))
return root, {f: f.relative_to(root).as_posix() for f in book_files}
def group_files_into_books(
book_files: list[Path],
*,
series_name: str | None,
author_name: str | None = None,
root: Path | None = None,
) -> list[BookGroup]:
"""Heuristically split on-disk files into books (see `plan_pack`).
`root` pins the release root when grouping a subset of a larger file set.
"""
if not book_files:
return []
_root, rel_by_path = _relative_paths(book_files, root)
path_by_rel = {rel: path for path, rel in rel_by_path.items()}
extensions = {f.suffix.lower().lstrip(".") for f in book_files}
plan = plan_pack(
[PackFile(rel) for rel in rel_by_path.values()],
supported_extensions=extensions,
series_name=series_name,
author_name=author_name,
root_depth=None if root is None else 0,
)
return [
BookGroup(
title=book.title,
series_position=book.series_position,
year=book.year,
files=[path_by_rel[rel] for rel in book.files],
)
for book in plan.books
]
def match_plan_to_files(
plan: list[PackBook],
book_files: list[Path],
*,
series_name: str | None = None,
author_name: str | None = None,
) -> list[BookGroup]:
"""Apply an approved plan to on-disk files.
Files match by release-relative path first, then by basename (archive extraction
and client save paths can shift the root), then by the on-disk basename being a
suffix of the planned name (sources that glue folder and file names together).
Book files the plan does not mention fall back to heuristic grouping so nothing
is silently dropped.
"""
if not book_files:
return []
root, rel_by_path = _relative_paths(book_files)
by_rel = {rel: path for path, rel in rel_by_path.items()}
by_name: dict[str, list[Path]] = {}
for path in book_files:
by_name.setdefault(path.name, []).append(path)
claimed: set[Path] = set()
groups: list[BookGroup] = []
for book in plan:
matched: list[Path] = []
for wanted in book.files:
wanted_rel = wanted.replace("\\", "/").lstrip("./")
candidate = by_rel.get(wanted_rel)
if candidate is None:
candidates = [
p for p in by_name.get(PurePosixPath(wanted_rel).name, []) if p not in claimed
]
candidate = candidates[0] if candidates else None
if candidate is None:
wanted_name = PurePosixPath(wanted_rel).name.lower()
candidates = [
p
for p in book_files
if p not in claimed and wanted_name.endswith(p.name.lower())
]
candidate = candidates[0] if len(candidates) == 1 else None
if candidate is not None and candidate not in claimed:
claimed.add(candidate)
matched.append(candidate)
if matched:
groups.append(
BookGroup(
title=book.title,
series_position=book.series_position,
year=book.year,
files=matched,
)
)
unmatched = [p for p in book_files if p not in claimed]
if unmatched:
groups.extend(
group_files_into_books(
unmatched, series_name=series_name, author_name=author_name, root=root
)
)
return groups
@@ -40,7 +40,6 @@ from .transfer import (
build_metadata_dict,
is_torrent_source,
process_directory,
resolve_book_groups,
resolve_hardlink_source,
should_hardlink,
transfer_book_files,
@@ -81,7 +80,6 @@ __all__ = [
"process_directory",
"record_step",
"resolve_custom_script_target",
"resolve_book_groups",
"resolve_hardlink_source",
"run_custom_script",
"safe_cleanup_path",
+1 -1
View File
@@ -52,7 +52,7 @@ def get_file_organization(*, is_audiobook: bool) -> str:
"""Get the file organization mode for the content type."""
key = "FILE_ORGANIZATION_AUDIOBOOK" if is_audiobook else "FILE_ORGANIZATION"
mode = _config_text(core_config.config.get(key, "rename")).strip().lower()
return mode if mode in ("none", "rename", "rename_and_group", "organize") else "rename"
return mode if mode in ("none", "rename", "organize") else "rename"
def get_template(*, is_audiobook: bool, organization_mode: str) -> str:
+2 -3
View File
@@ -7,7 +7,6 @@ from pathlib import Path
from typing import TYPE_CHECKING
from shelfmark.core.logger import setup_logger
from shelfmark.core.utils import AUDIOBOOK_FORMATS
from shelfmark.core.utils import is_audiobook as check_audiobook
from shelfmark.download.archive import ArchiveExtractionError, extract_archive, is_archive
from shelfmark.download.fs import run_blocking_io
@@ -143,7 +142,7 @@ def scan_directory_tree(
is_audiobook = check_audiobook(content_type)
if is_audiobook:
trackable_exts = {f".{fmt}" for fmt in AUDIOBOOK_FORMATS}
trackable_exts = {".m4b", ".mp3", ".m4a", ".flac", ".ogg", ".wma", ".aac", ".wav"}
else:
trackable_exts = {
".pdf",
@@ -368,7 +367,7 @@ def collect_staged_files(
is_audiobook = check_audiobook(task.content_type)
if is_audiobook:
trackable_exts = {f".{fmt}" for fmt in AUDIOBOOK_FORMATS}
trackable_exts = {".m4b", ".mp3", ".m4a", ".flac", ".ogg", ".wma", ".aac", ".wav"}
else:
trackable_exts = {
".pdf",
+1 -140
View File
@@ -2,7 +2,6 @@
from __future__ import annotations
import dataclasses
import os
from pathlib import Path
from typing import TYPE_CHECKING
@@ -13,12 +12,10 @@ from shelfmark.core.naming import (
assign_part_numbers,
build_library_path,
derive_primary_title,
normalize_language_code,
parse_naming_template,
sanitize_filename,
)
from shelfmark.core.utils import is_audiobook as check_audiobook
from shelfmark.download.archive import is_archive
from shelfmark.download.fs import (
atomic_copy,
atomic_hardlink,
@@ -27,7 +24,6 @@ from shelfmark.download.fs import (
)
from shelfmark.download.postprocess.policy import get_file_organization, get_template
from .packs import BookGroup, PackBook, group_files_into_books, match_plan_to_files
from .scan import collect_directory_files, scan_directory_tree
from .types import TransferPlan
from .workspace import safe_cleanup_path
@@ -67,7 +63,6 @@ def build_metadata_dict(task: DownloadTask) -> dict:
"Year": task.year,
"Series": task.series_name,
"SeriesPosition": task.series_position,
"Language": normalize_language_code(task.language),
"User": task.username,
}
@@ -163,24 +158,6 @@ def _transfer_single_file(
return atomic_move(source_path, dest_path, max_attempts=max_attempts), "move"
def _group_folder_name(source_root: Path | None) -> str:
"""Name the folder a grouped multi-file audiobook is transferred into.
A directory names the group directly. A file cannot hold several book files
on its own, so a non-directory source that produced more than one means
`collect_staged_files` extracted an archive: the stem is the release name and
the suffix is packaging, which is why `Book.zip` groups into `Book/` rather
than `Book.zip/` or, worse, not at all.
"""
if source_root is None:
return ""
if run_blocking_io(source_root.is_dir):
return sanitize_filename(source_root.name)
if is_archive(source_root):
return sanitize_filename(source_root.stem)
return ""
def transfer_book_files(
book_files: list[Path],
destination: Path,
@@ -190,7 +167,6 @@ def transfer_book_files(
is_torrent: bool,
preserve_source: bool = False,
organization_mode: str | None = None,
source_root: Path | None = None,
) -> tuple[list[Path], str | None, dict[str, int]]:
"""Transfer discovered book files into their final destination layout."""
if not book_files:
@@ -198,19 +174,6 @@ def transfer_book_files(
is_audiobook = check_audiobook(task.content_type)
organization_mode = organization_mode or get_file_organization(is_audiobook=is_audiobook)
groups = resolve_book_groups(task, book_files, organization_mode=organization_mode)
if groups is not None:
return _transfer_book_groups(
groups,
destination,
task,
use_hardlink=use_hardlink,
is_torrent=is_torrent,
preserve_source=preserve_source,
organization_mode=organization_mode,
)
max_attempts = _max_attempts_for_batch(len(book_files))
final_paths: list[Path] = []
@@ -273,13 +236,6 @@ def transfer_book_files(
return final_paths, None, op_counts
transfer_destination = destination
if is_audiobook and len(book_files) > 1 and organization_mode == "rename_and_group":
source_folder = _group_folder_name(source_root)
if source_folder:
transfer_destination = destination / source_folder
run_blocking_io(transfer_destination.mkdir, parents=True, exist_ok=True)
for book_file in book_files:
if len(book_files) == 1 and organization_mode != "none":
if not task.format:
@@ -298,7 +254,7 @@ def transfer_book_files(
else:
filename = book_file.name
dest_path = transfer_destination / filename
dest_path = destination / filename
final_path, op = _transfer_single_file(
book_file,
dest_path,
@@ -314,101 +270,6 @@ def transfer_book_files(
return final_paths, None, op_counts
def resolve_book_groups(
task: DownloadTask,
book_files: list[Path],
*,
organization_mode: str,
) -> list[BookGroup] | None:
"""Split a multi-book pack into per-book groups, or None to file as one book.
An approved `book_plan` wins; a bare `multi_book` flag falls back to heuristic
grouping. Organization `none` keeps files as-is, and a split that yields a single
group is not a pack at all.
"""
if organization_mode == "none" or not (task.book_plan or task.multi_book):
return None
if task.book_plan:
plan = [
PackBook(
title=str(entry.get("title") or ""),
series_position=entry.get("series_position"),
year=entry.get("year"),
files=list(entry.get("files") or []),
)
for entry in task.book_plan
if isinstance(entry, dict)
]
groups = match_plan_to_files(
plan, book_files, series_name=task.series_name, author_name=task.author
)
else:
groups = group_files_into_books(
book_files, series_name=task.series_name, author_name=task.author
)
return groups if len(groups) > 1 else None
def _transfer_book_groups(
groups: list[BookGroup],
destination: Path,
task: DownloadTask,
*,
use_hardlink: bool,
is_torrent: bool,
preserve_source: bool,
organization_mode: str,
) -> tuple[list[Path], str | None, dict[str, int]]:
"""Transfer each book of a pack through the normal single-book path.
Each book gets an isolated task copy (the single-file path mutates `task.format`)
carrying its own title, position and year; the searched book's position must not
leak onto its siblings, while author and series name apply to all of them.
"""
all_paths: list[Path] = []
totals: dict[str, int] = {"hardlink": 0, "copy": 0, "move": 0}
errors: list[str] = []
for group in groups:
book_task = dataclasses.replace(
task,
title=group.title or task.title,
year=str(group.year) if group.year is not None else None,
subtitle=None,
series_position=group.series_position,
multi_book=False,
book_plan=None,
)
paths, error, op_counts = transfer_book_files(
group.files,
destination,
book_task,
use_hardlink=use_hardlink,
is_torrent=is_torrent,
preserve_source=preserve_source,
organization_mode=organization_mode,
source_root=group.files[0].parent,
)
for op, count in op_counts.items():
totals[op] = totals.get(op, 0) + count
if error:
errors.append(f"{group.title}: {error}")
logger.warning("Task %s: pack book %r failed: %s", task.task_id, group.title, error)
continue
all_paths.extend(paths)
if not all_paths:
return [], "; ".join(errors) or "No book files found", totals
if errors:
logger.warning(
"Task %s: pack filed with %d failed book(s): %s",
task.task_id,
len(errors),
"; ".join(errors),
)
return all_paths, None, totals
def process_directory(
directory: Path,
ingest_dir: Path,
-153
View File
@@ -1,153 +0,0 @@
"""Boot-time warm-up of the direct-download source.
The first AA search after a cold start pays for the whole cold path at once: DNS
resolution, electing a live mirror, spinning up headless Chrome and solving the
DDoS-Guard challenge. That is tens of seconds with the user sat at the search box.
Running one throwaway search shortly after boot moves that cost off the user's first
search. It primes the DNS cache, elects (and quarantines) mirrors, and leaves the
clearance cookie in the bypasser's per-domain cache, so the first real search reuses
it instead of solving from scratch.
Runs on a daemon thread and swallows every failure: this is an optimisation, and a
source that is down at boot must not affect startup or health.
"""
from __future__ import annotations
import os
import threading
from shelfmark.core.config import config
from shelfmark.core.logger import setup_logger
logger = setup_logger(__name__)
# Delay before the warm-up fires. Long enough that it does not compete with the rest
# of startup (and with a container's own health probe) for the first request.
_DEFAULT_DELAY_SECONDS = 15.0
_DEFAULT_QUERY = "The Great Gatsby"
_warmup_thread: threading.Thread | None = None
_warmup_lock = threading.Lock()
# Set as soon as a real release search starts. The warm-up exists to pay the cold path
# *before* the user does; once they have beaten it to the box there is nothing left to
# pre-solve, and running anyway is actively harmful - the bypasser serializes on one
# browser, so the warm-up's solve goes in front of the search the user is watching. In
# the bundle on issue #1276 that cost a full minute of a 2m27s wait, on a container 16
# seconds old, for a throwaway "The Great Gatsby" query nobody asked for.
_user_search_seen = threading.Event()
def note_user_search() -> None:
"""Record that a real search has run, so a pending warm-up stands down."""
_user_search_seen.set()
def _as_bool(value: object, *, default: bool) -> bool:
"""Coerce a config value that may arrive as a string, bool or None."""
if value is None:
return default
if isinstance(value, str):
from shelfmark.config.env import string_to_bool
return string_to_bool(value)
return bool(value)
def _setting(key: str, default: object) -> object:
"""Read a warm-up setting, preferring the deployment environment.
These keys are not in the settings registry, and ``config.get`` only consults the
environment for keys it knows about - so reading config alone silently ignored
SEARCH_WARMUP_ENABLED and always returned the default. Check os.environ first so
the documented switches actually work.
"""
raw = os.environ.get(key)
if raw is not None and raw.strip():
return raw
return config.get(key, default)
def is_enabled() -> bool:
"""Whether the boot-time warm-up search should run."""
if not _as_bool(_setting("SEARCH_WARMUP_ENABLED", True), default=True):
return False
# Nothing to warm if the source is off, and no challenge to pre-solve without
# the bypasser - a plain search is fast enough not to need this.
if not _as_bool(_setting("DIRECT_DOWNLOAD_ENABLED", True), default=True):
logger.debug("Search warm-up skipped: direct download disabled")
return False
return True
def warmup_query() -> str:
"""The query used to warm the source."""
raw = _setting("SEARCH_WARMUP_QUERY", _DEFAULT_QUERY)
query = str(raw).strip() if raw else ""
return query or _DEFAULT_QUERY
def run_warmup() -> bool:
"""Run one warm-up search. Returns True if it produced results.
Never raises: every failure mode here is one the next real search would hit
anyway, and reporting it is the search path's job, not the warm-up's.
"""
from shelfmark.core.mirrors import has_aa_mirror_configuration
# Checked here rather than only at schedule time: the delay is what this races with,
# so the user's first search usually lands *during* the wait, not before it.
if _user_search_seen.is_set():
logger.info("Search warm-up skipped: a real search got there first")
return False
if not has_aa_mirror_configuration():
logger.debug("Search warm-up skipped: no Anna's Archive mirrors configured")
return False
query = warmup_query()
logger.info("Warming up direct download search (%r)", query)
try:
from shelfmark.core.models import SearchFilters
from shelfmark.release_sources.direct_download import search_books
results = search_books(query, SearchFilters())
except Exception:
# Broad by design: a warm-up must never take the app down, and the source
# raises everything from network errors to parse failures.
logger.warning("Search warm-up did not complete; first user search may be slow")
logger.debug("Search warm-up failure detail", exc_info=True)
return False
if results:
logger.info("Search warm-up complete: %s results, source is ready", len(results))
return True
logger.info("Search warm-up returned no results; source reachable but empty")
return False
def start(delay_seconds: float = _DEFAULT_DELAY_SECONDS) -> bool:
"""Schedule the warm-up on a daemon thread. Safe to call multiple times."""
global _warmup_thread
if not is_enabled():
return False
with _warmup_lock:
if _warmup_thread is not None and _warmup_thread.is_alive():
logger.debug("Search warm-up already scheduled")
return False
def _run() -> None:
run_warmup()
_warmup_thread = threading.Timer(delay_seconds, _run)
_warmup_thread.daemon = True
_warmup_thread.name = "SearchWarmup"
_warmup_thread.start()
logger.debug("Search warm-up scheduled in %ss", delay_seconds)
return True
+26 -64
View File
@@ -38,11 +38,7 @@ from shelfmark.config.env import (
string_to_bool,
)
from shelfmark.config.security import _migrate_security_settings
from shelfmark.config.settings import (
_SUPPORTED_BOOK_LANGUAGE,
migrate_audiobook_format_settings,
)
from shelfmark.core import search_deadline
from shelfmark.config.settings import _SUPPORTED_BOOK_LANGUAGE
from shelfmark.core.activity_view_state_service import ActivityViewStateService
from shelfmark.core.auth_modes import (
get_auth_check_admin_status,
@@ -63,7 +59,6 @@ from shelfmark.core.notifications import (
notify_user,
)
from shelfmark.core.prefix_middleware import PrefixMiddleware
from shelfmark.core.release_inspect_routes import register_release_inspect_routes
from shelfmark.core.request_helpers import (
coerce_bool,
emit_ws_event,
@@ -85,9 +80,8 @@ from shelfmark.core.requests_service import (
sync_delivery_states_from_queue_status,
)
from shelfmark.core.user_db import UserDB
from shelfmark.core.utils import AUDIOBOOK_FORMATS, normalize_base_path
from shelfmark.core.utils import normalize_base_path
from shelfmark.download import orchestrator as backend
from shelfmark.download import warmup
from shelfmark.release_sources import (
BrowseRecord,
Release,
@@ -124,7 +118,7 @@ BASE_PATH = normalize_base_path(normalize_optional_text(app_config.get("URL_BASE
app = Flask(__name__)
app.config["SEND_FILE_MAX_AGE_DEFAULT"] = 0 # Disable caching
app.config["APPLICATION_ROOT"] = BASE_PATH or "/"
wsgi_app = cast(Any, ProxyFix(app.wsgi_app, x_host=1, x_port=1))
wsgi_app = cast(Any, ProxyFix(app.wsgi_app))
if BASE_PATH:
wsgi_app = cast(Any, PrefixMiddleware(wsgi_app, BASE_PATH, bypass_paths={"/api/health"}))
app.wsgi_app = wsgi_app
@@ -174,9 +168,6 @@ except ImportError as e:
# Migrate legacy security settings if needed
_migrate_security_settings()
# Widen audiobook formats for installs that still carry the old m4b/mp3-only default
migrate_audiobook_format_settings()
# Initialize user database and register multi-user routes
# If CONFIG_DIR doesn't exist or is read-only, multi-user features will be disabled
_user_db_path = str(Path(os.environ.get("CONFIG_DIR", "/config")) / "users.db")
@@ -209,10 +200,6 @@ except (sqlite3.OperationalError, OSError) as e:
# Start download coordinator
backend.start()
# Pre-solve the direct-download source's protection challenge in the background so the
# first user search does not pay for a cold Chrome bypass. Never blocks startup.
warmup.start()
# Rate limiting for login attempts
# Map usernames to their failed-attempt counters and lockout timestamps.
failed_login_attempts: dict[str, dict[str, Any]] = {}
@@ -332,7 +319,19 @@ def get_auth_mode() -> str:
_AUDIOBOOK_CATEGORY_RANGE = (3030, 3049)
_AUDIOBOOK_FORMAT_HINTS = frozenset(AUDIOBOOK_FORMATS)
_AUDIOBOOK_FORMAT_HINTS = frozenset(
{
"m4b",
"mp3",
"m4a",
"flac",
"ogg",
"wma",
"aac",
"wav",
"opus",
}
)
def _contains_audiobook_format_hint(value: Any) -> bool:
@@ -1026,9 +1025,6 @@ def _serialize_release(release: Release) -> dict:
return result
register_release_inspect_routes(app, login_required)
@app.route("/api/releases/download", methods=["POST"])
@login_required
def api_download_release() -> Response | tuple[Response, int]:
@@ -1155,7 +1151,7 @@ def api_config() -> Response | tuple[Response, int]:
"build_version": BUILD_VERSION,
"release_version": RELEASE_VERSION,
"book_languages": _SUPPORTED_BOOK_LANGUAGE,
"default_language": app_config.get("BOOK_LANGUAGE", ["en"], user_id=db_user_id),
"default_language": app_config.BOOK_LANGUAGE,
"supported_formats": app_config.SUPPORTED_FORMATS,
"supported_audiobook_formats": app_config.SUPPORTED_AUDIOBOOK_FORMATS,
"search_mode": search_mode,
@@ -1167,9 +1163,6 @@ def api_config() -> Response | tuple[Response, int]:
"show_combined_selector": app_config.get(
"SHOW_COMBINED_SELECTOR", True, user_id=db_user_id
),
"force_combined_search": app_config.get(
"FORCE_COMBINED_SEARCH", False, user_id=db_user_id
),
"books_output_mode": app_config.get("BOOKS_OUTPUT_MODE", "folder"),
"auto_open_downloads_sidebar": app_config.get("AUTO_OPEN_DOWNLOADS_SIDEBAR", True),
"hardcover_auto_remove_on_download": app_config.get(
@@ -1180,12 +1173,6 @@ def api_config() -> Response | tuple[Response, int]:
[],
user_id=db_user_id,
),
# The client must not give up before this budget does. `/api/releases`
# answers a spent budget with a message naming the real cause (a protection
# challenge nobody could solve); a browser that aborted first replaces it
# with a generic network/proxy error and RELEASE_SEARCH_TIMEOUT becomes a
# setting the user can raise with no visible effect. See issue #1285.
"release_search_timeout": search_deadline.budget_seconds(),
"settings_enabled": _is_config_dir_writable(),
"onboarding_complete": _get_onboarding_complete(),
# Default sort orders
@@ -2851,7 +2838,6 @@ def api_releases() -> Response | tuple[Response, int]:
manual_query=query_text if source_query_filters is not None else manual_query,
indexers=indexers,
source_filters=source_query_filters,
user_id=db_user_id,
)
if plan.source_filters is not None:
@@ -2904,8 +2890,6 @@ def api_releases() -> Response | tuple[Response, int]:
if languages_param
else None
)
# Without an explicit filter the plan falls back to this user's default languages.
db_user_id = get_session_db_user_id(session)
# Content type for audiobook vs ebook search
content_type = request.args.get("content_type", "ebook").strip()
@@ -2953,10 +2937,6 @@ def api_releases() -> Response | tuple[Response, int]:
elif provider == "manual":
resolved_title = title_param or manual_query or "Manual Search"
resolved_author = author_param or ""
# The release modal sends `authors.join(', ')` as `author`, so the commas here
# are joins between contributors, not part of one name. This split is the only
# place that knows that, so `search_author` comes from it rather than from the
# joined text - see issue #1252.
authors = [a.strip() for a in resolved_author.split(",") if a.strip()]
book = BookMetadata(
@@ -2965,7 +2945,7 @@ def api_releases() -> Response | tuple[Response, int]:
provider_display_name="Manual Search",
title=resolved_title,
search_title=resolved_title,
search_author=authors[0] if authors else None,
search_author=resolved_author or None,
authors=authors,
)
else:
@@ -2998,36 +2978,18 @@ def api_releases() -> Response | tuple[Response, int]:
# Search only enabled sources
sources_to_search = [src["name"] for src in list_available_sources() if src["enabled"]]
# Search each source for releases.
#
# Under a wall-clock budget: this endpoint is synchronous, and the bypass path it
# can reach used to be allowed minutes per URL with nothing bounding the request
# as a whole. A search that ran into an unsolvable protection challenge therefore
# outlived every reverse proxy in front of it and surfaced to the user as
# "Server unavailable (504)" - a gateway timeout that blames their proxy for a
# challenge failure. The budget is shared across sources, so a stuck first source
# cannot spend the whole request on its own. See issue #1276.
# Search each source for releases
all_releases = []
errors = []
source_instances = {} # Keep source instances for column config
# A real search is under way, so a warm-up still sitting on its start-up delay
# should stand down rather than queue its throwaway solve in front of this one.
warmup.note_user_search()
with search_deadline.search_deadline():
for source_name in sources_to_search:
if search_deadline.expired():
logger.warning("Release search budget spent; %s not searched", source_name)
errors.append(f"{source_name}: {search_deadline.deadline_message()}")
continue
source, releases, error = _search_source_releases(source_name, book)
if source is not None:
source_instances[source_name] = source
all_releases.extend(releases)
if error is not None:
errors.append(error)
for source_name in sources_to_search:
source, releases, error = _search_source_releases(source_name, book)
if source is not None:
source_instances[source_name] = source
all_releases.extend(releases)
if error is not None:
errors.append(error)
# Convert Release objects to dicts
releases_data = [_serialize_release(release) for release in all_releases]
+14 -17
View File
@@ -24,12 +24,12 @@ Dataclass representing a book from a metadata provider:
```python
@dataclass
class BookMetadata:
provider: str # Internal provider name (e.g., "hardcover")
provider_id: str # ID in that provider's system
provider: str # Internal provider name (e.g., "hardcover")
provider_id: str # ID in that provider's system
title: str
# Optional fields
provider_display_name: str # Human-readable name (e.g., "Hardcover")
provider_display_name: str # Human-readable name (e.g., "Hardcover")
authors: List[str]
isbn_10: str
isbn_13: str
@@ -39,7 +39,7 @@ class BookMetadata:
publish_year: int
language: str
genres: List[str]
source_url: str # Link to book on provider's site
source_url: str # Link to book on provider's site
display_fields: List[DisplayField] # Provider-specific display data
```
@@ -50,9 +50,9 @@ Provider-specific metadata for UI cards (ratings, page counts, reader counts, et
```python
@dataclass
class DisplayField:
label: str # e.g., "Rating", "Pages", "Readers"
value: str # e.g., "4.5", "496", "8,041"
icon: str # Icon name: "star", "book", "users", "editions"
label: str # e.g., "Rating", "Pages", "Readers"
value: str # e.g., "4.5", "496", "8,041"
icon: str # Icon name: "star", "book", "users", "editions"
```
### MetadataSearchOptions
@@ -64,7 +64,7 @@ Unified search options that work across all providers:
class MetadataSearchOptions:
query: str
search_type: SearchType = SearchType.GENERAL # GENERAL, TITLE, AUTHOR, ISBN
language: str = None # ISO 639-1 code (e.g., "en")
language: str = None # ISO 639-1 code (e.g., "en")
sort: SortOrder = SortOrder.RELEVANCE
limit: int = 40
page: int = 1
@@ -88,10 +88,10 @@ All providers must implement this interface:
```python
class MetadataProvider(ABC):
name: str # Internal identifier
display_name: str # Human-readable name
requires_auth: bool # True if API key required
supported_sorts: List[SortOrder] # Supported sort options
name: str # Internal identifier
display_name: str # Human-readable name
requires_auth: bool # True if API key required
supported_sorts: List[SortOrder] # Supported sort options
@abstractmethod
def search(self, options: MetadataSearchOptions) -> List[BookMetadata]:
@@ -121,9 +121,9 @@ class MetadataProvider(ABC):
```python
from shelfmark.metadata_providers import register_provider
@register_provider("my_provider")
class MyProvider(MetadataProvider): ...
class MyProvider(MetadataProvider):
...
```
### Getting Providers
@@ -281,13 +281,11 @@ from shelfmark.config.env import (
METADATA_CACHE_BOOK_TTL,
)
@cacheable(ttl=METADATA_CACHE_SEARCH_TTL, key_prefix="myprovider:search")
def _search_cached(self, cache_key: str, options: MetadataSearchOptions):
# Cached search implementation
pass
@cacheable(ttl=METADATA_CACHE_BOOK_TTL, key_prefix="myprovider:book")
def get_book(self, book_id: str):
# Cached book lookup
@@ -304,7 +302,6 @@ from shelfmark.metadata_providers.openlibrary import RateLimiter
# 90 requests per 60 seconds
rate_limiter = RateLimiter(max_requests=90, window_seconds=60)
def make_request(self):
rate_limiter.wait_if_needed() # Blocks if rate limited
# ... make request
-3
View File
@@ -709,6 +709,3 @@ with suppress(ImportError):
with suppress(ImportError):
from shelfmark.metadata_providers import googlebooks as googlebooks
with suppress(ImportError):
from shelfmark.metadata_providers import moly as moly
+20 -127
View File
@@ -1,7 +1,6 @@
"""Hardcover.app metadata provider. Requires API key."""
import re
import time
from contextlib import suppress
from dataclasses import dataclass
from datetime import UTC, datetime
@@ -48,10 +47,6 @@ HARDCOVER_PAGE_SIZE = 25 # Hardcover API returns max 25 results per page
HARDCOVER_MIN_AUTHOR_PARTS = 2
HARDCOVER_MIN_TYPEAHEAD_QUERY_LENGTH = 2
HARDCOVER_MAX_SERIES_OPTIONS = 7
# Hardcover hands out short opaque tokens now ("hc_pat_...") instead of the ~500 char
# JWTs it used to, so the length floor only applies to keys without that prefix.
HARDCOVER_API_KEY_PREFIX = "hc_pat_"
HARDCOVER_BEARER_PREFIX_PATTERN = re.compile(r"^bearer\s+", re.IGNORECASE)
HARDCOVER_API_KEY_MIN_LENGTH = 100
HARDCOVER_LIST_URL_PATTERN = re.compile(
r"^/(?:@([\w.-]+)/)?lists?/([\w-]+)/?$",
@@ -322,7 +317,6 @@ query SearchFieldOptions(
fields: $fields,
weights: $weights
) {
error
results
}
}
@@ -541,19 +535,13 @@ SORT_MAPPING: dict[SortOrder, str] = {
SortOrder.OLDEST: "release_year:asc",
}
# `fields` becomes Typesense's `query_by`, but Hardcover keeps `num_typos` and
# `query_by_weights` as fixed-length presets per query_type. Passing a different
# number of fields than the preset expects makes Typesense reject the whole search,
# complaining that the number of num_typos values does not match the number of
# query_by fields. So a Book search may only ever narrow to *these five* names --
# a shorter list is rejected outright rather than searched, and any weights sent
# alongside must match one-for-one.
# Weights only bias ranking: a field weighted 0 still matches, so `fields` can no
# longer restrict which fields a Book query looks at.
BOOK_SEARCH_FIELDS = "title,alternative_titles,author_names,series_names,isbns"
BOOK_SEARCH_FIELD_COUNT = 5
BOOK_TITLE_WEIGHTS = "5,1,0,0,0"
BOOK_TITLE_AUTHOR_WEIGHTS = "5,1,3,0,0"
# Mapping from abstract search type to Hardcover fields parameter
SEARCH_TYPE_FIELDS: dict[SearchType, str] = {
SearchType.GENERAL: "title,isbns,series_names,author_names,alternative_titles",
SearchType.TITLE: "title,alternative_titles",
SearchType.AUTHOR: "author_names",
# ISBN is handled separately via search_by_isbn()
}
SERIES_SEARCH_FIELDS = "name,books,author_name"
SERIES_SEARCH_WEIGHTS = "2,1,1"
@@ -561,52 +549,10 @@ SERIES_SEARCH_SORT = "_text_match:desc,readers_count:desc"
AUTHOR_SUGGESTION_FIELDS = "name,name_personal,alternate_names"
AUTHOR_SUGGESTION_WEIGHTS = "4,3,2"
AUTHOR_SUGGESTION_SORT = "_text_match:desc,books_count:desc"
TITLE_SUGGESTION_FIELDS = BOOK_SEARCH_FIELDS
TITLE_SUGGESTION_WEIGHTS = "5,2,0,0,0"
TITLE_SUGGESTION_FIELDS = "title,alternative_titles"
TITLE_SUGGESTION_WEIGHTS = "5,2"
TITLE_SUGGESTION_SORT = "_text_match:desc,users_count:desc"
# Hardcover forwards `sort` to Typesense's `sort_by` and rejects the whole search
# if it does not like the value -- an unknown field, a bare field name with no
# direction, more than three keys. A rejected search comes back as HTTP 200 with
# no GraphQL errors and a null `results` body, which is otherwise indistinguishable
# from "nothing matched"; the reason only shows up in the sibling `error` field,
# so every search asks for it. Dropping `sort` from the request is the one shape
# Hardcover always accepts -- an empty string is a value like any other and has
# been rejected too -- so retry that way and keep the fallback sticky for a while
# rather than paying for a doomed request on every search.
SORT_FALLBACK_TTL = 900.0
_sort_fallback_until = 0.0
def _without_sort(variables: dict[str, Any]) -> dict[str, Any]:
"""Drop `sort` entirely so Hardcover applies its own default ordering."""
return {key: value for key, value in variables.items() if key != "sort"}
def _search_payload_rejected(result: dict[str, Any] | None) -> bool:
"""Report whether Hardcover answered a search with a null results body.
A search that genuinely matched nothing still returns a results object with
``found: 0``; only a rejected search nulls it out entirely.
"""
if not isinstance(result, dict):
return False
root = result.get("search", result)
if not isinstance(root, dict) or "results" not in root:
return False
return root["results"] is None
def _search_rejection_reason(result: dict[str, Any] | None) -> str:
"""Return Hardcover's explanation for a rejected search, if it sent one."""
if not isinstance(result, dict):
return ""
root = result.get("search", result)
if not isinstance(root, dict):
return ""
error = root.get("error")
return error.strip() if isinstance(error, str) else ""
def _combine_headline_description(headline: str | None, description: str | None) -> str | None:
"""Combine headline (tagline) and description into a single description."""
@@ -674,7 +620,7 @@ def _normalize_series_position(value: Any) -> float | None:
def _normalize_hardcover_api_key(value: object) -> str:
"""Normalize Hardcover API keys, stripping copied auth-header prefixes."""
normalized_value = normalize_optional_text(value) or ""
return HARDCOVER_BEARER_PREFIX_PATTERN.sub("", normalized_value.strip()).strip()
return normalized_value.removeprefix("Bearer ").strip()
def _normalize_search_text(value: str) -> str:
@@ -1040,15 +986,13 @@ class HardcoverProvider(MetadataProvider):
"""Build search query, fields, and weights based on provided values.
Returns (query, fields, weights) tuple. Fields/weights are None for general search.
A narrowed search still sends all of BOOK_SEARCH_FIELDS -- Hardcover rejects a
shorter list outright -- and leans on the weights to rank the wanted field first.
"""
if author and not title and not series:
return author, None, None
if title and not author and not series:
return title, BOOK_SEARCH_FIELDS, BOOK_TITLE_WEIGHTS
return title, "title,alternative_titles", "5,1"
if author and title and not series:
return f"{title} {author}", BOOK_SEARCH_FIELDS, BOOK_TITLE_AUTHOR_WEIGHTS
return f"{title} {author}", "title,alternative_titles,author_names", "5,1,3"
return default_query, None, None
def _detect_list_url(self, query: str) -> tuple[str | None, str] | None:
@@ -1256,7 +1200,7 @@ class HardcoverProvider(MetadataProvider):
if not self.api_key or len(normalized_query) < HARDCOVER_MIN_TYPEAHEAD_QUERY_LENGTH:
return []
result = self._execute_search_query(
result = self._execute_query(
SEARCH_FIELD_OPTIONS_QUERY,
{
"query": normalized_query,
@@ -1487,7 +1431,7 @@ class HardcoverProvider(MetadataProvider):
logger.debug("Invalid Hardcover series id field value: %s", normalized_value)
return None
result = self._execute_search_query(
result = self._execute_query(
SEARCH_FIELD_OPTIONS_QUERY,
{
"query": normalized_value,
@@ -2414,7 +2358,6 @@ class HardcoverProvider(MetadataProvider):
graphql_query = """
query SearchBooks($query: String!, $limit: Int!, $page: Int!, $sort: String, $fields: String, $weights: String) {
search(query: $query, query_type: "Book", per_page: $limit, page: $page, sort: $sort, fields: $fields, weights: $weights) {
error
results
}
}
@@ -2423,7 +2366,6 @@ class HardcoverProvider(MetadataProvider):
graphql_query = """
query SearchBooks($query: String!, $limit: Int!, $page: Int!, $sort: String) {
search(query: $query, query_type: "Book", per_page: $limit, page: $page, sort: $sort) {
error
results
}
}
@@ -2444,7 +2386,7 @@ class HardcoverProvider(MetadataProvider):
variables["weights"] = search_weights
try:
result = self._execute_search_query(graphql_query, variables)
result = self._execute_query(graphql_query, variables)
if not result:
logger.debug("Hardcover search: No result from API")
return SearchResult(books=[], page=options.page, total_found=0, has_more=False)
@@ -2686,7 +2628,7 @@ class HardcoverProvider(MetadataProvider):
raise RuntimeError(msg) from e
return None
except requests.HTTPError as e:
if e.response is not None and e.response.status_code == HTTPStatus.UNAUTHORIZED:
if e.response.status_code == HTTPStatus.UNAUTHORIZED:
logger.exception("Hardcover API key is invalid")
if raise_on_error:
msg = "Hardcover API key is invalid"
@@ -2712,54 +2654,6 @@ class HardcoverProvider(MetadataProvider):
raise RuntimeError(msg) from e
return None
def _execute_search_query(self, query: str, variables: dict[str, Any]) -> dict | None:
"""Execute a search query, retrying without ``sort`` if Hardcover rejects it.
Returns None when the search was rejected, so callers report an empty
result rather than silently treating a failure as "nothing matched".
"""
global _sort_fallback_until
sort = variables.get("sort")
if sort and time.monotonic() < _sort_fallback_until:
variables = _without_sort(variables)
sort = None
result = self._execute_query(query, variables)
if not _search_payload_rejected(result):
return result
reason = _search_rejection_reason(result)
if not sort:
logger.error(
"Hardcover rejected this search (query_type=%s, fields=%s): %s",
variables.get("queryType", "Book"),
variables.get("fields"),
reason or "no error message",
)
return None
retry = self._execute_query(query, _without_sort(variables))
if _search_payload_rejected(retry):
# The sort was not the culprit, so leave sorting alone for other searches.
logger.error(
"Hardcover rejected this search (query_type=%s, fields=%s) with and without "
"a sort order: %s",
variables.get("queryType", "Book"),
variables.get("fields"),
_search_rejection_reason(retry) or reason or "no error message",
)
return None
logger.warning(
"Hardcover rejected sort '%s' (%s); dropping the sort order from searches for %ss",
sort,
reason or "no error message",
int(SORT_FALLBACK_TTL),
)
_sort_fallback_until = time.monotonic() + SORT_FALLBACK_TTL
return retry
def _parse_search_result(self, item: dict) -> BookMetadata | None:
"""Parse a search result item into BookMetadata."""
try:
@@ -3023,13 +2917,12 @@ def _test_hardcover_connection(current_values: dict[str, Any] | None = None) ->
_save_connected_user(None, None)
return {"success": False, "message": "API key is required"}
is_prefixed_key = api_key.startswith(HARDCOVER_API_KEY_PREFIX)
if not is_prefixed_key and key_len < HARDCOVER_API_KEY_MIN_LENGTH:
if key_len < HARDCOVER_API_KEY_MIN_LENGTH:
return {
"success": False,
"message": (
f"API key seems too short ({key_len} chars). Expected a key starting "
f"with {HARDCOVER_API_KEY_PREFIX} or {HARDCOVER_API_KEY_MIN_LENGTH}+ chars."
f"API key seems too short ({key_len} chars). "
f"Expected {HARDCOVER_API_KEY_MIN_LENGTH}+ chars."
),
}
@@ -3136,7 +3029,7 @@ def hardcover_settings() -> list[SettingsField]:
PasswordField(
key="HARDCOVER_API_KEY",
label="API Key",
description="Get your API key from hardcover.app/account/api (starts with hc_pat_)",
description="Get your API key from hardcover.app/account/api",
required=True,
),
ActionButton(
-493
View File
@@ -1,493 +0,0 @@
"""Moly.hu metadata provider. Hungarian book catalog, no API key required.
Scraping approach (search URL, book-page structure, language mapping) adapted
from the Calibre Moly_hu plugin by Hoffer Csaba, Kloon, otapi, Dezso, Hokutya,
seeder and contributors (GPL v3, mobileread.com).
"""
import re
import threading
import time
import unicodedata
from collections import deque
from typing import Any, ClassVar
from urllib.parse import quote
import requests
from bs4 import BeautifulSoup, Tag
from shelfmark.core.cache import cacheable
from shelfmark.core.logger import setup_logger
from shelfmark.core.settings_registry import (
ActionButton,
CheckboxField,
HeadingField,
SettingsField,
register_settings,
)
from shelfmark.download.network import get_ssl_verify
from shelfmark.metadata_providers import (
BookMetadata,
DisplayField,
MetadataProvider,
MetadataSearchOptions,
SearchField,
SearchType,
SortOrder,
TextSearchField,
register_provider,
)
logger = setup_logger(__name__)
MOLY_BASE_URL = "https://moly.hu"
MOLY_BOOK_URL = f"{MOLY_BASE_URL}/konyvek/"
MOLY_SEARCH_URL = f"{MOLY_BASE_URL}/kereses?query="
# Be polite: moly.hu is a small community site
RATE_LIMIT_REQUESTS = 30
RATE_LIMIT_WINDOW_SECONDS = 60
REQUEST_HEADERS = {
"User-Agent": ("Mozilla/5.0 (X11; Linux x86_64; rv:128.0) Gecko/20100101 Firefox/128.0"),
"Accept-Language": "hu,en;q=0.7",
}
ISBN_13_LENGTH = 13
# Moly tags its foreign-language editions; everything else is Hungarian.
# Mapping from the Calibre Moly_hu plugin.
_LANGUAGE_TAG_MAP = {
"angol nyelvű": "en",
"n\xe9met nyelvű": "de",
"francia nyelvű": "fr",
"olasz nyelvű": "it",
"spanyol nyelvű": "es",
"orosz nyelvű": "ru",
"t\xf6r\xf6k nyelvű": "tr",
"g\xf6r\xf6g nyelvű": "el",
"k\xednai nyelvű": "zh",
"jap\xe1n nyelvű": "ja",
}
class RateLimiter:
"""Simple sliding window rate limiter."""
def __init__(self, max_requests: int, window_seconds: int) -> None:
"""Initialize rate limiter with max requests per time window."""
self.max_requests = max_requests
self.window_seconds = window_seconds
self.timestamps: deque[float] = deque()
self.lock = threading.Lock()
def wait_if_needed(self) -> None:
"""Block until a request is allowed (thread-safe)."""
wait_time = 0.0
with self.lock:
now = time.time()
cutoff = now - self.window_seconds
while self.timestamps and self.timestamps[0] < cutoff:
self.timestamps.popleft()
if len(self.timestamps) >= self.max_requests:
wait_time = self.timestamps[0] + self.window_seconds - now
if wait_time > 0:
logger.debug("Rate limited, waiting %0.2fs", wait_time)
time.sleep(wait_time)
with self.lock:
now = time.time()
cutoff = now - self.window_seconds
while self.timestamps and self.timestamps[0] < cutoff:
self.timestamps.popleft()
self.timestamps.append(time.time())
_rate_limiter = RateLimiter(RATE_LIMIT_REQUESTS, RATE_LIMIT_WINDOW_SECONDS)
def _clean_text(value: str | None) -> str | None:
"""Strip zero-width characters and collapse whitespace."""
if value is None:
return None
value = value.replace("​", "").replace("", "")
return " ".join(value.split())
def _normalize_for_match(value: str | None) -> str:
"""Accent-insensitive, punctuation-insensitive comparison form."""
if not value:
return ""
value = unicodedata.normalize("NFKD", value)
value = "".join(char for char in value if not unicodedata.combining(char))
value = "".join(char if char.isalnum() else " " for char in value)
return " ".join(value.lower().split())
def _absolute_url(url: str | None) -> str | None:
if not url:
return None
if url.startswith(("http://", "https://")):
return url
return MOLY_BASE_URL + url
def _valid_isbn(candidate: str) -> str | None:
"""Return a normalized ISBN-10/13 (digits, with optional X check digit), else None."""
digits = candidate.replace("-", "").strip()
if len(digits) == ISBN_13_LENGTH and digits.isdigit():
return digits
if len(digits) == 10 and re.fullmatch(r"\d{9}[\dXx]", digits):
return digits.upper()
return None
@register_provider("moly")
class MolyProvider(MetadataProvider):
"""Moly.hu metadata provider (HTML scraping, Hungarian catalog)."""
name = "moly"
display_name = "Moly.hu"
requires_auth = False
supported_sorts: ClassVar[tuple[SortOrder, ...]] = (SortOrder.RELEVANCE,)
search_fields: ClassVar[tuple[SearchField, ...]] = (
TextSearchField(
key="author",
label="Author",
description="Search by author name",
),
TextSearchField(
key="title",
label="Title",
description="Search by book title",
),
)
def __init__(self) -> None:
"""Initialize provider."""
self.session = requests.Session()
self.session.headers.update(REQUEST_HEADERS)
def is_available(self) -> bool:
"""Moly.hu needs no authentication."""
return True
def _fetch(self, url: str, timeout: int = 15) -> str | None:
_rate_limiter.wait_if_needed()
try:
response = self.session.get(url, timeout=timeout, verify=get_ssl_verify(MOLY_BASE_URL))
response.raise_for_status()
except requests.Timeout:
logger.warning("Moly.hu request timed out: %s", url)
return None
except requests.RequestException:
logger.exception("Moly.hu request failed: %s", url)
return None
return response.text
def search(self, options: MetadataSearchOptions) -> list[BookMetadata]:
"""Search moly.hu's site search."""
if options.search_type == SearchType.ISBN:
result = self.search_by_isbn(options.query)
return [result] if result else []
# Moly's search is a single ranked page; no server-side pagination.
if options.page > 1:
return []
author_value = (options.fields.get("author") or "").strip()
title_value = (options.fields.get("title") or "").strip()
terms = " ".join(t for t in (author_value, title_value) if t)
query = terms or options.query.strip()
if not query:
return []
fields_key = ":".join(f"{k}={v}" for k, v in sorted(options.fields.items()))
cache_key = f"{query}:{options.search_type.value}:{options.limit}:{fields_key}"
return self._search_cached(cache_key, query, options.limit) or []
@cacheable(ttl_key="METADATA_CACHE_SEARCH_TTL", ttl_default=300, key_prefix="moly:search")
def _search_cached(self, cache_key: str, query: str, limit: int) -> list[BookMetadata] | None:
# Return None (not []) on fetch failure so the failure is not cached.
html = self._fetch(MOLY_SEARCH_URL + quote(query.encode("utf-8")))
if html is None:
return None
soup = BeautifulSoup(html, "html.parser")
books: list[BookMetadata] = []
seen: set[str] = set()
for anchor in soup.select("#content div.search_area a.book_selector"):
href = anchor.get("href") or ""
match = re.search(r"/konyvek/([^/?#]+)", str(href))
if not match:
continue
slug = match.group(1)
if slug in seen:
continue
# No separator: moly wraps matched search terms in <strong> even
# mid-word ("Lis<strong>a</strong> Jewell"), so inserting one
# would split words at highlight boundaries.
text = _clean_text(anchor.get_text()) or ""
author, _, title = text.partition(":")
if not title:
# Result rows are "Author: Title"; skip anything else.
continue
author = author.strip()
title = title.strip()
seen.add(slug)
books.append(
BookMetadata(
provider=self.name,
provider_id=slug,
provider_display_name=self.display_name,
title=title,
authors=[author] if author else [],
cover_url=self._cover_for_result(soup, text),
source_url=MOLY_BOOK_URL + slug,
language="hu",
search_title=title,
search_author=author or None,
display_fields=self._result_display_fields(anchor),
)
)
if len(books) >= limit:
break
logger.info("Moly.hu search '%s' returned %s results", query, len(books))
return books
def _cover_for_result(self, soup: BeautifulSoup, result_text: str) -> str | None:
"""Find the search-result thumbnail whose alt matches 'Author: Title'."""
target = _normalize_for_match(result_text)
if not target:
return None
for img in soup.select("#content img.tooltip[alt]"):
if _normalize_for_match(str(img.get("alt") or "")) == target:
return _absolute_url(str(img.get("src") or "")) or None
return None
def _result_display_fields(self, anchor: Tag) -> list[DisplayField]:
fields: list[DisplayField] = []
parent = anchor.parent
if parent is None:
return fields
like = parent.select_one("span.like_count")
if like:
fields.append(
DisplayField(label="Rating", value=like.get_text(strip=True), icon="star")
)
series = parent.select_one('a[href*="/sorozatok/"]')
if series:
fields.append(
DisplayField(
label="Series",
value=series.get_text(strip=True).strip("()"),
icon="editions",
)
)
return fields
@cacheable(ttl_key="METADATA_CACHE_BOOK_TTL", ttl_default=600, key_prefix="moly:book")
def get_book(self, book_id: str) -> BookMetadata | None:
"""Get book details by moly.hu slug (e.g. 'mocsidzuki-mai-a-telihold-kavezo')."""
html = self._fetch(MOLY_BOOK_URL + quote(book_id))
if html is None:
return None
soup = BeautifulSoup(html, "html.parser")
title = self._parse_title(soup)
authors = [_clean_text(a.get_text()) or "" for a in soup.select("#content div.authors a")]
authors = [a for a in authors if a]
if not title or not authors:
logger.warning("Moly.hu book page missing title/authors: %s", book_id)
return None
isbn_13, isbn_10 = self._parse_isbns(soup)
series = self._parse_series(soup)
tags = [_clean_text(t.get_text()) or "" for t in soup.select("#book_tags a.tag")]
tags = [t for t in tags if t]
display_fields: list[DisplayField] = []
rating = soup.select_one("#content .rating .like_count")
if rating:
display_fields.append(
DisplayField(label="Rating", value=rating.get_text(strip=True), icon="star")
)
if series:
display_fields.append(DisplayField(label="Series", value=series, icon="editions"))
return BookMetadata(
provider=self.name,
provider_id=book_id,
provider_display_name=self.display_name,
title=title,
authors=authors,
isbn_13=isbn_13,
isbn_10=isbn_10,
cover_url=self._parse_cover(soup),
description=self._parse_description(soup),
publisher=self._parse_publisher(soup),
publish_year=self._parse_publish_year(soup),
language=self._parse_language(tags),
genres=tags,
source_url=MOLY_BOOK_URL + book_id,
search_title=title,
search_author=authors[0],
display_fields=display_fields,
)
@cacheable(ttl_key="METADATA_CACHE_BOOK_TTL", ttl_default=600, key_prefix="moly:isbn")
def search_by_isbn(self, isbn: str) -> BookMetadata | None:
"""Moly's site search resolves ISBN queries directly."""
isbn = isbn.replace("-", "").strip()
if not isbn:
return None
html = self._fetch(MOLY_SEARCH_URL + quote(isbn))
if html is None:
return None
soup = BeautifulSoup(html, "html.parser")
anchor = soup.select_one("#content div.search_area a.book_selector[href]")
if not anchor:
return None
match = re.search(r"/konyvek/([^/?#]+)", str(anchor.get("href")))
if not match:
return None
return self.get_book(match.group(1))
def _parse_title(self, soup: BeautifulSoup) -> str | None:
node = soup.select_one("#content .head_title h1 span.item")
if node:
# The series link is nested inside this span; only direct text
# belongs to the book title.
direct = "".join(node.find_all(string=True, recursive=False))
title = _clean_text(direct)
if title:
return title
node = soup.select_one("#content .book > span")
if node:
return _clean_text(node.get_text())
return None
def _parse_series(self, soup: BeautifulSoup) -> str | None:
node = soup.select_one('#content h1 a[href*="/sorozatok/"]')
if not node:
return None
return (_clean_text(node.get_text()) or "").strip("()") or None
def _parse_isbns(self, soup: BeautifulSoup) -> tuple[str | None, str | None]:
isbn_13 = isbn_10 = None
editions = soup.select("#content .items .edition") or soup.select("#content .items > div")
for edition in editions:
text = edition.get_text(" ")
for candidate in re.findall(r"(?<!\d)[\d-]{10,17}(?!\d)", text):
isbn = _valid_isbn(candidate)
if not isbn:
continue
if len(isbn) == ISBN_13_LENGTH and not isbn_13:
isbn_13 = isbn
elif len(isbn) != ISBN_13_LENGTH and not isbn_10:
isbn_10 = isbn
if isbn_13:
break
return isbn_13, isbn_10
def _parse_cover(self, soup: BeautifulSoup) -> str | None:
node = soup.select_one("#content .coverbox a.zoom[href]")
if node:
return _absolute_url(str(node.get("href")))
img = soup.select_one("#content .coverbox img[src]")
if img:
return _absolute_url(str(img.get("src")))
return None
def _parse_description(self, soup: BeautifulSoup) -> str | None:
node = soup.select_one("#content #full_description")
if node is None:
node = soup.select_one("#content div.text")
if node is None:
return None
spoiler_warning = "Vigyázat! Cselekményleírást tartalmaz."
parts = []
for text in node.stripped_strings:
cleaned = _clean_text(text) or ""
if cleaned.startswith(spoiler_warning):
cleaned = cleaned[len(spoiler_warning) :].strip()
if cleaned:
parts.append(cleaned)
return "\n".join(parts) or None
def _parse_publisher(self, soup: BeautifulSoup) -> str | None:
node = soup.select_one('#content .items .edition a[href*="/kiadok/"]')
if node:
return _clean_text(node.get_text())
return None
def _parse_publish_year(self, soup: BeautifulSoup) -> int | None:
editions = soup.select("#content .items .edition") or soup.select("#content .items > div")
for edition in editions:
match = re.search(r"\b(\d{4})\b", edition.get_text(" "))
if match:
return int(match.group(1))
return None
def _parse_language(self, tags: list[str]) -> str:
for tag in tags:
code = _LANGUAGE_TAG_MAP.get(tag.lower().strip())
if code:
return code
return "hu"
def _test_moly_connection() -> dict[str, Any]:
"""Test connectivity to moly.hu."""
try:
provider = MolyProvider()
response = provider.session.get(
MOLY_SEARCH_URL + quote("teszt"),
timeout=10,
verify=get_ssl_verify(MOLY_BASE_URL),
)
response.raise_for_status()
except requests.Timeout:
return {"success": False, "message": "Connection timed out"}
except requests.RequestException as e:
return {"success": False, "message": f"Connection failed: {e}"}
if "moly" in response.text.lower():
return {"success": True, "message": "Successfully connected to moly.hu"}
return {"success": False, "message": "Unexpected response from moly.hu"}
@register_settings("moly", "Moly.hu", icon="library", order=54, group="metadata_providers")
def moly_settings() -> list[SettingsField]:
"""Moly.hu metadata provider settings."""
return [
HeadingField(
key="moly_heading",
title="Moly.hu",
description=(
"Hungarian community book catalog with excellent coverage of "
"Hungarian editions and translations. No API key required."
),
link_url="https://moly.hu",
link_text="moly.hu",
),
CheckboxField(
key="MOLY_ENABLED",
label="Enable Moly.hu",
description="Enable Moly.hu as a metadata provider for book searches",
default=False,
),
ActionButton(
key="test_connection",
label="Test Connection",
description="Verify moly.hu is accessible",
style="primary",
callback=_test_moly_connection,
),
]
+3 -3
View File
@@ -214,7 +214,7 @@ class OpenLibraryProvider(MetadataProvider):
logger.warning("Open Library search timed out")
return []
except requests.HTTPError as e:
if e.response is not None and e.response.status_code == HTTPStatus.SERVICE_UNAVAILABLE:
if e.response.status_code == HTTPStatus.SERVICE_UNAVAILABLE:
logger.warning("Open Library service unavailable (503)")
else:
logger.exception("Open Library HTTP error")
@@ -253,7 +253,7 @@ class OpenLibraryProvider(MetadataProvider):
logger.warning("Open Library get_book timed out")
return None
except requests.HTTPError as e:
if e.response is not None and e.response.status_code == HTTPStatus.NOT_FOUND:
if e.response.status_code == HTTPStatus.NOT_FOUND:
logger.debug("Open Library work not found: %s", book_id)
else:
logger.exception("Open Library HTTP error")
@@ -314,7 +314,7 @@ class OpenLibraryProvider(MetadataProvider):
return self._parse_edition(edition, clean_isbn)
except requests.HTTPError as e:
if e.response is not None and e.response.status_code == HTTPStatus.NOT_FOUND:
if e.response.status_code == HTTPStatus.NOT_FOUND:
logger.debug("Open Library ISBN not found: %s", isbn)
else:
logger.exception("Open Library ISBN search HTTP error")
+2 -21
View File
@@ -13,7 +13,6 @@ if TYPE_CHECKING:
from shelfmark.core.models import DownloadTask
from shelfmark.core.search_plan import ReleaseSearchPlan
from shelfmark.download.postprocess.packs import PackFile
from shelfmark.metadata_providers import BookMetadata
@@ -164,7 +163,6 @@ class SortOption:
label: str # Display label in the sort dropdown
sort_key: str # Field to sort by on the Release object
default_direction: Literal["asc", "desc"] = "desc" # Which way "best first" runs
@dataclass
@@ -263,12 +261,7 @@ def serialize_column_config(config: ReleaseColumnConfig) -> dict[str, Any]:
# Include extra sort options (sort entries not tied to a column)
if config.extra_sort_options:
result["extra_sort_options"] = [
{
"label": opt.label,
"sort_key": opt.sort_key,
"default_direction": opt.default_direction,
}
for opt in config.extra_sort_options
{"label": opt.label, "sort_key": opt.sort_key} for opt in config.extra_sort_options
]
# Include action button if specified (replaces default expand search)
@@ -401,14 +394,6 @@ class DownloadHandler(ABC):
"""Return private queue-time fields needed for restart-safe retry."""
return {}
def list_files(self, release_data: dict[str, Any]) -> list[PackFile] | None:
"""Return the release's file list without downloading it.
Lets the UI review a multi-book pack before queueing. Return None when the
source cannot know the files ahead of time (magnet links, usenet, ...).
"""
return None
@abstractmethod
def cancel(self, task_id: str) -> bool:
"""Cancel an in-progress download."""
@@ -519,10 +504,6 @@ def browse_record_to_book_metadata(
"""Convert a source-native browse record into generic book metadata."""
resolved_title = title_override or str(record.title or "").strip() or "Unknown title"
resolved_author = author_override or str(record.author or "").strip()
# `author_override` is the frontend's display string, `authors.join(', ')` - every
# contributor, translators included. The split below is the only place that knows the
# commas were joins rather than part of a name, so `search_author` is taken from it
# rather than from the joined text. See issue #1252.
authors = [part.strip() for part in resolved_author.split(",") if part.strip()]
publish_year = None
@@ -539,7 +520,7 @@ def browse_record_to_book_metadata(
provider_display_name=get_source_display_name(record.source),
title=resolved_title,
search_title=resolved_title,
search_author=authors[0] if authors else None,
search_author=resolved_author or None,
authors=authors,
cover_url=record.preview,
description=record.description,
@@ -1,6 +1,6 @@
"""AudiobookBay download handler - resolves magnet links and uses shared client lifecycle."""
from typing import TYPE_CHECKING, Any
from typing import TYPE_CHECKING
from urllib.parse import urlparse
from shelfmark.core.config import config
@@ -22,7 +22,6 @@ if TYPE_CHECKING:
from collections.abc import Callable
from shelfmark.core.models import DownloadTask
from shelfmark.download.postprocess.packs import PackFile
logger = setup_logger(__name__)
DEFAULT_ABB_HOSTNAME = "audiobookbay.lu"
@@ -69,19 +68,6 @@ class AudiobookBayHandler(ExternalClientHandler):
return task_id
return None
def list_files(self, release_data: dict[str, Any]) -> list[PackFile] | None:
"""Read the torrent's file list off the detail page, without downloading."""
raw_url = release_data.get("download_url") or release_data.get("source_url")
detail_url = raw_url.strip() if isinstance(raw_url, str) else ""
hostname = _resolve_allowed_detail_hostname()
if not detail_url or not _detail_url_matches_host(detail_url, hostname):
logger.debug("Cannot list files for AudiobookBay release without a valid detail URL")
return None
detail_html = scraper.fetch_detail_html(detail_url, hostname)
if not detail_html:
return None
return scraper.extract_file_list(detail_html)
def _get_client(self, protocol: str) -> DownloadClient | None:
"""Compatibility shim so module-level patching still works in tests."""
return get_client(protocol)
+32 -128
View File
@@ -2,8 +2,7 @@
import re
import time
from threading import Lock
from urllib.parse import quote, quote_plus
from urllib.parse import quote
import requests
from bs4 import BeautifulSoup
@@ -11,8 +10,6 @@ from bs4 import BeautifulSoup
from shelfmark.core.config import config
from shelfmark.core.logger import setup_logger
from shelfmark.download import http as downloader
from shelfmark.download.postprocess.packs import PackFile
from shelfmark.release_sources.audiobookbay.utils import normalize_search_punctuation
logger = setup_logger(__name__)
@@ -34,13 +31,6 @@ FIRST_PAGE_SESSION_REFRESH_ATTEMPTS = 2
# Legacy search parameter used by older ABB flows
LEGACY_CATEGORY_QUERY = "undefined%2Cundefined"
# Detail pages are fetched once and shared by inspection (file list) and download
# (magnet link) so a "review then download" round trip costs ABB a single request.
DETAIL_PAGE_CACHE_TTL_SECONDS = 120.0
DETAIL_PAGE_CACHE_MAX_ENTRIES = 8
_detail_page_cache: dict[str, tuple[float, str]] = {}
_detail_page_cache_lock = Lock()
# Precompiled patterns used while parsing result cards
LANGUAGE_PATTERN = re.compile(r"Language:\s*([A-Za-z]+)")
POSTED_PATTERN = re.compile(r"Posted:\s*(\d+\s+[A-Za-z]+\s+\d{4})")
@@ -48,11 +38,6 @@ FORMAT_PATTERN = re.compile(r"Format:\s*([A-Za-z0-9]+)")
BITRATE_PATTERN = re.compile(r"Bitrate:\s*([\d]+\s*[A-Za-z/]+)")
SIZE_PATTERN = re.compile(r"File Size:\s*([\d.]+)\s*([A-Za-z]+)")
INFO_HASH_LABEL_PATTERN = re.compile(r"Info Hash", re.IGNORECASE)
FILE_ROW_SIZE_PATTERN = re.compile(
r"^(?P<name>.+?)\s+(?P<size>\d+(?:\.\d+)?)\s*(?P<unit>Bytes?|KBs?|MBs?|GBs?|TBs?)$",
re.IGNORECASE,
)
_FILE_SIZE_MULTIPLIERS = {"b": 1, "k": 1024, "m": 1024**2, "g": 1024**3, "t": 1024**4}
def _coerce_non_negative_float(value: object, default: float) -> float:
@@ -113,10 +98,8 @@ def _encode_search_query(query: str, *, exact_phrase: bool) -> str:
and not (search_query.startswith('"') and search_query.endswith('"'))
):
search_query = f'"{search_query}"'
# Keep ABB's space-as-'+' style, but percent-encode everything else: a bare
# '&' would otherwise start a new query parameter, '%' would open an invalid
# escape, and a literal '+' would arrive as a space.
return quote_plus(search_query)
# Keep ABB-friendly encoding style (spaces as '+') while percent-encoding quotes.
return search_query.replace('"', "%22").replace(" ", "+")
def _normalize_result_url(url: str, hostname: str) -> str:
@@ -170,9 +153,6 @@ def search_audiobookbay(
"""
results = []
# ABB matches the stored, untexturized title, so a curly apostrophe reaching
# the search returns nothing at all rather than merely ranking worse.
query = normalize_search_punctuation(query)
rate_limit_delay = _coerce_non_negative_float(config.get("ABB_RATE_LIMIT_DELAY", 1.0), 1.0)
session = requests.Session()
@@ -362,98 +342,6 @@ def search_audiobookbay(
return results
def _get_cached_detail_page(details_url: str) -> str | None:
with _detail_page_cache_lock:
entry = _detail_page_cache.get(details_url)
if entry is None:
return None
fetched_at, html = entry
if time.monotonic() - fetched_at > DETAIL_PAGE_CACHE_TTL_SECONDS:
del _detail_page_cache[details_url]
return None
return html
def _store_cached_detail_page(details_url: str, html: str) -> None:
with _detail_page_cache_lock:
_detail_page_cache[details_url] = (time.monotonic(), html)
while len(_detail_page_cache) > DETAIL_PAGE_CACHE_MAX_ENTRIES:
oldest = min(_detail_page_cache, key=lambda key: _detail_page_cache[key][0])
del _detail_page_cache[oldest]
def clear_detail_page_cache() -> None:
"""Drop cached detail pages (used by tests)."""
with _detail_page_cache_lock:
_detail_page_cache.clear()
def _fetch_detail_page_once(details_url: str, hostname: str) -> str:
session = requests.Session()
_bootstrap_abb_session(hostname, session, DETAIL_PAGE_RETRY_ATTEMPTS)
return _coerce_markup_to_html(
downloader.html_get_page(
details_url,
retry=DETAIL_PAGE_RETRY_ATTEMPTS,
use_bypasser=False,
allow_bypasser_fallback=False,
success_delay=0,
session=session,
)
)
def fetch_detail_html(details_url: str, hostname: str = "audiobookbay.lu") -> str:
"""Fetch a detail page (one retry with a fresh session), cached briefly per URL."""
cached = _get_cached_detail_page(details_url)
if cached is not None:
logger.debug("Reusing recently fetched detail page: %s", details_url)
return cached
detail_html = _fetch_detail_page_once(details_url, hostname)
if not detail_html:
detail_html = _fetch_detail_page_once(details_url, hostname)
if detail_html:
_store_cached_detail_page(details_url, detail_html)
return detail_html
def _parse_file_row(text: str) -> PackFile | None:
match = FILE_ROW_SIZE_PATTERN.match(text.strip())
if not match:
return None
multiplier = _FILE_SIZE_MULTIPLIERS[match.group("unit")[0].lower()]
return PackFile(match.group("name"), int(float(match.group("size")) * multiplier))
def extract_file_list(detail_html: str) -> list[PackFile] | None:
"""Read the torrent file rows off a detail page.
ABB renders the torrent's file table as single-cell rows between the
"This is a Multifile Torrent" marker (absent for single-file torrents) and the
"Combined File Size" row. Returns None when the page has no such table.
"""
soup = BeautifulSoup(detail_html, "html.parser")
rows: list[PackFile] = []
for row in soup.find_all("tr"):
cells = row.find_all("td")
if not cells:
continue
label = cells[0].get_text(" ", strip=True)
if label.lower().startswith("combined file size"):
return rows or None
if len(cells) != 1:
rows = [] # a two-column metadata row means we're not in the file table yet
continue
text = cells[0].get_text(" ", strip=True)
if "multifile torrent" in text.lower():
rows = []
continue
parsed = _parse_file_row(text)
if parsed is not None:
rows.append(parsed)
return None
def extract_magnet_link(details_url: str, hostname: str = "audiobookbay.lu") -> str | None:
"""Extract info hash and trackers from book detail page, then construct magnet link.
@@ -466,7 +354,35 @@ def extract_magnet_link(details_url: str, hostname: str = "audiobookbay.lu") ->
"""
try:
detail_html = fetch_detail_html(details_url, hostname)
session = requests.Session()
_bootstrap_abb_session(hostname, session, DETAIL_PAGE_RETRY_ATTEMPTS)
# Fetch detail page
detail_html = _coerce_markup_to_html(
downloader.html_get_page(
details_url,
retry=DETAIL_PAGE_RETRY_ATTEMPTS,
use_bypasser=False,
allow_bypasser_fallback=False,
success_delay=0,
session=session,
)
)
if not detail_html:
session = requests.Session()
_bootstrap_abb_session(hostname, session, DETAIL_PAGE_RETRY_ATTEMPTS)
detail_html = _coerce_markup_to_html(
downloader.html_get_page(
details_url,
retry=DETAIL_PAGE_RETRY_ATTEMPTS,
use_bypasser=False,
allow_bypasser_fallback=False,
success_delay=0,
session=session,
)
)
if not detail_html:
logger.warning("Failed to fetch details page")
return None
@@ -501,18 +417,6 @@ def extract_magnet_link(details_url: str, hostname: str = "audiobookbay.lu") ->
# Clean up info hash (remove whitespace, ensure uppercase)
info_hash = re.sub(r"\s+", "", info_hash).upper()
# Validate: SHA1 = 40 hex chars, SHA256 = 64 hex chars
if not re.match(r"^[0-9A-F]{40}$|^[0-9A-F]{64}$", info_hash):
logger.warning("Info Hash invalid (got %r), trying magnet fallback.", info_hash)
# Fallback: search entire page for a complete magnet link (e.g. posted in comments)
magnet_match = re.search(r"magnet:\?xt=urn:btih:([0-9a-fA-F]{40,64})", detail_html)
if magnet_match:
info_hash = magnet_match.group(1).upper()
logger.info("Found hash via magnet fallback: %s", info_hash)
else:
logger.warning("No valid magnet link found on page, giving up.")
return None
# 2. Extract Trackers
# Find all <td> containing udp:// or http://
trackers = []
@@ -9,7 +9,6 @@ if TYPE_CHECKING:
from shelfmark.metadata_providers import BookMetadata
from shelfmark.core.config import config
from shelfmark.core.languages import normalize_language
from shelfmark.core.logger import setup_logger
from shelfmark.release_sources import (
ColumnAlign,
@@ -23,11 +22,7 @@ from shelfmark.release_sources import (
register_source,
)
from shelfmark.release_sources.audiobookbay import scraper
from shelfmark.release_sources.audiobookbay.utils import (
normalize_hostname,
normalize_search_punctuation,
parse_size,
)
from shelfmark.release_sources.audiobookbay.utils import normalize_hostname, parse_size
logger = setup_logger(__name__)
MIN_RELEVANCE_QUERY_WORD_LENGTH = 2
@@ -48,6 +43,41 @@ def _coerce_positive_int(value: object, default: int) -> int:
# Map language names to ISO 639-1 codes (matching frontend color maps)
LANGUAGE_MAP = {
"english": "en",
"spanish": "es",
"french": "fr",
"german": "de",
"italian": "it",
"portuguese": "pt",
"russian": "ru",
"japanese": "ja",
"chinese": "zh",
"dutch": "nl",
"swedish": "sv",
"norwegian": "no",
"danish": "da",
"finnish": "fi",
"polish": "pl",
"czech": "cs",
"hungarian": "hu",
"korean": "ko",
"arabic": "ar",
"hebrew": "he",
"turkish": "tr",
"greek": "el",
"hindi": "hi",
"thai": "th",
"vietnamese": "vi",
"indonesian": "id",
"ukrainian": "uk",
"romanian": "ro",
"bulgarian": "bg",
"catalan": "ca",
"croatian": "hr",
"slovenian": "sl",
"serbian": "sr",
}
def _split_title_and_author(raw_title: str) -> tuple[str, str | None]:
@@ -89,9 +119,8 @@ def _map_language(language: str) -> str | None:
if not language:
return None
# Fall back to the raw value so an unrecognised language is still shown
# rather than silently dropped from the release row.
return normalize_language(language) or language.lower().strip()
lang_lower = language.lower().strip()
return LANGUAGE_MAP.get(lang_lower, lang_lower)
def _parse_bitrate_to_kbps(bitrate: str | None) -> int | None:
@@ -209,8 +238,8 @@ class AudiobookBaySource(ReleaseSource):
exact_phrase=exact_phrase,
)
# Fallback to broad matching if exact phrase returns nothing (manual or auto query).
if exact_phrase and not results:
# For auto-generated queries, fallback to broad matching if exact phrase returns nothing.
if exact_phrase and not results and not plan.manual_query:
logger.info(
"No exact phrase results, retrying AudiobookBay search without quotes"
)
@@ -231,12 +260,10 @@ class AudiobookBaySource(ReleaseSource):
deduped_queries[index + 1].lower(),
)
# Extract query words for relevance checking. Both sides of the
# comparison are punctuation-normalized: scraped titles carry the
# typographic forms WordPress renders, queries carry the ASCII ones.
# Extract query words for relevance checking
query_words = {
word.lower()
for word in normalize_search_punctuation(query_lower).split()
for word in query_lower.split()
if len(word) > MIN_RELEVANCE_QUERY_WORD_LENGTH
}
@@ -245,7 +272,7 @@ class AudiobookBaySource(ReleaseSource):
try:
raw_title = result["title"]
title, author = _split_title_and_author(raw_title)
title_for_filter = normalize_search_punctuation(raw_title).lower()
title_for_filter = raw_title.lower()
# Basic relevance check: ensure title contains at least one query word
# This filters out homepage "Latest" feed items that may leak through
@@ -261,7 +288,7 @@ class AudiobookBaySource(ReleaseSource):
size_str = result.get("size")
size_bytes = parse_size(size_str) if size_str else None
language_raw = result.get("language")
language_code = _map_language(language_raw) if language_raw else "en"
language_code = _map_language(language_raw) if language_raw else None
bitrate = result.get("bitrate")
bitrate_kbps = _parse_bitrate_to_kbps(bitrate)
@@ -2,63 +2,6 @@
import re
# WordPress texturizes punctuation on output only: a post stored as "The
# Stranger's Wife" is rendered as "The Stranger’s Wife". ABB's search matches the
# stored value, so a query carrying the typographic form matches nothing -- and
# because ABB ANDs its search terms, one such term empties the entire result set.
# Book metadata and phone keyboards both hand us the typographic forms, so map
# them back before they reach a search or a title comparison.
_ASCII_PUNCTUATION = str.maketrans(
{
# Single quotes
"‘": "'", # left single quotation mark
"’": "'", # right single quotation mark
"‚": "'", # single low-9 quotation mark
"‛": "'", # single high-reversed-9 quotation mark
"′": "'", # prime
"´": "'", # acute accent
"`": "'", # grave accent
# Double quotes
"“": '"', # left double quotation mark
"”": '"', # right double quotation mark
"„": '"', # double low-9 quotation mark
"‟": '"', # double high-reversed-9 quotation mark
"″": '"', # double prime
# Dashes
"‐": "-", # hyphen
"‑": "-", # non-breaking hyphen
"‒": "-", # figure dash
"–": "-", # en dash
"—": "-", # em dash
"―": "-", # horizontal bar
"−": "-", # minus sign
"﹘": "-", # small em dash
"﹣": "-", # small hyphen-minus
"-": "-", # fullwidth hyphen-minus
# Ellipsis
"…": "...", # horizontal ellipsis
}
)
def normalize_search_punctuation(text: str) -> str:
"""Replace typographic punctuation with the ASCII forms ABB stores.
Each character is mapped individually rather than collapsing runs, so an
ASCII "--" is left alone: only characters ABB cannot have stored are
rewritten.
Args:
text: A search query, or a scraped title being compared against one.
Returns:
The text with curly quotes, dashes and ellipses mapped to ASCII.
"""
if not text:
return text
return text.translate(_ASCII_PUNCTUATION)
def normalize_hostname(raw: str | None) -> str:
"""Normalize a user-supplied hostname for URL construction.
+37 -581
View File
@@ -3,14 +3,9 @@
import itertools
import json
import re
import threading
import time
import unicodedata
from contextlib import contextmanager
from contextvars import ContextVar
from dataclasses import replace
from http import HTTPStatus
from pathlib import Path
from typing import TYPE_CHECKING, ClassVar, NoReturn, TypedDict
from urllib.parse import quote, urlparse
@@ -18,11 +13,8 @@ import requests
from bs4 import BeautifulSoup, Tag
from bs4.element import NavigableString
from shelfmark.bypass.challenge import MAX_CHALLENGE_HTML_CHARS, challenge_marker
from shelfmark.config.env import DEBUG_SKIP_SOURCES, TMP_DIR
from shelfmark.core import search_deadline
from shelfmark.core.config import config
from shelfmark.core.languages import language_alias_map
from shelfmark.core.logger import setup_logger
from shelfmark.core.models import DownloadTask, SearchFilters, build_filename
from shelfmark.core.utils import CONTENT_TYPES, get_aa_content_type_dir
@@ -46,7 +38,7 @@ from shelfmark.release_sources import (
)
if TYPE_CHECKING:
from collections.abc import Callable, Iterable, Iterator
from collections.abc import Callable, Iterable
from pathlib import Path
from threading import Event
@@ -115,16 +107,6 @@ def _html_response_text(response: str | tuple[str, str]) -> str:
return response
def _html_response_url(response: str | tuple[str, str]) -> str | None:
"""The URL that actually answered, when the downloader was asked to report it.
None for the plain-string shape, so a caller can fall back to what it requested.
"""
if isinstance(response, tuple):
return response[1] or None
return None
def _attr_to_str(value: object) -> str | None:
"""Convert a BeautifulSoup attribute value to a plain string."""
if isinstance(value, str):
@@ -215,49 +197,6 @@ _SOURCE_FAILURE_THRESHOLD = 4
_MIN_VALID_FILE_SIZE = 10 * 1024
_AA_COUNTDOWN_MAX_SECONDS = 300
# --- Distant-path language detection ---
_DISTANT_PATH_EXTENSIONS = (
"epub",
"mobi",
"azw3",
"fb2",
"djvu",
"cbz",
"cbr",
"pdf",
"zip",
"rar",
"m4b",
"mp3",
)
_DISTANT_PATH_EXTENSION_PATTERN = "|".join(re.escape(e) for e in _DISTANT_PATH_EXTENSIONS)
_DISTANT_PATH_PATTERN = re.compile(
rf"(?:[A-Za-z0-9._-]+/)?[A-Za-z]:(?:\\|/)[^\n\r<>\"]+?\.(?:{_DISTANT_PATH_EXTENSION_PATTERN})\b",
re.IGNORECASE,
)
_DISTANT_PATH_FALLBACK_PATTERN = re.compile(
r"(?:[A-Za-z0-9._-]+/)?[A-Za-z]:(?:\\|/)[^\n\r<>\"]+",
re.IGNORECASE,
)
_BRACKETED_LANGUAGE_CODE_PATTERN = re.compile(
r"\[(?:bd[\s._-]*)?([A-Za-z]{2,3})\]",
re.IGNORECASE,
)
_KEYED_LANGUAGE_CODE_PATTERN = re.compile(
r"\b(?:bd|lang(?:uage)?)\s*[:._-]?\s*([A-Za-z]{2,3})\b",
re.IGNORECASE,
)
_LANGUAGE_CODE_TOKEN_PATTERN = re.compile(
r"(?:^|[\s_./\\\-\[(])([A-Za-z]{2,3})(?=$|[\s_./\\\-)\]])"
)
_LANGUAGE_NAME_TOKEN_PATTERN = re.compile(r"[a-z]{4,}(?:-[a-z0-9]+)?")
_LANGUAGE_ALIAS_TO_CODE: dict[str, str] | None = None
_LANGUAGE_ALIAS_LOCK = threading.Lock()
_LANGUAGE_PLACEHOLDERS = frozenset({"", "-", "--", "unknown", "unk", "n/a", "na"})
# Short codes that appear in common words — require bracket/key context to accept
_AMBIGUOUS_SHORT_LANGUAGE_CODES = frozenset({"de", "en", "it", "la", "no", "or", "is", "in"})
# Sources that require Cloudflare bypass
_CF_BYPASS_REQUIRED = frozenset({"aa-slow-nowait", "aa-slow-wait", "zlib", "welib"})
@@ -265,161 +204,6 @@ _CF_BYPASS_REQUIRED = frozenset({"aa-slow-nowait", "aa-slow-wait", "zlib", "weli
_AA_PAGE_SOURCES = frozenset({"aa-slow-nowait", "aa-slow-wait"})
def _is_language_from_path_enabled() -> bool:
return bool(config.get("DIRECT_DOWNLOAD_LANGUAGE_FROM_PATH", False))
def _normalize_language_token(value: str) -> str:
normalized = value.strip().lower()
for dash in ("‑", "–", "—", "−"):
normalized = normalized.replace(dash, "-")
return normalized
def _fold_text(value: str) -> str:
normalized = unicodedata.normalize("NFKD", value)
return "".join(c for c in normalized if not unicodedata.combining(c)).lower()
def _language_alias_to_code() -> dict[str, str]:
"""Alias to code map, delegating to the shared language data."""
global _LANGUAGE_ALIAS_TO_CODE
cached = _LANGUAGE_ALIAS_TO_CODE
if cached is not None:
return cached
with _LANGUAGE_ALIAS_LOCK:
cached = _LANGUAGE_ALIAS_TO_CODE
if cached is not None:
return cached
_LANGUAGE_ALIAS_TO_CODE = language_alias_map()
return _LANGUAGE_ALIAS_TO_CODE
def _extract_distant_path(row: Tag, *, enabled: bool) -> str | None:
"""Extract the Windows-style file path from an AA search result row."""
if not enabled:
return None
def _normalize_candidate(text: str) -> str:
normalized = re.sub(r"\s*([\\/])\s*", r"\1", text)
normalized = re.sub(r":\s*([\\/])", r":\1", normalized)
return re.sub(
r"\s+\.(epub|mobi|azw3|fb2|djvu|cbz|cbr|pdf|zip|rar|m4b|mp3)\b",
r".\1",
normalized,
flags=re.IGNORECASE,
)
candidates = [row.get_text(" ", strip=True)]
for cell in row.find_all("td"):
cell_text = cell.get_text(" ", strip=True)
if cell_text:
candidates.append(cell_text)
best: str | None = None
for text in candidates:
for match in _DISTANT_PATH_PATTERN.findall(_normalize_candidate(text)):
candidate = match.strip().rstrip(".,;")
if best is None or len(candidate) > len(best):
best = candidate
if best is not None:
return best
for text in candidates:
for match in _DISTANT_PATH_FALLBACK_PATTERN.findall(_normalize_candidate(text)):
candidate = match.strip().rstrip(".,;")
if best is None or len(candidate) > len(best):
best = candidate
return best
def _detect_language_from_distant_path(path: str | None) -> str | None:
"""Infer a language code from distant-path tags such as [BD FR] or [Fr]."""
if not path:
return None
aliases = _language_alias_to_code()
if not aliases:
return None
folded_path = _fold_text(path)
strong_candidates: list[str] = []
for code in _BRACKETED_LANGUAGE_CODE_PATTERN.findall(path):
normalized = _normalize_language_token(code)
if normalized in aliases:
strong_candidates.append(aliases[normalized])
for code in _KEYED_LANGUAGE_CODE_PATTERN.findall(path):
normalized = _normalize_language_token(code)
if normalized in aliases:
strong_candidates.append(aliases[normalized])
non_ambiguous = [c for c in strong_candidates if c not in _AMBIGUOUS_SHORT_LANGUAGE_CODES]
if non_ambiguous:
return non_ambiguous[0]
for token in _LANGUAGE_NAME_TOKEN_PATTERN.findall(folded_path):
normalized = _normalize_language_token(token)
if normalized in aliases:
candidate = aliases[normalized]
if candidate not in _AMBIGUOUS_SHORT_LANGUAGE_CODES:
return candidate
if strong_candidates:
return strong_candidates[0]
for code in _LANGUAGE_CODE_TOKEN_PATTERN.findall(path):
normalized = _normalize_language_token(code)
if normalized in _AMBIGUOUS_SHORT_LANGUAGE_CODES:
continue
if normalized in aliases:
return aliases[normalized]
return None
def _is_missing_or_placeholder_language(language: str | None) -> bool:
if language is None:
return True
return _normalize_language_token(language) in _LANGUAGE_PLACEHOLDERS
def _normalize_requested_languages(languages: list[str] | None) -> set[str]:
if not languages:
return set()
aliases = _language_alias_to_code()
normalized: set[str] = set()
for value in languages:
token = _normalize_language_token(str(value))
if not token or token == "all": # noqa: S105 - "all" is a language sentinel
continue
normalized.add(aliases.get(token, token))
return normalized
def _book_matches_requested_languages(book_language: str | None, requested: set[str]) -> bool:
"""Return True when a book's language matches the requested filter.
Books with unknown/missing language always pass — the server-side &lang= filter
already narrowed the result set, so dropping unlabelled rows hides valid results.
"""
if not requested:
return True
if not book_language:
return True
aliases = _language_alias_to_code()
normalized_book = aliases.get(
_normalize_language_token(book_language),
_normalize_language_token(book_language),
)
return normalized_book in requested
def _is_configured_zlib_link(url: str) -> bool:
"""Return True when a URL belongs to a configured Z-Library mirror."""
from shelfmark.core.mirrors import get_zlib_cookie_domains
@@ -553,237 +337,6 @@ class SearchUnavailableError(SourceUnavailableError):
"""Raised when Anna's Archive cannot be reached via any mirror/DNS."""
# Markers that prove a 200 really came from Anna's Archive, and markers that mean we
# are looking at a protection interstitial rather than the site. A page with neither
# is a domain that answers but is not AA - seized, parked or for sale.
#
# Deliberately structural rather than the domain name: a parking page's whole job is
# to display the domain it is squatting on, so "annas-archive" matches the very pages
# this is meant to catch. These paths only exist on the real site.
_AA_PAGE_MARKERS = (
"/md5/",
"aarecord",
"anna's archive",
"/dyn/",
"/datasets",
"/fast_download",
"/slow_download",
)
def _looks_like_aa_page(html: str) -> bool:
"""Whether ``html`` is recognisably Anna's Archive itself."""
lowered = html.lower()
return any(marker in lowered for marker in _AA_PAGE_MARKERS)
def _looks_like_challenge_page(html: str) -> bool:
"""Whether ``html`` is a protection interstitial rather than the site behind it.
Delegates to the shared detector rather than substring-matching here. A bare
"ddos-guard"/"cloudflare" scan flags the protected site's *own* pages: DDoS-Guard
links its endpoints on everything it fronts, and AA ships a `DDOS-GUARD` comment in
the inline JS on every page it serves. That misread every real AA response that was
not a results table as an unsolved challenge, and sent users off to fix a bypasser
that had just succeeded - see #1289/#1292. `challenge_marker` caps its scan at
64 KB, which is what separates a few-KB interstitial from the page behind it.
"""
return challenge_marker(html) is not None
# Pages already fetched during the search in flight, keyed by URL. Scoped to one
# DirectDownload.search() so nothing is carried between requests.
_search_page_cache: ContextVar[dict[str, tuple[str, Tag | None]] | None] = ContextVar(
"aa_search_page_cache", default=None
)
@contextmanager
def _search_page_reuse() -> Iterator[None]:
"""Fetch each distinct AA search URL at most once per search.
One search asks AA for the same URL more than once. The language-filter retry in
`search()` re-runs every title variant, and when DIRECT_DOWNLOAD_LANGUAGE_FROM_PATH
is on the requested language is applied locally instead of as `&lang=`, so both
passes build a byte-identical URL - the retry differs only in the filtering it does
to the response it already had. A repeat is not a cheap round trip either: AA is
behind DDoS-Guard, so each one is a fresh browser solve, tens of seconds that buy
nothing. See issue #1285.
"""
token = _search_page_cache.set({})
try:
yield
finally:
_search_page_cache.reset(token)
def _is_reusable_answer(result: tuple[str, Tag | None]) -> bool:
"""Whether a fetched page is an answer, rather than a giving-up worth retrying.
`_fetch_search_table_uncached` exists to rotate past mirrors that are not actually AA,
and when it runs out of them it *returns* instead of raising: a page with no results
table and no marker. Storing that would hand the language-filter retry - the pass this
cache exists for - a mirror set that may have recovered in between (DNS rotation, a
mirror coming back), turning a transient outage into "this book has no releases". A
real "No files found." is an answer and is worth keeping.
"""
html, tbody = result
return tbody is not None or "No files found." in html or _looks_like_aa_page(html)
# How much of an unreadable search page to quote in the debug log. Enough to carry the
# <head> - title, injected challenge scripts - without pasting a 180 KB page into a log
# file that ships inside the debug bundle.
_PAGE_FINGERPRINT_CHARS = 700
_TITLE_RE = re.compile(r"<title[^>]*>(.*?)</title>", re.IGNORECASE | re.DOTALL)
def _log_untabled_search_page(url: str, html: str) -> None:
"""Record why a search page with no results table is about to be classified.
#1289 cost a full investigation because the log said only "unsolved protection
challenge" while FlareSolverr said "Challenge solved!", and the debug bundle carries
no response bodies - there was no way to tell a real AA page from an interstitial
after the fact. These are the facts that would have settled it in one line: the size
(the 64 KB cap is what separates the two), which markers matched, and the head of
the document.
Diagnostics must never be the reason a search fails, so this swallows its own errors.
"""
try:
title_match = _TITLE_RE.search(html[: _PAGE_FINGERPRINT_CHARS * 4])
title = " ".join(title_match.group(1).split())[:120] if title_match else "<none>"
lowered = html.lower()
aa_markers = [marker for marker in _AA_PAGE_MARKERS if marker in lowered]
logger.info(
"Search page has no results table: %s (bytes=%d, title=%r, aa_markers=%s, "
"challenge_marker=%r, over_challenge_size_cap=%s)",
url,
len(html),
title,
aa_markers or "none",
challenge_marker(html),
len(html) > MAX_CHALLENGE_HTML_CHARS,
)
logger.debug(
"Untabled search page head (%d of %d bytes): %s",
min(len(html), _PAGE_FINGERPRINT_CHARS),
len(html),
html[:_PAGE_FINGERPRINT_CHARS],
)
except Exception:
logger.debug("Could not fingerprint the untabled search page", exc_info=True)
def _fetch_search_table(url: str, selector: network.AAMirrorSelector) -> tuple[str, Tag | None]:
"""Fetch the AA search page, reusing one already fetched during this search."""
cache = _search_page_cache.get()
if cache is not None and url in cache:
logger.debug("Reusing search page already fetched for this search: %s", url)
return cache[url]
result = _fetch_search_table_uncached(url, selector)
if cache is not None and _is_reusable_answer(result):
cache[url] = result
return result
def _fetch_search_table_uncached(
url: str, selector: network.AAMirrorSelector
) -> tuple[str, Tag | None]:
"""Fetch the AA search page, retrying past mirrors that are not actually AA.
A parked or seized domain answers 200 with a page that has no results table and no
"No files found." - indistinguishable from a broken search unless we check whether
the response looks like AA at all. Those mirrors are quarantined for the session so
later searches skip them instead of paying the timeout again.
"""
attempt_url = url
for _ in range(len(network.get_available_aa_urls()) or 1):
# Every mirror shares the protection, so once the search budget is gone another
# mirror is another full solve nobody is still waiting for.
if search_deadline.expired():
raise SearchUnavailableError(search_deadline.deadline_message())
# include_response_url is what makes the diagnostics below name the mirror that
# actually answered. html_get_page rotates mirrors and follows redirects on its
# own, so `attempt_url` is only where this iteration started: #1298's bundle
# reported the untabled page against annas-archive.gl when the body had come
# from .pk, which is precisely the triage cost #1289 added the line to remove.
response = downloader.html_get_page(
attempt_url,
selector=selector,
allow_bypasser_fallback=True,
include_response_url=True,
)
html = _html_response_text(response)
# Checked on the body, not on `response`: with include_response_url the give-up
# shape is the tuple ("", url), and a tuple is truthy.
if not html:
# Network/mirror exhaustion path bubbles up so API can notify clients.
# html_get_page records the concrete give-up reason on the selector; fall
# back to the generic line only if nothing was recorded.
detail = getattr(selector, "last_failure", None) or (
"Network restricted or mirrors are blocked."
)
raise SearchUnavailableError(f"Unable to reach download source. {detail}")
answered_url = _html_response_url(response) or attempt_url
soup = BeautifulSoup(html, "html.parser")
table = soup.find("table")
if isinstance(table, Tag):
return html, table
if table is not None:
msg = f"Expected results table tag, got {type(table).__name__}"
raise TypeError(msg)
if "No files found." in html:
# A real, genuinely empty answer from a healthy mirror.
return html, None
# A search page with no table is the one shape we cannot read off the response
# alone, and the response body is not in the debug bundle. Fingerprint it here
# so the next report says which branch fired and why, rather than costing
# another round of guesswork - see #1289.
_log_untabled_search_page(answered_url, html)
if _looks_like_aa_page(html):
# A real AA response in a shape the caller should report as drift. Checked
# ahead of the challenge branch: AA's own pages carry the protection's
# markers, so an interstitial is only the better explanation once the page
# has nothing of AA's about it. A genuine interstitial has no AA markers.
return html, None
if _looks_like_challenge_page(html):
# The bypass did not actually clear the protection - the interstitial is
# what came back. Rotating is pointless (every mirror shares the same
# protection) and reporting it as an empty result is worse: the user is
# told their query found nothing when the search never ran.
#
# The wording no longer blames the bypasser outright. In #1292 it was
# reachable and working, and the page it was handed was DDoS-Guard's manual
# CAPTCHA - so "check that the bypasser is working" was the one piece of
# advice guaranteed to waste the reporter's time. Name the marker instead
# and let the two causes be told apart.
msg = (
"Anna's Archive answered with a protection challenge that was not "
f"cleared (marker={challenge_marker(html)!r}). If the bypasser reports "
"solving it, the host is serving a manual CAPTCHA that no bypasser can "
"answer - try again shortly. Otherwise check that the bypasser is "
"reachable and working."
)
raise SearchUnavailableError(msg)
new_base, action = selector.next_mirror_or_rotate_dns(
fatal=True, reason="responded without an Anna's Archive page"
)
if action not in ("mirror", "dns") or not new_base:
return html, None
attempt_url = selector.rewrite(url)
logger.info("Retrying search on %s", new_base)
return "", None
def search_books(query: str, filters: SearchFilters) -> list[BrowseRecord]:
"""Search for books matching the query.
@@ -807,17 +360,9 @@ def search_books(query: str, filters: SearchFilters) -> list[BrowseRecord]:
filters_query = ""
path_language_enabled = _is_language_from_path_enabled()
requested_langs = _normalize_requested_languages(filters.lang)
# When path-language inference is on and a language is requested, skip the
# server-side &lang= filter: lgli files often have no AA language metadata
# and would be excluded before we can infer language from the distant path.
# Local filtering below handles the narrowing instead.
if not (path_language_enabled and requested_langs):
for value in filters.lang or []:
if value and value != "all":
filters_query += f"&lang={quote(value)}"
for value in filters.lang or []:
if value and value != "all":
filters_query += f"&lang={quote(value)}"
if filters.sort and filters.sort != "relevance":
filters_query += f"&sort={quote(filters.sort)}"
@@ -846,13 +391,20 @@ def search_books(query: str, filters: SearchFilters) -> list[BrowseRecord]:
f"{filters_query}"
)
# AA gates /search behind a DDoS-Guard JS challenge, which every mirror shares. Rotating
# to another mirror only collects another 403, so let the bypasser solve it.
html, tbody = _fetch_search_table(url, selector)
html = downloader.html_get_page(url, selector=selector, allow_bypasser_fallback=False)
if not html:
# Network/mirror exhaustion path bubbles up so API can notify clients
msg = "Unable to reach download source. Network restricted or mirrors are blocked."
raise SearchUnavailableError(msg)
if "No files found." in html:
logger.info("No books found for query: %s", query)
return []
soup = BeautifulSoup(_html_response_text(html), "html.parser")
tbody = soup.find("table")
if tbody is None:
if "No files found." in html:
logger.info("No books found for query: %s", query)
return []
logger.warning("No results table found for query: %s", query)
msg = "No books found. Please try another query."
raise RuntimeError(msg)
@@ -866,9 +418,6 @@ def search_books(query: str, filters: SearchFilters) -> list[BrowseRecord]:
if book:
books.append(book)
if path_language_enabled and requested_langs:
books = [b for b in books if _book_matches_requested_languages(b.language, requested_langs)]
supported_formats = _get_supported_formats()
books.sort(
@@ -896,14 +445,11 @@ def get_book_info(book_id: str, *, fetch_download_count: bool = True) -> BrowseR
"""
url = f"{network.get_aa_base_url()}/md5/{book_id}"
selector = network.AAMirrorSelector()
# Same challenge as search: the detail page is gated on every mirror, so bypass it.
html = downloader.html_get_page(url, selector=selector, allow_bypasser_fallback=True)
html = downloader.html_get_page(url, selector=selector, allow_bypasser_fallback=False)
if not html:
detail = getattr(selector, "last_failure", None) or (
"Network restricted or mirrors are blocked."
)
raise SearchUnavailableError(f"Unable to reach download source. {detail}")
msg = "Unable to reach download source. Network restricted or mirrors are blocked."
raise SearchUnavailableError(msg)
soup = BeautifulSoup(_html_response_text(html), "html.parser")
@@ -925,23 +471,10 @@ def _parse_search_result_row(row: Tag) -> BrowseRecord | None:
if not record_id:
return None
path_language_enabled = _is_language_from_path_enabled()
distant_path = _extract_distant_path(row, enabled=path_language_enabled)
preview_img = cells[0].find("img")
preview = _get_attr(preview_img, "src") if isinstance(preview_img, Tag) else None
title_span = cells[1].find("span")
if isinstance(title_span, Tag):
# AA nests related-edition spans inside the main title span — take only direct text.
direct = " ".join(
str(c).strip()
for c in title_span.children
if isinstance(c, NavigableString) and str(c).strip()
).strip()
title = direct or _first_stripped_text(title_span)
else:
title = None
title = _first_stripped_text(cells[1].find("span"))
author = _first_stripped_text(cells[2].find("span"))
publisher = _first_stripped_text(cells[3].find("span"))
year = _first_stripped_text(cells[4].find("span"))
@@ -950,19 +483,18 @@ def _parse_search_result_row(row: Tag) -> BrowseRecord | None:
file_format = _first_stripped_text(cells[9].find("span"))
size = _first_stripped_text(cells[10].find("span"))
# Only title and format are truly required — lgli rows often have sparse metadata
if title is None or file_format is None:
if (
title is None
or author is None
or publisher is None
or year is None
or language is None
or content is None
or file_format is None
or size is None
):
return None
# Skip entries where the title is a catalog format descriptor, not a real title
# e.g. "Book/Online Audio", "Print book" — lgli metadata pollution
if title and "/" in title and len(title) < 40 and not any(c.isdigit() for c in title):
return None
if path_language_enabled and _is_missing_or_placeholder_language(language):
detected = _detect_language_from_distant_path(distant_path)
language = detected or "unknown"
return BrowseRecord(
id=record_id,
title=title,
@@ -975,7 +507,6 @@ def _parse_search_result_row(row: Tag) -> BrowseRecord | None:
content=content.lower() if content else None,
format=file_format.lower() if file_format else None,
size=size,
download_path=distant_path,
)
except (AttributeError, IndexError, KeyError, TypeError) as e:
logger.error_trace(f"Error parsing search result row: {e}")
@@ -1127,9 +658,6 @@ def _parse_book_info_page(
if fetch_download_count:
try:
summary_url = f"{network.get_aa_base_url()}/dyn/md5/summary/{book_id}"
# Unlike search and the detail page above, this one stays off the bypasser: a
# download count is decoration on the details modal, not worth holding the
# modal open for a browser solve. If it is gated, drop it and move on.
summary_response = downloader.html_get_page(
summary_url, selector=network.AAMirrorSelector(), allow_bypasser_fallback=False
)
@@ -1701,9 +1229,6 @@ def _get_download_url(
return downloader.get_absolute_url(link, url)
_AA_COUNTDOWN_MAX_RETRIES = 3
def _extract_slow_download_url(
soup: BeautifulSoup,
link: str,
@@ -1712,7 +1237,6 @@ def _extract_slow_download_url(
status_callback: Callable[[str, str | None], None] | None,
selector: network.AAMirrorSelector,
source_context: str | None = None,
_countdown_attempts: int = 0,
) -> str:
"""Extract download URL from AA slow download pages."""
html_str = str(soup)
@@ -1777,14 +1301,6 @@ def _extract_slow_download_url(
countdown_seconds = _extract_countdown_seconds(soup, html_str)
if countdown_seconds > 0:
if _countdown_attempts >= _AA_COUNTDOWN_MAX_RETRIES:
logger.warning(
"Countdown retry limit (%s) reached for %s, giving up",
_AA_COUNTDOWN_MAX_RETRIES,
title,
)
return ""
max_countdown_seconds = 600
sleep_time = min(countdown_seconds, max_countdown_seconds)
if countdown_seconds > max_countdown_seconds:
@@ -1793,13 +1309,7 @@ def _extract_slow_download_url(
countdown_seconds,
max_countdown_seconds,
)
logger.info(
"AA waitlist: %ss for %s (attempt %s/%s)",
sleep_time,
title,
_countdown_attempts + 1,
_AA_COUNTDOWN_MAX_RETRIES,
)
logger.info("AA waitlist: %ss for %s", sleep_time, title)
# Live countdown with status updates
for remaining in range(sleep_time, 0, -1):
@@ -1820,31 +1330,12 @@ def _extract_slow_download_url(
if status_callback and source_context:
status_callback("resolving", f"{source_context} - Fetching")
html = downloader.html_get_page(
link, selector=selector, cancel_flag=cancel_flag, status_callback=status_callback
)
if not html:
return ""
new_soup = BeautifulSoup(_html_response_text(html), "html.parser")
return _extract_slow_download_url(
new_soup,
link,
title,
cancel_flag,
status_callback,
selector,
source_context,
_countdown_attempts + 1,
return _get_download_url(
link, title, cancel_flag, status_callback, selector, source_context
)
link_texts = [a.get_text(strip=True)[:50] for a in soup.find_all("a", href=True)[:10]]
logger.warning("No download URL found. First 10 links: %s", link_texts)
# A bypassed page with no AA download links often means the network served a wrong
# page (e.g. an ISP block page) instead of Anna's Archive. Probe for DNS interference
# so we can give the user an actionable hint instead of a generic failure.
host = urlparse(link).hostname or ""
if host:
network.note_possible_dns_interference(host)
return ""
@@ -2058,22 +1549,6 @@ class DirectDownloadSource(ReleaseSource):
) -> list[Release]:
"""Search for releases using the book's metadata.
The whole fan-out runs under one page cache, so a URL built twice by different
passes is fetched once. See `_search_page_reuse`.
"""
with _search_page_reuse():
return self._search(book, plan, expand_search=expand_search, content_type=content_type)
def _search(
self,
book: BookMetadata,
plan: ReleaseSearchPlan,
*,
expand_search: bool = False,
content_type: str = "ebook",
) -> list[Release]:
"""Search for releases using the book's metadata.
Priority: ISBN search first (most precise), then title+author fallback.
For non-English languages, uses localized titles from book.titles_by_language.
@@ -2138,12 +1613,6 @@ class DirectDownloadSource(ReleaseSource):
query = f"{title} {author}".strip()
if not query:
continue
# `except Exception` below keeps this loop going past a failed variant, which
# is right for a parse error and wrong for a spent budget: without this the
# variants queue up behind each other and the request outlives the caller.
if search_deadline.expired():
logger.info("Release search budget spent; skipping remaining title variants")
break
logger.debug("Searching direct_download: title_author='%s', langs=%s", query, langs)
filters = SearchFilters(lang=langs if langs is not None else [])
@@ -2157,11 +1626,7 @@ class DirectDownloadSource(ReleaseSource):
except Exception:
logger.exception("Search error")
if (
not all_results
and any(langs for _, langs in searches)
and not search_deadline.expired()
):
if not all_results and any(langs for _, langs in searches):
logger.debug(
"No title+author results with language filter, retrying without language filter"
)
@@ -2169,9 +1634,6 @@ class DirectDownloadSource(ReleaseSource):
query = f"{title} {author}".strip()
if not query:
continue
if search_deadline.expired():
logger.info("Release search budget spent; skipping remaining retries")
break
logger.debug("Searching direct_download: title_author='%s', langs=[]", query)
try:
@@ -2184,6 +1646,7 @@ class DirectDownloadSource(ReleaseSource):
except Exception:
logger.exception("Search error")
logger.info("Found %s releases via title+author", len(all_results))
return [_browse_record_to_release(record) for record in all_results]
def is_available(self) -> bool:
@@ -2306,14 +1769,7 @@ class DirectDownloadHandler(DownloadHandler):
return None
if not success_url:
if network.dns_interference_detected():
status_callback(
"error",
"All sources failed - your network/ISP appears to be blocking "
"Anna's Archive. Enable DNS-over-HTTPS in settings.",
)
else:
status_callback("error", "All download sources failed")
status_callback("error", "All download sources failed")
return None
# Return temp path - orchestrator handles post-processing (archive extraction, ingest)
+46 -21
View File
@@ -13,6 +13,7 @@ from typing import Any
from shelfmark.config import env
from shelfmark.core.logger import setup_logger
from shelfmark.core.utils import is_audiobook as check_audiobook
from shelfmark.release_sources import Release, ReleaseProtocol
logger = setup_logger(__name__)
@@ -55,6 +56,12 @@ def _coerce_timestamp(value: object) -> float:
return 0.0
def _generate_cache_key(provider: str, provider_id: str, content_type: str | None = None) -> str:
"""Generate a cache key from provider, provider_id, and content type."""
normalized_content_type = "audiobook" if check_audiobook(content_type) else "ebook"
return f"{provider}:{provider_id}:{normalized_content_type}"
def _load_cache() -> dict[str, Any]:
"""Load cache from disk."""
try:
@@ -96,17 +103,17 @@ def _dict_to_release(data: dict[str, Any]) -> Release:
def get_cached_results(
cache_key: str,
provider: str,
provider_id: str,
content_type: str | None = None,
ttl_seconds: int | None = None,
) -> dict[str, Any] | None:
"""Get the cached IRC answer for a query identity (server:channel:query).
The cache stores the whole answer (releases for all content types) under the query
identity, so it is not isolated by book or content type. Callers filter by content
type after reading.
"""Get cached search results for a book.
Args:
cache_key: Query identity (e.g. "server:channel:query")
provider: Metadata provider name (e.g., "hardcover", "openlibrary")
provider_id: Book ID in the provider's system
content_type: Search content type for cache isolation
ttl_seconds: Cache TTL in seconds (from settings)
Returns:
@@ -120,6 +127,8 @@ def get_cached_results(
ttl_value = config.get("IRC_CACHE_TTL", DEFAULT_CACHE_TTL)
ttl_seconds = _coerce_cache_ttl(ttl_value, DEFAULT_CACHE_TTL)
cache_key = _generate_cache_key(provider, provider_id, content_type)
with _cache_lock:
cache = _load_cache()
entry = cache.get("entries", {}).get(cache_key)
@@ -132,9 +141,10 @@ def get_cached_results(
age = time.time() - cached_at
if ttl_seconds != 0 and age > ttl_seconds:
title = entry.get("title", cache_key)
logger.debug(
"IRC cache expired for '%s' (age: %.0fs > TTL: %ss)",
entry.get("title", cache_key),
title,
age,
ttl_seconds,
)
@@ -143,36 +153,44 @@ def get_cached_results(
# Convert dicts back to Release objects
releases = [_dict_to_release(r) for r in entry.get("releases", [])]
online_servers = entry.get("online_servers", [])
title = entry.get("title", "")
logger.info(
"IRC cache hit for '%s' (%s releases, age: %.0fs)",
entry.get("title", ""),
title,
len(releases),
age,
)
return {
"releases": releases,
"online_servers": entry.get("online_servers", []),
"online_servers": online_servers,
"cached_at": cached_at,
}
def cache_results(
cache_key: str,
provider: str,
provider_id: str,
title: str,
releases: list[Release],
content_type: str | None = None,
online_servers: list[str] | None = None,
) -> None:
"""Cache the whole IRC answer for a query identity.
"""Cache search results for a book.
Args:
cache_key: Query identity (e.g. "server:channel:query")
title: Query text (for logging/display)
releases: All Release objects from the search (every content type)
provider: Metadata provider name
provider_id: Book ID in the provider's system
title: Book title (for logging/display)
releases: List of Release objects from search
content_type: Search content type for cache isolation
online_servers: List of online server nicks (optional)
"""
cache_key = _generate_cache_key(provider, provider_id, content_type)
with _cache_lock:
cache = _load_cache()
@@ -180,6 +198,9 @@ def cache_results(
cache["entries"] = {}
cache["entries"][cache_key] = {
"provider": provider,
"provider_id": provider_id,
"content_type": "audiobook" if check_audiobook(content_type) else "ebook",
"title": title,
"releases": [_release_to_dict(r) for r in releases],
"online_servers": list(online_servers) if online_servers else [],
@@ -190,23 +211,27 @@ def cache_results(
logger.info("Cached %s IRC releases for '%s'", len(releases), title)
def invalidate_cache(cache_key: str) -> bool:
def invalidate_cache(provider: str, provider_id: str, content_type: str | None = None) -> bool:
"""Remove a specific entry from the cache.
Args:
cache_key: Query identity to remove
provider: Metadata provider name
provider_id: Book ID in the provider's system
content_type: Search content type for cache isolation
Returns:
True if entry was found and removed
"""
cache_key = _generate_cache_key(provider, provider_id, content_type)
with _cache_lock:
cache = _load_cache()
entries = cache.get("entries", {})
entry = cache.get("entries", {}).get(cache_key)
title = entry.get("title", cache_key) if entry else cache_key
if cache_key in entries:
title = entries[cache_key].get("title", cache_key)
del entries[cache_key]
if cache_key in cache.get("entries", {}):
del cache["entries"][cache_key]
_save_cache(cache)
logger.info("Invalidated IRC cache for '%s'", title)
return True
+32 -68
View File
@@ -25,8 +25,6 @@ logger = setup_logger(__name__)
# Timing
SOCKET_TIMEOUT = 300.0 # 5 minutes - long because we wait for DCC offers
RECV_BUFFER = 4096
# How often a deadline-bound read wakes up to re-check the clock
POLL_INTERVAL = 2.0
# IRC channel user prefixes that indicate elevated status (ops, voice, etc.)
# These are the download bots/servers
@@ -250,22 +248,11 @@ class IRCClient:
# 366 = RPL_ENDOFNAMES - channel join is complete
if msg.command == "366":
if not self.online_servers:
# Joining a channel that doesn't exist on this network
# silently creates an empty one, so an empty name list is
# the only hint that the channel name is wrong.
logger.warning(
"Joined #%s but no servers are online - the channel may "
"be empty or not exist on %s",
channel,
self.server,
)
else:
logger.info(
"Joined #%s - %s servers online",
channel,
len(self.online_servers),
)
logger.info(
"Joined #%s - %s servers online",
channel,
len(self.online_servers),
)
return
# Check for errors (e.g., banned, channel doesn't exist)
@@ -309,47 +296,27 @@ class IRCClient:
data = f"{message}\r\n".encode()
self._socket.sendall(data)
def _recv_lines(self, deadline: float | None = None) -> Iterator[str]:
"""Receive and yield complete CRLF-delimited IRC lines.
A deadline stops the read once it passes, even if nothing ever arrives.
Callers time out by watching the messages they receive, so on a channel
with no traffic at all there is nothing to watch: the recv would just
keep blocking for SOCKET_TIMEOUT and retrying forever.
"""
def _recv_lines(self) -> Iterator[str]:
"""Receive and yield complete CRLF-delimited IRC lines."""
sock = self._require_socket()
original_timeout = sock.gettimeout()
while True:
# Check if we have a complete line in buffer
while "\r\n" in self._buffer:
line, self._buffer = self._buffer.split("\r\n", 1)
if line:
yield line
try:
while True:
# Check if we have a complete line in buffer
while "\r\n" in self._buffer:
line, self._buffer = self._buffer.split("\r\n", 1)
if line:
yield line
if deadline is not None:
remaining = deadline - time.time()
if remaining <= 0:
return
# Wake up often enough to notice the deadline pass
sock.settimeout(min(remaining, POLL_INTERVAL))
# Read more data
try:
data = sock.recv(RECV_BUFFER)
if not data:
return # Connection closed
self._buffer += data.decode("utf-8", errors="replace")
except TimeoutError:
continue # Keep waiting (the deadline is re-checked above)
except OSError as e:
logger.warning("Socket error: %s", e)
return # Connection error
finally:
if deadline is not None:
with suppress(OSError):
sock.settimeout(original_timeout)
# Read more data
try:
data = sock.recv(RECV_BUFFER)
if not data:
return # Connection closed
self._buffer += data.decode("utf-8", errors="replace")
except TimeoutError:
continue # Keep waiting
except OSError as e:
logger.warning("Socket error: %s", e)
return # Connection error
def _parse_message(self, line: str) -> IRCMessage:
"""Parse an IRC message line into components.
@@ -460,14 +427,9 @@ class IRCClient:
return False
return True
def read_messages(
self,
*,
auto_handle: bool = True,
deadline: float | None = None,
) -> Iterator[IRCMessage]:
def read_messages(self, *, auto_handle: bool = True) -> Iterator[IRCMessage]:
"""Read and yield IRC messages, optionally auto-handling PING/VERSION."""
for line in self._recv_lines(deadline):
for line in self._recv_lines():
msg = self._parse_message(line)
# Auto-handle certain events
@@ -491,9 +453,13 @@ class IRCClient:
) -> DCCOffer | None:
"""Wait for a DCC SEND offer. Returns None on timeout or no results."""
target_event = IRCEvent.SEARCH_RESULT if result_type else IRCEvent.BOOK_RESULT
deadline = time.time() + timeout
start = time.time()
for msg in self.read_messages():
if time.time() - start > timeout:
logger.warning("Timeout waiting for DCC offer")
return None
for msg in self.read_messages(deadline=deadline):
if msg.event == target_event:
if not self._is_allowed_dcc_sender(msg, expected_senders):
continue
@@ -525,8 +491,6 @@ class IRCClient:
count = match.group(1)
logger.info("Found %s matches", count)
if time.time() >= deadline:
logger.warning("Timeout waiting for DCC offer")
return None
@property
+23 -64
View File
@@ -11,7 +11,6 @@ from typing import TYPE_CHECKING
from shelfmark.core.config import config
from shelfmark.core.logger import setup_logger
from shelfmark.core.utils import ARCHIVE_FORMATS, AUDIOBOOK_FORMATS
from shelfmark.core.utils import is_audiobook as check_audiobook
if TYPE_CHECKING:
@@ -19,8 +18,11 @@ if TYPE_CHECKING:
logger = setup_logger(__name__)
# Ebook formats recognized in IRC result lines.
EBOOK_FORMATS = (
# All recognized formats for parsing IRC result lines.
# This comprehensive list is used to identify file extensions in results.
# User-configured formats are used separately for filtering.
ALL_RECOGNIZED_FORMATS = {
# Ebook formats
"epub",
"mobi",
"azw3",
@@ -39,17 +41,19 @@ EBOOK_FORMATS = (
"cbz",
"cdr",
"jpg",
)
# All recognized formats for parsing IRC result lines.
# This comprehensive list is used to identify file extensions in results.
# User-configured formats are used separately for filtering.
# Ordered longest-first so that scanning a line matches "azw3" before "azw" and "docx"
# before "doc". It used to be a set, which made the winning format for a line naming more
# than one extension depend on set iteration order, and therefore vary between restarts.
ALL_RECOGNIZED_FORMATS = tuple(
sorted({*EBOOK_FORMATS, *ARCHIVE_FORMATS, *AUDIOBOOK_FORMATS}, key=len, reverse=True)
)
"rar",
"zip",
# Audiobook formats
"m4b",
"mp3",
"m4a",
"flac",
"ogg",
"wma",
"aac",
"wav",
"opus",
}
def _normalize_config_formats(raw_formats: object) -> set[str]:
@@ -80,22 +84,13 @@ def _get_supported_formats(content_type: str | None = None) -> set[str]:
# Regex to parse result lines
# Format: !Server Author - Title.format ::INFO:: size
#
# The extension is matched against the known formats rather than a bare \w+. A bare \w+
# happily matched the decimal point in the size, so a line with no file extension parsed
# as format="5mb" out of "::INFO:: 620.5MB" - taking the title and size down with it, and
# leaving the result to be discarded by every format filter downstream. Restricting the
# alternation makes such a line fall through to SIMPLE_RESULT_REGEX and come back as
# "unknown", which is what the rest of the parser already expects.
_FORMAT_ALTERNATION = "|".join(re.escape(fmt) for fmt in ALL_RECOGNIZED_FORMATS)
RESULT_LINE_REGEX = re.compile(
r"^!(\S+)\s+" # !ServerName
r"(.+?)\s+-\s+" # Author Name -
rf"(.+?)\.({_FORMAT_ALTERNATION})\b" # Title.format
r"(.+?)\.(\w+)" # Title.format
r"(?:\s+::INFO::\s*(.+?))?" # Optional ::INFO:: metadata
r"(?:\s+::HASH::\s*(\S+))?" # Optional ::HASH::
r"\s*$",
re.IGNORECASE,
r"\s*$"
)
# Simpler fallback pattern
@@ -192,54 +187,18 @@ def parse_result_line(line: str) -> SearchResult | None:
return None
# Words that mark an archive as holding an audiobook rather than an ebook. Multi-file
# audiobooks ship as .rar/.zip, so for those the extension says nothing about the content
# and the release name is the only evidence there is.
_AUDIOBOOK_MARKER_REGEX = re.compile(
r"\b(?:audio ?books?|unabridged|abridged|narrat(?:ed|or)|audible|\d+ ?kbps|"
+ "|".join(re.escape(fmt) for fmt in AUDIOBOOK_FORMATS)
+ r")\b",
re.IGNORECASE,
)
_AUDIOBOOK_FORMAT_SET = frozenset(AUDIOBOOK_FORMATS)
_EBOOK_FORMAT_SET = frozenset(EBOOK_FORMATS)
def detect_content_type(result: SearchResult) -> str:
"""Classify a parsed result as an audiobook or an ebook.
Extension alone is not enough. It settles the plain cases, but the common audiobook
release is a .rar or .zip of MP3s, which is indistinguishable by extension from an
ebook archive - so for containers (and for lines with no usable extension) the
release name decides.
"""
if result.format in _AUDIOBOOK_FORMAT_SET:
return "audiobook"
if result.format in _EBOOK_FORMAT_SET:
return "ebook"
return "audiobook" if _AUDIOBOOK_MARKER_REGEX.search(result.full_line) else "ebook"
def parse_results_file(content: str, content_type: str | None = None) -> list[SearchResult]:
"""Parse a search results file into SearchResult objects."""
results = []
supported = _get_supported_formats(content_type)
requested = "audiobook" if check_audiobook(content_type) else "ebook"
for line in content.splitlines():
result = parse_result_line(line)
if not result:
continue
# Classify first, then apply the user's format filter within that bucket. Doing it
# the other way round is what lost audiobooks entirely: an audiobook .rar matched
# neither the ebook nor the audiobook format list, so it fell out of both.
if detect_content_type(result) != requested:
continue
if result.format in supported or result.format == "unknown":
if result and (result.format in supported or result.format == "unknown"):
# Filter to user's configured formats
results.append(result)
logger.info("Parsed %s %s results from search file", len(results), requested)
logger.info("Parsed %s results from search file", len(results))
return results
+2 -45
View File
@@ -72,10 +72,7 @@ def irc_settings() -> list[SettingsField]:
key="IRC_CHANNEL",
label="Channel",
placeholder="e.g. ebooks",
description=(
"Channel name without the # prefix. Used for all searches unless a "
"separate audiobook channel is configured below."
),
description="Channel name without the # prefix",
required=True,
env_supported=True,
),
@@ -91,47 +88,7 @@ def irc_settings() -> list[SettingsField]:
key="IRC_SEARCH_BOT",
label="Search bot",
placeholder="e.g. search",
description=(
"The search bot to address queries to (required). Searches are sent as "
'"@<bot> <query>".'
),
required=True,
env_supported=True,
),
HeadingField(
key="audiobook_heading",
title="Audiobooks",
description=(
"Most networks index audiobooks in the same channel as ebooks, so leaving "
"these blank is the right setting for almost everyone. On irc.irchighway.net "
"the audiobooks are in #ebooks and #bookz is effectively inactive — pointing "
"this at an empty channel just returns no results. Only fill these in when "
"your network really does index audiobooks elsewhere (Undernet's #bookz, for "
"example). Audiobooks are usually posted as archives, so keep ZIP and RAR "
"enabled under Supported Audiobook Formats or the releases are filtered out."
),
),
TextField(
key="IRC_AUDIOBOOK_CHANNEL",
label="Audiobook channel",
placeholder="e.g. bookz",
description=(
"Optional. Channel name (without the # prefix) for networks that index "
"audiobooks separately, such as Undernet's bookz. Leave blank (the usual "
"setting) to search the main channel above for audiobooks too."
),
required=False,
env_supported=True,
),
TextField(
key="IRC_AUDIOBOOK_SEARCH_BOT",
label="Audiobook search bot",
placeholder="e.g. search",
description=(
"Optional. Search bot for the audiobook channel. Leave blank to reuse "
"the main search bot above. Only used when an audiobook channel is set."
),
required=False,
description="The search bot to query for results",
env_supported=True,
),
HeadingField(
+54 -164
View File
@@ -15,8 +15,6 @@ if TYPE_CHECKING:
from shelfmark.api.websocket import ws_manager
from shelfmark.core.config import config
from shelfmark.core.logger import setup_logger
from shelfmark.core.search_plan import pick_search_author
from shelfmark.core.utils import is_audiobook
from shelfmark.release_sources import (
ColumnColorHint,
ColumnRenderType,
@@ -90,17 +88,6 @@ def _emit_status(message: str, phase: str = "searching") -> None:
MIN_SEARCH_INTERVAL = 15.0
_last_search_time: float = 0
# Anti-spam budget: the exact same message may only be posted to the channel a limited
# number of times within a rolling window. This stops a retry/refresh loop from flooding
# the channel with the same line over and over, while still allowing a few genuine retries
# (a search that came back empty can be tried again, and Refresh works until the budget runs
# out). Normal use never hits this: successful searches are served from the result cache
# without re-posting at all.
MAX_IDENTICAL_SENDS = 3
IDENTICAL_SEND_WINDOW_SECONDS = 24 * 60 * 60 # 24 hours
# message-send-key -> timestamps of recent posts of that exact message
_recent_message_sends: dict[str, list[float]] = {}
def _enforce_rate_limit() -> None:
"""Ensure minimum time between searches."""
@@ -115,36 +102,6 @@ def _enforce_rate_limit() -> None:
_last_search_time = time.time()
def _query_identity(server: str, channel: str, query: str) -> str:
"""Stable identity for a query on a given IRC server-channel.
Used as BOTH the result-cache key and the per-query send-counter key, so the same
query shares one cached answer and one send budget regardless of which book or
content type triggered it.
"""
return f"{server.casefold()}:{channel.casefold()}:{query.strip().casefold()}"
def _recent_send_count(key: str) -> int:
"""Number of times this exact message was posted within the rolling window."""
cutoff = time.time() - IDENTICAL_SEND_WINDOW_SECONDS
timestamps = [ts for ts in _recent_message_sends.get(key, []) if ts > cutoff]
if timestamps:
_recent_message_sends[key] = timestamps
else:
_recent_message_sends.pop(key, None)
return len(timestamps)
def _record_message_sent(key: str) -> None:
"""Record that an exact message was just posted to the channel."""
now = time.time()
cutoff = now - IDENTICAL_SEND_WINDOW_SECONDS
timestamps = [ts for ts in _recent_message_sends.get(key, []) if ts > cutoff]
timestamps.append(now)
_recent_message_sends[key] = timestamps
@register_source("irc")
class IRCReleaseSource(ReleaseSource):
"""Search IRC channels for ebook and audiobook releases."""
@@ -160,16 +117,11 @@ class IRCReleaseSource(ReleaseSource):
self._online_servers: set[str] | None = None
def is_available(self) -> bool:
"""Check if IRC is configured (server, channel, nick, and search bot are set).
The search bot is required: without it we would post bare queries straight
to the channel, which reads as spam and gets the nick banned.
"""
"""Check if IRC is configured (server, channel, and nick are set)."""
server = _config_text("IRC_SERVER")
channel = _config_text("IRC_CHANNEL")
nick = _config_text("IRC_NICK")
search_bot = _config_text("IRC_SEARCH_BOT")
return bool(server and channel and nick and search_bot)
return bool(server and channel and nick)
def get_column_config(self) -> ReleaseColumnConfig:
"""Configure UI columns for IRC results."""
@@ -227,12 +179,25 @@ class IRCReleaseSource(ReleaseSource):
logger.debug("IRC source is disabled, skipping search")
return []
# Check cache first (unless expand_search/refresh is requested)
if not expand_search:
cached = get_cached_results(book.provider, book.provider_id, content_type=content_type)
if cached:
_emit_status("Using cached results", phase="complete")
self._online_servers = set(cached.get("online_servers", []))
return cached["releases"]
# Build search query
query = plan.primary_query or self._build_query(book)
if not query:
logger.warning("No search query could be built")
return []
logger.info("IRC search: %s", query)
# Enforce rate limit
_enforce_rate_limit()
# Get IRC settings
server = _config_text("IRC_SERVER")
port = _config_port("IRC_PORT", 6697)
@@ -241,67 +206,6 @@ class IRCReleaseSource(ReleaseSource):
nick = _config_text("IRC_NICK")
search_bot = _config_text("IRC_SEARCH_BOT")
# A few networks index audiobooks in a separate channel from ebooks (Undernet's
# #bookz, say). When an audiobook channel is configured and an audiobook was
# requested, route the search there (with its own search bot if set). Otherwise
# fall back to the main channel/bot — that is the common case, since most networks
# (irchighway included) serve both formats from the one channel.
if is_audiobook(content_type):
audiobook_channel = _config_text("IRC_AUDIOBOOK_CHANNEL")
if audiobook_channel:
channel = audiobook_channel
audiobook_search_bot = _config_text("IRC_AUDIOBOOK_SEARCH_BOT")
if audiobook_search_bot:
search_bot = audiobook_search_bot
# Never post an unaddressed query to the channel. A bare book title looks like
# spam to everyone else in the channel and gets the nick banned. Searches must
# be addressed to a search bot ("@<bot> <query>").
if not search_bot:
logger.warning(
"IRC search bot not configured; refusing to post unaddressed query to channel"
)
_emit_status("IRC search bot not configured", phase="error")
return []
# One identity per query on this server-channel. The result cache and the send
# counter are both keyed on it: the SAME query shares one cached answer and one
# send budget regardless of which book/content type triggered it, while different
# queries are independent (searching 100 different books posts 100 messages).
requested = "audiobook" if is_audiobook(content_type) else "ebook"
query_key = _query_identity(server, channel, query)
# Serve the cached whole answer for an identical query (unless this is a refresh).
if not expand_search:
cached = get_cached_results(query_key)
if cached:
_emit_status("Using cached results", phase="complete")
self._online_servers = set(cached.get("online_servers", []))
return self._filter_by_content_type(cached["releases"], requested)
# Anti-spam cap: the exact same query may only be POSTED a limited number of times
# per window, even via refresh. Beyond that, serve whatever is cached rather than
# re-posting the identical message to the channel.
if _recent_send_count(query_key) >= MAX_IDENTICAL_SENDS:
logger.info(
"IRC query hit %s-send limit in window, not re-posting: %s",
MAX_IDENTICAL_SENDS,
query,
)
_emit_status(
"Search limit reached for this query — showing latest results", phase="complete"
)
cached = get_cached_results(query_key)
if cached:
self._online_servers = set(cached.get("online_servers", []))
return self._filter_by_content_type(cached["releases"], requested)
return []
logger.info("IRC search: %s", query)
# Enforce rate limit
_enforce_rate_limit()
client = None
try:
# Get or reuse IRC connection
@@ -317,27 +221,28 @@ class IRCReleaseSource(ReleaseSource):
# Capture online servers (elevated users in channel)
self._online_servers = client.online_servers
# Send search request (always addressed to the search bot, never bare)
client.send_message(f"#{channel}", f"@{search_bot} {query}")
_record_message_sent(query_key)
# Send search request
search_msg = f"@{search_bot} {query}" if search_bot else query
client.send_message(f"#{channel}", search_msg)
# Wait for results DCC - this is the long wait.
# Don't restrict the sender to the trigger bot's nick: many channels answer an
# "@search" from a differently-named results bot. The DCC endpoint/filename are
# still validated, and wait_for_dcc falls back to the channel's server list.
# Wait for results DCC - this is the long wait
_emit_status(f"Connected to #{channel} - Waiting for results...", phase="searching")
offer = client.wait_for_dcc(timeout=60.0, result_type=True)
online_servers = list(self._online_servers) if self._online_servers else None
wait_kwargs = {"expected_senders": {search_bot}} if search_bot else {}
offer = client.wait_for_dcc(timeout=60.0, result_type=True, **wait_kwargs)
if not offer:
logger.info("No search results received")
_emit_status("No results found", phase="complete")
# Release connection for reuse (don't close it)
connection_manager.release_connection(client)
# Cache the (empty) answer under the query identity so an identical query
# is served from cache instead of re-posting.
cache_results(query_key, query, [], online_servers=online_servers)
# Cache empty result to avoid repeated failed searches
cache_results(
book.provider,
book.provider_id,
book.title,
[],
content_type=content_type,
online_servers=list(self._online_servers) if self._online_servers else None,
)
return []
# Download results file
@@ -355,22 +260,19 @@ class IRCReleaseSource(ReleaseSource):
# Release connection for reuse (don't close it)
connection_manager.release_connection(client)
# A single "@search" returns one file containing every format. Parse the whole
# answer (both ebooks and audiobooks) and cache it under the query identity, so
# requesting the other content type is served from cache without re-posting.
ebook_releases = self._convert_to_releases(
parse_results_file(content, content_type="ebook"), content_type="ebook"
)
audiobook_releases = self._convert_to_releases(
parse_results_file(content, content_type="audiobook"), content_type="audiobook"
)
# Convert to Release objects
results = parse_results_file(content, content_type=content_type)
releases = self._convert_to_releases(results, content_type=content_type)
# Cache results
cache_results(
query_key,
query,
ebook_releases + audiobook_releases,
online_servers=online_servers,
book.provider,
book.provider_id,
book.title,
releases,
content_type=content_type,
online_servers=list(self._online_servers) if self._online_servers else None,
)
releases = audiobook_releases if requested == "audiobook" else ebook_releases
except DCCError as e:
logger.exception("DCC error during search")
@@ -395,13 +297,11 @@ class IRCReleaseSource(ReleaseSource):
if book.search_title or book.title:
parts.append(book.search_title or book.title)
# Only ever the first author: both metadata fields can arrive holding every
# contributor joined with ", ", and an IRC query carrying an author plus two
# translators matches nothing. The choice between them - and the narrowing - is
# `pick_search_author`, shared with the search plan so this cannot drift from it
# again. See issue #1252.
author = pick_search_author(book)
if author:
if book.search_author:
parts.append(book.search_author)
elif book.authors:
# Use first author
author = book.authors[0] if isinstance(book.authors, list) else book.authors
parts.append(author)
return " ".join(parts)
@@ -431,15 +331,14 @@ class IRCReleaseSource(ReleaseSource):
"m4b": 0,
"mp3": 1,
"m4a": 2,
"mp4": 3,
"flac": 4,
"opus": 5,
"ogg": 6,
"aac": 7,
"wav": 8,
"wma": 9,
"rar": 10,
"zip": 11,
"flac": 3,
"opus": 4,
"ogg": 5,
"aac": 6,
"wav": 7,
"wma": 8,
"rar": 9,
"zip": 10,
}
def _convert_to_releases(
@@ -491,15 +390,6 @@ class IRCReleaseSource(ReleaseSource):
return releases
@staticmethod
def _filter_by_content_type(releases: list[Release], requested: str) -> list[Release]:
"""Pick the requested content type out of a cached whole answer.
The cache stores releases for every content type under one query identity; each
release is tagged with its content type (defaulting to ebook when missing).
"""
return [release for release in releases if (release.content_type or "ebook") == requested]
@staticmethod
def _parse_size(size_str: str) -> int | None:
"""Parse human-readable size (e.g., '1.2MB', '500K') to bytes."""
+4
View File
@@ -14,6 +14,10 @@ from shelfmark.release_sources.prowlarr.torznab import parse_torznab_xml
logger = setup_logger(__name__)
# Newznab standard book category IDs
NEWZNAB_BOOKS = 7000
NEWZNAB_AUDIOBOOKS = 3030
class NewznabClient:
"""Client for any Newznab-compatible indexer API."""
+8 -93
View File
@@ -8,8 +8,6 @@ from shelfmark.core.settings_registry import (
HeadingField,
PasswordField,
SettingsField,
TableField,
TagListField,
TextField,
register_settings,
)
@@ -17,36 +15,12 @@ from shelfmark.core.utils import normalize_http_url
def _test_newznab_connection(current_values: dict[str, Any] | None = None) -> dict[str, Any]:
"""Test all named Newznab connections, or the legacy connection as fallback."""
"""Test the Newznab connection using current form values."""
from shelfmark.core.config import config
from shelfmark.release_sources.newznab.api import NewznabClient
from shelfmark.release_sources.newznab.source import _parse_indexer_rows
current_values = current_values or {}
raw_indexers = current_values.get("NEWZNAB_INDEXERS")
if raw_indexers is None:
raw_indexers = config.get("NEWZNAB_INDEXERS", [])
indexers = _parse_indexer_rows(raw_indexers)
if indexers:
details: list[str] = []
all_successful = True
for name, url, api_key in indexers:
try:
success, message = NewznabClient(url, api_key).test_connection()
except Exception as e: # noqa: BLE001 — surface unexpected errors to the UI
success, message = False, f"Connection failed: {e!s}"
all_successful = all_successful and success
details.append(f"{name}: {message}")
summary = (
f"Connected to all {len(indexers)} indexers"
if all_successful
else "One or more Newznab indexers failed"
)
return {"success": all_successful, "message": summary, "details": details}
raw_url = str(current_values.get("NEWZNAB_URL") or config.get("NEWZNAB_URL", "") or "")
api_key = str(current_values.get("NEWZNAB_API_KEY") or config.get("NEWZNAB_API_KEY", "") or "")
@@ -89,88 +63,29 @@ def newznab_config_settings() -> list[SettingsField]:
default=False,
description="Enable searching for books via a Newznab-compatible indexer",
),
TableField(
key="NEWZNAB_INDEXERS",
label="Named Indexers",
description=(
"Add each Newznab-compatible indexer separately. The configured name is shown "
"beside every result from that indexer."
),
columns=[
{
"key": "name",
"label": "Name",
"type": "text",
"placeholder": "NZBGeek",
},
{
"key": "url",
"label": "URL",
"type": "text",
"placeholder": "https://api.nzbgeek.info",
},
{
"key": "api_key",
"label": "API Key",
"type": "password",
"placeholder": "Optional",
},
],
default=[],
add_label="Add Indexer",
empty_message=(
"No named indexers configured. The legacy single-indexer fields below are used "
"as a fallback."
),
show_when={"field": "NEWZNAB_ENABLED", "value": True},
),
TextField(
key="NEWZNAB_URL",
label="Legacy Newznab URL",
description="Used only when the named indexer list is empty",
label="Newznab URL",
description="Base URL of your Newznab indexer or aggregator",
placeholder="http://nzbhydra:5076",
required=False,
required=True,
show_when={"field": "NEWZNAB_ENABLED", "value": True},
),
PasswordField(
key="NEWZNAB_API_KEY",
label="Legacy API Key",
description="Used only with the legacy Newznab URL",
label="API Key",
description="Your Newznab API key (leave blank if not required)",
required=False,
show_when={"field": "NEWZNAB_ENABLED", "value": True},
),
ActionButton(
key="test_newznab",
label="Test Connections",
description="Verify every named indexer, or the legacy connection when the list is empty",
label="Test Connection",
description="Verify your Newznab configuration",
style="primary",
callback=_test_newznab_connection,
show_when={"field": "NEWZNAB_ENABLED", "value": True},
),
TagListField(
key="NEWZNAB_EBOOK_CATEGORIES",
label="Ebook Categories",
description=(
"Newznab category IDs searched for ebooks. Most indexers use the standard 7000, "
"but some use custom IDs. Leave empty to use 7000."
),
placeholder="7000",
default=["7000"],
normalize_urls=False,
show_when={"field": "NEWZNAB_ENABLED", "value": True},
),
TagListField(
key="NEWZNAB_AUDIOBOOK_CATEGORIES",
label="Audiobook Categories",
description=(
"Newznab category IDs searched for audiobooks. Most indexers use the standard "
"3030, but some use custom IDs. Leave empty to use 3030."
),
placeholder="3030",
default=["3030"],
normalize_urls=False,
show_when={"field": "NEWZNAB_ENABLED", "value": True},
),
CheckboxField(
key="NEWZNAB_AUTO_EXPAND",
label="Auto-expand search on no results",
+42 -211
View File
@@ -2,12 +2,8 @@
from __future__ import annotations
import re
import time
from dataclasses import dataclass
from hashlib import sha256
from typing import TYPE_CHECKING, ClassVar
from urllib.parse import urlparse
if TYPE_CHECKING:
from shelfmark.core.search_plan import ReleaseSearchPlan
@@ -43,137 +39,15 @@ from shelfmark.release_sources.prowlarr.source import (
logger = setup_logger(__name__)
# Standard Newznab category IDs, used when the indexer's categories aren't configured.
_DEFAULT_AUDIOBOOK_CATS = [3030]
_DEFAULT_BOOK_CATS = [7000]
# Newznab category IDs
_AUDIOBOOK_CATS = [3030]
_BOOK_CATS = [7000]
# Reuse the same timeout constant as Prowlarr.
NEWZNAB_SEARCH_TIMEOUT_SECONDS = _SEARCH_TIMEOUT
@dataclass(frozen=True)
class _NamedClient:
"""A configured Newznab connection and its stable cache namespace."""
name: str
connection_id: str
client: NewznabClient
def _parse_indexer_rows(raw: object) -> list[tuple[str, str, str]]:
"""Normalize structured Newznab indexer settings.
Invalid/incomplete rows are ignored so one partially edited row cannot disable
the other configured indexers.
"""
if not isinstance(raw, list):
return []
indexers: list[tuple[str, str, str]] = []
seen_connections: set[tuple[str, str]] = set()
for row in raw:
if not isinstance(row, dict):
continue
raw_url = str(row.get("url") or "").strip()
url = normalize_http_url(raw_url)
if not url:
if raw_url:
logger.warning("Newznab: ignoring indexer row with invalid URL '%s'", raw_url)
continue
api_key = str(row.get("api_key") or "").strip()
connection_key = (url, api_key)
if connection_key in seen_connections:
continue
seen_connections.add(connection_key)
configured_name = str(row.get("name") or "").strip()
hostname = urlparse(url).hostname or ""
name = configured_name or hostname or "Newznab"
indexers.append((name, url, api_key))
return indexers
def _parse_category_ids(raw: object) -> list[int]:
"""Parse a configured category setting into Newznab category IDs.
Accepts a list of values or a comma/whitespace separated string. Entries that
aren't positive integers are skipped, and duplicates are dropped.
"""
if raw is None:
return []
values = list(raw) if isinstance(raw, (list, tuple)) else [raw]
category_ids: list[int] = []
for value in values:
for token in re.split(r"[,\s]+", str(value).strip()):
if not token:
continue
try:
category_id = int(token)
except ValueError:
logger.warning("Newznab: ignoring invalid category ID '%s'", token)
continue
if category_id > 0 and category_id not in category_ids:
category_ids.append(category_id)
return category_ids
def _configured_categories(content_type: str) -> list[int]:
"""Return the categories to search for a content type, falling back to defaults."""
if content_type == "audiobook":
key, defaults = "NEWZNAB_AUDIOBOOK_CATEGORIES", _DEFAULT_AUDIOBOOK_CATS
else:
key, defaults = "NEWZNAB_EBOOK_CATEGORIES", _DEFAULT_BOOK_CATS
return _parse_category_ids(config.get(key, None)) or list(defaults)
def _result_category_ids(categories: object) -> set[int]:
"""Extract numeric category IDs from a result's categories field."""
if not isinstance(categories, (list, tuple)):
return set()
category_ids: set[int] = set()
for cat in categories:
raw = cat.get("id") if isinstance(cat, dict) else cat
try:
category_ids.add(int(raw)) # type: ignore[arg-type]
except TypeError, ValueError:
continue
return category_ids
def _resolve_content_type(
categories: object,
content_type: str,
searched_categories: list[int] | None,
) -> str:
"""Resolve a result's content type, honouring custom indexer categories.
Indexers using non-standard IDs (e.g. 7100 for ebooks) fall outside the standard
ranges, so trust the searched content type when the result carries a category we
explicitly asked for.
"""
category_list = list(categories) if isinstance(categories, (list, tuple)) else []
detected = _detect_content_type_from_categories(category_list, content_type)
if (
detected == "other"
and searched_categories
and _result_category_ids(category_list) & set(searched_categories)
):
return "audiobook" if content_type == "audiobook" else "book"
return detected
def _newznab_result_to_release(
result: dict,
content_type: str = "ebook",
searched_categories: list[int] | None = None,
) -> Release:
def _newznab_result_to_release(result: dict, content_type: str = "ebook") -> Release:
"""Convert a parsed Newznab XML result dict to a Release object."""
raw_title = result.get("title", "Unknown")
size_bytes = result.get("size")
@@ -193,11 +67,8 @@ def _newznab_result_to_release(
else None
)
# Namespace IDs from named connections so identical GUIDs returned by two
# indexers cannot overwrite one another in the private release cache.
raw_source_id = result.get("guid") or f"newznab:{hash(raw_title)}"
connection_id = str(result.get("_newznab_connection_id") or "").strip()
source_id = f"newznab:{connection_id}:{raw_source_id}" if connection_id else raw_source_id
# Build source_id from GUID
source_id = result.get("guid") or f"newznab:{hash(raw_title)}"
# Cache the raw result for the handler
cache_release(source_id, result)
@@ -254,7 +125,7 @@ def _newznab_result_to_release(
indexer=indexer,
seeders=seeders if is_torrent else None,
peers=peers_display,
content_type=_resolve_content_type(categories, content_type, searched_categories),
content_type=_detect_content_type_from_categories(categories, content_type),
extra={
"publish_date": result.get("publishDate"),
"categories": categories,
@@ -322,7 +193,6 @@ class NewznabSource(ReleaseSource):
)
def _get_client(self) -> NewznabClient | None:
"""Build the legacy single-indexer client."""
raw_url = str(config.get("NEWZNAB_URL", "") or "")
api_key = str(config.get("NEWZNAB_API_KEY", "") or "")
@@ -335,28 +205,6 @@ class NewznabSource(ReleaseSource):
return NewznabClient(url, api_key or "")
def _get_clients(self) -> list[_NamedClient]:
"""Build named clients, falling back to the legacy single connection."""
configured = _parse_indexer_rows(config.get("NEWZNAB_INDEXERS", []))
if configured:
clients: list[_NamedClient] = []
for name, url, api_key in configured:
digest = sha256(f"{name}\0{url}\0{api_key}".encode()).hexdigest()[:16]
clients.append(
_NamedClient(
name=name,
connection_id=digest,
client=NewznabClient(url, api_key),
)
)
return clients
legacy_client = self._get_client()
if legacy_client is None:
return []
legacy_name = str(config.get("NEWZNAB_NAME", "") or "").strip() or "Newznab"
return [_NamedClient(name=legacy_name, connection_id="legacy", client=legacy_client)]
def search(
self,
book: BookMetadata,
@@ -366,8 +214,8 @@ class NewznabSource(ReleaseSource):
content_type: str = "ebook",
) -> list[Release]:
"""Search the Newznab indexer for releases matching the book."""
clients = self._get_clients()
if not clients:
client = self._get_client()
if not client:
logger.warning("Newznab not configured - skipping search")
return []
@@ -382,7 +230,12 @@ class NewznabSource(ReleaseSource):
return []
# Category selection — omit categories when expanding search
categories = None if expand_search else _configured_categories(content_type)
if expand_search:
categories = None
elif content_type == "audiobook":
categories = [3030]
else:
categories = [7000]
auto_expand = config.get("NEWZNAB_AUTO_EXPAND", False)
deadline = time.monotonic() + NEWZNAB_SEARCH_TIMEOUT_SECONDS
@@ -397,60 +250,40 @@ class NewznabSource(ReleaseSource):
all_results: list[dict] = []
try:
for connection in clients:
try:
for idx, query in enumerate(queries, start=1):
_check_timeout()
if len(queries) > 1:
logger.debug(
"Newznab [%s] query %d/%d: '%s'",
connection.name,
idx,
len(queries),
query,
)
for idx, query in enumerate(queries, start=1):
_check_timeout()
if len(queries) > 1:
logger.debug("Newznab query %d/%d: '%s'", idx, len(queries), query)
raw = connection.client.search(query=query, categories=categories)
raw = client.search(query=query, categories=categories)
# Auto-expand: retry without category filter if no results
if not raw and categories and auto_expand:
_check_timeout()
logger.info(
"Newznab [%s]: no results for '%s' with category filter, "
"auto-expanding",
connection.name,
query,
)
raw = connection.client.search(query=query, categories=None)
# Auto-expand: retry without category filter if no results
if not raw and categories and auto_expand:
_check_timeout()
logger.info(
"Newznab: no results for '%s' with category filter, auto-expanding",
query,
)
raw = client.search(query=query, categories=None)
for raw_result in raw:
r = dict(raw_result)
# Aggregators can identify the underlying indexer. Plain feeds
# generally cannot, so use the user-configured connection name.
r["indexer"] = r.get("indexer") or connection.name
r["_newznab_connection_id"] = connection.connection_id
key = (
connection.connection_id,
r.get("guid")
or r.get("downloadUrl")
or f"{r.get('indexer')}:{r.get('title')}",
)
if key in seen_keys:
continue
seen_keys.add(key)
all_results.append(r)
except TimeoutError:
raise
except Exception:
logger.exception("Newznab search failed for %s", connection.name)
for r in raw:
key = (
r.get("guid")
or r.get("downloadUrl")
or f"{r.get('indexer')}:{r.get('title')}"
)
if key in seen_keys:
continue
seen_keys.add(key)
all_results.append(r)
except TimeoutError as e:
logger.warning("Newznab search timed out: %s", e)
except Exception:
logger.exception("Newznab search failed")
return []
results = [_newznab_result_to_release(r, content_type, categories) for r in all_results]
if plan.indexers:
selected_indexers = set(plan.indexers)
results = [r for r in results if r.indexer in selected_indexers]
results = [_newznab_result_to_release(r, content_type) for r in all_results]
if results:
nzb_count = sum(1 for r in results if r.protocol == ReleaseProtocol.NZB)
@@ -472,7 +305,5 @@ class NewznabSource(ReleaseSource):
def is_available(self) -> bool:
if not config.get("NEWZNAB_ENABLED", False):
return False
if _parse_indexer_rows(config.get("NEWZNAB_INDEXERS", [])):
return True
url = normalize_http_url(str(config.get("NEWZNAB_URL", "") or ""))
return bool(url)
+14 -116
View File
@@ -7,7 +7,6 @@ from typing import Any, TypedDict
import requests
from shelfmark.core.config import config
from shelfmark.core.logger import setup_logger
from shelfmark.core.utils import normalize_http_url
from shelfmark.download.network import get_ssl_verify
@@ -19,19 +18,6 @@ logger = setup_logger(__name__)
_HTTP_STATUS_UNAUTHORIZED = HTTPStatus.UNAUTHORIZED
_BOOK_CATEGORY_RANGE_START = 7000
_BOOK_CATEGORY_RANGE_END = 8000
# Prowlarr's own JSON endpoints (status, indexer list) read local state and answer
# in milliseconds, so they keep a short timeout. A Torznab search is different: it
# is Prowlarr proxying a live request to the tracker, which for a Cloudflare-fronted
# indexer means waiting on FlareSolverr to solve a challenge. A cold challenge
# routinely runs past a minute, so indexer searches get their own, longer budget.
DEFAULT_INDEXER_TIMEOUT_SECONDS = 90
MIN_INDEXER_TIMEOUT_SECONDS = 5
MAX_INDEXER_TIMEOUT_SECONDS = 300
# Connecting to Prowlarr itself is a LAN hop; only the read is allowed to be slow.
_CONNECT_TIMEOUT_SECONDS = 10.0
_PROWLARR_CLIENT_ERRORS = (
requests.exceptions.RequestException,
OSError,
@@ -41,37 +27,6 @@ _PROWLARR_CLIENT_ERRORS = (
)
class ProwlarrSearchError(RuntimeError):
"""A Torznab search could not be completed.
Deliberately distinct from an empty result list. Reporting a failed search as
"this indexer has nothing" is what turns a slow FlareSolverr challenge into
"No releases found for this book" in the UI (#1249), and it also makes the
auto-expand retry fire a second request on top of the one still running.
"""
def resolve_indexer_timeout(timeout: object = None) -> int:
"""Resolve the per-indexer search timeout, falling back to config.
Out-of-range and unparsable values are clamped rather than rejected: this
feeds an HTTP timeout, and a bad setting should not take searching down.
"""
if timeout is None:
timeout = config.get("PROWLARR_INDEXER_TIMEOUT", DEFAULT_INDEXER_TIMEOUT_SECONDS)
resolved = coerce_int_like(timeout)
if resolved is None:
logger.warning(
"Invalid PROWLARR_INDEXER_TIMEOUT %r - using %ss",
timeout,
DEFAULT_INDEXER_TIMEOUT_SECONDS,
)
return DEFAULT_INDEXER_TIMEOUT_SECONDS
return max(MIN_INDEXER_TIMEOUT_SECONDS, min(MAX_INDEXER_TIMEOUT_SECONDS, resolved))
class IndexerSeedSettings(TypedDict, total=False):
ratio_limit: float
seeding_time_limit_minutes: int
@@ -122,23 +77,11 @@ def _get_field_value(fields: object, name: str) -> object | None:
class ProwlarrClient:
"""Client for interacting with the Prowlarr API."""
def __init__(
self, url: str, api_key: str, timeout: int = 30, indexer_timeout: int | None = None
) -> None:
"""Initialize the API client with base URL, key, and timeouts.
Args:
url: Prowlarr base URL.
api_key: Prowlarr API key.
timeout: Timeout for Prowlarr's own JSON endpoints.
indexer_timeout: Timeout for Torznab searches, which Prowlarr proxies
out to the tracker. Defaults to PROWLARR_INDEXER_TIMEOUT.
"""
def __init__(self, url: str, api_key: str, timeout: int = 30) -> None:
"""Initialize the API client with base URL, key, and timeout."""
self.base_url = normalize_http_url(url)
self.api_key = api_key
self.timeout = timeout
self.indexer_timeout = resolve_indexer_timeout(indexer_timeout)
self._session = requests.Session()
self._session.headers.update(
{
@@ -181,12 +124,10 @@ class ProwlarrClient:
msg = f"Invalid JSON response: {e}"
raise ValueError(msg) from e
except requests.exceptions.HTTPError as e:
status_code = e.response.status_code if e.response is not None else "unknown"
reason = e.response.reason if e.response is not None else "unknown"
logger.exception(
"Prowlarr API HTTP error: %s %s",
status_code,
reason,
e.response.status_code,
e.response.reason,
)
raise
except requests.exceptions.RequestException:
@@ -215,55 +156,36 @@ class ProwlarrClient:
logger.info("Prowlarr connection successful: version %s", version)
return True, f"Connected to Prowlarr {version}"
def get_indexers(self, *, raise_on_error: bool = False) -> list[dict[str, Any]]:
"""Get all configured indexers.
Args:
raise_on_error: When True, propagate API failures instead of
returning an empty list. Callers that must distinguish
"no indexers" from "the request failed" should set this.
"""
def get_indexers(self) -> list[dict[str, Any]]:
"""Get all configured indexers."""
try:
return _normalize_json_object_list(
self._request("GET", "/api/v1/indexer"),
context="Prowlarr indexer list",
)
except _PROWLARR_CLIENT_ERRORS:
if raise_on_error:
raise
logger.exception("Failed to get indexers")
return []
def get_enabled_indexers_detailed(
self, *, raise_on_error: bool = False
) -> list[dict[str, Any]]:
def get_enabled_indexers_detailed(self) -> list[dict[str, Any]]:
"""Get enabled indexers, including implementation metadata.
Note: Prowlarr indexer "name" is user-configurable; prefer
"implementation"/"implementationName" for stable identification.
"""
indexers = self.get_indexers(raise_on_error=raise_on_error)
indexers = self.get_indexers()
return [idx for idx in indexers if idx.get("enable", False)]
def get_enriched_indexer_ids(
self,
*,
restrict_to: list[int] | None = None,
indexers: list[dict[str, Any]] | None = None,
) -> list[int]:
def get_enriched_indexer_ids(self, *, restrict_to: list[int] | None = None) -> list[int]:
"""Return enabled indexer IDs that benefit from extra Torznab handling.
Args:
restrict_to: Optional list of candidate indexer IDs to consider.
indexers: Optional already-fetched enabled indexer list, so callers
that need the full records for other reasons can avoid a second
round trip.
"""
enriched_ids: list[int] = []
for idx in indexers if indexers is not None else self.get_enabled_indexers_detailed():
for idx in self.get_enabled_indexers_detailed():
idx_id_int = coerce_int_like(idx.get("id"))
if idx_id_int is None:
continue
@@ -290,17 +212,10 @@ class ProwlarrClient:
Prowlarr exposes seedTime in minutes, which is also the unit expected by
torrent clients.
Raises:
requests.exceptions.RequestException (and other client errors) when
the indexer list cannot be fetched. An empty dict strictly means
"no share limits are configured", never "the request failed" -
callers rely on this to avoid silently dropping seed limits.
"""
settings_by_indexer: dict[int, IndexerSeedSettings] = {}
for idx in self.get_enabled_indexers_detailed(raise_on_error=True):
for idx in self.get_enabled_indexers_detailed():
idx_id_int = coerce_int_like(idx.get("id"))
if idx_id_int is None:
continue
@@ -364,12 +279,6 @@ class ProwlarrClient:
This returns richer fields (e.g., author/booktitle, torznab tags like
FreeLeech) than the JSON /api/v1/search endpoint.
Raises:
ProwlarrSearchError: The search could not be completed. An empty list
strictly means the indexer answered with no matches, never that
the request timed out or errored.
"""
if not query:
return []
@@ -392,7 +301,7 @@ class ProwlarrClient:
response = self._session.get(
url=url,
params=params,
timeout=(_CONNECT_TIMEOUT_SECONDS, self.indexer_timeout),
timeout=self.timeout,
headers={
# Override the session default JSON accept header.
"Accept": "application/rss+xml, application/xml;q=0.9, */*;q=0.8"
@@ -410,20 +319,9 @@ class ProwlarrClient:
for r in results:
if r.get("indexerId") is None:
r["indexerId"] = int(indexer_id)
except requests.exceptions.Timeout as e:
logger.warning(
"Prowlarr Torznab search for indexer %s timed out after %ss. An indexer "
"behind FlareSolverr can need far longer than that on a cold Cloudflare "
"challenge - raise PROWLARR_INDEXER_TIMEOUT if this keeps happening.",
indexer_id,
self.indexer_timeout,
)
msg = f"indexer {indexer_id} did not respond within {self.indexer_timeout}s"
raise ProwlarrSearchError(msg) from e
except Exception as e:
except Exception:
logger.exception("Prowlarr Torznab search failed for indexer %s", indexer_id)
msg = f"indexer {indexer_id} search failed: {e}"
raise ProwlarrSearchError(msg) from e
return []
else:
return results
+16 -226
View File
@@ -1,15 +1,10 @@
"""Prowlarr download handler - resolves releases and delegates lifecycle to shared clients."""
from typing import TYPE_CHECKING, Any
from urllib.parse import urlparse
import requests
from shelfmark.core.config import config
from shelfmark.core.logger import setup_logger
from shelfmark.core.request_helpers import normalize_optional_text
from shelfmark.core.search_plan import build_release_search_plan
from shelfmark.core.utils import normalize_http_url
from shelfmark.download.clients import (
DownloadClient,
get_client,
@@ -28,40 +23,20 @@ from shelfmark.download.clients.base_handler import (
DownloadRequest,
ExternalClientHandler,
)
from shelfmark.download.clients.torrent_utils import (
extract_file_list_from_torrent,
extract_torrent_info,
)
from shelfmark.metadata_providers import BookMetadata
from shelfmark.release_sources import register_handler
from shelfmark.release_sources.prowlarr.api import IndexerSeedSettings, ProwlarrClient
from shelfmark.release_sources.prowlarr.cache import cache_release, get_release, remove_release
from shelfmark.release_sources.prowlarr.source import ProwlarrSource
from shelfmark.release_sources.prowlarr.cache import get_release, remove_release
from shelfmark.release_sources.prowlarr.utils import (
build_source_id,
coerce_int_like,
get_preferred_download_url,
get_protocol,
sanitize_download_url,
)
if TYPE_CHECKING:
from collections.abc import Callable
from shelfmark.core.models import DownloadTask
from shelfmark.download.postprocess.packs import PackFile
logger = setup_logger(__name__)
# Errors that ProwlarrClient can raise when fetching indexer settings.
_SEED_SETTINGS_FALLBACK_ERRORS = (
requests.exceptions.RequestException,
OSError,
RuntimeError,
TypeError,
ValueError,
)
__all__ = [
"ProwlarrHandler",
"POLL_INTERVAL",
@@ -74,11 +49,6 @@ __all__ = [
POLL_INTERVAL = _DEFAULT_POLL_INTERVAL
COMPLETED_PATH_RETRY_INTERVAL = _DEFAULT_COMPLETED_PATH_RETRY_INTERVAL
COMPLETED_PATH_MAX_ATTEMPTS = _DEFAULT_COMPLETED_PATH_MAX_ATTEMPTS
EXPIRED_LINK_REFRESH_ERROR = (
"The indexer download link expired and the release could not be refreshed. "
"Search again for a fresh result."
)
HASH_DETECTION_ERROR = "Could not determine torrent hash from URL"
def _coerce_positive_minutes(raw_minutes: object) -> int | None:
@@ -92,65 +62,6 @@ def _coerce_positive_minutes(raw_minutes: object) -> int | None:
class ProwlarrHandler(ExternalClientHandler):
"""Handler for Prowlarr downloads via configured torrent or usenet client."""
@staticmethod
def _build_prowlarr_client() -> ProwlarrClient | None:
"""Build a ProwlarrClient from config, or None if not configured."""
raw_url = config.get("PROWLARR_URL", "")
raw_api_key = config.get("PROWLARR_API_KEY", "")
url = normalize_optional_text(raw_url) if isinstance(raw_url, str) else None
api_key = normalize_optional_text(raw_api_key) if isinstance(raw_api_key, str) else None
if not url or not api_key:
return None
normalized_url = normalize_http_url(url)
if not normalized_url:
return None
return ProwlarrClient(normalized_url, api_key)
def _fetch_seed_settings_fallback(self, raw_indexer_id: object) -> IndexerSeedSettings | None:
"""Fetch share limits for one indexer directly from Prowlarr.
Used when the cached release is missing its search-time seed-limit
enrichment so that transient failures during search cannot cause a
torrent to be added without its configured share limits.
"""
indexer_id = coerce_int_like(raw_indexer_id)
if indexer_id is None:
return None
client = self._build_prowlarr_client()
if client is None:
return None
try:
settings = client.get_indexer_seed_settings(restrict_to=[indexer_id])
except _SEED_SETTINGS_FALLBACK_ERRORS:
logger.warning(
"Grab-time seed settings fallback failed for indexerId=%s",
indexer_id,
exc_info=True,
)
return None
return settings.get(indexer_id)
def list_files(self, release_data: dict[str, Any]) -> list[PackFile] | None:
"""List a cached torrent release's files from its .torrent, without downloading.
Magnet-only and usenet releases cannot be listed ahead of time.
"""
source_id = str(release_data.get("source_id") or "")
prowlarr_result = get_release(source_id) if source_id else None
if not prowlarr_result or get_protocol(prowlarr_result) != "torrent":
return None
download_url = sanitize_download_url(str(prowlarr_result.get("downloadUrl") or "").strip())
if not download_url or download_url.startswith("magnet:"):
return None
expected_hash = str(prowlarr_result.get("infoHash") or "").strip() or None
info = extract_torrent_info(download_url, expected_hash=expected_hash)
if not info.torrent_data:
return None
return extract_file_list_from_torrent(info.torrent_data)
def _get_client(self, protocol: str) -> DownloadClient | None:
"""Compatibility shim so module-level patching still works in tests."""
return get_client(protocol)
@@ -170,30 +81,18 @@ class ProwlarrHandler(ExternalClientHandler):
def build_retry_resolution_fields(self, release_data: dict[str, Any]) -> dict[str, Any]:
source_id = normalize_optional_text(release_data.get("source_id"))
extra = release_data.get("extra")
if not isinstance(extra, dict):
extra = {}
if source_id is None:
return {}
retry_source_context: dict[str, Any] = {}
indexer_id = release_data.get("indexer_id") or extra.get("indexer_id")
if indexer_id is not None:
retry_source_context["indexer_id"] = indexer_id
indexer = normalize_optional_text(release_data.get("indexer") or extra.get("indexer"))
if indexer is not None and indexer.lower() != "unknown":
retry_source_context["indexer"] = indexer
info_url = normalize_optional_text(release_data.get("info_url") or extra.get("info_url"))
if info_url is not None:
retry_source_context["info_url"] = info_url
if source_id is not None:
retry_source_context["source_id"] = source_id
prowlarr_result = get_release(source_id)
if prowlarr_result is None:
return {}
return {
"retry_download_url": None,
"retry_download_protocol": None,
"retry_source_context": retry_source_context,
"retry_download_url": normalize_optional_text(
get_preferred_download_url(prowlarr_result)
),
"retry_download_protocol": normalize_optional_text(get_protocol(prowlarr_result)),
}
@classmethod
@@ -240,12 +139,13 @@ class ProwlarrHandler(ExternalClientHandler):
# Look up the cached release
prowlarr_result = get_release(task.task_id)
if not prowlarr_result:
logger.info("Prowlarr release cache miss, refreshing: %s", task.task_id)
prowlarr_result = self._refresh_release(task)
if prowlarr_result is None:
logger.warning("Prowlarr release refresh failed: %s", task.task_id)
status_callback("error", EXPIRED_LINK_REFRESH_ERROR)
restored_request = self._restore_download_request_from_task(task)
if restored_request is None:
logger.warning("Release cache miss: %s", task.task_id)
status_callback("error", "Release not found in cache (may have expired)")
return None
logger.info("Restored Prowlarr download request for retry: %s", task.task_id)
return restored_request
# Extract download URL
download_url = get_preferred_download_url(prowlarr_result)
@@ -271,28 +171,6 @@ class ProwlarrHandler(ExternalClientHandler):
seeding_time_limit = _coerce_positive_minutes(raw_configured_seed_time)
ratio_limit = float(raw_configured_ratio) if raw_configured_ratio is not None else None
# Fallback: search-time enrichment can be missing when the indexer
# settings fetch transiently failed during the search (#795).
# Re-resolve the limits from Prowlarr at grab time so torrents are
# never sent to the client without their configured share limits.
if seeding_time_limit is None and ratio_limit is None and protocol == "torrent":
fallback = self._fetch_seed_settings_fallback(prowlarr_result.get("indexerId"))
if fallback:
seeding_time_limit = _coerce_positive_minutes(
fallback.get("seeding_time_limit_minutes")
)
raw_ratio = fallback.get("ratio_limit")
ratio_limit = float(raw_ratio) if raw_ratio is not None else None
if seeding_time_limit is None and ratio_limit is None and protocol == "torrent":
logger.warning(
"Prowlarr seed preferences are enabled but no share limits "
"could be resolved for release '%s' (indexerId=%s); the "
"torrent will use the client's global limits",
release_name,
prowlarr_result.get("indexerId"),
)
return DownloadRequest(
url=download_url,
protocol=protocol,
@@ -302,94 +180,6 @@ class ProwlarrHandler(ExternalClientHandler):
ratio_limit=ratio_limit,
)
def _refresh_release(self, task: DownloadTask) -> dict[str, Any] | None:
"""Re-query Prowlarr and cache the exact original release if it still exists."""
title = normalize_optional_text(task.title)
if title is None:
return None
context = getattr(task, "retry_source_context", None)
if not isinstance(context, dict):
context = {}
indexer = normalize_optional_text(context.get("indexer"))
book = BookMetadata(
provider="shelfmark",
provider_id=task.task_id,
title=title,
authors=[task.author] if task.author else [],
search_title=title,
search_author=task.author,
)
# No language default here on purpose: this re-finds one exact release by its
# guid, and Prowlarr does not filter on plan.languages anyway.
plan = build_release_search_plan(
book,
indexers=[indexer] if indexer is not None else None,
)
source = ProwlarrSource()
results = source.search(book, plan, content_type=task.content_type or "ebook")
for release in results:
raw_release = get_release(release.source_id)
if raw_release is None:
continue
if not self._raw_release_matches_task(raw_release, task.task_id):
continue
cache_release(task.task_id, raw_release)
logger.info("Refreshed Prowlarr release: %s", task.task_id)
return raw_release
return None
@staticmethod
def _raw_release_matches_task(raw_release: dict[str, Any], task_id: str) -> bool:
wanted = normalize_optional_text(task_id)
if wanted is None:
return False
bare = [
identity
for identity in (
normalize_optional_text(raw_release.get("guid")),
normalize_optional_text(raw_release.get("infoUrl")),
)
if identity is not None
]
identities = [*bare, build_source_id(raw_release)]
indexer_id = coerce_int_like(raw_release.get("indexerId"))
if indexer_id is not None:
identities.extend(f"{indexer_id}:{identity}" for identity in bare)
return wanted in identities
def _refresh_download_request_after_add_failure(
self,
*,
task: DownloadTask,
request: DownloadRequest,
error: Exception,
status_callback: Callable[[str, str | None], None],
) -> DownloadRequest | None:
"""Refresh once when a cached Prowlarr torrent proxy URL has expired."""
if request.protocol != "torrent":
return None
if HASH_DETECTION_ERROR not in str(error):
return None
parsed = urlparse(request.url)
if parsed.scheme.lower() not in {"http", "https"}:
return None
logger.info("Refreshing stale Prowlarr torrent URL for %s", task.task_id)
remove_release(task.task_id)
refreshed_request = self._resolve_download(task, status_callback)
if refreshed_request is None:
raise RuntimeError(EXPIRED_LINK_REFRESH_ERROR) from error
return refreshed_request
def _on_download_complete(self, task: DownloadTask) -> None:
"""Remove completed release from the Prowlarr cache."""
remove_release(task.task_id)
@@ -10,18 +10,12 @@ from shelfmark.core.settings_registry import (
CheckboxField,
HeadingField,
MultiSelectField,
NumberField,
PasswordField,
SettingsField,
TextField,
register_settings,
)
from shelfmark.core.utils import normalize_http_url
from shelfmark.release_sources.prowlarr.api import (
DEFAULT_INDEXER_TIMEOUT_SECONDS,
MAX_INDEXER_TIMEOUT_SECONDS,
MIN_INDEXER_TIMEOUT_SECONDS,
)
# ==================== Dynamic Options Loaders ====================
@@ -189,20 +183,6 @@ def prowlarr_config_settings() -> list[SettingsField]:
default=[],
show_when={"field": "PROWLARR_ENABLED", "value": True},
),
NumberField(
key="PROWLARR_INDEXER_TIMEOUT",
label="Indexer Search Timeout (seconds)",
description=(
"How long to wait for a single indexer to answer a search. Indexers behind "
"FlareSolverr can need 90 seconds or more while a cold Cloudflare challenge "
"is solved; raise this if searches come back empty and the Prowlarr log "
"shows the search still running."
),
default=DEFAULT_INDEXER_TIMEOUT_SECONDS,
min_value=MIN_INDEXER_TIMEOUT_SECONDS,
max_value=MAX_INDEXER_TIMEOUT_SECONDS,
show_when={"field": "PROWLARR_ENABLED", "value": True},
),
CheckboxField(
key="PROWLARR_AUTO_EXPAND",
label="Auto-expand search on no results",
@@ -210,18 +190,6 @@ def prowlarr_config_settings() -> list[SettingsField]:
description="Automatically retry search without category filtering if no results are found",
show_when={"field": "PROWLARR_ENABLED", "value": True},
),
CheckboxField(
key="PROWLARR_COLLAPSE_DUPLICATES",
label="Show one row per release",
default=True,
description=(
"Collapse a release that several indexer entries returned down to a single row, "
"keeping the entry with the best Prowlarr priority. Turn this off to see every "
"entry that carried it, which is what makes results from filter-specific entries "
"(freeleech and the like) visible."
),
show_when={"field": "PROWLARR_ENABLED", "value": True},
),
CheckboxField(
key="PROWLARR_USE_SEED_PREFERENCES",
label="Use Prowlarr seed preferences",
+106 -389
View File
@@ -2,22 +2,16 @@
import re
import time
from dataclasses import dataclass
from threading import Lock
from typing import TYPE_CHECKING, ClassVar, NoReturn
import requests
if TYPE_CHECKING:
from shelfmark.core.search_plan import ReleaseSearchPlan
from shelfmark.metadata_providers import BookMetadata
from shelfmark.core.config import config
from shelfmark.core.languages import normalize_language
from shelfmark.core.logger import setup_logger
from shelfmark.core.request_helpers import normalize_optional_text
from shelfmark.core.search_plan import ReleaseSearchVariant
from shelfmark.core.utils import AUDIOBOOK_FORMATS as CORE_AUDIOBOOK_FORMATS
from shelfmark.core.utils import normalize_http_url
from shelfmark.release_sources import (
ColumnAlign,
@@ -31,19 +25,11 @@ from shelfmark.release_sources import (
ReleaseProtocol,
ReleaseSource,
SortOption,
SourceUnavailableError,
register_source,
)
from shelfmark.release_sources.prowlarr.api import (
IndexerSeedSettings,
ProwlarrClient,
ProwlarrSearchError,
)
from shelfmark.release_sources.prowlarr.api import IndexerSeedSettings, ProwlarrClient
from shelfmark.release_sources.prowlarr.cache import cache_release
from shelfmark.release_sources.prowlarr.utils import (
AUTHOR_UNKNOWN,
author_affinity,
build_source_id,
coerce_float_like,
coerce_int_like,
get_protocol,
@@ -55,14 +41,6 @@ _SIZE_UNIT_BASE = 1024
_TWO_FORMATS = 2
_PROWLARR_SOURCE_ERRORS = (AttributeError, OSError, RuntimeError, TypeError, ValueError)
# Prowlarr indexer priority is 1-50 and lower is preferred; unknown sorts last.
_UNRANKED_INDEXER_RANK = 51
# Errors that can surface from a ProwlarrClient call that talks to Prowlarr. The
# client raises requests exceptions (subclasses of OSError via IOError lineage
# is not guaranteed), so include RequestException explicitly.
_PROWLARR_REQUEST_ERRORS = (*_PROWLARR_SOURCE_ERRORS, requests.exceptions.RequestException)
def _raise_timeout_error(message: str) -> NoReturn:
raise TimeoutError(message)
@@ -83,148 +61,6 @@ def _coerce_indexer_id(value: object) -> int | None:
return coerce_int_like(value)
def _identity_text(value: object) -> str | None:
"""Trimmed text for an identity field, or None when there is nothing usable."""
if isinstance(value, str):
return value.strip() or None
if isinstance(value, (int, float)) and not isinstance(value, bool):
return str(value)
return None
def _release_identity(result: dict) -> str | None:
"""Identify the underlying release, independent of which indexer surfaced it.
Strong identifiers only. Title is deliberately excluded because matching on
it here would merge two genuinely different releases that happen to share a
name, and every caller of this either drops or overwrites a row on a match.
Returns None when nothing identifies the result.
"""
for field in ("guid", "downloadUrl", "magnetUrl", "infoUrl"):
identity = _identity_text(result.get(field))
if identity is not None:
return identity
return None
def _result_dedup_key(result: dict) -> tuple[int | None, str] | None:
"""Dedup key for a raw Prowlarr result, or None if it cannot be identified.
One tracker is often configured in Prowlarr as several indexer entries that
differ only by a server-side search filter, say a "freeleech only" entry
alongside an unfiltered one. Those entries return the same guid for the same
torrent, so keying on the guid alone throws away the filtered entry's copy
and with it the only signal that the release matched the filter. Including
the indexer id keeps the entries distinct.
Title is an acceptable last resort here, unlike in _release_identity, because
the indexer id is part of the key: it only ever collapses a literal repeat
from one indexer, never two rows from different entries.
"""
identity = _release_identity(result) or _identity_text(result.get("title"))
if identity is None:
return None
return (_coerce_indexer_id(result.get("indexerId")), identity)
def _build_indexer_priority(indexers: list[dict]) -> dict[int, int]:
"""Map indexer id to the priority configured in Prowlarr. Lower is preferred.
Users already rank their indexers in Prowlarr, and on trackers configured as
several entries that ranking is usually the meaningful one: a "freeleech
only" entry is typically given a better priority than the unfiltered entry
beside it. Reusing it avoids asking for the same ordering a second time.
"""
priority: dict[int, int] = {}
for indexer in indexers:
indexer_id = _coerce_indexer_id(indexer.get("id"))
if indexer_id is None:
continue
rank = coerce_int_like(indexer.get("priority"))
if rank is not None:
priority[indexer_id] = rank
return priority
def _drop_unknown_indexer_ids(
selected_ids: list[int] | None, indexers: list[dict]
) -> list[int] | None:
"""Keep only selected indexer ids Prowlarr still serves.
An indexer removed or disabled in Prowlarr stays in the saved selection,
where settings can no longer show it - so it cannot be unselected, and every
search keeps querying an indexer that is gone (#1283). Dropping it here
keeps the saved selection intact for an indexer that comes back.
"""
if selected_ids is None:
return None
live_ids = {
indexer_id
for indexer in indexers
if (indexer_id := _coerce_indexer_id(indexer.get("id"))) is not None
}
kept = [indexer_id for indexer_id in selected_ids if indexer_id in live_ids]
stale = [indexer_id for indexer_id in selected_ids if indexer_id not in live_ids]
if stale:
logger.warning(
"Skipping selected Prowlarr indexers that are no longer enabled in Prowlarr: %s",
stale,
)
return kept
def _rank_for_indexer_id(indexer_id: object, priority: dict[int, int]) -> int:
"""Preference rank for an indexer id. Lower wins, unknown ranks last."""
coerced = _coerce_indexer_id(indexer_id)
if coerced is None:
return _UNRANKED_INDEXER_RANK
return priority.get(coerced, _UNRANKED_INDEXER_RANK)
def _indexer_rank(result: dict, priority: dict[int, int]) -> int:
"""Preference rank of the indexer that surfaced a raw result."""
return _rank_for_indexer_id(result.get("indexerId"), priority)
def _release_indexer_rank(release: Release, priority: dict[int, int]) -> int:
"""Preference rank of the indexer that surfaced a converted release."""
return _rank_for_indexer_id(release.extra.get("indexer_id"), priority)
def _collapse_duplicate_indexer_results(
results: list[dict], priority: dict[int, int]
) -> list[dict]:
"""Reduce a release to a single row, keeping the preferred indexer entry.
Opt-in behaviour for users who want one row per torrent. Ties keep the
result that was queried first, and the winner holds the loser's position so
the overall result order stays stable.
"""
position_by_identity: dict[str, int] = {}
kept: list[dict] = []
for result in results:
identity = _release_identity(result)
if identity is None:
kept.append(result)
continue
existing_position = position_by_identity.get(identity)
if existing_position is None:
position_by_identity[identity] = len(kept)
kept.append(result)
continue
if _indexer_rank(result, priority) < _indexer_rank(kept[existing_position], priority):
kept[existing_position] = result
return kept
def _parse_size(size_bytes: int | None) -> str | None:
"""Convert bytes to human-readable size string."""
if size_bytes is None or size_bytes <= 0:
@@ -261,44 +97,59 @@ EBOOK_FORMATS = [
]
# Common audiobook formats
AUDIOBOOK_FORMATS = list(CORE_AUDIOBOOK_FORMATS)
AUDIOBOOK_FORMATS = ["m4b", "mp3", "m4a", "flac", "ogg", "wma", "aac", "wav", "opus"]
# Combined list for format detection (audiobook formats first for priority)
ALL_BOOK_FORMATS = AUDIOBOOK_FORMATS + EBOOK_FORMATS
# Map 3-char MAM language codes to 2-char ISO codes used by frontend color maps
MAM_LANGUAGE_MAP = {
"eng": "en",
"ita": "it",
"spa": "es",
"fra": "fr",
"fre": "fr",
"ger": "de",
"deu": "de",
"por": "pt",
"rus": "ru",
"jpn": "ja",
"jap": "ja",
"chi": "zh",
"zho": "zh",
"dut": "nl",
"nld": "nl",
"swe": "sv",
"nor": "no",
"dan": "da",
"fin": "fi",
"pol": "pl",
"cze": "cs",
"ces": "cs",
"hun": "hu",
"kor": "ko",
"ara": "ar",
"heb": "he",
"tur": "tr",
"gre": "el",
"ell": "el",
"hin": "hi",
"tha": "th",
"vie": "vi",
"ind": "id",
"ukr": "uk",
"rom": "ro",
"ron": "ro",
"bul": "bg",
"cat": "ca",
"hrv": "hr",
"slv": "sl",
"srp": "sr",
}
# Backend safeguard: cap total Prowlarr search time per request.
PROWLARR_SEARCH_TIMEOUT_SECONDS = 120.0
# The overall budget has to leave room for at least a couple of indexers to spend
# their full per-indexer timeout, otherwise raising PROWLARR_INDEXER_TIMEOUT for a
# Cloudflare-fronted tracker just moves the cutoff here. Capped short of the
# gunicorn worker timeout (300s) so the worker is never the thing that gives up.
_MAX_SEARCH_BUDGET_SECONDS = 240.0
def _search_budget_seconds(indexer_timeout: int) -> float:
"""Total time one Prowlarr search may spend, scaled to the per-indexer timeout."""
return min(
_MAX_SEARCH_BUDGET_SECONDS,
max(PROWLARR_SEARCH_TIMEOUT_SECONDS, indexer_timeout * 2.0),
)
@dataclass
class _IndexerSearchOutcome:
"""What one pass over the target indexers produced.
Separates "every indexer answered, none had this book" from "the indexers
never answered", which the caller has to tell apart before it decides to
auto-expand or to report the search as failed.
"""
results: list[dict]
attempted: int = 0
failed: int = 0
last_error: str | None = None
def _extract_format(title: str) -> str | None:
"""Extract ebook/audiobook format from release title (extension, bracketed, or standalone)."""
@@ -342,32 +193,25 @@ def _extract_mam_language(raw_title: str) -> str | None:
for token in tokens:
lang_code = token.lower()
resolved = normalize_language(lang_code)
if resolved is not None:
return resolved
if lang_code in MAM_LANGUAGE_MAP:
return MAM_LANGUAGE_MAP[lang_code]
return None
def _split_mam_formats(raw_title: str) -> tuple[list[str], list[str]]:
"""Split the format tokens of a MyAnonamouse title into (recognized, unrecognized).
def _extract_mam_formats(raw_title: str) -> list[str]:
"""Extract a list of formats from MyAnonamouse titles.
Prowlarr's MAM parser appends a structured bracket segment like:
[ENG / EPUB MOBI PDF]
We only trust this structured segment (and do not attempt generic title
heuristics for other indexers).
Tokens after the "/" that Shelfmark does not know as a book or audiobook format
(e.g. ``[ENG / AVI]``) are returned separately so the UI can warn that the release
will download but cannot be processed, instead of showing a bare content-type icon
that looks like an ordinary result.
"""
if not raw_title:
return [], []
return []
format_set = set(ALL_BOOK_FORMATS)
first_unrecognized: list[str] | None = None
for bracket in re.findall(r"\[([^\]]+)\]", raw_title):
if "/" not in bracket:
continue
@@ -376,26 +220,15 @@ def _split_mam_formats(raw_title: str) -> tuple[list[str], list[str]]:
tokens = re.findall(r"[A-Za-z0-9]+", after_slash)
formats: list[str] = []
unrecognized: list[str] = []
for token in tokens:
fmt = token.lower()
if fmt in format_set:
if fmt not in formats:
formats.append(fmt)
elif fmt not in unrecognized:
unrecognized.append(fmt)
if fmt in format_set and fmt not in formats:
formats.append(fmt)
if formats:
return formats, unrecognized
if unrecognized and first_unrecognized is None:
first_unrecognized = unrecognized
return formats
return [], first_unrecognized or []
def _extract_mam_formats(raw_title: str) -> list[str]:
"""Extract the recognized formats from a MyAnonamouse title (see _split_mam_formats)."""
return _split_mam_formats(raw_title)[0]
return []
def _formats_display(formats: list[str]) -> str | None:
@@ -534,7 +367,6 @@ def _prowlarr_result_to_release(
format_detected: str | None = None
formats: list[str] = []
unrecognized_formats: list[str] = []
formats_display: str | None = None
language_detected: str | None = None
if enable_format_detection:
@@ -542,12 +374,13 @@ def _prowlarr_result_to_release(
if book_title:
title = book_title
formats, unrecognized_formats = _split_mam_formats(str(raw_title or ""))
formats = _extract_mam_formats(str(raw_title or ""))
format_detected = formats[0] if formats else None
formats_display = _formats_display(formats)
language_detected = _extract_mam_language(str(raw_title or ""))
source_id = build_source_id(result)
# Build the source_id from GUID or generate from indexer + title
source_id = result.get("guid") or f"{indexer}:{hash(raw_title)}"
# Cache the raw Prowlarr result so handler can look it up by source_id
cache_release(source_id, result)
@@ -604,45 +437,12 @@ def _prowlarr_result_to_release(
"info_hash": result.get("infoHash"),
"formats": formats or None,
"formats_display": formats_display,
# Format tokens the indexer declared but Shelfmark can't process (e.g. a MAM
# "[ENG / AVI]"). Lets the UI warn instead of showing a bare content icon.
"unrecognized_formats": unrecognized_formats or None,
# Raw torznab attributes for rich tooltips (enriched indexers)
"torznab_attrs": result.get("torznabAttrs"),
},
)
# Last successfully fetched per-indexer share limits. Used as a fallback when
# a transient Prowlarr API failure prevents fetching fresh settings during a
# search, so results are never silently cached without seed limits (#795).
_seed_settings_lock = Lock()
_last_known_seed_settings: dict[int, IndexerSeedSettings] = {}
def _fetch_indexer_seed_settings(
client: ProwlarrClient,
indexer_ids: list[int] | None,
) -> dict[int, IndexerSeedSettings]:
"""Fetch per-indexer share limits, falling back to last-known-good on failure."""
try:
fetched = client.get_indexer_seed_settings(restrict_to=indexer_ids)
except _PROWLARR_REQUEST_ERRORS:
with _seed_settings_lock:
fallback = dict(_last_known_seed_settings)
logger.warning(
"Failed to fetch Prowlarr indexer seed settings; "
"falling back to last known settings for %s indexer(s)",
len(fallback),
exc_info=True,
)
return fallback
with _seed_settings_lock:
_last_known_seed_settings.update(fetched)
return fetched
def _apply_indexer_seed_settings(
result: dict,
indexer_seed_settings: dict[int, IndexerSeedSettings],
@@ -774,11 +574,6 @@ class ProwlarrSource(ReleaseSource):
],
extra_sort_options=[
SortOption(label="Peers", sort_key="seeders"),
SortOption(
label="Indexer priority",
sort_key="extra.indexer_priority",
default_direction="asc",
),
],
grid_template="minmax(0,2fr) minmax(140px,1fr) 50px 50px 90px 80px",
leading_cell=LeadingCellConfig(
@@ -983,143 +778,91 @@ class ProwlarrSource(ReleaseSource):
try:
auto_expand_enabled = config.get("PROWLARR_AUTO_EXPAND", False)
search_budget = _search_budget_seconds(client.indexer_timeout)
deadline = time.monotonic() + search_budget
try:
enabled_indexers = client.get_enabled_indexers_detailed(raise_on_error=True)
except _PROWLARR_REQUEST_ERRORS as e:
# Prowlarr itself is unreachable. Swallowing this leaves the search
# with no indexers to query, which the UI renders as "No releases
# found for this book" - the same lie as a swallowed timeout (#1249).
msg = f"could not reach Prowlarr: {e}"
raise SourceUnavailableError(msg) from e
indexer_ids = _drop_unknown_indexer_ids(indexer_ids, enabled_indexers)
indexer_priority = _build_indexer_priority(enabled_indexers)
deadline = time.monotonic() + PROWLARR_SEARCH_TIMEOUT_SECONDS
# Some indexers benefit from title+author queries and extra format detection.
enriched_indexer_ids = client.get_enriched_indexer_ids(
restrict_to=indexer_ids, indexers=enabled_indexers
)
enriched_indexer_ids = client.get_enriched_indexer_ids(restrict_to=indexer_ids)
enriched_indexer_ids_set = set(enriched_indexer_ids)
indexer_seed_settings = (
_fetch_indexer_seed_settings(client, indexer_ids)
client.get_indexer_seed_settings(restrict_to=indexer_ids)
if config.get("PROWLARR_USE_SEED_PREFERENCES", False)
else {}
)
def _check_timeout() -> None:
if time.monotonic() > deadline:
_raise_timeout_error(f"Prowlarr search timed out after {int(search_budget)}s")
_raise_timeout_error(
f"Prowlarr search timed out after {int(PROWLARR_SEARCH_TIMEOUT_SECONDS)}s"
)
def search_indexers(query: str, cats: list[int] | None) -> _IndexerSearchOutcome:
"""Search indexers with given categories via Torznab/Newznab.
Every indexer gets the same title-only query. Enriched indexers used
to be sent "{title} {author}", but an indexer that ANDs its search
terms (MyAnonamouse) returns nothing whenever the metadata provider
spells the author differently to the tracker - "Timothy Ferriss" vs
"Tim Ferriss" - and the UI reports the book as missing (#1293). The
author still decides ordering below, where a spelling difference
costs a release its position rather than its existence.
"""
outcome = _IndexerSearchOutcome(results=[])
def search_indexers(
query: str, cats: list[int] | None, *, enriched_query: str | None = None
) -> list[dict]:
"""Search indexers with given categories via Torznab/Newznab."""
results: list[dict] = []
target_indexer_ids = self._get_search_indexer_ids(client, indexer_ids, cats)
if not target_indexer_ids:
return outcome
return results
for indexer_id in target_indexer_ids:
_check_timeout()
outcome.attempted += 1
try:
raw = client.torznab_search(
indexer_id=indexer_id,
query=query,
categories=cats,
search_type="book",
)
except ProwlarrSearchError as e:
# One unreachable indexer must not sink the others, but it
# is not "no results" either - record it so the caller can
# report a failed search instead of an empty one.
outcome.failed += 1
outcome.last_error = str(e)
continue
indexer_query = (
enriched_query
if indexer_id in enriched_indexer_ids_set and enriched_query
else query
)
raw = client.torznab_search(
indexer_id=indexer_id,
query=indexer_query,
categories=cats,
search_type="book",
)
if raw:
outcome.results.extend(raw)
results.extend(raw)
return outcome
return results
seen_keys: set[tuple[int | None, str]] = set()
seen_keys: set[str] = set()
all_results: list[dict] = []
attempted_searches = 0
failed_searches = 0
last_search_error: str | None = None
for idx, variant in enumerate(variants, start=1):
_check_timeout()
query = variant.title
enriched_query = variant.query # title + author
if len(variants) > 1:
logger.debug("Prowlarr query %s/%s: '%s'", idx, len(variants), query)
outcome = search_indexers(query=query, cats=categories)
raw_results = search_indexers(
query=query, cats=categories, enriched_query=enriched_query
)
# Auto-expand: if no results with categories and auto-expand enabled, retry without.
# Only when every indexer actually answered: a failed search says nothing about
# whether the category filter is what hid the book, and retrying it stacks a second
# request on an indexer that is still busy solving a Cloudflare challenge (#1249).
if (
not outcome.results
and not outcome.failed
and categories
and auto_expand_enabled
):
# Auto-expand: if no results with categories and auto-expand enabled, retry without
if not raw_results and categories and auto_expand_enabled:
_check_timeout()
logger.info(
"Prowlarr: no results for query '%s' with category filter, auto-expanding search",
query,
)
expanded = search_indexers(query=query, cats=None)
outcome.results = expanded.results
outcome.attempted += expanded.attempted
outcome.failed += expanded.failed
outcome.last_error = expanded.last_error or outcome.last_error
raw_results = search_indexers(
query=query, cats=None, enriched_query=enriched_query
)
self.last_search_type = "expanded"
attempted_searches += outcome.attempted
failed_searches += outcome.failed
last_search_error = outcome.last_error or last_search_error
for r in outcome.results:
key = _result_dedup_key(r)
if key is not None:
if key in seen_keys:
continue
seen_keys.add(key)
all_results.append(r)
if failed_searches:
logger.warning(
"Prowlarr: %s of %s indexer searches failed (%s)",
failed_searches,
attempted_searches,
last_search_error,
)
if config.get("PROWLARR_COLLAPSE_DUPLICATES", True):
before_collapse = len(all_results)
all_results = _collapse_duplicate_indexer_results(all_results, indexer_priority)
if len(all_results) != before_collapse:
logger.debug(
"Prowlarr: collapsed %s duplicate result(s) across indexer entries",
before_collapse - len(all_results),
for r in raw_results:
key = (
r.get("guid")
or r.get("downloadUrl")
or r.get("magnetUrl")
or r.get("infoUrl")
or f"{r.get('indexerId')}:{r.get('title')}"
)
if key in seen_keys:
continue
seen_keys.add(key)
all_results.append(r)
results: list[Release] = []
enriched_source_ids: set[str] = set()
affinity_by_source_id: dict[str, int] = {}
# A manual query is the user's own words; ranking it against the
# metadata author would second-guess what they typed.
wanted_author = "" if plan.manual_query else plan.author
for raw_result in all_results:
result_with_seed_settings = _apply_indexer_seed_settings(
@@ -1136,26 +879,13 @@ class ProwlarrSource(ReleaseSource):
content_type,
enable_format_detection=is_enriched,
)
if idx_id_int is not None and idx_id_int in indexer_priority:
release.extra["indexer_priority"] = indexer_priority[idx_id_int]
results.append(release)
affinity_by_source_id[release.source_id] = author_affinity(
wanted_author, release.extra.get("author")
)
if is_enriched:
enriched_source_ids.add(release.source_id)
# Indexer priority first: it is an explicit user preference. Author
# agreement then orders what one indexer returned, so the editions that
# match the requested author lead and the rest stay reachable below.
results.sort(
key=lambda r: (
_release_indexer_rank(r, indexer_priority),
affinity_by_source_id.get(r.source_id, AUTHOR_UNKNOWN),
0 if r.source_id in enriched_source_ids else 1,
)
)
# Sort results: enriched indexers first, then others
results.sort(key=lambda r: 0 if r.source_id in enriched_source_ids else 1)
if results:
torrent_count = sum(1 for r in results if r.protocol == ReleaseProtocol.TORRENT)
@@ -1172,10 +902,6 @@ class ProwlarrSource(ReleaseSource):
else:
logger.debug("Prowlarr: no results found")
except SourceUnavailableError:
# Already carries its own message for the caller to surface; the blanket
# handler below would turn it back into a silent empty result.
raise
except TimeoutError as e:
logger.warning("Prowlarr search timed out: %s", e)
raise
@@ -1183,15 +909,6 @@ class ProwlarrSource(ReleaseSource):
logger.exception("Prowlarr search failed")
return []
else:
# An empty list is the UI's "No releases found for this book", so it has
# to mean the indexers answered and had nothing. When they failed instead,
# say so rather than blaming the book (#1249).
if not results and failed_searches:
msg = (
f"{failed_searches} of {attempted_searches} indexer searches failed "
f"({last_search_error})"
)
raise SourceUnavailableError(msg)
return results
def is_available(self) -> bool:
@@ -14,20 +14,6 @@ if TYPE_CHECKING:
_INTEGER_LIKE_PATTERN = re.compile(r"^[+-]?\d+$")
_FLOAT_LIKE_PATTERN = re.compile(r"^[+-]?(?:\d+(?:\.\d*)?|\.\d+)$")
_AUTHOR_TOKEN_PATTERN = re.compile(r"\w+", re.UNICODE)
_AUTHOR_NOISE_TOKENS = frozenset(
{"jr", "sr", "ii", "iii", "iv", "phd", "md", "dr", "mr", "mrs", "ms", "et", "al", "and", "the"}
)
# Ordering tiers for author agreement between the requested book and what an
# indexer reported. Lower sorts first.
AUTHOR_MATCH = 0
AUTHOR_UNKNOWN = 1
AUTHOR_MISMATCH = 2
# A mononym ("Homer") can only ever agree on one token; a longer name needs a
# given name and a surname to agree before it counts as the same person.
_AUTHOR_TOKENS_REQUIRED = 2
def coerce_int_like(value: object) -> int | None:
@@ -46,71 +32,6 @@ def coerce_int_like(value: object) -> int | None:
return int(normalized)
def _author_tokens(value: object) -> list[str]:
"""Split an author string into comparable lowercase name tokens."""
if not isinstance(value, str):
return []
tokens = [token.lower() for token in _AUTHOR_TOKEN_PATTERN.findall(value)]
return [token for token in tokens if token not in _AUTHOR_NOISE_TOKENS]
def _author_tokens_compatible(wanted: str, offered: str) -> bool:
"""Treat an abbreviated given name as the name it abbreviates."""
return wanted == offered or wanted.startswith(offered) or offered.startswith(wanted)
def author_affinity(wanted: object, offered: object) -> int:
"""Rank how far an indexer's author field is from the requested author.
Shelfmark ranks on this rather than filtering on it, so a wrong verdict only
costs a release its position in the list, never its visibility. That is what
makes the loose token comparison safe: "Tim"/"Timothy" and "T."/"Timothy"
agree, while a transliteration ("Dostoevsky"/"Dostoyevsky") is merely sorted
last instead of being hidden.
Three-way on purpose: an indexer that reports no author at all must not sort
below one that reports a wrong author, so "no metadata" ranks between
agreement and disagreement rather than counting as either.
"""
wanted_tokens = _author_tokens(wanted)
offered_tokens = _author_tokens(offered)
if not wanted_tokens or not offered_tokens:
return AUTHOR_UNKNOWN
matched = sum(
1
for wanted_token in wanted_tokens
if any(
_author_tokens_compatible(wanted_token, offered_token)
for offered_token in offered_tokens
)
)
required = min(_AUTHOR_TOKENS_REQUIRED, len(wanted_tokens))
return AUTHOR_MATCH if matched >= required else AUTHOR_MISMATCH
def build_source_id(result: dict) -> str:
"""Build the Release.source_id for a raw Prowlarr result.
Qualified by the indexer id because one tracker is often configured in
Prowlarr as several indexer entries that differ only by a server-side search
filter, and those entries return the same guid for the same torrent. Without
the qualifier the entries collide in the release cache and a grab routes
through whichever entry happened to cache last.
"""
guid = result.get("guid")
if guid:
base = str(guid)
else:
indexer = result.get("indexer", "Unknown")
base = f"{indexer}:{hash(result.get('title', 'Unknown'))}"
indexer_id = coerce_int_like(result.get("indexerId"))
if indexer_id is None:
return base
return f"{indexer_id}:{base}"
def coerce_float_like(value: object) -> float | None:
"""Return a float for float-like config/API values, else None."""
if isinstance(value, bool):
-9
View File
@@ -35,15 +35,6 @@
"typescript/no-misused-promises": "error",
"typescript/no-non-null-assertion": "error",
"typescript/only-throw-error": "error",
// React Compiler advisories, enforced everywhere with no per-file exemptions.
// The violations inherited from the oxlint 1.70 -> 1.80 bump are all resolved:
// three by widening a dependency to the object the compiler infers, and seven
// by an `oxlint-disable-next-line` that says, at the callsite, why the flagged
// dependency is load-bearing - five are re-run triggers that are never read,
// and two are values the callback genuinely uses.
"react/preserve-manual-memoization": "error",
"react/exhaustive-effect-dependencies": "error",
"react/memo-dependencies": "error",
"react/no-danger": "error",
"react/no-clone-element": "error",
"react/no-react-children": "error",

Some files were not shown because too many files have changed in this diff Show More