Merge branch 'main' into chore/ruff-lint-cleanup

Resolved conflicts in src/consts.py and src/endpoints.py:
- consts.py: take theirs (CHALLENGE_TITLES removed, browser_locale added,
  CaptchaType import no longer needed — detection is now library-based)
- endpoints.py: merge both refactors — keep theirs' detect_cloudflare_challenge
  + page_html capture, reapply my helper extraction (setup_routes,
  _navigate_and_solve, _solve_challenge, _wait_for_networkidle,
  build_response_content, _fetch_pdf_content) on top
This commit is contained in:
ThePhaseless
2026-08-11 00:20:55 +02:00
6 changed files with 71 additions and 44 deletions
+7 -5
View File
@@ -9,7 +9,7 @@ on:
schedule:
- cron: "25 0 * * *"
push:
branches: ["*"]
branches: ["main"]
# Publish semver tags as releases.
tags: ["v*.*.*"]
paths:
@@ -65,10 +65,12 @@ jobs:
with:
context: .
platforms: linux/amd64
cache-from: type=gha,scope=x64
cache-from: type=gha,scope=amd64
pull: true
cache-to: type=gha,mode=max,scope=x64
cache-to: type=gha,mode=max,scope=amd64
target: test
build-args: |
GITHUB_BUILD=true
build:
needs: test
@@ -135,8 +137,8 @@ jobs:
tags: ${{ steps.meta.outputs.tags }}
labels: ${{ steps.meta.outputs.labels }}
platforms: ${{ matrix.platform }}
cache-from: type=gha,scope=${{ matrix.platform }}
cache-to: type=gha,mode=max,scope=${{ matrix.platform }}
cache-from: type=gha,scope=${{ steps.vars.outputs.SURFIX }}
cache-to: type=gha,mode=max,scope=${{ steps.vars.outputs.SURFIX }}
build-args: |
GITHUB_BUILD=true
VERSION=${{ github.ref_type == 'tag' && github.ref_name || github.sha }}
+7 -8
View File
@@ -3,19 +3,16 @@
# cannot install firefox deps for (no libgtk-3 -> camoufox fails to launch).
FROM ubuntu:24.04 AS base
ARG GITHUB_BUILD=false \
VERSION
ARG GITHUB_BUILD=false
ENV GITHUB_BUILD=${GITHUB_BUILD}\
VERSION=${VERSION}\
DEBIAN_FRONTEND=noninteractive \
PYTHONUNBUFFERED=1 \
# prevents python creating .pyc files
PYTHONDONTWRITEBYTECODE=1 \
UV_LINK_MODE=copy \
PORT=8191 \
XDG_CACHE_HOME=/cache \
HOME=/tmp
HOME=/home/byparr
RUN apt-get update &&\
apt-get install -y --no-install-recommends curl ca-certificates git tini &&\
@@ -47,9 +44,9 @@ RUN mkdir -p /cache &&\
COPY . .
# Make app and cache world-readable; cache must be writable for runtime browser/profile data
RUN chmod -R o+rX /app /cache &&\
chmod -R o+w /cache
RUN mkdir -p /home/byparr &&\
chmod -R o+rX /app &&\
chmod -R a+rwX /cache /home/byparr
FROM app AS test
RUN \
@@ -57,6 +54,8 @@ RUN \
uv run pytest --retries 3
FROM app
ARG VERSION
ENV VERSION=${VERSION}
USER 1000
EXPOSE $PORT
HEALTHCHECK --interval=15m --timeout=30s --start-period=5s --retries=3 CMD curl "http://127.0.0.1:${PORT}/health"
+7
View File
@@ -17,6 +17,13 @@
| `PROXY_USERNAME` | None | Username for proxy authentication. |
| `PROXY_PASSWORD` | None | Password for proxy authentication. |
| `OWUI_API_KEY` | None | Bearer token for `/load` endpoint authentication. Must match `EXTERNAL_WEB_LOADER_API_KEY` in Open WebUI. |
| `BROWSER_LOCALE` | None | Override the browser's language with a [BCP-47](https://www.rfc-editor.org/rfc/bcp/bcp47.txt) tag, e.g. `en-US`, `de-DE`, `fr-FR`. When unset, the locale is derived from the egress country. |
#### Browser language
Set `BROWSER_LOCALE` to a [BCP-47](https://www.rfc-editor.org/rfc/bcp/bcp47.txt) language tag like `en-US`, `de-DE`, `fr-FR`, `pl-PL`, or `zh-CN` to fix the browser's language and `Accept-Language` header. When unset, Byparr derives the locale from the egress country (e.g. a French proxy → `fr-FR`), keeping the browser language consistent with the exit IP.
Valid tags are maintained in the [IANA Language Subtag Registry](https://www.iana.org/assignments/language-subtag-registry/language-subtag-registry). For a friendlier list, see [List of ISO 639-1 codes](https://en.wikipedia.org/wiki/List_of_ISO_639-1_codes) (language) combined with an [ISO 3166-1 alpha-2](https://en.wikipedia.org/wiki/ISO_3166-1_alpha-2) region code for the full tag, e.g. `pt-BR`.
## Proxy Recommendation
+2 -10
View File
@@ -1,7 +1,6 @@
import logging
import sys
from playwright_captcha import CaptchaType
from pydantic_settings import BaseSettings, SettingsConfigDict
@@ -23,6 +22,7 @@ class Settings(BaseSettings):
block_media: bool = False
return_only_cookies: bool = False
owui_api_key: str | None = None
browser_locale: str | None = None
settings = Settings()
@@ -43,12 +43,4 @@ BLOCK_MEDIA = settings.block_media
RETURN_ONLY_COOKIES = settings.return_only_cookies
OWUI_API_KEY = settings.owui_api_key
CHALLENGE_TITLES_MAP: dict[CaptchaType, list[str]] = {
# Cloudflare
CaptchaType.CLOUDFLARE_INTERSTITIAL: ["Just a moment..."],
}
CHALLENGE_TITLES = [
title for titles in CHALLENGE_TITLES_MAP.values() for title in titles
]
BROWSER_LOCALE = settings.browser_locale
+46 -20
View File
@@ -9,8 +9,10 @@ from fastapi import APIRouter, Depends, HTTPException
from fastapi.responses import RedirectResponse
from playwright.async_api import TimeoutError as PlaywrightTimeoutError
from playwright_captcha import CaptchaType
from playwright_captcha.solvers.click.cloudflare.utils.detection import (
detect_cloudflare_challenge,
)
from src.consts import CHALLENGE_TITLES
from src.models import (
HealthcheckResponse,
LinkRequest,
@@ -73,7 +75,9 @@ async def read_item(request: LinkRequest, dep: BrowserDep) -> LinkResponse:
final_url = await setup_routes(request, dep)
try:
page_request, status = await load_page_and_solve(dep, request, timer)
challenge_detected, page_html, page_request, status = (
await _navigate_and_solve(dep, request, timer)
)
except (TimeoutError, PlaywrightTimeoutError) as e:
logger.error("Timed out while loading the page or solving the challenge")
raise HTTPException(
@@ -82,9 +86,10 @@ async def read_item(request: LinkRequest, dep: BrowserDep) -> LinkResponse:
) from e
cookies = await dep.context.cookies()
content_type, response_content = await build_response_content(
dep, request, page_request
dep, request, page_request,
challenge_detected=challenge_detected,
page_html=page_html,
)
return LinkResponse(
@@ -106,8 +111,8 @@ async def setup_routes(request: LinkRequest, dep: BrowserDep) -> str | None:
"""
Install request routes for media blocking and CSP stripping.
Returns a mutable holder for the final URL captured during navigation;
callers read it after the page settles.
Returns the final URL captured during navigation; callers read it after
the page settles.
"""
if request.block_media:
@@ -148,26 +153,33 @@ async def setup_routes(request: LinkRequest, dep: BrowserDep) -> str | None:
return final_url
async def load_page_and_solve(
dep: BrowserDep, request: LinkRequest, timer: TimeoutTimer
) -> tuple[object | None, HTTPStatus]:
async def _navigate_and_solve(
dep: BrowserDep,
request: LinkRequest,
timer: TimeoutTimer,
) -> tuple[bool, str | None, object, HTTPStatus]:
"""Navigate to the URL, then solve a challenge or wait for network idle."""
page_html: str | None = None
page_request = await dep.page.goto(
request.url, timeout=timer.remaining() * 1000
)
status = HTTPStatus.OK if page_request is None else HTTPStatus(page_request.status)
status = page_request.status if page_request else HTTPStatus.OK
await dep.page.wait_for_load_state(
state="domcontentloaded", timeout=timer.remaining() * 1000
)
if await dep.page.title() in CHALLENGE_TITLES:
await _solve_challenge(dep, timer)
status = HTTPStatus.OK
logger.debug("Challenge solved successfully.")
else:
challenge_active = (
await detect_cloudflare_challenge(dep.page, "interstitial")
or await detect_cloudflare_challenge(dep.page, "turnstile")
)
if not challenge_active:
page_html = await dep.page.content()
await _wait_for_networkidle(dep, timer)
return False, page_html, page_request, status
return page_request, status
await _solve_challenge(dep, timer)
status = HTTPStatus.OK
return True, page_html, page_request, status
async def _solve_challenge(dep: BrowserDep, timer: TimeoutTimer) -> None:
@@ -182,6 +194,7 @@ async def _solve_challenge(dep: BrowserDep, timer: TimeoutTimer) -> None:
),
timeout=timer.remaining(),
)
logger.debug("Challenge solved successfully.")
async def _wait_for_networkidle(dep: BrowserDep, timer: TimeoutTimer) -> None:
@@ -192,12 +205,18 @@ async def _wait_for_networkidle(dep: BrowserDep, timer: TimeoutTimer) -> None:
)
except PlaywrightTimeoutError:
logger.info(
"networkidle timed out after domcontentloaded; continuing with loaded page"
"networkidle timed out after domcontentloaded; "
"continuing with loaded page"
)
async def build_response_content(
dep: BrowserDep, request: LinkRequest, page_request: object | None
dep: BrowserDep,
request: LinkRequest,
page_request: object,
*,
challenge_detected: bool,
page_html: str | None,
) -> tuple[str, str]:
"""Build (content_type, response_content) from the settled page."""
if request.return_only_cookies:
@@ -208,14 +227,21 @@ async def build_response_content(
):
return await _fetch_pdf_content(dep)
return "text/html", await dep.page.content()
response_content = (
page_html
if page_html is not None and not challenge_detected
else await dep.page.content()
)
return "text/html", response_content
async def _fetch_pdf_content(dep: BrowserDep) -> tuple[str, str]:
"""Fetch raw PDF bytes as base64, falling back to viewer HTML on failure."""
try:
fetch_response = await dep.page.request.fetch(dep.page.url)
response_content = base64.b64encode(await fetch_response.body()).decode("ascii")
response_content = base64.b64encode(
await fetch_response.body()
).decode("ascii")
except Exception:
logger.exception("Failed to fetch PDF bytes, falling back to viewer HTML")
return "text/html", await dep.page.content()
+2 -1
View File
@@ -13,6 +13,7 @@ from playwright_captcha import (
from pydantic import BaseModel, Field
from src.consts import (
BROWSER_LOCALE,
LOG_LEVEL,
MAX_ATTEMPTS,
PROXY_PASSWORD,
@@ -94,7 +95,7 @@ async def get_browser(
headless=True,
proxy=proxy_config,
humanize=True,
locale="auto",
locale=BROWSER_LOCALE or "auto",
) as browser_raw:
# InvisiblePlaywright yields a Browser instance
browser = cast("Browser", browser_raw)