diff --git a/.github/workflows/docker-publish.yml b/.github/workflows/docker-publish.yml index 6180075..7ef231d 100644 --- a/.github/workflows/docker-publish.yml +++ b/.github/workflows/docker-publish.yml @@ -9,7 +9,7 @@ on: schedule: - cron: "25 0 * * *" push: - branches: ["*"] + branches: ["main"] # Publish semver tags as releases. tags: ["v*.*.*"] paths: @@ -65,10 +65,12 @@ jobs: with: context: . platforms: linux/amd64 - cache-from: type=gha,scope=x64 + cache-from: type=gha,scope=amd64 pull: true - cache-to: type=gha,mode=max,scope=x64 + cache-to: type=gha,mode=max,scope=amd64 target: test + build-args: | + GITHUB_BUILD=true build: needs: test @@ -135,8 +137,8 @@ jobs: tags: ${{ steps.meta.outputs.tags }} labels: ${{ steps.meta.outputs.labels }} platforms: ${{ matrix.platform }} - cache-from: type=gha,scope=${{ matrix.platform }} - cache-to: type=gha,mode=max,scope=${{ matrix.platform }} + cache-from: type=gha,scope=${{ steps.vars.outputs.SURFIX }} + cache-to: type=gha,mode=max,scope=${{ steps.vars.outputs.SURFIX }} build-args: | GITHUB_BUILD=true VERSION=${{ github.ref_type == 'tag' && github.ref_name || github.sha }} diff --git a/Dockerfile b/Dockerfile index 9b02061..920c52e 100644 --- a/Dockerfile +++ b/Dockerfile @@ -3,19 +3,16 @@ # cannot install firefox deps for (no libgtk-3 -> camoufox fails to launch). FROM ubuntu:24.04 AS base -ARG GITHUB_BUILD=false \ - VERSION +ARG GITHUB_BUILD=false ENV GITHUB_BUILD=${GITHUB_BUILD}\ - VERSION=${VERSION}\ - DEBIAN_FRONTEND=noninteractive \ PYTHONUNBUFFERED=1 \ # prevents python creating .pyc files PYTHONDONTWRITEBYTECODE=1 \ UV_LINK_MODE=copy \ PORT=8191 \ XDG_CACHE_HOME=/cache \ - HOME=/tmp + HOME=/home/byparr RUN apt-get update &&\ apt-get install -y --no-install-recommends curl ca-certificates git tini &&\ @@ -47,9 +44,9 @@ RUN mkdir -p /cache &&\ COPY . . -# Make app and cache world-readable; cache must be writable for runtime browser/profile data -RUN chmod -R o+rX /app /cache &&\ - chmod -R o+w /cache +RUN mkdir -p /home/byparr &&\ + chmod -R o+rX /app &&\ + chmod -R a+rwX /cache /home/byparr FROM app AS test RUN \ @@ -57,6 +54,8 @@ RUN \ uv run pytest --retries 3 FROM app +ARG VERSION +ENV VERSION=${VERSION} USER 1000 EXPOSE $PORT HEALTHCHECK --interval=15m --timeout=30s --start-period=5s --retries=3 CMD curl "http://127.0.0.1:${PORT}/health" diff --git a/README.md b/README.md index 74898c4..16e4884 100644 --- a/README.md +++ b/README.md @@ -17,6 +17,13 @@ | `PROXY_USERNAME` | None | Username for proxy authentication. | | `PROXY_PASSWORD` | None | Password for proxy authentication. | | `OWUI_API_KEY` | None | Bearer token for `/load` endpoint authentication. Must match `EXTERNAL_WEB_LOADER_API_KEY` in Open WebUI. | +| `BROWSER_LOCALE` | None | Override the browser's language with a [BCP-47](https://www.rfc-editor.org/rfc/bcp/bcp47.txt) tag, e.g. `en-US`, `de-DE`, `fr-FR`. When unset, the locale is derived from the egress country. | + +#### Browser language + +Set `BROWSER_LOCALE` to a [BCP-47](https://www.rfc-editor.org/rfc/bcp/bcp47.txt) language tag like `en-US`, `de-DE`, `fr-FR`, `pl-PL`, or `zh-CN` to fix the browser's language and `Accept-Language` header. When unset, Byparr derives the locale from the egress country (e.g. a French proxy → `fr-FR`), keeping the browser language consistent with the exit IP. + +Valid tags are maintained in the [IANA Language Subtag Registry](https://www.iana.org/assignments/language-subtag-registry/language-subtag-registry). For a friendlier list, see [List of ISO 639-1 codes](https://en.wikipedia.org/wiki/List_of_ISO_639-1_codes) (language) combined with an [ISO 3166-1 alpha-2](https://en.wikipedia.org/wiki/ISO_3166-1_alpha-2) region code for the full tag, e.g. `pt-BR`. ## Proxy Recommendation diff --git a/src/consts.py b/src/consts.py index 912cf9f..eb30c7c 100644 --- a/src/consts.py +++ b/src/consts.py @@ -1,7 +1,6 @@ import logging import sys -from playwright_captcha import CaptchaType from pydantic_settings import BaseSettings, SettingsConfigDict @@ -23,6 +22,7 @@ class Settings(BaseSettings): block_media: bool = False return_only_cookies: bool = False owui_api_key: str | None = None + browser_locale: str | None = None settings = Settings() @@ -43,12 +43,4 @@ BLOCK_MEDIA = settings.block_media RETURN_ONLY_COOKIES = settings.return_only_cookies OWUI_API_KEY = settings.owui_api_key - -CHALLENGE_TITLES_MAP: dict[CaptchaType, list[str]] = { - # Cloudflare - CaptchaType.CLOUDFLARE_INTERSTITIAL: ["Just a moment..."], -} - -CHALLENGE_TITLES = [ - title for titles in CHALLENGE_TITLES_MAP.values() for title in titles -] +BROWSER_LOCALE = settings.browser_locale diff --git a/src/endpoints.py b/src/endpoints.py index c29c6e3..dd9785a 100644 --- a/src/endpoints.py +++ b/src/endpoints.py @@ -9,8 +9,10 @@ from fastapi import APIRouter, Depends, HTTPException from fastapi.responses import RedirectResponse from playwright.async_api import TimeoutError as PlaywrightTimeoutError from playwright_captcha import CaptchaType +from playwright_captcha.solvers.click.cloudflare.utils.detection import ( + detect_cloudflare_challenge, +) -from src.consts import CHALLENGE_TITLES from src.models import ( HealthcheckResponse, LinkRequest, @@ -73,7 +75,9 @@ async def read_item(request: LinkRequest, dep: BrowserDep) -> LinkResponse: final_url = await setup_routes(request, dep) try: - page_request, status = await load_page_and_solve(dep, request, timer) + challenge_detected, page_html, page_request, status = ( + await _navigate_and_solve(dep, request, timer) + ) except (TimeoutError, PlaywrightTimeoutError) as e: logger.error("Timed out while loading the page or solving the challenge") raise HTTPException( @@ -82,9 +86,10 @@ async def read_item(request: LinkRequest, dep: BrowserDep) -> LinkResponse: ) from e cookies = await dep.context.cookies() - content_type, response_content = await build_response_content( - dep, request, page_request + dep, request, page_request, + challenge_detected=challenge_detected, + page_html=page_html, ) return LinkResponse( @@ -106,8 +111,8 @@ async def setup_routes(request: LinkRequest, dep: BrowserDep) -> str | None: """ Install request routes for media blocking and CSP stripping. - Returns a mutable holder for the final URL captured during navigation; - callers read it after the page settles. + Returns the final URL captured during navigation; callers read it after + the page settles. """ if request.block_media: @@ -148,26 +153,33 @@ async def setup_routes(request: LinkRequest, dep: BrowserDep) -> str | None: return final_url -async def load_page_and_solve( - dep: BrowserDep, request: LinkRequest, timer: TimeoutTimer -) -> tuple[object | None, HTTPStatus]: +async def _navigate_and_solve( + dep: BrowserDep, + request: LinkRequest, + timer: TimeoutTimer, +) -> tuple[bool, str | None, object, HTTPStatus]: """Navigate to the URL, then solve a challenge or wait for network idle.""" + page_html: str | None = None page_request = await dep.page.goto( request.url, timeout=timer.remaining() * 1000 ) - status = HTTPStatus.OK if page_request is None else HTTPStatus(page_request.status) + status = page_request.status if page_request else HTTPStatus.OK await dep.page.wait_for_load_state( state="domcontentloaded", timeout=timer.remaining() * 1000 ) - if await dep.page.title() in CHALLENGE_TITLES: - await _solve_challenge(dep, timer) - status = HTTPStatus.OK - logger.debug("Challenge solved successfully.") - else: + challenge_active = ( + await detect_cloudflare_challenge(dep.page, "interstitial") + or await detect_cloudflare_challenge(dep.page, "turnstile") + ) + if not challenge_active: + page_html = await dep.page.content() await _wait_for_networkidle(dep, timer) + return False, page_html, page_request, status - return page_request, status + await _solve_challenge(dep, timer) + status = HTTPStatus.OK + return True, page_html, page_request, status async def _solve_challenge(dep: BrowserDep, timer: TimeoutTimer) -> None: @@ -182,6 +194,7 @@ async def _solve_challenge(dep: BrowserDep, timer: TimeoutTimer) -> None: ), timeout=timer.remaining(), ) + logger.debug("Challenge solved successfully.") async def _wait_for_networkidle(dep: BrowserDep, timer: TimeoutTimer) -> None: @@ -192,12 +205,18 @@ async def _wait_for_networkidle(dep: BrowserDep, timer: TimeoutTimer) -> None: ) except PlaywrightTimeoutError: logger.info( - "networkidle timed out after domcontentloaded; continuing with loaded page" + "networkidle timed out after domcontentloaded; " + "continuing with loaded page" ) async def build_response_content( - dep: BrowserDep, request: LinkRequest, page_request: object | None + dep: BrowserDep, + request: LinkRequest, + page_request: object, + *, + challenge_detected: bool, + page_html: str | None, ) -> tuple[str, str]: """Build (content_type, response_content) from the settled page.""" if request.return_only_cookies: @@ -208,14 +227,21 @@ async def build_response_content( ): return await _fetch_pdf_content(dep) - return "text/html", await dep.page.content() + response_content = ( + page_html + if page_html is not None and not challenge_detected + else await dep.page.content() + ) + return "text/html", response_content async def _fetch_pdf_content(dep: BrowserDep) -> tuple[str, str]: """Fetch raw PDF bytes as base64, falling back to viewer HTML on failure.""" try: fetch_response = await dep.page.request.fetch(dep.page.url) - response_content = base64.b64encode(await fetch_response.body()).decode("ascii") + response_content = base64.b64encode( + await fetch_response.body() + ).decode("ascii") except Exception: logger.exception("Failed to fetch PDF bytes, falling back to viewer HTML") return "text/html", await dep.page.content() diff --git a/src/utils.py b/src/utils.py index 5a23d86..e34bca9 100644 --- a/src/utils.py +++ b/src/utils.py @@ -13,6 +13,7 @@ from playwright_captcha import ( from pydantic import BaseModel, Field from src.consts import ( + BROWSER_LOCALE, LOG_LEVEL, MAX_ATTEMPTS, PROXY_PASSWORD, @@ -94,7 +95,7 @@ async def get_browser( headless=True, proxy=proxy_config, humanize=True, - locale="auto", + locale=BROWSER_LOCALE or "auto", ) as browser_raw: # InvisiblePlaywright yields a Browser instance browser = cast("Browser", browser_raw)