CF BYPASS (#24)

The Calibre dependency was due to the script testing for validity of the
downloaded file, as often they would be corrupted from aa. But CWA is
already doing that, so we are just having redundant code here.

For the cloudflarebypasser, I basically run my own version now, instead
of depending on an external library, this way we have better control for
debugging and on the docker image.

Fixes #18, #33, #27, #48, #65, #78, #86, #88, #89

---------

Co-authored-by: mik593 <91991279+mik593@users.noreply.github.com>
This commit is contained in:
CaliBrain
2025-03-16 02:25:15 -04:00
committed by GitHub
co-authored by mik593
parent 17fdb87749
commit 3a92c5de78
19 changed files with 688 additions and 264 deletions
+31 -54
View File
@@ -8,24 +8,22 @@ from typing import Optional
from urllib.parse import urlparse
from tqdm import tqdm
import cloudflare_bypasser
from logger import setup_logger
from config import MAX_RETRY, DEFAULT_SLEEP, CLOUDFLARE_PROXY, USE_CF_BYPASS
from config import MAX_RETRY, DEFAULT_SLEEP, USE_CF_BYPASS, PROXIES
logger = setup_logger(__name__)
"""Configure urllib opener with appropriate headers."""
opener = urllib.request.build_opener()
opener.addheaders = [
('User-agent', 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) '
'AppleWebKit/537.36 (KHTML, like Gecko) '
'Chrome/129.0.0.0 Safari/537.3')
]
urllib.request.install_opener(opener)
def setup_urllib_opener():
"""Configure urllib opener with appropriate headers."""
opener = urllib.request.build_opener()
opener.addheaders = [
('User-agent', 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) '
'AppleWebKit/537.36 (KHTML, like Gecko) '
'Chrome/129.0.0.0 Safari/537.3')
]
urllib.request.install_opener(opener)
setup_urllib_opener()
def html_get_page(url: str, retry: int = MAX_RETRY, skip_404: bool = False, skip_403: bool = False) -> str:
def html_get_page(url: str, retry: int = MAX_RETRY, use_bypasser: bool = False) -> str:
"""Fetch HTML content from a URL with retry mechanism.
Args:
@@ -37,25 +35,37 @@ def html_get_page(url: str, retry: int = MAX_RETRY, skip_404: bool = False, skip
str: HTML content if successful, None otherwise
"""
try:
if use_bypasser and USE_CF_BYPASS:
logger.info(f"GET Using Cloudflare Bypasser for: {url}")
response = cloudflare_bypasser.get(url)
logger.debug(f"Cloudflare Bypasser response: {response}")
if response:
return str(response.html)
else:
raise requests.exceptions.RequestException("Failed to bypass Cloudflare")
logger.info(f"GET: {url}")
response = requests.get(url)
response = requests.get(url, proxies=PROXIES)
response.raise_for_status()
logger.debug(f"Success getting: {url}")
time.sleep(1)
return response.text
return str(response.text)
except requests.exceptions.RequestException as e:
if retry == 0:
logger.error(f"Failed to fetch page: {url}, error: {e}")
return ""
if skip_404 and response.status_code == 404:
if response.status_code == 404:
logger.warning(f"404 error for URL: {url}")
return ""
if skip_403 and response.status_code == 403:
if response.status_code == 403:
logger.warning(f"403 error for URL: {url}. Should retry using cloudflare bypass.")
return ""
if use_bypasser:
return ""
use_bypasser = True
sleep_time = DEFAULT_SLEEP * (MAX_RETRY - retry + 1)
@@ -63,40 +73,7 @@ def html_get_page(url: str, retry: int = MAX_RETRY, skip_404: bool = False, skip
f"Retrying GET {url} in {sleep_time} seconds due to error: {e}"
)
time.sleep(sleep_time)
return html_get_page(url, retry - 1)
def html_get_page_cf(url: str, retry: int = MAX_RETRY) -> str:
"""Fetch HTML content through Cloudflare proxy.
Args:
url: Target URL
retry: Number of retry attempts
Returns:
str: HTML content if successful, None otherwise
"""
if USE_CF_BYPASS == False:
logger.warning("Cloudflare bypass is disabled, trying without it.")
return html_get_page(url, retry, skip_403=True)
try:
logger.info(f"GET_CF: {url}")
response = requests.get(
f"{CLOUDFLARE_PROXY}/html?url={url}&retries=3"
)
time.sleep(1)
return response.text
except Exception as e:
if retry == 0:
logger.error(f"Failed to fetch page through CF: {url}, error: {e}")
return ""
sleep_time = DEFAULT_SLEEP * (MAX_RETRY - retry + 1)
logger.warning(
f"Retrying GET_CF {url} in {sleep_time} seconds due to error: {e}"
)
time.sleep(sleep_time)
return html_get_page_cf(url, retry - 1)
return html_get_page(url, retry - 1, use_bypasser)
def download_url(link: str, size: str = "") -> Optional[BytesIO]:
"""Download content from URL into a BytesIO buffer.
@@ -109,7 +86,7 @@ def download_url(link: str, size: str = "") -> Optional[BytesIO]:
"""
try:
logger.info(f"Downloading from: {link}")
response = requests.get(link, stream=True)
response = requests.get(link, stream=True, proxies=PROXIES)
response.raise_for_status()
total_size : float = 0.0