diff --git a/app.py b/app.py index ac9b102..b6edea4 100644 --- a/app.py +++ b/app.py @@ -167,6 +167,7 @@ def api_search() -> Union[Response, Tuple[Response, int]]: lang (str): Book Language sort (str): Order to sort results content (str): Content type of book + format (str): File format filter (pdf, epub, mobi, azw3, fb2, djvu, cbz, cbr) Returns: flask.Response: JSON array of matching books or error response. @@ -180,6 +181,7 @@ def api_search() -> Union[Response, Tuple[Response, int]]: lang = request.args.getlist('lang'), sort = request.args.get('sort'), content = request.args.getlist('content'), + format = request.args.getlist('format'), ) if not query and not any(vars(filters).values()): diff --git a/book_manager.py b/book_manager.py index b67201a..5209e82 100644 --- a/book_manager.py +++ b/book_manager.py @@ -9,76 +9,85 @@ from bs4 import BeautifulSoup, Tag, NavigableString, ResultSet import downloader from logger import setup_logger from config import SUPPORTED_FORMATS, BOOK_LANGUAGE, AA_BASE_URL -from env import AA_DONATOR_KEY, USE_CF_BYPASS +from env import AA_DONATOR_KEY, USE_CF_BYPASS from models import BookInfo, SearchFilters logger = setup_logger(__name__) + def search_books(query: str, filters: SearchFilters) -> List[BookInfo]: """Search for books matching the query. - + Args: query: Search term (ISBN, title, author, etc.) - + Returns: List[BookInfo]: List of matching books - + Raises: Exception: If no books found or parsing fails """ query_html = quote(query) if filters.isbn: - #ISBNs are included in query string - isbns = " || ".join([f"('isbn13:{isbn}' || 'isbn10:{isbn}')" for isbn in filters.isbn]) + # ISBNs are included in query string + isbns = " || ".join( + [f"('isbn13:{isbn}' || 'isbn10:{isbn}')" for isbn in filters.isbn] + ) query_html = quote(f"({isbns}) {query}") - + filters_query = "" - + for value in filters.lang or BOOK_LANGUAGE: if value != "all": filters_query += f"&lang={quote(value)}" - + if filters.sort: filters_query += f"&sort={quote(filters.sort)}" - + if filters.content: for value in filters.content: filters_query += f"&content={quote(value)}" + # Handle format filter + formats_to_use = filters.format if filters.format else SUPPORTED_FORMATS + index = 1 for filter_type, filter_values in vars(filters).items(): - if filter_type == 'author' or filter_type == 'title' and filter_values: + if filter_type == "author" or filter_type == "title" and filter_values: for value in filter_values: - filters_query += f"&termtype_{index}={filter_type}&termval_{index}={quote(value)}" + filters_query += ( + f"&termtype_{index}={filter_type}&termval_{index}={quote(value)}" + ) index += 1 url = ( f"{AA_BASE_URL}" f"/search?index=&page=1&display=table" f"&acc=aa_download&acc=external_download" - f"&ext={'&ext='.join(SUPPORTED_FORMATS)}&q={query_html}" - f"{filters_query}" + f"&ext={'&ext='.join(formats_to_use)}" + f"&q={query_html}" + f"{filters_query}" ) html = downloader.html_get_page(url) if not html: raise Exception("Failed to fetch search results") - + if "No files found." in html: logger.info(f"No books found for query: {query}") raise Exception("No books found. Please try another query.") - soup = BeautifulSoup(html, 'html.parser') - tbody: Tag | NavigableString | None = soup.find('table') - + soup = BeautifulSoup(html, "html.parser") + tbody: Tag | NavigableString | None = soup.find("table") + if not tbody: logger.warning(f"No results table found for query: {query}") raise Exception("No books found. Please try another query.") books = [] - if isinstance(tbody, Tag): - for line_tr in tbody.find_all('tr'): + if isinstance(tbody, Tag): + for line_tr in tbody.find_all("tr"): try: book = _parse_search_result_row(line_tr) if book: @@ -93,75 +102,73 @@ def search_books(query: str, filters: SearchFilters) -> List[BookInfo]: else len(SUPPORTED_FORMATS) ) ) - + return books + def _parse_search_result_row(row: Tag) -> Optional[BookInfo]: """Parse a single search result row into a BookInfo object.""" try: - cells = row.find_all('td') - preview_img = cells[0].find('img') - preview = preview_img['src'] if preview_img else None - + cells = row.find_all("td") + preview_img = cells[0].find("img") + preview = preview_img["src"] if preview_img else None + return BookInfo( - id=row.find_all('a')[0]['href'].split('/')[-1], + id=row.find_all("a")[0]["href"].split("/")[-1], preview=preview, - title=cells[1].find('span').next, - author=cells[2].find('span').next, - publisher=cells[3].find('span').next, - year=cells[4].find('span').next, - language=cells[7].find('span').next, - format=cells[9].find('span').next.lower(), - size=cells[10].find('span').next + title=cells[1].find("span").next, + author=cells[2].find("span").next, + publisher=cells[3].find("span").next, + year=cells[4].find("span").next, + language=cells[7].find("span").next, + format=cells[9].find("span").next.lower(), + size=cells[10].find("span").next, ) except Exception as e: logger.error_trace(f"Error parsing search result row: {e}") return None + def get_book_info(book_id: str) -> BookInfo: """Get detailed information for a specific book. - + Args: book_id: Book identifier (MD5 hash) - + Returns: BookInfo: Detailed book information """ url = f"{AA_BASE_URL}/md5/{book_id}" html = downloader.html_get_page(url) - + if not html: raise Exception(f"Failed to fetch book info for ID: {book_id}") - soup = BeautifulSoup(html, 'html.parser') + soup = BeautifulSoup(html, "html.parser") return _parse_book_info_page(soup, book_id) + def _parse_book_info_page(soup: BeautifulSoup, book_id: str) -> BookInfo: """Parse the book info page HTML into a BookInfo object.""" - data = soup.select_one('body > main > div:nth-of-type(1)') - + data = soup.select_one("body > main > div:nth-of-type(1)") + if not data: raise Exception(f"Failed to parse book info for ID: {book_id}") - + preview: str = "" - node = data.select_one( - 'div:nth-of-type(1) > img' - ) + node = data.select_one("div:nth-of-type(1) > img") if node: - preview_value = node.get('src', "") + preview_value = node.get("src", "") if isinstance(preview_value, list): preview = preview_value[0] else: - preview = preview_value + preview = preview_value # Find the start of book information - divs = data.find_all('div') - start_div_id = next( - (i for i, div in enumerate(divs) if "🔍" in div.text), - 3 - ) + divs = data.find_all("div") + start_div_id = next((i for i, div in enumerate(divs) if "🔍" in div.text), 3) format_div = divs[start_div_id - 1].text format_parts = format_div.split(".") @@ -171,40 +178,60 @@ def _parse_book_info_page(soup: BeautifulSoup, book_id: str) -> BookInfo: format = None size = next( - (token.strip() for token in format_div.split(",") - if token.strip() and token.strip()[0].isnumeric()), - None + ( + token.strip() + for token in format_div.split(",") + if token.strip() and token.strip()[0].isnumeric() + ), + None, ) - every_url = soup.find_all('a') + every_url = soup.find_all("a") slow_urls_no_waitlist = set() slow_urls_with_waitlist = set() external_urls_libgen = set() external_urls_z_lib = set() - for url in every_url: try: if url.parent.text.strip().lower().startswith("option #"): if url.text.strip().lower().startswith("slow partner server"): - if url.next is not None and url.next.next is not None and "waitlist" in url.next.next.strip().lower(): + if ( + url.next is not None + and url.next.next is not None + and "waitlist" in url.next.next.strip().lower() + ): internal_text = url.next.next.strip().lower() if "no waitlist" in internal_text: - slow_urls_no_waitlist.add(url['href']) + slow_urls_no_waitlist.add(url["href"]) else: - slow_urls_with_waitlist.add(url['href']) - elif url.next is not None and url.next.next is not None and "click “GET” at the top" in url.next.next.text.strip(): - external_urls_libgen.add(url['href']) + slow_urls_with_waitlist.add(url["href"]) + elif ( + url.next is not None + and url.next.next is not None + and "click “GET” at the top" in url.next.next.text.strip() + ): + external_urls_libgen.add(url["href"]) elif url.text.strip().lower().startswith("z-lib"): - if ".onion/" not in url['href']: - external_urls_z_lib.add(url['href']) + if ".onion/" not in url["href"]: + external_urls_z_lib.add(url["href"]) except: pass if USE_CF_BYPASS: - urls = list(slow_urls_no_waitlist) + list(external_urls_libgen) + list(slow_urls_with_waitlist) + list(external_urls_z_lib) + urls = ( + list(slow_urls_no_waitlist) + + list(external_urls_libgen) + + list(slow_urls_with_waitlist) + + list(external_urls_z_lib) + ) else: - urls = list(external_urls_libgen) + list(external_urls_z_lib) + list(slow_urls_no_waitlist) + list(slow_urls_with_waitlist) + urls = ( + list(external_urls_libgen) + + list(external_urls_z_lib) + + list(slow_urls_no_waitlist) + + list(slow_urls_with_waitlist) + ) for i in range(len(urls)): urls[i] = downloader.get_absolute_url(AA_BASE_URL, urls[i]) @@ -218,11 +245,11 @@ def _parse_book_info_page(soup: BeautifulSoup, book_id: str) -> BookInfo: author=divs[start_div_id + 2].next, format=format, size=size, - download_urls=urls + download_urls=urls, ) # Extract additional metadata - info = _extract_book_metadata(divs[start_div_id + 3:]) + info = _extract_book_metadata(divs[start_div_id + 3 :]) book_info.info = info # Set language and year from metadata if available @@ -233,12 +260,15 @@ def _parse_book_info_page(soup: BeautifulSoup, book_id: str) -> BookInfo: return book_info -def _extract_book_metadata(metadata_divs: Union[ResultSet[Tag], List[Tag]]) -> Dict[str, List[str]]: + +def _extract_book_metadata( + metadata_divs: Union[ResultSet[Tag], List[Tag]], +) -> Dict[str, List[str]]: """Extract metadata from book info divs.""" - info : Dict[str, List[str]] = {} - + info: Dict[str, List[str]] = {} + # Process the first set of metadata - sub_data = metadata_divs[0].find_all('div') + sub_data = metadata_divs[0].find_all("div") for i in range(0, len(sub_data) - 1, 2): key = sub_data[i].next value = sub_data[i + 1].next @@ -250,8 +280,8 @@ def _extract_book_metadata(metadata_divs: Union[ResultSet[Tag], List[Tag]]) -> D # Find elements where aria-label="code tabs" meta_spans: List[Tag] = [] for div in metadata_divs: - if div.find_all('div', {'aria-label': 'code tabs'}): - meta_spans = div.find_all('span') + if div.find_all("div", {"aria-label": "code tabs"}): + meta_spans = div.find_all("span") break for i in range(0, len(meta_spans) - 1, 2): key = meta_spans[i].next @@ -262,21 +292,28 @@ def _extract_book_metadata(metadata_divs: Union[ResultSet[Tag], List[Tag]]) -> D # Filter relevant metadata relevant_prefixes = [ - "ISBN-", "ALTERNATIVE", "ASIN", "Goodreads", "Language", "Year" + "ISBN-", + "ALTERNATIVE", + "ASIN", + "Goodreads", + "Language", + "Year", ] return { - k.strip(): v for k, v in info.items() + k.strip(): v + for k, v in info.items() if any(k.lower().startswith(prefix.lower()) for prefix in relevant_prefixes) and "filename" not in k.lower() } + def download_book(book_info: BookInfo, book_path: Path) -> bool: """Download a book from available sources. - + Args: book_id: Book identifier (MD5 hash) title: Book title for logging - + Returns: Optional[BytesIO]: Book content buffer if successful """ @@ -287,10 +324,11 @@ def download_book(book_info: BookInfo, book_path: Path) -> bool: # If AA_DONATOR_KEY is set, use the fast download URL. Else try other sources. if AA_DONATOR_KEY != "": - download_links.insert(0, - f"{AA_BASE_URL}/dyn/api/fast_download.json?md5={book_info.id}&key={AA_DONATOR_KEY}" + download_links.insert( + 0, + f"{AA_BASE_URL}/dyn/api/fast_download.json?md5={book_info.id}&key={AA_DONATOR_KEY}", ) - + for link in download_links: try: download_url = _get_download_url(link, book_info.title) @@ -305,45 +343,46 @@ def download_book(book_info: BookInfo, book_path: Path) -> bool: f.write(data.getbuffer()) logger.info(f"Writing `{book_info.title}` successfully") return True - + except Exception as e: logger.error_trace(f"Failed to download from {link}: {e}") continue - + return False + def _get_download_url(link: str, title: str) -> str: """Extract actual download URL from various source pages.""" url = "" - + if link.startswith(f"{AA_BASE_URL}/dyn/api/fast_download.json"): page = downloader.html_get_page(link) url = json.loads(page).get("download_url") else: html = downloader.html_get_page(link) - + if html == "": return "" - - soup = BeautifulSoup(html, 'html.parser') - + + soup = BeautifulSoup(html, "html.parser") + if link.startswith("https://z-lib."): - download_link = soup.find_all('a', href=True, class_="addDownloadedBook") + download_link = soup.find_all("a", href=True, class_="addDownloadedBook") if download_link: - url = download_link[0]['href'] + url = download_link[0]["href"] elif link.startswith(f"{AA_BASE_URL}/slow_download/"): - download_links = soup.find_all('a', href=True, string="📚 Download now") + download_links = soup.find_all("a", href=True, string="📚 Download now") if not download_links: - countdown = soup.find_all('span', class_="js-partner-countdown") + countdown = soup.find_all("span", class_="js-partner-countdown") if countdown: sleep_time = int(countdown[0].text) logger.info(f"Waiting {sleep_time}s for {title}") time.sleep(sleep_time) url = _get_download_url(link, title) else: - url = download_links[0]['href'] + url = download_links[0]["href"] else: - url = soup.find_all('a', string="GET")[0]['href'] + url = soup.find_all("a", string="GET")[0]["href"] return downloader.get_absolute_url(link, url) diff --git a/models.py b/models.py index a30b562..50fb77f 100644 --- a/models.py +++ b/models.py @@ -127,4 +127,5 @@ class SearchFilters: title: Optional[List[str]] = None lang: Optional[List[str]] = None sort: Optional[str] = None - content: Optional[List[str]] = None \ No newline at end of file + content: Optional[List[str]] = None + format: Optional[List[str]] = None \ No newline at end of file diff --git a/static/js/main.js b/static/js/main.js index cf38ad8..8e0aadb 100644 --- a/static/js/main.js +++ b/static/js/main.js @@ -51,7 +51,7 @@ document.addEventListener('DOMContentLoaded', () => { download: '/request/api/download', status: '/request/api/status' }; - const FILTERS = ['isbn', 'author', 'title', 'lang' , 'sort', "content"]; + const FILTERS = ['isbn', 'author', 'title', 'lang' , 'sort', "content", "format"]; // Utility Functions const utils = { @@ -289,14 +289,22 @@ document.addEventListener('DOMContentLoaded', () => { } FILTERS.forEach(filterType => { - const inputs = document.querySelectorAll(`[id^="${filterType}-input"]`); - - inputs.forEach(input => { - const value = input.value.trim(); - if (value) { - queryParams.push(`${filterType}=${encodeURIComponent(value.trim())}`); - } - }); + if (filterType === 'format') { + // Handle format checkboxes + const checkboxes = document.querySelectorAll(`[id^="${filterType}-"]:checked`); + checkboxes.forEach(checkbox => { + queryParams.push(`${filterType}=${encodeURIComponent(checkbox.value)}`); + }); + } else { + const inputs = document.querySelectorAll(`[id^="${filterType}-input"]`); + + inputs.forEach(input => { + const value = input.value.trim(); + if (value) { + queryParams.push(`${filterType}=${encodeURIComponent(value.trim())}`); + } + }); + } }); return queryParams.join('&'); diff --git a/templates/index.html b/templates/index.html index 8ab9c0a..8dd7db4 100644 --- a/templates/index.html +++ b/templates/index.html @@ -117,6 +117,19 @@ +
+ +
+ + + + + + + + +
+