From e1c3f057abc1973c41901f41cd59fe859df9282d Mon Sep 17 00:00:00 2001 From: CaliBrain Date: Mon, 21 Sep 2026 00:15:41 -0400 Subject: [PATCH] fix: bypass recordings, welib wrong-md5 links, footer build sha (#1364) (#1373) Debug screen recordings never started. Every bypass logged "Capturearea 1540x1050 at position 0.0 outside the screen size 1440x1880".We ask ffmpeg for the fingerprint screen size plus margin, the size wealso pass SeleniumBase as xvfb_metrics. SeleniumBase builds thatdisplay with use_xauth=True, the image ships no xauth binary, so itfalls back to a fixed 1440x1880 Xvfb and the requested size neverexists. Drop -video_size so x11grab records the whole screen, whateversize it turned out to be. welib could hand back a link for a different book. Welib answers/md5/ with a search for that md5; when it does not have the file,the resolver took the first "Download" on the results page (md5a2c1dc0c... resolved to auto_download/9c8cf85d...). On an /md5/page a GET/Download link is now only taken when its href names thatmd5; otherwise the source is reported as not having the file. Alsoremoves _get_download_urls_from_welib and _is_source_enabled: themd5-template branch in _get_urls_for_source always handles welib first,so that resolver could never run. The footer showed the build date instead of the commit. CI stampsBUILD_VERSION as - (pr- for PR images) and thefooter kept its first seven characters, so dev images read "Shelfmarkmain (2026-09)". Take the trailing commit sha instead: "main (1a5b37d)".The full BUILD_VERSION stays in the hover title --- shelfmark/bypass/internal_bypasser.py | 11 +-- .../direct_download/annas_archive.py | 99 +++++-------------- src/frontend/src/components/Footer.tsx | 10 +- src/frontend/src/tests/buildVersion.test.ts | 30 ++++++ src/frontend/src/utils/buildVersion.ts | 15 +++ .../test_ffmpeg_recording_diagnostics.py | 25 +++++ tests/direct_download/test_welib_md5_match.py | 65 ++++++++++++ 7 files changed, 169 insertions(+), 86 deletions(-) create mode 100644 src/frontend/src/tests/buildVersion.test.ts create mode 100644 src/frontend/src/utils/buildVersion.ts create mode 100644 tests/direct_download/test_welib_md5_match.py diff --git a/shelfmark/bypass/internal_bypasser.py b/shelfmark/bypass/internal_bypasser.py index fe9c6024..f1eeb4df 100644 --- a/shelfmark/bypass/internal_bypasser.py +++ b/shelfmark/bypass/internal_bypasser.py @@ -1491,17 +1491,16 @@ def _start_ffmpeg_recording(display: str) -> None: timestamp = datetime.now(UTC).strftime("%y%m%d-%H%M%S") output_file = RECORDING_DIR / f"screen_recording_{timestamp}.mp4" - screen_width, screen_height = get_screen_size() - display_width = screen_width + 100 - display_height = screen_height + 150 - + # No -video_size: x11grab then captures the whole screen, whatever size it is. The + # size we ask SeleniumBase for (xvfb_metrics) is not the size we get. It builds that + # display with use_xauth=True, the image ships no xauth binary, so it falls back to a + # fixed 1440x1880 Xvfb. Asking ffmpeg for the fingerprint size plus margin then asked + # for an area larger than the screen, and every recording died at startup (#1364). ffmpeg_cmd = [ "ffmpeg", "-y", "-f", "x11grab", - "-video_size", - f"{display_width}x{display_height}", "-i", display, "-c:v", diff --git a/shelfmark/release_sources/direct_download/annas_archive.py b/shelfmark/release_sources/direct_download/annas_archive.py index 55eed505..d8eb31c2 100644 --- a/shelfmark/release_sources/direct_download/annas_archive.py +++ b/shelfmark/release_sources/direct_download/annas_archive.py @@ -131,19 +131,34 @@ def _find_first_anchor_with_text( text: str, *, contains: bool = False, + href_contains: str | None = None, ) -> Tag | None: - """Find the first anchor whose text matches the requested value.""" + """Find the first anchor whose text matches the requested value. + + With ``href_contains``, anchors whose href does not include it are skipped. + """ expected = text.lower() for anchor in container.find_all("a", href=True): anchor_text = anchor.get_text(strip=True) if not anchor_text: continue + if href_contains and href_contains.lower() not in (get_attr(anchor, "href") or "").lower(): + continue candidate = anchor_text.lower() if candidate == expected or (contains and expected in candidate): return anchor return None +_MD5_PAGE_PATH = re.compile(r"/md5/([0-9a-f]{32})(?:/|$)", re.IGNORECASE) + + +def _md5_from_page_url(url: str) -> str | None: + """Return the md5 an ``/md5/`` page URL is for, or None for any other URL.""" + match = _MD5_PAGE_PATH.search(urlparse(url).path) + return match.group(1).lower() if match else None + + def _find_text_node(container: BeautifulSoup | Tag, needle: str) -> NavigableString | None: """Find a text node containing a case-insensitive substring.""" expected = needle.lower() @@ -417,17 +432,6 @@ def _get_source_priority() -> list[SourcePriorityEntry]: return fast_sources + slow_sources -def _is_source_enabled(source_id: str) -> bool: - """Check if a source is enabled in the priority config. - - Returns False for unknown sources. - """ - for item in _get_source_priority(): - if item["id"] == source_id: - return item.get("enabled", True) - return False - - def get_unavailable_reason() -> str | None: """Return a user-facing reason when Direct Download cannot be used.""" from shelfmark.core import mirrors @@ -1314,17 +1318,6 @@ def _get_urls_for_source( urls.append(url) return urls - # Welib - fetch page and parse for slow_download links - if source_id == "welib": - if status_callback: - status_callback("resolving", "Fetching welib sources") - return _get_download_urls_from_welib( - book_info.id, - selector=selector, - cancel_flag=cancel_flag, - status_callback=status_callback, - ) - # AA page sources - fetch AA page if not already done if source_id in _AA_PAGE_SOURCES: if not urls_by_source: @@ -1404,53 +1397,6 @@ def _try_download_url( return download_url -def _get_download_urls_from_welib( - book_id: str, - selector: network.AAMirrorSelector | None = None, - cancel_flag: Event | None = None, - status_callback: Callable[[str, str | None], None] | None = None, -) -> list[str]: - """Get download URLs from welib.org (bypasser required).""" - from shelfmark.core import mirrors - - if not _is_source_enabled("welib"): - return [] - template = mirrors.get_welib_url_template() - if not template: - return [] - url = template.format(md5=book_id) - logger.info("Fetching welib download URLs for %s", book_id) - try: - html = downloader.html_get_page( - url, - use_bypasser=True, - selector=selector or network.AAMirrorSelector(), - cancel_flag=cancel_flag, - status_callback=status_callback, - ) - except ( - SearchUnavailableError, - requests.exceptions.RequestException, - RuntimeError, - ValueError, - TypeError, - AttributeError, - ) as exc: - logger.error_trace(f"Welib fetch failed for {book_id}: {exc}") - return [] - if not html: - logger.warning("Welib page empty for %s", book_id) - return [] - - soup = BeautifulSoup(html_response_text(html), "html.parser") - links = [ - downloader.get_absolute_url(url, href) - for a in soup.find_all("a", href=True) - if (href := get_attr(a, "href")) and "/slow_download/" in href - ] - return list(dict.fromkeys(links)) # Dedupe while preserving order - - def _extract_libgen_download_url(link: str, cancel_flag: Event | None = None) -> str: """Extract download URL from Libgen ads.php page using direct HTTP.""" if cancel_flag and cancel_flag.is_set(): @@ -1680,14 +1626,19 @@ def _get_download_url( ) else: - get_btn = _find_first_anchor_with_text(soup, "GET") or _find_first_anchor_with_text( - soup, "Download" - ) + # Welib answers /md5/ with a search for that md5. When it does not have the + # file, the first "Download" on that page belongs to whichever book ranked first, + # so a link is only taken when it names the md5 we asked for (#1364). + md5 = _md5_from_page_url(link) + get_btn = _find_first_anchor_with_text( + soup, "GET", href_contains=md5 + ) or _find_first_anchor_with_text(soup, "Download", href_contains=md5) if get_btn: url = get_attr(get_btn, "href") or "" + elif md5: + logger.info("No download link for md5 %s on %s", md5, link) else: logger.warning("Unknown source type, couldn't find download link: %s", link) - url = "" return downloader.get_absolute_url(link, url) diff --git a/src/frontend/src/components/Footer.tsx b/src/frontend/src/components/Footer.tsx index 8eddb935..b04d5244 100644 --- a/src/frontend/src/components/Footer.tsx +++ b/src/frontend/src/components/Footer.tsx @@ -1,3 +1,5 @@ +import { shortBuildId } from '../utils/buildVersion'; + interface FooterProps { buildVersion?: string; releaseVersion?: string; @@ -8,11 +10,7 @@ export const Footer = ({ buildVersion, releaseVersion, debug }: FooterProps) => // Determine version display - show "dev" if no version is set const versionDisplay = releaseVersion && releaseVersion !== 'N/A' ? releaseVersion : 'dev'; - // Truncate long build versions (e.g., git hashes) to 7 chars - let truncatedBuild: string | null = null; - if (buildVersion && buildVersion !== 'N/A') { - truncatedBuild = buildVersion.length > 7 ? buildVersion.slice(0, 7) : buildVersion; - } + const buildId = shortBuildId(buildVersion); return (