From 719a5b9625f6cb177e85a5dc1ae7d84bbd24e854 Mon Sep 17 00:00:00 2001 From: vansh-nagar Date: Sat, 26 Sep 2026 13:32:33 +0530 Subject: [PATCH 1/2] fix: hand DiamWall 513 challenges to bypasser --- shelfmark/bypass/challenge.py | 11 ++- shelfmark/download/http.py | 23 +++++++ tests/download/test_http_challenge_513.py | 83 +++++++++++++++++++++++ 3 files changed, 116 insertions(+), 1 deletion(-) create mode 100644 tests/download/test_http_challenge_513.py diff --git a/shelfmark/bypass/challenge.py b/shelfmark/bypass/challenge.py index b24a0225e..ebca3e497 100644 --- a/shelfmark/bypass/challenge.py +++ b/shelfmark/bypass/challenge.py @@ -21,6 +21,10 @@ "could not verify your browser automatically", ] +OTHER_CHALLENGE_INDICATORS = [ + "diamwall", +] + # Markers that exist only in raw markup: the bypassers scan rendered innerText, where # a script src or a never appears. The title match is scoped to the tag on # purpose - hosts word the rest of that sentence differently, and matching "checking @@ -46,7 +50,12 @@ def challenge_marker(html: str) -> str | None: if not html or len(html) > MAX_CHALLENGE_HTML_CHARS: return None lowered = html.lower() - for marker in (*_RAW_HTML_MARKERS, *DDOS_GUARD_INDICATORS, *CLOUDFLARE_INDICATORS): + for marker in ( + *_RAW_HTML_MARKERS, + *DDOS_GUARD_INDICATORS, + *CLOUDFLARE_INDICATORS, + *OTHER_CHALLENGE_INDICATORS, + ): if marker in lowered: return marker return None diff --git a/shelfmark/download/http.py b/shelfmark/download/http.py index 079f16fad..3df87e231 100644 --- a/shelfmark/download/http.py +++ b/shelfmark/download/http.py @@ -42,6 +42,8 @@ _HTTP_STATUS_NOT_FOUND = HTTPStatus.NOT_FOUND _HTTP_STATUS_RATE_LIMITED = HTTPStatus.TOO_MANY_REQUESTS _HTTP_STATUS_SERVICE_UNAVAILABLE = HTTPStatus.SERVICE_UNAVAILABLE +# DiamWall uses this non-standard status for its browser-verification page. +_HTTP_STATUS_CHALLENGE = 513 _HTTP_STATUS_OK = HTTPStatus.OK _HTTP_STATUS_RANGE_NOT_SATISFIABLE = HTTPStatus.REQUESTED_RANGE_NOT_SATISFIABLE _HTTP_STATUS_PARTIAL_CONTENT = HTTPStatus.PARTIAL_CONTENT @@ -619,6 +621,27 @@ def _redirect_loop_handoff(bypass_url: str) -> str | tuple[str, str]: current_url, ) + if response.status_code == _HTTP_STATUS_CHALLENGE: + marker = _response_challenge_marker(response) + if marker and _bypass_handoff_allowed(): + if cookies: + logger.debug( + "513 challenge with cookies presented; purging: %s", current_url + ) + _purge_clearance(current_url) + logger.info( + "513 challenge detected (%s); switching to bypasser: %s", + marker, + current_url, + ) + return _run_bypasser(current_url) + if marker: + logger.debug( + "513 challenge (%s) but no bypasser handoff available: %s", + marker, + current_url, + ) + if is_aa_url and response.is_redirect: location = response.headers.get("Location", "") if not location: diff --git a/tests/download/test_http_challenge_513.py b/tests/download/test_http_challenge_513.py new file mode 100644 index 000000000..f2c448b09 --- /dev/null +++ b/tests/download/test_http_challenge_513.py @@ -0,0 +1,83 @@ +"""Tests for handing a DiamWall 513 challenge to the browser bypasser.""" + +import requests + + +class _FakeResponse: + def __init__(self, text: str) -> None: + self.status_code = 513 + self.url = "https://z-lib.gd/book/abc" + self.text = text + self.cookies: dict[str, str] = {} + self.headers = {"Content-Type": "text/html; charset=utf-8"} + self.is_redirect = False + + def raise_for_status(self) -> None: + error = requests.exceptions.HTTPError("513 Error") + error.response = self + raise error + + +def _neutralize_network(monkeypatch, http) -> None: + monkeypatch.setattr(http, "_apply_cf_bypass", lambda _url, _headers: {}) + monkeypatch.setattr(http, "get_proxies", lambda _url: {}) + monkeypatch.setattr(http, "get_ssl_verify", lambda _url: True) + monkeypatch.setattr(http.network, "should_rotate_dns_for_url", lambda _url: False) + monkeypatch.setattr(http.time, "sleep", lambda _seconds: None) + monkeypatch.setattr(http, "_bypass_grace_seconds", lambda: 1.0) + + +def test_diamwall_513_is_handed_to_the_bypasser(monkeypatch) -> None: + import shelfmark.download.http as http + + _neutralize_network(monkeypatch, http) + monkeypatch.setattr(http, "_is_cf_bypass_enabled", lambda: True) + + attempts: list[str] = [] + bypassed: list[str] = [] + monkeypatch.setattr( + http.requests, + "get", + lambda url, **_kwargs: ( + attempts.append(url) + or _FakeResponse( + "<html><title>DiamWallBrowser verification" + ) + ), + ) + monkeypatch.setattr( + http, + "get_bypassed_page", + lambda url, *_args, **_kwargs: bypassed.append(url) or "book page", + ) + + url = "https://z-lib.gd/book/abc" + html = http.html_get_page(url, retry=10, success_delay=0) + + assert html == "book page" + assert attempts == [url] + assert bypassed == [url] + + +def test_plain_513_does_not_start_a_browser(monkeypatch) -> None: + import shelfmark.download.http as http + + _neutralize_network(monkeypatch, http) + monkeypatch.setattr(http, "_is_cf_bypass_enabled", lambda: True) + + bypassed: list[str] = [] + monkeypatch.setattr( + http.requests, + "get", + lambda _url, **_kwargs: _FakeResponse("Unknown server error"), + ) + monkeypatch.setattr( + http, + "get_bypassed_page", + lambda url, *_args, **_kwargs: bypassed.append(url) or "", + ) + + html = http.html_get_page("https://z-lib.gd/book/abc", retry=1, success_delay=0) + + assert html == "" + assert bypassed == [] From 0449cd97ae8f2795915ef21281180b7353828b6f Mon Sep 17 00:00:00 2001 From: vansh-nagar Date: Sat, 26 Sep 2026 13:53:18 +0530 Subject: [PATCH 2/2] fix: teach browser bypasser about DiamWall --- shelfmark/bypass/internal_bypasser.py | 18 +++++++++++++++--- tests/bypass/test_internal_bypasser.py | 21 +++++++++++++++++++++ 2 files changed, 36 insertions(+), 3 deletions(-) diff --git a/shelfmark/bypass/internal_bypasser.py b/shelfmark/bypass/internal_bypasser.py index f1eeb4dfb..27154ce55 100644 --- a/shelfmark/bypass/internal_bypasser.py +++ b/shelfmark/bypass/internal_bypasser.py @@ -27,7 +27,11 @@ from seleniumbase.undetected.cdp_driver.connection import ProtocolException from shelfmark.bypass import BypassCancelledError -from shelfmark.bypass.challenge import CLOUDFLARE_INDICATORS, DDOS_GUARD_INDICATORS +from shelfmark.bypass.challenge import ( + CLOUDFLARE_INDICATORS, + DDOS_GUARD_INDICATORS, + OTHER_CHALLENGE_INDICATORS, +) from shelfmark.bypass.cookie_store import ( clear_cf_cookies, export_store, @@ -443,7 +447,7 @@ def _has_cloudflare_patterns(body: str, url: str) -> bool: async def _detect_challenge_type(page: Any) -> str: - """Detect challenge type: 'cloudflare', 'ddos_guard', or 'none'.""" + """Detect challenge type: 'cloudflare', 'ddos_guard', 'other', or 'none'.""" title, body, current_url = await _get_page_info(page) # DDOS-Guard indicators @@ -456,6 +460,10 @@ async def _detect_challenge_type(page: Any) -> str: logger.debug("Cloudflare indicator found: '%s'", found) return "cloudflare" + if found := _check_indicators(title, body, OTHER_CHALLENGE_INDICATORS): + logger.debug("Other challenge indicator found: '%s'", found) + return "other" + # Check URL patterns if _has_cloudflare_patterns(body, current_url): return "cloudflare" @@ -490,7 +498,11 @@ def _is_bypassed_content( return True # Check for protection indicators (means NOT bypassed) - if _check_indicators(title, body, CLOUDFLARE_INDICATORS + DDOS_GUARD_INDICATORS): + if _check_indicators( + title, + body, + CLOUDFLARE_INDICATORS + DDOS_GUARD_INDICATORS + OTHER_CHALLENGE_INDICATORS, + ): return False # Cloudflare URL patterns diff --git a/tests/bypass/test_internal_bypasser.py b/tests/bypass/test_internal_bypasser.py index 443479d69..d3c3cb2cb 100644 --- a/tests/bypass/test_internal_bypasser.py +++ b/tests/bypass/test_internal_bypasser.py @@ -6,6 +6,27 @@ import pytest +def test_diamwall_is_detected_until_the_challenge_clears(): + import shelfmark.bypass.internal_bypasser as internal_bypasser + + class DiamWallPage: + async def get_title(self): + return "DiamWall" + + async def evaluate(self, _expression): + return "Please wait while DiamWall checks your browser" + + async def get_current_url(self): + return "https://example.com" + + assert asyncio.run(internal_bypasser._detect_challenge_type(DiamWallPage())) == "other" + assert not internal_bypasser._is_bypassed_content( + "DiamWall", + "Please wait while DiamWall checks your browser", + "https://example.com", + ) + + def test_bypass_tries_all_methods_before_abort(monkeypatch): """Regression test for issue #524: don't abort before cycling through bypass methods.""" import shelfmark.bypass.internal_bypasser as internal_bypasser