Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 10 additions & 1 deletion shelfmark/bypass/challenge.py
Original file line number Diff line number Diff line change
Expand Up @@ -21,6 +21,10 @@
"could not verify your browser automatically",
]

OTHER_CHALLENGE_INDICATORS = [
"diamwall",
]
Comment thread
vansh-nagar marked this conversation as resolved.

# Markers that exist only in raw markup: the bypassers scan rendered innerText, where
# a script src or a <title> never appears. The title match is scoped to the tag on
# purpose - hosts word the rest of that sentence differently, and matching "checking
Expand All @@ -46,7 +50,12 @@ def challenge_marker(html: str) -> str | None:
if not html or len(html) > MAX_CHALLENGE_HTML_CHARS:
return None
lowered = html.lower()
for marker in (*_RAW_HTML_MARKERS, *DDOS_GUARD_INDICATORS, *CLOUDFLARE_INDICATORS):
for marker in (
*_RAW_HTML_MARKERS,
*DDOS_GUARD_INDICATORS,
*CLOUDFLARE_INDICATORS,
*OTHER_CHALLENGE_INDICATORS,
):
if marker in lowered:
return marker
return None
18 changes: 15 additions & 3 deletions shelfmark/bypass/internal_bypasser.py
Original file line number Diff line number Diff line change
Expand Up @@ -27,7 +27,11 @@
from seleniumbase.undetected.cdp_driver.connection import ProtocolException

from shelfmark.bypass import BypassCancelledError
from shelfmark.bypass.challenge import CLOUDFLARE_INDICATORS, DDOS_GUARD_INDICATORS
from shelfmark.bypass.challenge import (
CLOUDFLARE_INDICATORS,
DDOS_GUARD_INDICATORS,
OTHER_CHALLENGE_INDICATORS,
)
from shelfmark.bypass.cookie_store import (
clear_cf_cookies,
export_store,
Expand Down Expand Up @@ -443,7 +447,7 @@ def _has_cloudflare_patterns(body: str, url: str) -> bool:


async def _detect_challenge_type(page: Any) -> str:
"""Detect challenge type: 'cloudflare', 'ddos_guard', or 'none'."""
"""Detect challenge type: 'cloudflare', 'ddos_guard', 'other', or 'none'."""
title, body, current_url = await _get_page_info(page)

# DDOS-Guard indicators
Expand All @@ -456,6 +460,10 @@ async def _detect_challenge_type(page: Any) -> str:
logger.debug("Cloudflare indicator found: '%s'", found)
return "cloudflare"

if found := _check_indicators(title, body, OTHER_CHALLENGE_INDICATORS):
logger.debug("Other challenge indicator found: '%s'", found)
return "other"

# Check URL patterns
if _has_cloudflare_patterns(body, current_url):
return "cloudflare"
Expand Down Expand Up @@ -490,7 +498,11 @@ def _is_bypassed_content(
return True

# Check for protection indicators (means NOT bypassed)
if _check_indicators(title, body, CLOUDFLARE_INDICATORS + DDOS_GUARD_INDICATORS):
if _check_indicators(
title,
body,
CLOUDFLARE_INDICATORS + DDOS_GUARD_INDICATORS + OTHER_CHALLENGE_INDICATORS,
):
return False

# Cloudflare URL patterns
Expand Down
23 changes: 23 additions & 0 deletions shelfmark/download/http.py
Original file line number Diff line number Diff line change
Expand Up @@ -42,6 +42,8 @@
_HTTP_STATUS_NOT_FOUND = HTTPStatus.NOT_FOUND
_HTTP_STATUS_RATE_LIMITED = HTTPStatus.TOO_MANY_REQUESTS
_HTTP_STATUS_SERVICE_UNAVAILABLE = HTTPStatus.SERVICE_UNAVAILABLE
# DiamWall uses this non-standard status for its browser-verification page.
_HTTP_STATUS_CHALLENGE = 513
_HTTP_STATUS_OK = HTTPStatus.OK
_HTTP_STATUS_RANGE_NOT_SATISFIABLE = HTTPStatus.REQUESTED_RANGE_NOT_SATISFIABLE
_HTTP_STATUS_PARTIAL_CONTENT = HTTPStatus.PARTIAL_CONTENT
Expand Down Expand Up @@ -619,6 +621,27 @@ def _redirect_loop_handoff(bypass_url: str) -> str | tuple[str, str]:
current_url,
)

if response.status_code == _HTTP_STATUS_CHALLENGE:
marker = _response_challenge_marker(response)
if marker and _bypass_handoff_allowed():
if cookies:
logger.debug(
"513 challenge with cookies presented; purging: %s", current_url
)
_purge_clearance(current_url)
logger.info(
"513 challenge detected (%s); switching to bypasser: %s",
marker,
current_url,
)
return _run_bypasser(current_url)
if marker:
logger.debug(
"513 challenge (%s) but no bypasser handoff available: %s",
marker,
current_url,
)

if is_aa_url and response.is_redirect:
location = response.headers.get("Location", "")
if not location:
Expand Down
21 changes: 21 additions & 0 deletions tests/bypass/test_internal_bypasser.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,27 @@
import pytest


def test_diamwall_is_detected_until_the_challenge_clears():
import shelfmark.bypass.internal_bypasser as internal_bypasser

class DiamWallPage:
async def get_title(self):
return "DiamWall"

async def evaluate(self, _expression):
return "Please wait while DiamWall checks your browser"

async def get_current_url(self):
return "https://example.com"

assert asyncio.run(internal_bypasser._detect_challenge_type(DiamWallPage())) == "other"
assert not internal_bypasser._is_bypassed_content(
"DiamWall",
"Please wait while DiamWall checks your browser",
"https://example.com",
)


def test_bypass_tries_all_methods_before_abort(monkeypatch):
"""Regression test for issue #524: don't abort before cycling through bypass methods."""
import shelfmark.bypass.internal_bypasser as internal_bypasser
Expand Down
83 changes: 83 additions & 0 deletions tests/download/test_http_challenge_513.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,83 @@
"""Tests for handing a DiamWall 513 challenge to the browser bypasser."""

import requests


class _FakeResponse:
def __init__(self, text: str) -> None:
self.status_code = 513
self.url = "https://z-lib.gd/book/abc"
self.text = text
self.cookies: dict[str, str] = {}
self.headers = {"Content-Type": "text/html; charset=utf-8"}
self.is_redirect = False

def raise_for_status(self) -> None:
error = requests.exceptions.HTTPError("513 Error")
error.response = self
raise error


def _neutralize_network(monkeypatch, http) -> None:
monkeypatch.setattr(http, "_apply_cf_bypass", lambda _url, _headers: {})
monkeypatch.setattr(http, "get_proxies", lambda _url: {})
monkeypatch.setattr(http, "get_ssl_verify", lambda _url: True)
monkeypatch.setattr(http.network, "should_rotate_dns_for_url", lambda _url: False)
monkeypatch.setattr(http.time, "sleep", lambda _seconds: None)
monkeypatch.setattr(http, "_bypass_grace_seconds", lambda: 1.0)


def test_diamwall_513_is_handed_to_the_bypasser(monkeypatch) -> None:
import shelfmark.download.http as http

_neutralize_network(monkeypatch, http)
monkeypatch.setattr(http, "_is_cf_bypass_enabled", lambda: True)

attempts: list[str] = []
bypassed: list[str] = []
monkeypatch.setattr(
http.requests,
"get",
lambda url, **_kwargs: (
attempts.append(url)
or _FakeResponse(
"<html><title>DiamWall</title><body>Browser verification</body></html>"
)
),
)
monkeypatch.setattr(
http,
"get_bypassed_page",
lambda url, *_args, **_kwargs: bypassed.append(url) or "<html>book page</html>",
)

url = "https://z-lib.gd/book/abc"
html = http.html_get_page(url, retry=10, success_delay=0)

assert html == "<html>book page</html>"
assert attempts == [url]
assert bypassed == [url]


def test_plain_513_does_not_start_a_browser(monkeypatch) -> None:
import shelfmark.download.http as http

_neutralize_network(monkeypatch, http)
monkeypatch.setattr(http, "_is_cf_bypass_enabled", lambda: True)

bypassed: list[str] = []
monkeypatch.setattr(
http.requests,
"get",
lambda _url, **_kwargs: _FakeResponse("<html><body>Unknown server error</body></html>"),
)
monkeypatch.setattr(
http,
"get_bypassed_page",
lambda url, *_args, **_kwargs: bypassed.append(url) or "",
)

html = http.html_get_page("https://z-lib.gd/book/abc", retry=1, success_delay=0)

assert html == ""
assert bypassed == []