Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
28 changes: 28 additions & 0 deletions crawl4ai/antibot_detector.py
Original file line number Diff line number Diff line change
Expand Up @@ -188,6 +188,34 @@ def _structural_integrity_check(html: str) -> Tuple[bool, str]:
return False, ""


def effective_status(
status_code: Optional[int],
redirected_status_code: Optional[int] = None,
) -> Optional[int]:
"""
The status code that actually produced the HTML we are holding.

`AsyncCrawlResponse.status_code` / `CrawlResult.status_code` deliberately
carry the **first** hop of a redirect chain (a 3xx), while the body always
comes from the **last** hop; the final status is kept separately as
`redirected_status_code`. Block detection must be given the last hop —
otherwise a site that redirects apex -> www and then serves a 403 block
page is judged on its 301, no status rule fires, and the block page is
returned as successful content.

`redirected_status_code` is None on non-HTTP paths (raw:, file://,
js_only), so fall back to `status_code` there.

Args:
status_code: First hop of the redirect chain (or the only response).
redirected_status_code: Final hop, when the request was redirected.

Returns:
The status code to judge the response body by.
"""
return redirected_status_code or status_code


def is_blocked(
status_code: Optional[int],
html: str,
Expand Down
28 changes: 24 additions & 4 deletions crawl4ai/async_webcrawler.py
Original file line number Diff line number Diff line change
Expand Up @@ -52,7 +52,7 @@
compute_head_fingerprint,
)
from .cache_validator import CacheValidator, CacheValidationResult
from .antibot_detector import is_blocked
from .antibot_detector import is_blocked, effective_status


class AsyncWebCrawler:
Expand Down Expand Up @@ -509,8 +509,17 @@ async def arun(
_blocked = False
_block_reason = ""
else:
# Judge the body by the status that produced
# it — the LAST hop. status_code holds the
# first hop (a 3xx on any redirect), which no
# status rule in is_blocked() can ever match.
_blocked, _block_reason = is_blocked(
async_response.status_code, html)
effective_status(
async_response.status_code,
async_response.redirected_status_code,
),
html,
)

_crawl_stats["proxies_used"].append({
"proxy": _proxy.server if _proxy else None,
Expand Down Expand Up @@ -554,7 +563,13 @@ async def arun(
if _fallback_fn and not _done and not _is_raw_url:
_needs_fallback = (
crawl_result is None # All proxies threw exceptions
or is_blocked(crawl_result.status_code, crawl_result.html or "")[0]
or is_blocked(
effective_status(
crawl_result.status_code,
crawl_result.redirected_status_code,
),
crawl_result.html or "",
)[0]
)
if _needs_fallback:
self.logger.warning(
Expand Down Expand Up @@ -627,7 +642,12 @@ async def arun(
_has_download = bool(getattr(crawl_result, "downloaded_files", None))
if not _fallback_succeeded and not _is_raw_url and not _has_download:
_blocked, _block_reason = is_blocked(
crawl_result.status_code, crawl_result.html or "")
effective_status(
crawl_result.status_code,
crawl_result.redirected_status_code,
),
crawl_result.html or "",
)
if _blocked:
crawl_result.success = False
crawl_result.error_message = f"Blocked by anti-bot protection: {_block_reason}"
Expand Down
2 changes: 2 additions & 0 deletions docs/md_v2/advanced/anti-bot-and-fallback.md
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,8 @@ After each crawl attempt, Crawl4AI inspects the HTTP status code and HTML conten

Detection uses structural HTML markers (specific element IDs, script sources, form actions) rather than generic keywords to minimize false positives. A normal page that happens to mention "CAPTCHA" or "Cloudflare" in its content will not be flagged.

**Redirects:** the status code used for detection is the **final** hop of the redirect chain — `redirected_status_code`, falling back to `status_code` when there was no redirect. `CrawlResult.status_code` deliberately reports the *first* hop (see [CrawlResult](../api/crawl-result.md) §1.3/1.4), so a site that redirects `example.com → www.example.com` and then serves a `403` block page is correctly flagged even though `status_code` is `301`.

When all attempts fail and blocking is still detected, the result is returned with `success=False` and `error_message` describing the block reason.

## Configuration Options
Expand Down
153 changes: 153 additions & 0 deletions tests/proxy/test_antibot_redirect_status.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,153 @@
"""
Block detection must judge the FINAL hop of a redirect chain.

`CrawlResult.status_code` deliberately carries the first hop of a redirect
chain (issue #660, documented in docs/md_v2/api/crawl-result.md §1.3), while
the HTML always comes from the last hop, kept separately as
`redirected_status_code` (PR #1435, §1.4). `async_webcrawler` used to hand the
first hop to `antibot_detector.is_blocked()`, so a 301 was judged instead of
the 403 it led to — and no status rule in `is_blocked` can fire on a 3xx.

Effect: any site that redirects (apex -> www, http -> https) and then serves a
4xx/5xx block page was returned as `success: True` with the block page as its
content. Note `AsyncHTTPCrawlerStrategy` already sets `status_code` to the
post-redirect status, so the same URL produced different verdicts depending on
which crawler strategy ran.

No browser or network needed.

pytest tests/proxy/test_antibot_redirect_status.py -q
"""

import asyncio

from crawl4ai import AsyncWebCrawler, CrawlerRunConfig
from crawl4ai.antibot_detector import effective_status, is_blocked
from crawl4ai.models import AsyncCrawlResponse

BLOCK_PAGE = (
"<!DOCTYPE html><html><head><title>403 Forbidden</title></head><body>"
"<h1>Error 403 Forbidden</h1><p>Forbidden</p><h3>Error 54113</h3>"
"<p>Details: cache-ams-eham8680049-AMS</p><hr><p>Varnish cache server</p>"
"</body></html>"
)

REAL_PAGE = (
"<!DOCTYPE html><html><head><title>Example Co</title></head><body>"
"<h1>Example Co</h1><p>Contact: info@example.com</p>"
"<p>Phone +1 555 0100. Address: 1 Example Street, Springfield.</p>"
"<ul><li>Services</li><li>References</li><li>Contact</li></ul>"
"<p>Example Co is a company that does example things for example people.</p>"
"</body></html>"
)


class _FakeStrategy:
"""Returns a canned AsyncCrawlResponse; counts page loads."""

def __init__(self, response):
self.response = response
self.calls = 0

def update_user_agent(self, ua):
pass

async def crawl(self, url, config=None, **kwargs):
self.calls += 1
return self.response


def _run(coro):
return asyncio.get_event_loop_policy().new_event_loop().run_until_complete(coro)


def _response(html, status_code, redirected_status_code=None, redirected_url=None):
return AsyncCrawlResponse(
html=html,
response_headers={},
status_code=status_code,
redirected_status_code=redirected_status_code,
redirected_url=redirected_url,
)


async def _crawl(response, url="https://example.com", max_retries=1):
strategy = _FakeStrategy(response)
crawler = AsyncWebCrawler(crawler_strategy=strategy)
crawler.ready = True # skip start(): no browser is launched
result = await crawler.arun(url, config=CrawlerRunConfig(max_retries=max_retries))
return result, strategy


def test_effective_status_prefers_the_final_hop():
assert effective_status(301, 403) == 403
assert effective_status(302, 200) == 200


def test_effective_status_falls_back_without_a_redirect():
# raw:, file:// and js_only paths leave redirected_status_code as None.
assert effective_status(200, None) == 200
assert effective_status(None, None) is None
assert effective_status(403, 403) == 403


def test_block_page_behind_a_redirect_is_detected():
async def main():
result, _ = await _crawl(
_response(BLOCK_PAGE, 301, 403, "https://www.example.com/")
)
assert result.success is False
assert result.error_message.startswith("Blocked by anti-bot protection:")
assert "403" in result.error_message
# status_code semantics are unchanged: still the first hop.
assert result.status_code == 301
assert result.redirected_status_code == 403

_run(main())


def test_block_page_without_a_redirect_still_detected():
async def main():
result, _ = await _crawl(_response(BLOCK_PAGE, 403, 403))
assert result.success is False
assert "Blocked by anti-bot protection:" in result.error_message

_run(main())


def test_benign_redirect_is_not_a_false_positive():
async def main():
result, strategy = await _crawl(
_response(REAL_PAGE, 301, 200, "https://www.example.com/"),
url="http://example.com",
)
assert result.success is True
assert not result.error_message
assert strategy.calls == 1 # no retry burned on a good page

_run(main())


def test_no_redirect_is_a_no_op():
async def main():
result, strategy = await _crawl(_response(REAL_PAGE, 200, 200))
assert result.success is True
assert strategy.calls == 1

_run(main())


def test_raw_url_still_skips_block_detection():
async def main():
result, _ = await _crawl(
_response(BLOCK_PAGE, 200, None), url="raw:" + BLOCK_PAGE
)
assert result.success is True

_run(main())


def test_detector_itself_is_unchanged():
# Only the argument changed; is_blocked's own rules are untouched.
assert is_blocked(301, BLOCK_PAGE)[0] is False
assert is_blocked(403, BLOCK_PAGE)[0] is True