Skip to content
14 changes: 7 additions & 7 deletions src/crawlee/crawlers/_abstract_http/_abstract_http_crawler.py
Original file line number Diff line number Diff line change
Expand Up @@ -289,7 +289,7 @@ async def _make_http_request(self, context: BasicCrawlingContext) -> AsyncGenera
async def _handle_status_code_response(
self, context: HttpCrawlingContext
) -> AsyncGenerator[HttpCrawlingContext, None]:
"""Validate the HTTP status code and raise appropriate exceptions if needed.
"""Record rate limiting and validate the HTTP status code, raising appropriate exceptions if needed.

Args:
context: The current crawling context containing the HTTP response.
Expand All @@ -303,13 +303,13 @@ async def _handle_status_code_response(
The original crawling context if no errors are detected.
"""
status_code = context.http_response.status_code
self._record_rate_limit_status_code(
status_code,
request_url=context.request.url,
retry_after_header=context.http_response.headers.get('retry-after'),
)
if self._retry_on_blocked:
self._raise_for_session_blocked_status_code(
context.session,
status_code,
request_url=context.request.url,
retry_after_header=context.http_response.headers.get('retry-after'),
)
self._raise_for_session_blocked_status_code(context.session, status_code)
self._raise_for_error_status_code(status_code)
yield context

Expand Down
64 changes: 36 additions & 28 deletions src/crawlee/crawlers/_basic/_basic_crawler.py
Original file line number Diff line number Diff line change
Expand Up @@ -1621,51 +1621,59 @@ def _raise_for_error_status_code(self, status_code: int) -> None:
if is_status_code_server_error(status_code) and not is_ignored_status:
raise HttpStatusCodeError('Error status code returned', status_code)

def _raise_for_session_blocked_status_code(
def _record_rate_limit_status_code(
self,
session: Session | None,
status_code: int,
*,
request_url: str,
retry_after_header: str | None = None,
) -> None:
"""Raise an exception if the given status code indicates the session is blocked.
"""Record a 429 Too Many Requests response so the request's domain gets a backoff.

If the status code is 429 (Too Many Requests), the domain is recorded as rate-limited in the
`ThrottlingRequestManager` for per-domain backoff.
Rate limiting is independent of session blocking, so this runs for every response regardless of
`retry_on_blocked`.

Args:
session: The session used for the request. If `None`, no check is performed.
status_code: The HTTP status code to check.
request_url: The request URL, used for per-domain rate limit tracking.
retry_after_header: The value of the `Retry-After` response header, if present.

Raises:
SessionError: If the status code indicates the session is blocked.
"""
if status_code == HTTPStatus.TOO_MANY_REQUESTS:
if isinstance(self._request_manager, ThrottlingRequestManager):
retry_after = parse_retry_after_header(retry_after_header)
if not self._request_manager.record_domain_delay(request_url, retry_after=retry_after):
domain = (URL(request_url).host or '').lower().removesuffix('.')
if domain:
self._logger_once.log(
f'Received an HTTP 429 (Too Many Requests) response from domain "{domain}", but it is '
f'not in the `ThrottlingRequestManager.domains` list. Per-domain backoff will not be '
f'applied for this domain. Add it to `domains=` to enable throttling.',
key=f'unconfigured_throttle_domain:{domain}',
level=logging.WARNING,
)
else:
if status_code != HTTPStatus.TOO_MANY_REQUESTS:
return

if not isinstance(self._request_manager, ThrottlingRequestManager):
self._logger_once.log(
'Received an HTTP 429 (Too Many Requests) response, but the crawler is not using '
'`ThrottlingRequestManager`. Per-domain backoff and `Retry-After` headers will not be honored. '
'To enable per-domain rate limiting, configure the crawler to use `ThrottlingRequestManager` '
'as the request manager.',
key='no_throttling_manager_on_429',
level=logging.WARNING,
)
return

retry_after = parse_retry_after_header(retry_after_header)
if not self._request_manager.record_domain_delay(request_url, retry_after=retry_after):
domain = (URL(request_url).host or '').lower().removesuffix('.')
if domain:
self._logger_once.log(
'Received an HTTP 429 (Too Many Requests) response, but the crawler is not using '
'`ThrottlingRequestManager`. Per-domain backoff and `Retry-After` headers will not be honored. '
'To enable per-domain rate limiting, configure the crawler to use `ThrottlingRequestManager` '
'as the request manager.',
key='no_throttling_manager_on_429',
f'Received an HTTP 429 (Too Many Requests) response from domain "{domain}", but it is '
f'not in the `ThrottlingRequestManager.domains` list. Per-domain backoff will not be '
f'applied for this domain. Add it to `domains=` to enable throttling.',
key=f'unconfigured_throttle_domain:{domain}',
level=logging.WARNING,
)

def _raise_for_session_blocked_status_code(self, session: Session | None, status_code: int) -> None:
"""Raise an exception if the given status code indicates the session is blocked.

Args:
session: The session used for the request. If `None`, no check is performed.
status_code: The HTTP status code to check.

Raises:
SessionError: If the status code indicates the session is blocked.
"""
if session is not None and session.is_blocked_status_code(
status_code=status_code,
ignore_http_error_status_codes=self._ignore_http_error_status_codes,
Expand Down
14 changes: 7 additions & 7 deletions src/crawlee/crawlers/_playwright/_playwright_crawler.py
Original file line number Diff line number Diff line change
Expand Up @@ -494,7 +494,7 @@ async def extract_links(
return extract_links

async def _handle_status_code_response(self, context: TPostNavContext) -> AsyncGenerator[TPostNavContext, None]:
"""Validate the HTTP status code and raise appropriate exceptions if needed.
"""Record rate limiting and validate the HTTP status code, raising appropriate exceptions if needed.

Args:
context: The current crawling context containing the response.
Expand All @@ -508,13 +508,13 @@ async def _handle_status_code_response(self, context: TPostNavContext) -> AsyncG
The original crawling context if no errors are detected.
"""
status_code = context.response.status
self._record_rate_limit_status_code(
status_code,
request_url=context.request.url,
retry_after_header=context.response.headers.get('retry-after'),
)
if self._retry_on_blocked:
self._raise_for_session_blocked_status_code(
context.session,
status_code,
request_url=context.request.url,
retry_after_header=context.response.headers.get('retry-after'),
)
self._raise_for_session_blocked_status_code(context.session, status_code)
self._raise_for_error_status_code(status_code)
yield context

Expand Down
16 changes: 5 additions & 11 deletions tests/unit/crawlers/_basic/test_basic_crawler.py
Original file line number Diff line number Diff line change
Expand Up @@ -2517,8 +2517,8 @@ async def test_warn_no_throttling_manager_once_on_429(caplog: pytest.LogCaptureF
"""A 429 from a crawler without ThrottlingRequestManager logs a recommendation, only once per instance."""
crawler = BasicCrawler(configure_logging=False)
with caplog.at_level(logging.WARNING, logger='crawlee'):
crawler._raise_for_session_blocked_status_code(session=None, status_code=429, request_url='https://a.test/')
crawler._raise_for_session_blocked_status_code(session=None, status_code=429, request_url='https://b.test/')
crawler._record_rate_limit_status_code(429, request_url='https://a.test/')
crawler._record_rate_limit_status_code(429, request_url='https://b.test/')

matching = [
r for r in caplog.records if 'ThrottlingRequestManager' in r.getMessage() and 'HTTP 429' in r.getMessage()
Expand All @@ -2537,15 +2537,9 @@ async def test_warn_unconfigured_throttle_domain_once_per_domain(caplog: pytest.
crawler = BasicCrawler(configure_logging=False, request_manager=throttler)

with caplog.at_level(logging.WARNING, logger='crawlee'):
crawler._raise_for_session_blocked_status_code(
session=None, status_code=429, request_url='https://A.example.com/page1'
)
crawler._raise_for_session_blocked_status_code(
session=None, status_code=429, request_url='https://a.example.com/page2'
)
crawler._raise_for_session_blocked_status_code(
session=None, status_code=429, request_url='https://other.example.com/page1'
)
crawler._record_rate_limit_status_code(429, request_url='https://A.example.com/page1')
crawler._record_rate_limit_status_code(429, request_url='https://a.example.com/page2')
crawler._record_rate_limit_status_code(429, request_url='https://other.example.com/page1')

matching = [
r
Expand Down
39 changes: 39 additions & 0 deletions tests/unit/crawlers/_http/test_http_crawler.py
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
from __future__ import annotations

import json
from datetime import timedelta
from typing import TYPE_CHECKING
from unittest.mock import AsyncMock, Mock
from urllib.parse import parse_qs, urlencode
Expand All @@ -9,6 +10,7 @@

from crawlee import ConcurrencySettings, Request, RequestState
from crawlee.crawlers import HttpCrawler
from crawlee.request_loaders import ThrottlingRequestManager
from crawlee.sessions import SessionPool
from crawlee.statistics import Statistics
from crawlee.storages import RequestQueue
Expand Down Expand Up @@ -686,3 +688,40 @@ async def failed_request_handler(context: BasicCrawlingContext, _error: Exceptio
}

await queue.drop()


@pytest.mark.parametrize(
'retry_on_blocked',
[
pytest.param(True, id='retry_on_blocked'),
pytest.param(False, id='no_retry_on_blocked'),
],
)
async def test_records_429_regardless_of_retry_on_blocked(
mock_request_handler: AsyncMock,
server_url: URL,
*,
retry_on_blocked: bool,
) -> None:
"""Rate limiting is a separate concern from session blocking, so a 429 must be recorded either way."""
domain = server_url.host or ''
inner = await RequestQueue.open(alias='throttle-429-inner')
throttler = ThrottlingRequestManager(
inner,
domains=[domain],
request_manager_opener=RequestQueue.open,
# Long enough that the assertion below cannot race the backoff expiring.
base_delay=timedelta(seconds=30),
)
crawler = HttpCrawler(
request_handler=mock_request_handler,
request_manager=throttler,
retry_on_blocked=retry_on_blocked,
max_request_retries=0,
# Without this, a 429 retires the session and the rotation retries walk the backoff up to `max_delay`.
max_session_rotations=0,
)

await crawler.run([str(server_url / 'status/429')])

assert throttler._is_domain_throttled(domain)
Loading