From 3ae5ab40c1bd4cb668de04097dfb1bdff235315e Mon Sep 17 00:00:00 2001 From: Adrian Chaves Date: Tue, 30 Jun 2026 13:32:57 +0200 Subject: [PATCH] WIP --- docs/topics/api.rst | 2 + docs/topics/throttling.rst | 92 +++++-------------- scrapy/crawler.py | 6 ++ scrapy/settings/default_settings.py | 24 ++++- .../templates/project/module/settings.py.tmpl | 2 +- scrapy/throttling.py | 2 +- 6 files changed, 58 insertions(+), 70 deletions(-) diff --git a/docs/topics/api.rst b/docs/topics/api.rst index 19082d9d7..edfc37959 100644 --- a/docs/topics/api.rst +++ b/docs/topics/api.rst @@ -93,6 +93,8 @@ how you :ref:`configure the downloader middlewares modify the downloader and scheduler behaviour, although this is an advanced use and this API is not yet stable. + .. autoattribute:: throttler + .. attribute:: spider Spider currently being crawled. This is an instance of the spider class diff --git a/docs/topics/throttling.rst b/docs/topics/throttling.rst index 13e9c0e50..d15557baa 100644 --- a/docs/topics/throttling.rst +++ b/docs/topics/throttling.rst @@ -74,14 +74,15 @@ behavior for specific domains [1]_. It is a dict that maps scope names to :class:`~scrapy.throttling.ThrottlingScopeConfig` dicts. -Its default value allows faster crawling of the testing website using during +Its default value allows faster crawling of the testing websites used during the :ref:`tutorial ` while maintaining conservative defaults for other domains: .. code-block:: python THROTTLING_SCOPES = { - "quotes.toscrape.com": {"concurrency": 16, "delay": 0.0}, + "books.toscrape.com": {"concurrency": 16, "delay": 0.1}, + "quotes.toscrape.com": {"concurrency": 16, "delay": 0.1}, } Additional keys like ``"jitter"`` and ``"backoff"`` can be used here and are @@ -132,8 +133,7 @@ How backoff works ----------------- Every :ref:`throttling scope ` keeps a current delay that -starts at its configured value (``0`` by default, or :setting:`DOWNLOAD_DELAY` -for the default per-domain scopes). +starts at its configured ``"delay"`` (:setting:`DOWNLOAD_DELAY` by default). A **backoff trigger** is a response whose status code is in :setting:`BACKOFF_HTTP_CODES` or a download exception whose type is in @@ -533,10 +533,6 @@ method, define a :class:`dict` where keys are throttling scopes and values are :class:`float` values that indicate the expected quota consumption (it does not need to be exact). -Everything else being equal, :class:`~scrapy.pqueues.ScrapyPriorityQueue` will -prioritize requests that consume a higher portion of the available throttling -quota, to minimize the risk of those requests getting stuck. - .. _custom-throttling-scope-managers: Customizing throttling scope managers @@ -575,64 +571,22 @@ setting: }, } -A scope manager only needs to implement the -:class:`~scrapy.throttling.ThrottlingScopeManagerProtocol`. For example, the -following manager enforces a fixed quota per time window without any delay or -exponential backoff, similar to a fixed-window rate limiter: +Most custom scope managers subclass the default +:class:`~scrapy.throttling.ThrottlingScopeManager` and override only the methods +whose behavior they want to change; implementing the +:class:`~scrapy.throttling.ThrottlingScopeManagerProtocol` from scratch is also +supported. For example, this manager disables exponential :ref:`backoff +`, so a scope relies solely on its configured delay and quota: .. code-block:: python :caption: ``myproject/throttling.py`` - import time + from scrapy.throttling import ThrottlingScopeManager - class FixedWindowScopeManager: - def __init__(self, crawler, config): - self.limit = config["quota"] - self.window = config.get("window", 60.0) - self.reset_at = None - self.used = 0.0 - self.active = 0 - - @classmethod - def from_crawler(cls, crawler, config): - return cls(crawler, config) - - def can_send(self, now=None, amount=None): - now = now if now is not None else time.monotonic() - if self.reset_at is not None and now >= self.reset_at: - self.reset_at, self.used = None, 0.0 - if self.used and self.used + (amount or 0) > self.limit: - return self.reset_at - now - return 0.0 - - def record_sent(self, now=None, amount=None): - now = now if now is not None else time.monotonic() - if self.reset_at is None: - self.reset_at = now + self.window - self.used += amount or 0 - self.active += 1 - - def record_done(self, now=None): - self.active = max(0, self.active - 1) - - def record_backoff(self, delay=None, now=None): - pass # this manager does not back off - - def reconcile_quota(self, consumed=None, remaining=None, now=None): - if remaining is not None: - self.used = max(0.0, self.limit - remaining) - elif consumed is not None: - self.used = max(0.0, self.used + consumed) - - def set_base_delay(self, delay): - pass - - def set_concurrency(self, concurrency): - pass - - def is_idle(self, now, max_idle): - return self.active == 0 + class FixedWindowScopeManager(ThrottlingScopeManager): + def record_backoff(self, *args, **kwargs): + pass # never back off .. _throttling-aware-scheduler: @@ -685,12 +639,13 @@ Alternative approaches include: :caption: ``settings.py`` import tldextract + from scrapy.throttling import ThrottlingManager, scope_cache from scrapy.utils.httpobj import urlparse_cached - class MyThrottlingManager: - - def get_request_scopes(self, request): + class MyThrottlingManager(ThrottlingManager): + @scope_cache + async def get_scopes(self, request): extracted = tldextract.extract(request.url) if extracted.domain and extracted.suffix: return f"{extracted.domain}.{extracted.suffix}" @@ -715,12 +670,13 @@ Alternative approaches include: :caption: ``settings.py`` import tldextract + from scrapy.throttling import ThrottlingManager, scope_cache from scrapy.utils.httpobj import urlparse_cached - class MyThrottlingManager: - - def get_request_scopes(self, request): + class MyThrottlingManager(ThrottlingManager): + @scope_cache + async def get_scopes(self, request): extracted = tldextract.extract(request.url) if not (extracted.domain and extracted.suffix): return urlparse_cached(request).netloc @@ -870,7 +826,7 @@ Cost-capped throttling Imagine you are using an API that charges different requests differently, e.g. based on the features used, and you want to limit how much you spend per time -window (:setting:`BACKOFF_WINDOW`). You can use :ref:`throttling quotas +window (:setting:`THROTTLING_WINDOW`). You can use :ref:`throttling quotas ` for that: - Implement a :ref:`throttling manager ` that: @@ -1150,6 +1106,8 @@ API .. autoclass:: scrapy.throttling.BackoffConfig +.. autoclass:: scrapy.throttling.RampupConfig + .. autofunction:: scrapy.throttling.scope_cache .. autofunction:: scrapy.throttling.add_scope .. autofunction:: scrapy.throttling.update_scope_backoff diff --git a/scrapy/crawler.py b/scrapy/crawler.py index 28bb2bf83..8d94f9cf4 100644 --- a/scrapy/crawler.py +++ b/scrapy/crawler.py @@ -85,6 +85,12 @@ class Crawler: self.logformatter: LogFormatter | None = None self.request_fingerprinter: RequestFingerprinterProtocol | None = None self.throttler: ThrottlingManagerProtocol | None = None + """The throttling manager of this crawler, an instance of + :setting:`THROTTLING_MANAGER`. + + It is ``None`` until the crawl starts. Components can use it to inspect + or drive :ref:`throttling ` at run time, e.g. through + :meth:`~scrapy.throttling.ThrottlingManagerProtocol.delay_scope`.""" self.spider: Spider | None = None self.engine: ExecutionEngine | None = None diff --git a/scrapy/settings/default_settings.py b/scrapy/settings/default_settings.py index 6d8cdf154..0b09dfd53 100644 --- a/scrapy/settings/default_settings.py +++ b/scrapy/settings/default_settings.py @@ -32,6 +32,13 @@ __all__ = [ "AWS_SESSION_TOKEN", "AWS_USE_SSL", "AWS_VERIFY", + "BACKOFF_DELAY_FACTOR", + "BACKOFF_EXCEPTIONS", + "BACKOFF_HTTP_CODES", + "BACKOFF_JITTER", + "BACKOFF_MAX_DELAY", + "BACKOFF_MIN_DELAY", + "BACKOFF_WINDOW", "BOT_NAME", "CLOSESPIDER_ERRORCOUNT", "CLOSESPIDER_ITEMCOUNT", @@ -51,6 +58,7 @@ __all__ = [ "DEFAULT_DROPITEM_LOG_LEVEL", "DEFAULT_ITEM_CLASS", "DEFAULT_REQUEST_HEADERS", + "DELAYED_REQUESTS_WARN_THRESHOLD", "DEPTH_LIMIT", "DEPTH_PRIORITY", "DEPTH_STATS_VERBOSE", @@ -167,6 +175,7 @@ __all__ = [ "PERIODIC_LOG_DELTA", "PERIODIC_LOG_STATS", "PERIODIC_LOG_TIMING_ENABLED", + "RAMPUP_BACKOFF_TARGET", "RANDOMIZE_DOWNLOAD_DELAY", "REACTOR_THREADPOOL_MAXSIZE", "REDIRECT_ENABLED", @@ -209,6 +218,16 @@ __all__ = [ "TELNETCONSOLE_PORT", "TELNETCONSOLE_USERNAME", "TEMPLATES_DIR", + "THROTTLING_DEBUG", + "THROTTLING_MANAGER", + "THROTTLING_ROBOTSTXT_MAX_DELAY", + "THROTTLING_ROBOTSTXT_OBEY", + "THROTTLING_SCOPES", + "THROTTLING_SCOPE_CONCURRENCY", + "THROTTLING_SCOPE_LIMIT", + "THROTTLING_SCOPE_MANAGER", + "THROTTLING_SCOPE_MAX_IDLE", + "THROTTLING_WINDOW", "TWISTED_DNS_RESOLVER", "TWISTED_REACTOR", "TWISTED_REACTOR_ENABLED", @@ -594,7 +613,10 @@ TEMPLATES_DIR = str((Path(__file__).parent / ".." / "templates").resolve()) THROTTLING_MANAGER = "scrapy.throttling.ThrottlingManager" THROTTLING_SCOPE_MANAGER = "scrapy.throttling.ThrottlingScopeManager" -THROTTLING_SCOPES = {} +THROTTLING_SCOPES = { + "books.toscrape.com": {"concurrency": 16, "delay": 0.1}, + "quotes.toscrape.com": {"concurrency": 16, "delay": 0.1}, +} THROTTLING_WINDOW = 60.0 THROTTLING_ROBOTSTXT_OBEY = True THROTTLING_ROBOTSTXT_MAX_DELAY = 60.0 diff --git a/scrapy/templates/project/module/settings.py.tmpl b/scrapy/templates/project/module/settings.py.tmpl index 71c7596f5..64b27db68 100644 --- a/scrapy/templates/project/module/settings.py.tmpl +++ b/scrapy/templates/project/module/settings.py.tmpl @@ -13,7 +13,7 @@ NEWSPIDER_MODULE = "$project_name.spiders" ROBOTSTXT_OBEY = True # Throttle crawls to be polite to websites: -THROTTLING_SCOPE_CONCURRENCY = 1 +CONCURRENT_REQUESTS_PER_DOMAIN = 1 DOWNLOAD_DELAY = 1 # Set settings whose default value is deprecated to a future-proof value: diff --git a/scrapy/throttling.py b/scrapy/throttling.py index 2d7d14e15..3d88c5ea6 100644 --- a/scrapy/throttling.py +++ b/scrapy/throttling.py @@ -528,7 +528,7 @@ def scope_cache(f: _GetScopesMethod) -> _GetScopesMethod: .. code-block:: python from scrapy.utils.httpobj import urlparse_cached - from scrapy.utils.throttling import scope_cache + from scrapy.throttling import scope_cache class MyThrottlingManager: