From 5427080f4892fb8f33cc159fdcd6f8ce8d4c140e Mon Sep 17 00:00:00 2001 From: Adrian Date: Sun, 9 Aug 2026 18:37:24 +0200 Subject: [PATCH] Create a page on optimization (#7938) --- docs/conf.py | 5 + docs/index.rst | 6 +- docs/requirements.in | 1 + docs/requirements.txt | 3 + docs/topics/broad-crawls.rst | 195 ------------------- docs/topics/optimize.rst | 353 +++++++++++++++++++++++++++++++++++ docs/topics/settings.rst | 1 + 7 files changed, 366 insertions(+), 198 deletions(-) delete mode 100644 docs/topics/broad-crawls.rst create mode 100644 docs/topics/optimize.rst diff --git a/docs/conf.py b/docs/conf.py index 1b41adaad..ad55231bc 100644 --- a/docs/conf.py +++ b/docs/conf.py @@ -31,9 +31,14 @@ extensions = [ "sphinx_scrapy", "scrapyfixautodoc", # Must be after "sphinx.ext.autodoc" "sphinx.ext.coverage", + "sphinx_reredirects", "sphinx_rtd_dark_mode", ] +redirects = { + "topics/broad-crawls": "optimize.html#broad-crawls", +} + templates_path = ["_templates"] exclude_patterns = ["build", "Thumbs.db", ".DS_Store"] diff --git a/docs/index.rst b/docs/index.rst index 688cab81b..d45b2c208 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -152,7 +152,7 @@ Solving specific problems topics/contracts topics/practices topics/security - topics/broad-crawls + topics/optimize topics/developer-tools topics/dynamic-content topics/leaks @@ -180,8 +180,8 @@ Solving specific problems Understand the security implications of Scrapy defaults and how to harden them. -:doc:`topics/broad-crawls` - Tune Scrapy for crawling a lot domains in parallel. +:doc:`topics/optimize` + Find the bottleneck of your crawls and learn how to address it. :doc:`topics/developer-tools` Learn how to scrape with your browser's developer tools. diff --git a/docs/requirements.in b/docs/requirements.in index 3783dd1dc..77514642a 100644 --- a/docs/requirements.in +++ b/docs/requirements.in @@ -3,6 +3,7 @@ pydantic scrapy-spider-metadata sphinx sphinx-notfound-page +sphinx-reredirects sphinx-rtd-theme sphinx-rtd-dark-mode sphinx-scrapy @ git+https://github.com/scrapy/sphinx-scrapy.git@0.8.10 diff --git a/docs/requirements.txt b/docs/requirements.txt index 0f5969401..b0a6d0b04 100644 --- a/docs/requirements.txt +++ b/docs/requirements.txt @@ -134,6 +134,7 @@ sphinx==9.1.0 # sphinx-llms-txt # sphinx-markdown-builder # sphinx-notfound-page + # sphinx-reredirects # sphinx-rtd-theme # sphinx-scrapy # sphinxcontrib-jquery @@ -147,6 +148,8 @@ sphinx-markdown-builder @ git+https://github.com/zytedata/sphinx-markdown-builde # via sphinx-scrapy sphinx-notfound-page==1.1.0 # via -r docs/requirements.in +sphinx-reredirects==1.1.0 + # via -r docs/requirements.in sphinx-rtd-dark-mode==1.3.0 # via -r docs/requirements.in sphinx-rtd-theme==3.1.0 diff --git a/docs/topics/broad-crawls.rst b/docs/topics/broad-crawls.rst deleted file mode 100644 index d6b9fd6f9..000000000 --- a/docs/topics/broad-crawls.rst +++ /dev/null @@ -1,195 +0,0 @@ -.. _topics-broad-crawls: - -============ -Broad Crawls -============ - -Scrapy defaults are optimized for crawling specific sites. These sites are -often handled by a single Scrapy spider, although this is not necessary or -required (for example, there are generic spiders that handle any given site -thrown at them). - -In addition to this "focused crawl", there is another common type of crawling -which covers a large (potentially unlimited) number of domains, and is only -limited by time or other arbitrary constraint, rather than stopping when the -domain was crawled to completion or when there are no more requests to perform. -These are called "broad crawls" and is the typical crawlers employed by search -engines. - -These are some common properties often found in broad crawls: - -* they crawl many domains (often, unbounded) instead of a specific set of sites - -* they don't necessarily crawl domains to completion, because it would be - impractical (or impossible) to do so, and instead limit the crawl by time or - number of pages crawled - -* they are simpler in logic (as opposed to very complex spiders with many - extraction rules) because data is often post-processed in a separate stage - -* they crawl many domains concurrently, which allows them to achieve faster - crawl speeds by not being limited by any particular site constraint (each site - is crawled slowly to respect politeness, but many sites are crawled in - parallel) - -As said above, Scrapy default settings are optimized for focused crawls, not -broad crawls. However, due to its asynchronous architecture, Scrapy is very -well suited for performing fast broad crawls. This page summarizes some things -you need to keep in mind when using Scrapy for doing broad crawls, along with -concrete suggestions of Scrapy settings to tune in order to achieve an -efficient broad crawl. - -.. _broad-crawls-scheduler-priority-queue: - -.. _broad-crawls-concurrency: - -Increase concurrency -==================== - -Concurrency is the number of requests that are processed in parallel. There is -a global limit (:setting:`CONCURRENT_REQUESTS`) and an additional limit that -can be set per domain (:setting:`CONCURRENT_REQUESTS_PER_DOMAIN`). - -The default global concurrency limit in Scrapy is not suitable for crawling -many different domains in parallel, so you will want to increase it. How much -to increase it will depend on how much CPU and memory your crawler will have -available. - -A good starting point is ``100``: - -.. code-block:: python - - CONCURRENT_REQUESTS = 100 - -But the best way to find out is by doing some trials and identifying at what -concurrency your Scrapy process gets CPU bounded. For optimum performance, you -should pick a concurrency where CPU usage is at 80-90%. - -Increasing concurrency also increases memory usage. If memory usage is a -concern, you might need to lower your global concurrency limit accordingly. - - -Increase Twisted IO thread pool maximum size -============================================ - -Currently Scrapy does DNS resolution in a blocking way with usage of thread -pool. With higher concurrency levels the crawling could be slow or even fail -hitting DNS resolver timeouts. Possible solution to increase the number of -threads handling DNS queries. The DNS queue will be processed faster speeding -up establishing of connection and crawling overall. - -To increase maximum thread pool size use: - -.. code-block:: python - - REACTOR_THREADPOOL_MAXSIZE = 20 - -Setup your own DNS -================== - -If you have multiple crawling processes and single central DNS, it can act -like DoS attack on the DNS server resulting to slow down of entire network or -even blocking your machines. To avoid this setup your own DNS server with -local cache and upstream to some large DNS like OpenDNS or Verizon. - -Reduce log level -================ - -When doing broad crawls you are often only interested in the crawl rates you -get and any errors found. These stats are reported by Scrapy when using the -``INFO`` log level. In order to save CPU (and log storage requirements) you -should not use ``DEBUG`` log level when performing large broad crawls in -production. Using ``DEBUG`` level when developing your (broad) crawler may be -fine though. - -To set the log level use: - -.. code-block:: python - - LOG_LEVEL = "INFO" - -Disable cookies -=============== - -Disable cookies unless you *really* need. Cookies are often not needed when -doing broad crawls (search engine crawlers ignore them), and they improve -performance by saving some CPU cycles and reducing the memory footprint of your -Scrapy crawler. - -To disable cookies use: - -.. code-block:: python - - COOKIES_ENABLED = False - -Disable retries -=============== - -Retrying failed HTTP requests can slow down the crawls substantially, especially -when sites causes are very slow (or fail) to respond, thus causing a timeout -error which gets retried many times, unnecessarily, preventing crawler capacity -to be reused for other domains. - -To disable retries use: - -.. code-block:: python - - RETRY_ENABLED = False - -Reduce download timeout -======================= - -Unless you are crawling from a very slow connection (which shouldn't be the -case for broad crawls) reduce the download timeout so that stuck requests are -discarded quickly and free up capacity to process the next ones. - -To reduce the download timeout use: - -.. code-block:: python - - DOWNLOAD_TIMEOUT = 15 - -Disable redirects -================= - -Consider disabling redirects, unless you are interested in following them. When -doing broad crawls it's common to save redirects and resolve them when -revisiting the site at a later crawl. This also help to keep the number of -request constant per crawl batch, otherwise redirect loops may cause the -crawler to dedicate too many resources on any specific domain. - -To disable redirects use: - -.. code-block:: python - - REDIRECT_ENABLED = False - -.. _broad-crawls-bfo: - -Crawl in BFO order -================== - -:ref:`Scrapy crawls in DFO order by default `. - -In broad crawls, however, page crawling tends to be faster than page -processing. As a result, unprocessed early requests stay in memory until the -final depth is reached, which can significantly increase memory usage. - -:ref:`Crawl in BFO order ` instead to save memory. - - -Be mindful of memory leaks -========================== - -If your broad crawl shows a high memory usage, in addition to :ref:`crawling in -BFO order `, :ref:`lowering concurrency -` and :ref:`delaying start request iteration -` you should :ref:`debug your memory leaks -`. - - -Install a specific Twisted reactor -================================== - -If the crawl is exceeding the system's capabilities, you might want to try -installing a specific Twisted reactor, via the :setting:`TWISTED_REACTOR` setting. diff --git a/docs/topics/optimize.rst b/docs/topics/optimize.rst new file mode 100644 index 000000000..cea49fca3 --- /dev/null +++ b/docs/topics/optimize.rst @@ -0,0 +1,353 @@ +.. _optimize: + +============ +Optimization +============ + +A crawl goes as fast as its slowest part allows. :ref:`Find out which part that +is ` before changing any setting. + +:ref:`Broad crawls ` have their own set of recommended +adjustments. + +.. _optimize-bottleneck: + +Finding the bottleneck +====================== + +The bottleneck depends on the spider: on the same machine, one crawl can be +limited by its own parsing code and another by the target website. So measure +the crawl that you want to optimize. + +:class:`~scrapy.extensions.logstats.LogStats` reports crawl speed every +:setting:`LOGSTATS_INTERVAL` seconds: + +.. code-block:: text + + [scrapy.extensions.logstats] INFO: Crawled 1200 pages (at 60 pages/min), scraped 1150 items (at 58 items/min) + +A rate that stays flat as you raise :setting:`CONCURRENT_REQUESTS` means +something else is the limit. + + +Reading the engine status +------------------------- + +The :ref:`telnet console ` reports, through ``est()``, +what every part of the engine is doing at a given moment: + +.. code-block:: text + + len(engine.downloader.active) : 16 + len(engine._slot.scheduler.mqs) : 92 + len(engine.scraper.slot.active) : 0 + engine.scraper.slot.active_size : 0 + engine.scraper.slot.needs_backout() : False + +Take a few readings at different points of the crawl: + +- ``len(engine.downloader.active)`` stays at :setting:`CONCURRENT_REQUESTS`: + the downloader is the limit. You are waiting on the network or on the + target website. See :ref:`optimize-concurrency`. + +- ``len(engine.downloader.active)`` stays below + :setting:`CONCURRENT_REQUESTS` while the scheduler queues (``mqs``, + ``dqs``) hold requests: something throttles those requests before they + reach the downloader, usually :setting:`CONCURRENT_REQUESTS_PER_DOMAIN`, + :setting:`DOWNLOAD_DELAY` or :ref:`AutoThrottle `. + +- Both the downloader and the scheduler queues stay near empty: your spider + is not producing requests fast enough. A crawl that walks pagination one + page at a time cannot use more concurrency than it creates. See + :ref:`optimize-requests`. + +- ``needs_backout()`` is ``True``, or ``active_size`` approaches + :setting:`SCRAPER_SLOT_MAX_ACTIVE_SIZE`: responses arrive faster than your + callbacks and :ref:`item pipelines ` handle them. The + bottleneck is your own code. + +- ``len(engine._slot.scheduler.mqs)`` grows without settling: the crawl + discovers requests faster than it downloads them. This is what makes long + crawls run out of memory. + + +Reading resource usage +---------------------- + +CPU + Scrapy runs in a single process, and everything except DNS resolution and + code you explicitly move to a thread runs in a single thread. One CPU core + is the ceiling; a process sitting at 100% of a core is CPU-bound no matter + how many cores the machine has. + + Use a sampling profiler, such as py-spy_, to find out which code is + spending that CPU. :ref:`Selectors ` and item pipelines + are the usual answer. + + .. _py-spy: https://github.com/benfred/py-spy + +Memory + The :ref:`memory usage extension ` records + :stat:`memusage/startup` and :stat:`memusage/max`. A :stat:`memusage/max` + far above :stat:`memusage/startup` is expected; what matters is whether it + keeps growing for as long as the crawl runs. + + Growth that tracks ``len(engine._slot.scheduler.mqs)`` is a scheduling + problem, covered in :ref:`optimize-memory`. Growth that does not is a + :ref:`memory leak `. + +Network + Compare :stat:`downloader/response_bytes` over the crawl time against your + available bandwidth. Saturated bandwidth caps concurrency regardless of any + setting. + + DNS resolution is separate: it runs on a thread pool of + :setting:`REACTOR_THREADPOOL_MAXSIZE` threads, and results are cached + (:setting:`DNSCACHE_ENABLED`, :setting:`DNSCACHE_SIZE`). It only becomes a + limit of its own when there are many different domains to resolve, as in + :ref:`broad crawls `, where it shows up as slow starts and + DNS timeouts. + +Disk + :ref:`Feed exports ` write to disk on most crawls, + although item data is usually small enough for that not to matter. The ones + to suspect are + :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware` and + the :ref:`media pipelines `, which write whole + responses, and :setting:`JOBDIR`, which writes every scheduled request. + + +.. _optimize-concurrency: + +Sending more requests at a time +=============================== + +:setting:`CONCURRENT_REQUESTS` caps how many requests are being downloaded at +any given moment, :setting:`CONCURRENT_REQUESTS_PER_DOMAIN` caps how many of +those may target the same domain, and :setting:`DOWNLOAD_DELAY` sets a minimum +wait between two consecutive requests to the same domain. A project generated by +:command:`startproject` gets one request per second per domain out of these. + +Raise them to crawl a single website faster, and see +:ref:`broad-crawls-concurrency` to spread requests across many websites +instead. + +The limit that matters, though, is the one the target website tolerates. +Exceeding it gets you throttled, served errors or banned, all of which make the +crawl slower than a lower concurrency would have been. To find that limit: + +- Read the :ref:`robots.txt ` file of the website. Scrapy + does not act on its ``Crawl-delay`` and ``Request-rate`` directives, so when + they are present, translate them into :setting:`DOWNLOAD_DELAY` and + concurrency settings yourself. + +- Check the traffic that the website already gets, using a service like + `SimilarWeb`_ or `Cloudflare Radar`_. A rate that is a rounding error next + to what the website serves anyway is unlikely to be a problem for it. + + .. _SimilarWeb: https://www.similarweb.com/ + .. _Cloudflare Radar: https://radar.cloudflare.com/ + +- Look for a documented way in. An API, a bulk export or a search endpoint is + both faster for you and cheaper for the website than crawling its pages, and + the terms of service may state a rate. + +- Crawl when the website is idle, in its own timezone, so that the capacity + you take is capacity nobody else wanted. + +- Raise concurrency gradually and watch the website respond. + :stat:`downloader/response_status_count/{status_code}` counts for 429, 503 + or the ban page of the website, growing :stat:`retry/count`, or a + :ref:`download latency ` that climbs as you push harder, + all mean you have gone past the limit. + + +.. _optimize-requests: + +Producing requests faster +========================= + +A spider that discovers its requests one response at a time keeps the +downloader idle no matter how high you set :setting:`CONCURRENT_REQUESTS`. To +put more requests in the scheduler earlier: + +- Request every page at once when you can work out how many there are, e.g. + from a page count or from a result count and a page size in the first + response, instead of following a link to the next page on every response. + +- Get URLs from a source that lists many of them at once, such as a sitemap + or a search or export endpoint of the target website. For a crawl that + needs nothing else, :class:`~scrapy.spiders.SitemapSpider` reads sitemaps + for you. + +- Raise the :attr:`~scrapy.Request.priority` of pagination requests, so that + they are downloaded before the requests that they compete with, and + discover the rest of the crawl sooner. + +Each of these trades memory for speed: a request produced before the downloader +can take it waits in the scheduler, or on disk if you set :setting:`JOBDIR`. +Pushed far enough, they turn memory or disk into your new bottleneck, which is +why :ref:`optimize-memory` recommends the reverse of the last point. + + +.. _optimize-resources: + +Lowering resource usage +======================= + +.. _optimize-memory: + +Lowering memory usage +--------------------- + +- Lower :setting:`SCRAPER_SLOT_MAX_ACTIVE_SIZE`. + +- Lower :setting:`DOWNLOAD_MAXSIZE`, which allows a single response to take up + to 1 GiB of memory by default, multiplied by your concurrency. Set + :setting:`DOWNLOAD_WARNSIZE` first to find out whether the website actually + serves responses that big. + +- Lower the number of :ref:`scheduled requests ` held in + memory: + + - Increase the :attr:`~scrapy.Request.priority` of requests whose + :attr:`~scrapy.Request.callback` cannot yield additional requests. + + For example, the following spider uses a higher priority (1) for book + requests than for pagination requests: + + .. code-block:: python + + from scrapy import Spider + + + class BooksToScrapeComSpider(Spider): + name = "books_toscrape_com" + start_urls = [ + "http://books.toscrape.com/catalogue/category/books/mystery_3/index.html" + ] + + def parse(self, response): + next_page_links = response.css(".next a") + yield from response.follow_all(next_page_links) + book_links = response.css("article a") + yield from response.follow_all(book_links, callback=self.parse_book, priority=1) + + def parse_book(self, response): + yield { + "name": response.css("h1::text").get(), + "price": response.css(".price_color::text").re_first("£(.*)"), + "url": response.url, + } + + .. note:: If the number of request-yielding, low-priority requests + scheduled at any given time is lower than concurrency settings + (:setting:`CONCURRENT_REQUESTS_PER_DOMAIN` or + :setting:`CONCURRENT_REQUESTS`), as in the example above, this can + slow down your crawl by turning those requests into a bottleneck. + + - If you have many :ref:`start requests `, consider + :ref:`delaying their iteration `. + + - Set :setting:`JOBDIR` to offload all scheduled requests to disk. + +- Be on the lookout for :ref:`memory leaks `. + + +Lowering network usage +---------------------- + +- Install brotli_ and zstandard_ to support brotli-compressed_ and + zstd-compressed_ responses. + + .. _brotli-compressed: https://www.ietf.org/rfc/rfc7932.txt + .. _brotli: https://pypi.org/project/Brotli/ + .. _zstd-compressed: https://www.ietf.org/rfc/rfc8478.txt + .. _zstandard: https://pypi.org/project/zstandard/ + +- Enable :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware` + while developing your spider, so that re-runs do not download the same + responses again. + + +Lowering CPU usage +------------------ + +- Set :setting:`LOG_LEVEL` to ``"INFO"`` or higher. + +- Restrict what you parse. A :ref:`selector ` over a + smaller part of the response, or a single query whose result you reuse, + beats repeated queries over the whole document. + + +Other tips +---------- + +- Try :ref:`using the asyncio reactor ` with uvloop_ as + :ref:`custom event loop `, i.e. setting + :setting:`ASYNCIO_EVENT_LOOP` to ``"uvloop.Loop"``. + + .. _uvloop: https://github.com/MagicStack/uvloop + + Alternatively, try :ref:`switching to a non-asyncio reactor + `. + +- Disable unused :ref:`components `. + + For example, set :setting:`COOKIES_ENABLED` to ``False`` unless you need + cookies. + +- Split the crawl across separate processes to use more than one CPU core. + See :ref:`distributed-crawls`. + + +.. _broad-crawls: +.. _topics-broad-crawls: + +Speeding up broad crawls +======================== + +While Scrapy is well suited for **broad crawls**, i.e. crawls that target many +websites, the default :ref:`settings ` are optimized for +crawls targeting a single website. + +For broad crawls, consider these adjustments: + +- .. _broad-crawls-concurrency: + + Increase the global concurrency: + + - Set :setting:`CONCURRENT_REQUESTS` as close to + :setting:`CONCURRENT_REQUESTS_PER_DOMAIN` × [number of target domains] + (e.g. 8 × 10 domains = 80 concurrent requests) as your CPU and memory + allow. + + - Increase :setting:`SCRAPER_SLOT_MAX_ACTIVE_SIZE` when increasing + :setting:`CONCURRENT_REQUESTS` stops making a difference. + +- .. _broad-crawls-bfo: + + If memory is a bottleneck, see if :ref:`crawling in BFO order ` lowers + memory usage. + +- Improve DNS resolution speed: + + - Set up your own DNS server, with a local cache and upstream to a `large + DNS server`_, to avoid slowing down your network. + + .. _large DNS server: https://en.wikipedia.org/wiki/Public_recursive_name_server#Notable_public_DNS_service_operators + + - Increase :setting:`REACTOR_THREADPOOL_MAXSIZE` to the minimum value + that avoids DNS resolution timeouts and makes a noticeable positive + impact in crawl speed. + +- Lower the negative impact of some responses: + + - Set :setting:`RETRY_ENABLED` to ``False`` or, if you need retries, + consider lowering :setting:`RETRY_TIMES`. + + - Lower :setting:`DOWNLOAD_TIMEOUT` to a more reasonable value, to + discard stuck requests more quickly. + + - Set :setting:`REDIRECT_ENABLED` to ``False`` unless you want to follow + redirects. diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index ec222d999..0ea60e9cf 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -1913,6 +1913,7 @@ Type of in-memory queue used by the scheduler. Other available type is: .. setting:: SCHEDULER_PRIORITY_QUEUE +.. _broad-crawls-scheduler-priority-queue: SCHEDULER_PRIORITY_QUEUE ------------------------