Merge remote-tracking branch 'origin/master' into httpcache-scope

This commit is contained in:
Adrian Chaves 2026-08-11 00:39:30 +02:00
commit 780ab92345
150 changed files with 4586 additions and 3302 deletions

View File

@ -86,6 +86,7 @@ jobs:
- python-version: pypy3.11-7.3.20
env:
TOXENV: pypy3-extra-deps
coverage: true
- python-version: "3.14"
env:
TOXENV: botocore

View File

@ -16,8 +16,11 @@ jobs:
tests:
name: tests
runs-on: ubuntu-latest
timeout-minutes: 30
env:
PYTEST_ADDOPTS: -n auto --no-cov
# A development branch of a dependency can make a test hang forever, so
# tests get a time limit here that they do not need elsewhere.
PYTEST_ADDOPTS: -n auto --no-cov --timeout=120
TOXENV: vcs-deps
UV_PYTHON_PREFERENCE: only-system
steps:

View File

@ -27,7 +27,7 @@ repos:
hooks:
- id: sphinx-lint
- repo: https://github.com/scrapy/sphinx-scrapy
rev: 0.8.10
rev: 0.8.11
hooks:
- id: sphinx-scrapy
- repo: https://github.com/zizmorcore/zizmor-pre-commit

View File

@ -137,14 +137,6 @@ def source_role(
return [node], []
def issue_role(
name, rawtext, text: str, lineno, inliner, options=None, content=None
) -> tuple[list[Any], list[Any]]:
ref = "https://github.com/scrapy/scrapy/issues/" + text
node = nodes.reference(rawtext, "issue " + text, refuri=ref)
return [node], []
def commit_role(
name, rawtext, text: str, lineno, inliner, options=None, content=None
) -> tuple[list[Any], list[Any]]:
@ -164,7 +156,6 @@ def rev_role(
def setup(app: Sphinx) -> dict[str, Any]:
app.add_role("source", source_role)
app.add_role("commit", commit_role)
app.add_role("issue", issue_role)
app.add_role("rev", rev_role)
app.add_node(

View File

@ -31,9 +31,14 @@ extensions = [
"sphinx_scrapy",
"scrapyfixautodoc", # Must be after "sphinx.ext.autodoc"
"sphinx.ext.coverage",
"sphinx_reredirects",
"sphinx_rtd_dark_mode",
]
redirects = {
"topics/broad-crawls": "optimize.html#broad-crawls",
}
templates_path = ["_templates"]
exclude_patterns = ["build", "Thumbs.db", ".DS_Store"]

View File

@ -141,7 +141,7 @@ middleware with a :ref:`custom downloader middleware
- If you can meet the installation requirements, use pyre2_ instead of
Pythons re_ to compile your URL-filtering regular expression. See
:issue:`1908`.
:gh:`1908`.
See also `other suggestions at StackOverflow
<https://stackoverflow.com/q/36440681>`__.
@ -292,7 +292,7 @@ Does Scrapy manage cookies automatically?
Yes, Scrapy receives and keeps track of cookies sent by servers, and sends them
back on subsequent requests, like any regular web browser does.
For more info see :ref:`topics-request-response` and :ref:`cookies-mw`.
For more info see :ref:`cookies`.
How can I see the cookies being sent and received from Scrapy?
--------------------------------------------------------------
@ -419,7 +419,7 @@ Running ``runspider`` I get ``error: No spider found in file: <filename>``
This may happen if your Scrapy project has a spider module with a name that
conflicts with the name of one of the `Python standard library modules`_, such
as ``csv.py`` or ``os.py``, or any `Python package`_ that you have installed.
See :issue:`2680`.
See :gh:`2680`.
.. _has been reported: https://github.com/scrapy/scrapy/issues/2905

View File

@ -78,6 +78,7 @@ Basic concepts
topics/item-pipeline
topics/feed-exports
topics/request-response
topics/cookies
topics/link-extractors
topics/settings
topics/exceptions
@ -109,6 +110,9 @@ Basic concepts
:doc:`topics/request-response`
Understand the classes used to represent HTTP requests and responses.
:doc:`topics/cookies`
Send and receive cookies.
:doc:`topics/link-extractors`
Convenient classes to extract links to follow from pages.
@ -152,7 +156,7 @@ Solving specific problems
topics/contracts
topics/practices
topics/security
topics/broad-crawls
topics/optimize
topics/developer-tools
topics/dynamic-content
topics/leaks
@ -180,8 +184,8 @@ Solving specific problems
Understand the security implications of Scrapy defaults and how to harden
them.
:doc:`topics/broad-crawls`
Tune Scrapy for crawling a lot domains in parallel.
:doc:`topics/optimize`
Find the bottleneck of your crawls and learn how to address it.
:doc:`topics/developer-tools`
Learn how to scrape with your browser's developer tools.

View File

@ -111,8 +111,6 @@ The following extras are available:
- Provides
* - ``bpython``
- :ref:`bpython shell <shell-config>`
* - ``brotli``
- :ref:`Brotli response decompression <http-compression>`
* - ``gcs``
- :ref:`Google Cloud Storage <topics-feed-storage-gcs>` for
:ref:`feed exports <topics-feed-exports>` and

File diff suppressed because it is too large Load Diff

View File

@ -3,6 +3,7 @@ pydantic
scrapy-spider-metadata
sphinx
sphinx-notfound-page
sphinx-reredirects
sphinx-rtd-theme
sphinx-rtd-dark-mode
sphinx-scrapy @ git+https://github.com/scrapy/sphinx-scrapy.git@0.8.10
sphinx-scrapy @ git+https://github.com/scrapy/sphinx-scrapy.git@0.8.11

View File

@ -134,6 +134,7 @@ sphinx==9.1.0
# sphinx-llms-txt
# sphinx-markdown-builder
# sphinx-notfound-page
# sphinx-reredirects
# sphinx-rtd-theme
# sphinx-scrapy
# sphinxcontrib-jquery
@ -147,13 +148,15 @@ sphinx-markdown-builder @ git+https://github.com/zytedata/sphinx-markdown-builde
# via sphinx-scrapy
sphinx-notfound-page==1.1.0
# via -r docs/requirements.in
sphinx-reredirects==1.1.0
# via -r docs/requirements.in
sphinx-rtd-dark-mode==1.3.0
# via -r docs/requirements.in
sphinx-rtd-theme==3.1.0
# via
# -r docs/requirements.in
# sphinx-rtd-dark-mode
sphinx-scrapy @ git+https://github.com/scrapy/sphinx-scrapy.git@fe176adc1a8577601bc3fa39b590ebed71a7e9b8
sphinx-scrapy @ git+https://github.com/scrapy/sphinx-scrapy.git@6f8e5e0bbd171a857da480f7188f2a205041cb60
# via -r docs/requirements.in
sphinx-sitemap==2.9.0
# via sphinx-scrapy

View File

@ -35,6 +35,13 @@ how you :ref:`configure the downloader middlewares
:class:`scrapy.Spider` subclass and a
:class:`scrapy.settings.Settings` object.
The :attr:`engine`, :attr:`extensions`, :attr:`logformatter`,
:attr:`request_fingerprinter` and :attr:`stats` attributes get their value
when the crawl starts, and raise :exc:`RuntimeError` when read before that.
.. versionchanged:: VERSION
Those attributes used to be ``None`` before getting their value.
.. attribute:: request_fingerprinter
The request fingerprint builder of this crawler.

View File

@ -1,195 +0,0 @@
.. _topics-broad-crawls:
============
Broad Crawls
============
Scrapy defaults are optimized for crawling specific sites. These sites are
often handled by a single Scrapy spider, although this is not necessary or
required (for example, there are generic spiders that handle any given site
thrown at them).
In addition to this "focused crawl", there is another common type of crawling
which covers a large (potentially unlimited) number of domains, and is only
limited by time or other arbitrary constraint, rather than stopping when the
domain was crawled to completion or when there are no more requests to perform.
These are called "broad crawls" and is the typical crawlers employed by search
engines.
These are some common properties often found in broad crawls:
* they crawl many domains (often, unbounded) instead of a specific set of sites
* they don't necessarily crawl domains to completion, because it would be
impractical (or impossible) to do so, and instead limit the crawl by time or
number of pages crawled
* they are simpler in logic (as opposed to very complex spiders with many
extraction rules) because data is often post-processed in a separate stage
* they crawl many domains concurrently, which allows them to achieve faster
crawl speeds by not being limited by any particular site constraint (each site
is crawled slowly to respect politeness, but many sites are crawled in
parallel)
As said above, Scrapy default settings are optimized for focused crawls, not
broad crawls. However, due to its asynchronous architecture, Scrapy is very
well suited for performing fast broad crawls. This page summarizes some things
you need to keep in mind when using Scrapy for doing broad crawls, along with
concrete suggestions of Scrapy settings to tune in order to achieve an
efficient broad crawl.
.. _broad-crawls-scheduler-priority-queue:
.. _broad-crawls-concurrency:
Increase concurrency
====================
Concurrency is the number of requests that are processed in parallel. There is
a global limit (:setting:`CONCURRENT_REQUESTS`) and an additional limit that
can be set per domain (:setting:`CONCURRENT_REQUESTS_PER_DOMAIN`).
The default global concurrency limit in Scrapy is not suitable for crawling
many different domains in parallel, so you will want to increase it. How much
to increase it will depend on how much CPU and memory your crawler will have
available.
A good starting point is ``100``:
.. code-block:: python
CONCURRENT_REQUESTS = 100
But the best way to find out is by doing some trials and identifying at what
concurrency your Scrapy process gets CPU bounded. For optimum performance, you
should pick a concurrency where CPU usage is at 80-90%.
Increasing concurrency also increases memory usage. If memory usage is a
concern, you might need to lower your global concurrency limit accordingly.
Increase Twisted IO thread pool maximum size
============================================
Currently Scrapy does DNS resolution in a blocking way with usage of thread
pool. With higher concurrency levels the crawling could be slow or even fail
hitting DNS resolver timeouts. Possible solution to increase the number of
threads handling DNS queries. The DNS queue will be processed faster speeding
up establishing of connection and crawling overall.
To increase maximum thread pool size use:
.. code-block:: python
REACTOR_THREADPOOL_MAXSIZE = 20
Setup your own DNS
==================
If you have multiple crawling processes and single central DNS, it can act
like DoS attack on the DNS server resulting to slow down of entire network or
even blocking your machines. To avoid this setup your own DNS server with
local cache and upstream to some large DNS like OpenDNS or Verizon.
Reduce log level
================
When doing broad crawls you are often only interested in the crawl rates you
get and any errors found. These stats are reported by Scrapy when using the
``INFO`` log level. In order to save CPU (and log storage requirements) you
should not use ``DEBUG`` log level when performing large broad crawls in
production. Using ``DEBUG`` level when developing your (broad) crawler may be
fine though.
To set the log level use:
.. code-block:: python
LOG_LEVEL = "INFO"
Disable cookies
===============
Disable cookies unless you *really* need. Cookies are often not needed when
doing broad crawls (search engine crawlers ignore them), and they improve
performance by saving some CPU cycles and reducing the memory footprint of your
Scrapy crawler.
To disable cookies use:
.. code-block:: python
COOKIES_ENABLED = False
Disable retries
===============
Retrying failed HTTP requests can slow down the crawls substantially, especially
when sites causes are very slow (or fail) to respond, thus causing a timeout
error which gets retried many times, unnecessarily, preventing crawler capacity
to be reused for other domains.
To disable retries use:
.. code-block:: python
RETRY_ENABLED = False
Reduce download timeout
=======================
Unless you are crawling from a very slow connection (which shouldn't be the
case for broad crawls) reduce the download timeout so that stuck requests are
discarded quickly and free up capacity to process the next ones.
To reduce the download timeout use:
.. code-block:: python
DOWNLOAD_TIMEOUT = 15
Disable redirects
=================
Consider disabling redirects, unless you are interested in following them. When
doing broad crawls it's common to save redirects and resolve them when
revisiting the site at a later crawl. This also help to keep the number of
request constant per crawl batch, otherwise redirect loops may cause the
crawler to dedicate too many resources on any specific domain.
To disable redirects use:
.. code-block:: python
REDIRECT_ENABLED = False
.. _broad-crawls-bfo:
Crawl in BFO order
==================
:ref:`Scrapy crawls in DFO order by default <faq-bfo-dfo>`.
In broad crawls, however, page crawling tends to be faster than page
processing. As a result, unprocessed early requests stay in memory until the
final depth is reached, which can significantly increase memory usage.
:ref:`Crawl in BFO order <faq-bfo-dfo>` instead to save memory.
Be mindful of memory leaks
==========================
If your broad crawl shows a high memory usage, in addition to :ref:`crawling in
BFO order <broad-crawls-bfo>`, :ref:`lowering concurrency
<broad-crawls-concurrency>` and :ref:`delaying start request iteration
<start-requests-lazy>` you should :ref:`debug your memory leaks
<topics-leaks>`.
Install a specific Twisted reactor
==================================
If the crawl is exceeding the system's capabilities, you might want to try
installing a specific Twisted reactor, via the :setting:`TWISTED_REACTOR` setting.

138
docs/topics/cookies.rst Normal file
View File

@ -0,0 +1,138 @@
.. _cookies:
.. _cookies-mw:
=======
Cookies
=======
Scrapy keeps track of the cookies that websites set and sends them back on
later requests to those websites, just like a web browser does. That is the job
of :class:`~scrapy.downloadermiddlewares.cookies.CookiesMiddleware`, which is
enabled by default.
Setting cookies on a request
============================
.. invisible-code-block: python
from scrapy import Request
Use the ``cookies`` parameter of :class:`~scrapy.Request` to send cookies of
your own, either as a dict:
.. code-block:: python
request = Request(
url="https://example.com",
cookies={"currency": "USD", "country": "UY"},
)
Or as a list of dicts, which also lets you set cookie attributes:
.. code-block:: python
request = Request(
url="https://example.com",
cookies=[
{
"name": "currency",
"value": "USD",
"domain": "example.com",
"path": "/currency",
"secure": True,
},
],
)
Setting attributes is only useful if the cookies are stored for later requests,
i.e. if :reqmeta:`dont_merge_cookies` is not enabled.
.. caution:: Cookies set through the ``Cookie`` header are not handled by
:class:`~scrapy.downloadermiddlewares.cookies.CookiesMiddleware`, which
drops that header.
.. caution:: When a cookie name or value is a byte sequence that is not UTF-8
encoded, the cookie is dropped and a warning is logged. See
:ref:`topics-logging-advanced-customization` to customize the logging
behavior.
.. reqmeta:: cookiejar
Multiple cookie sessions per spider
===================================
By default all requests share a single cookie jar (session). To use different
ones, pass an identifier in the :reqmeta:`cookiejar` request meta key:
.. skip: next
.. code-block:: python
for i, url in enumerate(urls):
yield Request(url, meta={"cookiejar": i}, callback=self.parse_page)
The :reqmeta:`cookiejar` meta key is not "sticky", so you need to keep passing
it along on subsequent requests:
.. code-block:: python
def parse_page(self, response):
return Request(
"https://example.com/otherpage",
meta={"cookiejar": response.meta["cookiejar"]},
callback=self.parse_other_page,
)
.. reqmeta:: dont_merge_cookies
Skipping the cookie jar for a request
=====================================
Set the :reqmeta:`dont_merge_cookies` request meta key to ``True`` to keep a
request from touching the cookie jar in either direction: no stored cookie is
sent with the request, and no cookie received in the response is stored. The
cookies of the request itself are ignored as well.
.. setting:: COOKIES_ENABLED
COOKIES_ENABLED
===============
Default: ``True``
Whether to enable :class:`~scrapy.downloadermiddlewares.cookies.CookiesMiddleware`.
If disabled, no cookies are sent to web servers.
.. setting:: COOKIES_DEBUG
COOKIES_DEBUG
=============
Default: ``False``
If enabled, Scrapy logs all cookies sent in requests (i.e. the ``Cookie``
header) and all cookies received in responses (i.e. the ``Set-Cookie``
header)::
2011-04-06 14:35:10-0300 [scrapy.core.engine] INFO: Spider opened
2011-04-06 14:35:10-0300 [scrapy.downloadermiddlewares.cookies] DEBUG: Sending cookies to: <GET http://www.diningcity.com/netherlands/index.html>
Cookie: clientlanguage_nl=en_EN
2011-04-06 14:35:14-0300 [scrapy.downloadermiddlewares.cookies] DEBUG: Received cookies from: <200 http://www.diningcity.com/netherlands/index.html>
Set-Cookie: JSESSIONID=B~FA4DC0C496C8762AE4F1A620EAB34F38; Path=/
Set-Cookie: ip_isocode=US
Set-Cookie: clientlanguage_nl=en_EN; Expires=Thu, 07-Apr-2011 21:21:34 GMT; Path=/
2011-04-06 14:49:50-0300 [scrapy.core.engine] DEBUG: Crawled (200) <GET http://www.diningcity.com/netherlands/index.html> (referer: None)
[...]
CookiesMiddleware
=================
.. module:: scrapy.downloadermiddlewares.cookies
:synopsis: Cookies Downloader Middleware
.. autoclass:: CookiesMiddleware

View File

@ -130,17 +130,23 @@ using different handlers.
Here is a comparison of some features of the built-in HTTP handlers, see the
individual handler docs for more differences:
================== ================= ===================== ====================
Feature H2DownloadHandler HTTP11DownloadHandler HttpxDownloadHandler
================== ================= ===================== ====================
Requires asyncio No No Yes
Requires a reactor Yes Yes No
HTTP/1.1 No Yes Yes
HTTP/2 Yes No Yes
TLS implementation ``cryptography`` ``cryptography`` Stdlib ``ssl``
HTTP proxies No Yes Yes
SOCKS proxies No No Yes
================== ================= ===================== ====================
=================== ================= ===================== ====================
Feature H2DownloadHandler HTTP11DownloadHandler HttpxDownloadHandler
=================== ================= ===================== ====================
Requires asyncio No No Yes
Requires a reactor Yes Yes No
HTTP/1.1 No Yes Yes
HTTP/2 Yes No Yes
TLS implementation ``cryptography`` ``cryptography`` Stdlib ``ssl``
HTTP proxies No Yes Yes
SOCKS proxies No No Yes
Bad header handling Not applicable Skip bad Fail
=================== ================= ===================== ====================
Bad header handling is what a handler does when a response has a bad header
line, e.g. one with no colon in it, which some servers send. Handlers that skip
bad header lines, like web browsers do, still parse the header lines that follow
them; other handlers also lose those, or cannot download such responses at all.
You can find additional HTTP download handlers in the
scrapy-download-handlers-incubator_ package. This package is made by the Scrapy
@ -191,6 +197,7 @@ Features and limitations
HTTP proxies No (not implemented)
SOCKS proxies No (not supported by the library)
HTTP/2 Yes
Bad header handling Not applicable (HTTP/2 only)
``response.certificate`` :class:`twisted.internet.ssl.Certificate` object
Per-request ``bindaddress`` Yes
TLS implementation ``pyOpenSSL``/``cryptography``
@ -239,11 +246,16 @@ Features and limitations
HTTP proxies Yes
SOCKS proxies No (not supported by the library)
HTTP/2 No (implemented as a separate handler)
Bad header handling Skip bad, like web browsers do
``response.certificate`` :class:`twisted.internet.ssl.Certificate` object
Per-request ``bindaddress`` Yes
TLS implementation ``pyOpenSSL``/``cryptography``
=========================== ================================================
.. versionchanged:: VERSION
Bad header lines with no colon in them are now skipped, instead of making
the whole response impossible to download.
Other limitations:
- IPv6 support requires setting :setting:`TWISTED_DNS_RESOLVER`
@ -297,6 +309,7 @@ Features and limitations
HTTP proxies Yes
SOCKS proxies Yes (SOCKS5)
HTTP/2 Yes
Bad header handling Fail (not supported by the library)
``response.certificate`` DER bytes
Per-request ``bindaddress`` No (not supported by the library)
TLS implementation Standard library ``ssl``

View File

@ -156,6 +156,61 @@ defines one or more of these methods:
:param exception: the raised exception
:type exception: an ``Exception`` object
.. _mw-download:
Downloading a request from a downloader middleware
==================================================
A downloader middleware can download a request of its own while it processes
another one, e.g. to fetch something that the request it is processing needs.
The built-in :ref:`robots.txt middleware <topics-dlmw-robots>` does that: it
holds each request while it downloads the ``robots.txt`` file of its website.
Use :meth:`crawler.engine.download_async()
<scrapy.core.engine.ExecutionEngine.download_async>` for that:
.. code-block:: python
from scrapy import Request
from scrapy.http.request import NO_CALLBACK
class TokenMiddleware:
def __init__(self, crawler):
self.crawler = crawler
self.token = None
@classmethod
def from_crawler(cls, crawler):
return cls(crawler)
async def process_request(self, request):
if request.meta.get("dont_obey_robotstxt"):
return
if self.token is None:
response = await self.crawler.engine.download_async(
Request(
"https://example.com/token",
callback=NO_CALLBACK,
meta={"dont_obey_robotstxt": True},
)
)
self.token = response.text
request.headers["Authorization"] = self.token
Requests that you download this way go through the downloader middleware chain
as well, including your own middleware and the :ref:`robots.txt middleware
<topics-dlmw-robots>`, which holds a request until the ``robots.txt`` file of
its website arrives. Be careful not to introduce deadlocks: a request that you
download must not end up waiting for the request that is waiting for it. Hence
:reqmeta:`dont_obey_robotstxt` above, which makes both middlewares let the token
request through.
While the first token response is in transit, ``process_request`` runs for other
requests as well, and the middleware above downloads a token for each of them.
Cache the task that downloads the token, and not only its result, to download
the token only once.
.. _topics-downloader-middleware-ref:
Built-in downloader middleware reference
@ -169,106 +224,10 @@ middleware, see the :ref:`downloader middleware usage guide
For a list of the components enabled by default (and their orders) see the
:setting:`DOWNLOADER_MIDDLEWARES_BASE` setting.
.. _cookies-mw:
CookiesMiddleware
-----------------
.. module:: scrapy.downloadermiddlewares.cookies
:synopsis: Cookies Downloader Middleware
.. class:: CookiesMiddleware
This middleware enables working with sites that require cookies, such as
those that use sessions. It keeps track of cookies sent by web servers, and
sends them back on subsequent requests (from that spider), just like web
browsers do.
.. caution:: When non-UTF8 encoded byte sequences are passed to a
:class:`~scrapy.Request`, the ``CookiesMiddleware`` will log
a warning. Refer to :ref:`topics-logging-advanced-customization`
to customize the logging behaviour.
.. caution:: Cookies set via the ``Cookie`` header are not considered by the
:ref:`cookies-mw`. If you need to set cookies for a request, use the
:class:`Request.cookies <scrapy.Request>` parameter. This is a known
current limitation that is being worked on.
The following settings can be used to configure the cookie middleware:
* :setting:`COOKIES_ENABLED`
* :setting:`COOKIES_DEBUG`
.. reqmeta:: cookiejar
Multiple cookie sessions per spider
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
There is support for keeping multiple cookie sessions per spider by using the
:reqmeta:`cookiejar` Request meta key. By default it uses a single cookie jar
(session), but you can pass an identifier to use different ones.
For example:
.. skip: next
.. code-block:: python
for i, url in enumerate(urls):
yield scrapy.Request(url, meta={"cookiejar": i}, callback=self.parse_page)
Keep in mind that the :reqmeta:`cookiejar` meta key is not "sticky". You need to keep
passing it along on subsequent requests. For example:
.. code-block:: python
def parse_page(self, response):
# do some processing
return scrapy.Request(
"http://www.example.com/otherpage",
meta={"cookiejar": response.meta["cookiejar"]},
callback=self.parse_other_page,
)
.. setting:: COOKIES_ENABLED
COOKIES_ENABLED
~~~~~~~~~~~~~~~
Default: ``True``
Whether to enable the cookies middleware. If disabled, no cookies will be sent
to web servers.
Notice that despite the value of :setting:`COOKIES_ENABLED` setting if
``Request.``:reqmeta:`meta['dont_merge_cookies'] <dont_merge_cookies>`
evaluates to ``True`` the request cookies will **not** be sent to the
web server and received cookies in :class:`~scrapy.http.Response` will
**not** be merged with the existing cookies.
For more detailed information see the ``cookies`` parameter in
:class:`~scrapy.Request`.
.. setting:: COOKIES_DEBUG
COOKIES_DEBUG
~~~~~~~~~~~~~
Default: ``False``
If enabled, Scrapy will log all cookies sent in requests (i.e. ``Cookie``
header) and all cookies received in responses (i.e. ``Set-Cookie`` header).
Here's an example of a log with :setting:`COOKIES_DEBUG` enabled::
2011-04-06 14:35:10-0300 [scrapy.core.engine] INFO: Spider opened
2011-04-06 14:35:10-0300 [scrapy.downloadermiddlewares.cookies] DEBUG: Sending cookies to: <GET http://www.diningcity.com/netherlands/index.html>
Cookie: clientlanguage_nl=en_EN
2011-04-06 14:35:14-0300 [scrapy.downloadermiddlewares.cookies] DEBUG: Received cookies from: <200 http://www.diningcity.com/netherlands/index.html>
Set-Cookie: JSESSIONID=B~FA4DC0C496C8762AE4F1A620EAB34F38; Path=/
Set-Cookie: ip_isocode=US
Set-Cookie: clientlanguage_nl=en_EN; Expires=Thu, 07-Apr-2011 21:21:34 GMT; Path=/
2011-04-06 14:49:50-0300 [scrapy.core.engine] DEBUG: Crawled (200) <GET http://www.diningcity.com/netherlands/index.html> (referer: None)
[...]
See :ref:`cookies`.
DefaultHeadersMiddleware
@ -773,14 +732,13 @@ HttpCompressionMiddleware
.. class:: HttpCompressionMiddleware
This middleware allows compressed (gzip, deflate) traffic to be
This middleware allows compressed (gzip, deflate, `brotli`_) traffic to be
sent/received from web sites.
This middleware also supports decoding `brotli-compressed`_ responses with
the :ref:`brotli <extras>` extra, and `zstd-compressed`_
responses with the :ref:`zstd <extras>` extra.
This middleware also supports decoding `zstd-compressed`_ responses with
the :ref:`zstd <extras>` extra.
.. _brotli-compressed: https://www.ietf.org/rfc/rfc7932.txt
.. _brotli: https://www.ietf.org/rfc/rfc7932.txt
.. _zstd-compressed: https://www.ietf.org/rfc/rfc8478.txt
@ -873,40 +831,9 @@ OffsiteMiddleware
.. module:: scrapy.downloadermiddlewares.offsite
:synopsis: Offsite Middleware
.. class:: OffsiteMiddleware
.. autoclass:: OffsiteMiddleware
.. versionadded:: 2.11.2
Filters out Requests for URLs outside the domains covered by the spider.
This middleware filters out every request whose host names aren't in the
spider's :attr:`~scrapy.Spider.allowed_domains` attribute.
All subdomains of any domain in the list are also allowed.
E.g. the rule ``www.example.org`` will also allow ``bob.www.example.org``
but not ``www2.example.com`` nor ``example.com``.
When your spider returns a request for a domain not belonging to those
covered by the spider, this middleware will log a debug message similar to
this one::
DEBUG: Filtered offsite request to 'offsite.example': <GET http://offsite.example/some/page.html>
To avoid filling the log with too much noise, it will only print one of
these messages for each new domain filtered. So, for example, if another
request for ``offsite.example`` is filtered, no log message will be
printed. But if a request for ``other.example`` is filtered, a message
will be printed (but only for the first request filtered).
If the spider doesn't define an
:attr:`~scrapy.Spider.allowed_domains` attribute, or the
attribute is empty, the offsite middleware will allow all requests.
.. reqmeta:: allow_offsite
If the request has the :attr:`~scrapy.Request.dont_filter` attribute set to
``True`` or :attr:`Request.meta <scrapy.Request.meta>` has ``allow_offsite``
set to ``True``, then the OffsiteMiddleware will allow the request even if
its domain is not listed in allowed domains.
.. automethod:: should_follow
RedirectMiddleware
------------------

View File

@ -136,6 +136,70 @@ Example:
return f"$ {str(value)}"
return super().serialize_field(field, name, value)
.. _custom-exporters:
Writing your own item exporter
==============================
To write an item exporter, subclass :class:`BaseItemExporter` and implement
:meth:`~BaseItemExporter.export_item`, where
:meth:`~BaseItemExporter.get_serialized_fields` gives you the ``(name, value)``
pairs to export.
To make your exporter available to the :ref:`feed exports
<topics-feed-exports>`, list it in the :setting:`FEED_EXPORTERS` setting. Feed
exports :ref:`build <from-crawler>` it with the output file as the first
positional argument, and with the ``fields``, ``encoding`` and ``indent``
:ref:`feed options <feed-options>` and every key of ``item_export_kwargs`` as
keyword arguments, so your ``__init__`` method must forward unknown keyword
arguments to :class:`BaseItemExporter`.
The file object belongs to whoever opened it, i.e. to the feed storage in the
case of feed exports, which also closes it. If you need a text file, for
example to use :func:`csv.writer` or another Python API that does not accept a
binary file, wrap it with :class:`io.TextIOWrapper` and call
:meth:`~io.TextIOBase.detach` on the wrapper in
:meth:`~BaseItemExporter.finish_exporting`; otherwise the wrapper closes the
underlying file when it is garbage-collected.
For example, the following item exporter writes items as blocks of
``name: value`` lines:
.. code-block:: python
from io import TextIOWrapper
from scrapy.exporters import BaseItemExporter
class TextItemExporter(BaseItemExporter):
def __init__(self, file, item_separator="\n", **kwargs):
super().__init__(**kwargs)
self.item_separator = item_separator
self.stream = TextIOWrapper(
file, encoding=self.encoding or "utf-8", write_through=True
)
def export_item(self, item):
for name, value in self.get_serialized_fields(item):
print(f"{name}: {value}", file=self.stream)
self.stream.write(self.item_separator)
def finish_exporting(self):
self.stream.detach()
To use it as the ``txt`` feed format:
.. code-block:: python
FEED_EXPORTERS = {"txt": "myproject.exporters.TextItemExporter"}
FEEDS = {
"items.txt": {
"format": "txt",
"item_export_kwargs": {"item_separator": "---\n"},
},
}
.. _topics-exporters-reference:
Built-in Item Exporters reference
@ -168,6 +232,8 @@ BaseItemExporter
Exports the given item. This method must be implemented in subclasses.
.. automethod:: BaseItemExporter.get_serialized_fields
.. method:: serialize_field(field, name, value)
Return the serialized value for the given field. You can override this

View File

@ -374,8 +374,8 @@ This extension periodically logs rich stat data as a JSON object::
"elapsed": 360.008903,
"log_interval": 60.0,
"log_interval_real": 60.006694,
"start_time": "2023-08-03 23:24:57",
"utcnow": "2023-08-03 23:30:57"
"start_time": "2023-08-03T23:24:57.148903+00:00",
"utcnow": "2023-08-03T23:30:57.157806+00:00"
}
}

View File

@ -104,7 +104,8 @@ storage backend types which are defined by the URI scheme.
The storages backends supported out of the box are:
- :ref:`topics-feed-storage-fs`
- :ref:`topics-feed-storage-ftp`
- :ref:`feed-storage-ftp`
- :ref:`feed-storage-ftps`
- :ref:`topics-feed-storage-s3` (requires the :ref:`s3 <extras>` extra)
- :ref:`topics-feed-storage-gcs` (requires the :ref:`gcs <extras>` extra)
- :ref:`topics-feed-storage-stdout`
@ -168,6 +169,7 @@ you specify a path (e.g. ``/tmp/export.csv``).
Alternatively you can also use a :class:`pathlib.Path` object.
.. _topics-feed-storage-ftp:
.. _feed-storage-ftp:
FTP
---
@ -178,6 +180,9 @@ The feeds are stored in a FTP server.
- Example URI: ``ftp://user:pass@ftp.example.com/path/to/export.csv``
- Required external libraries: none
FTP sends credentials and data in cleartext. Use :ref:`feed-storage-ftps`
instead where possible.
FTP supports two different connection modes: `active or passive
<https://stackoverflow.com/a/1699163>`_. Scrapy uses the passive connection
mode by default. To use the active connection mode instead, set the
@ -192,6 +197,28 @@ storage backend is: ``True``.
This storage backend uses :ref:`delayed file delivery <delayed-file-delivery>`.
.. _feed-storage-ftps:
FTPS
----
The feeds are stored in a FTP server, over a TLS connection, with the
certificate of the server verified.
.. versionadded:: VERSION
- URI scheme: ``ftps``
- Example URI: ``ftps://user:pass@ftp.example.com/path/to/export.csv``
- Required external libraries: none
See :ref:`feed-storage-ftp` for connection modes, the ``overwrite`` default and
file delivery.
.. note:: For SFTP, an unrelated protocol built on SSH, use
`scrapy-feedexporter-sftp
<https://github.com/scrapy-plugins/scrapy-feedexporter-sftp>`_.
.. _topics-feed-storage-s3:
S3
@ -502,7 +529,7 @@ as a fallback value if that key is not provided for a specific feed definition:
- :ref:`topics-feed-storage-fs`: ``False``
- :ref:`topics-feed-storage-ftp`: ``True``
- :ref:`feed-storage-ftp` and :ref:`feed-storage-ftps`: ``True``
.. note:: Some FTP servers may not support appending to files (the
``APPE`` FTP command).
@ -624,6 +651,7 @@ Default:
"s3": "scrapy.extensions.feedexport.S3FeedStorage",
"gs": "scrapy.extensions.feedexport.GCSFeedStorage",
"ftp": "scrapy.extensions.feedexport.FTPFeedStorage",
"ftps": "scrapy.extensions.feedexport.FTPFeedStorage",
}
A dict containing the built-in feed storage backends supported by Scrapy. You

View File

@ -47,6 +47,13 @@ Additionally, they may also implement the following methods:
This method is called when the spider is opened.
.. versionchanged:: VERSION
Added support for :exc:`~scrapy.exceptions.CloseSpider`.
It may raise :exc:`~scrapy.exceptions.CloseSpider` to close the spider before
it starts crawling, e.g. if a resource that the pipeline needs is
unavailable.
.. method:: close_spider(self)
This method is called when the spider is closed, before the

View File

@ -178,6 +178,37 @@ By overriding ``file_path`` like this:
For more information about the ``file_path`` method, see :ref:`topics-media-pipeline-override`.
.. _file-naming-response:
Naming files after the response
-------------------------------
``file_path`` also receives the ``response``, which allows naming files after
response data. For example, to determine the file extension from the
``Content-Type`` header, for URLs that do not end in a file name:
.. code-block:: python
import mimetypes
from scrapy.pipelines.files import FilesPipeline
class ContentTypeFilesPipeline(FilesPipeline):
def file_path(self, request, response=None, info=None, *, item=None):
path = super().file_path(request, response, info, item=item)
if response is None:
return path
content_type = response.headers["Content-Type"].decode()
return path + (mimetypes.guess_extension(content_type) or "")
This requires setting :setting:`FILES_EXPIRES` to ``0``. To find out whether a
file has already been downloaded, Scrapy calls ``file_path`` before the
download, with ``response`` set to ``None``, and checks the age of the file at
the resulting path. A path that depends on the response can never match that
check, and :setting:`FILES_EXPIRES` set to ``0`` disables it, at the cost of
downloading every file on every run.
.. _topics-supported-storage:
Supported Storage
@ -543,7 +574,7 @@ See here the methods that you can override in your custom Files Pipeline:
return "files/" + PurePosixPath(urlparse_cached(request).path).name
Similarly, you can use the ``item`` to determine the file path based on some item
property.
property, or the ``response``, see :ref:`file-naming-response`.
By default the :meth:`file_path` method returns
``full/<request URL hash>.<extension>``.
@ -693,7 +724,7 @@ See here the methods that you can override in your custom Images Pipeline:
return "files/" + PurePosixPath(urlparse_cached(request).path).name
Similarly, you can use the ``item`` to determine the file path based on some item
property.
property, or the ``response``, see :ref:`file-naming-response`.
By default the :meth:`file_path` method returns
``full/<request URL hash>.<extension>``.

353
docs/topics/optimize.rst Normal file
View File

@ -0,0 +1,353 @@
.. _optimize:
============
Optimization
============
A crawl goes as fast as its slowest part allows. :ref:`Find out which part that
is <optimize-bottleneck>` before changing any setting.
:ref:`Broad crawls <broad-crawls>` have their own set of recommended
adjustments.
.. _optimize-bottleneck:
Finding the bottleneck
======================
The bottleneck depends on the spider: on the same machine, one crawl can be
limited by its own parsing code and another by the target website. So measure
the crawl that you want to optimize.
:class:`~scrapy.extensions.logstats.LogStats` reports crawl speed every
:setting:`LOGSTATS_INTERVAL` seconds:
.. code-block:: text
[scrapy.extensions.logstats] INFO: Crawled 1200 pages (at 60 pages/min), scraped 1150 items (at 58 items/min)
A rate that stays flat as you raise :setting:`CONCURRENT_REQUESTS` means
something else is the limit.
Reading the engine status
-------------------------
The :ref:`telnet console <topics-telnetconsole>` reports, through ``est()``,
what every part of the engine is doing at a given moment:
.. code-block:: text
len(engine.downloader.active) : 16
len(engine._slot.scheduler.mqs) : 92
len(engine.scraper.slot.active) : 0
engine.scraper.slot.active_size : 0
engine.scraper.slot.needs_backout() : False
Take a few readings at different points of the crawl:
- ``len(engine.downloader.active)`` stays at :setting:`CONCURRENT_REQUESTS`:
the downloader is the limit. You are waiting on the network or on the
target website. See :ref:`optimize-concurrency`.
- ``len(engine.downloader.active)`` stays below
:setting:`CONCURRENT_REQUESTS` while the scheduler queues (``mqs``,
``dqs``) hold requests: something throttles those requests before they
reach the downloader, usually :setting:`CONCURRENT_REQUESTS_PER_DOMAIN`,
:setting:`DOWNLOAD_DELAY` or :ref:`AutoThrottle <topics-autothrottle>`.
- Both the downloader and the scheduler queues stay near empty: your spider
is not producing requests fast enough. A crawl that walks pagination one
page at a time cannot use more concurrency than it creates. See
:ref:`optimize-requests`.
- ``needs_backout()`` is ``True``, or ``active_size`` approaches
:setting:`SCRAPER_SLOT_MAX_ACTIVE_SIZE`: responses arrive faster than your
callbacks and :ref:`item pipelines <topics-item-pipeline>` handle them. The
bottleneck is your own code.
- ``len(engine._slot.scheduler.mqs)`` grows without settling: the crawl
discovers requests faster than it downloads them. This is what makes long
crawls run out of memory.
Reading resource usage
----------------------
CPU
Scrapy runs in a single process, and everything except DNS resolution and
code you explicitly move to a thread runs in a single thread. One CPU core
is the ceiling; a process sitting at 100% of a core is CPU-bound no matter
how many cores the machine has.
Use a sampling profiler, such as py-spy_, to find out which code is
spending that CPU. :ref:`Selectors <topics-selectors>` and item pipelines
are the usual answer.
.. _py-spy: https://github.com/benfred/py-spy
Memory
The :ref:`memory usage extension <topics-extensions-ref-memusage>` records
:stat:`memusage/startup` and :stat:`memusage/max`. A :stat:`memusage/max`
far above :stat:`memusage/startup` is expected; what matters is whether it
keeps growing for as long as the crawl runs.
Growth that tracks ``len(engine._slot.scheduler.mqs)`` is a scheduling
problem, covered in :ref:`optimize-memory`. Growth that does not is a
:ref:`memory leak <topics-leaks>`.
Network
Compare :stat:`downloader/response_bytes` over the crawl time against your
available bandwidth. Saturated bandwidth caps concurrency regardless of any
setting.
DNS resolution is separate: it runs on a thread pool of
:setting:`REACTOR_THREADPOOL_MAXSIZE` threads, and results are cached
(:setting:`DNSCACHE_ENABLED`, :setting:`DNSCACHE_SIZE`). It only becomes a
limit of its own when there are many different domains to resolve, as in
:ref:`broad crawls <broad-crawls>`, where it shows up as slow starts and
DNS timeouts.
Disk
:ref:`Feed exports <topics-feed-exports>` write to disk on most crawls,
although item data is usually small enough for that not to matter. The ones
to suspect are
:class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware` and
the :ref:`media pipelines <topics-media-pipeline>`, which write whole
responses, and :setting:`JOBDIR`, which writes every scheduled request.
.. _optimize-concurrency:
Sending more requests at a time
===============================
:setting:`CONCURRENT_REQUESTS` caps how many requests are being downloaded at
any given moment, :setting:`CONCURRENT_REQUESTS_PER_DOMAIN` caps how many of
those may target the same domain, and :setting:`DOWNLOAD_DELAY` sets a minimum
wait between two consecutive requests to the same domain. A project generated by
:command:`startproject` gets one request per second per domain out of these.
Raise them to crawl a single website faster, and see
:ref:`broad-crawls-concurrency` to spread requests across many websites
instead.
The limit that matters, though, is the one the target website tolerates.
Exceeding it gets you throttled, served errors or banned, all of which make the
crawl slower than a lower concurrency would have been. To find that limit:
- Read the :ref:`robots.txt <topics-dlmw-robots>` file of the website. Scrapy
does not act on its ``Crawl-delay`` and ``Request-rate`` directives, so when
they are present, translate them into :setting:`DOWNLOAD_DELAY` and
concurrency settings yourself.
- Check the traffic that the website already gets, using a service like
`SimilarWeb`_ or `Cloudflare Radar`_. A rate that is a rounding error next
to what the website serves anyway is unlikely to be a problem for it.
.. _SimilarWeb: https://www.similarweb.com/
.. _Cloudflare Radar: https://radar.cloudflare.com/
- Look for a documented way in. An API, a bulk export or a search endpoint is
both faster for you and cheaper for the website than crawling its pages, and
the terms of service may state a rate.
- Crawl when the website is idle, in its own timezone, so that the capacity
you take is capacity nobody else wanted.
- Raise concurrency gradually and watch the website respond.
:stat:`downloader/response_status_count/{status_code}` counts for 429, 503
or the ban page of the website, growing :stat:`retry/count`, or a
:ref:`download latency <download-latency>` that climbs as you push harder,
all mean you have gone past the limit.
.. _optimize-requests:
Producing requests faster
=========================
A spider that discovers its requests one response at a time keeps the
downloader idle no matter how high you set :setting:`CONCURRENT_REQUESTS`. To
put more requests in the scheduler earlier:
- Request every page at once when you can work out how many there are, e.g.
from a page count or from a result count and a page size in the first
response, instead of following a link to the next page on every response.
- Get URLs from a source that lists many of them at once, such as a sitemap
or a search or export endpoint of the target website. For a crawl that
needs nothing else, :class:`~scrapy.spiders.SitemapSpider` reads sitemaps
for you.
- Raise the :attr:`~scrapy.Request.priority` of pagination requests, so that
they are downloaded before the requests that they compete with, and
discover the rest of the crawl sooner.
Each of these trades memory for speed: a request produced before the downloader
can take it waits in the scheduler, or on disk if you set :setting:`JOBDIR`.
Pushed far enough, they turn memory or disk into your new bottleneck, which is
why :ref:`optimize-memory` recommends the reverse of the last point.
.. _optimize-resources:
Lowering resource usage
=======================
.. _optimize-memory:
Lowering memory usage
---------------------
- Lower :setting:`SCRAPER_SLOT_MAX_ACTIVE_SIZE`.
- Lower :setting:`DOWNLOAD_MAXSIZE`, which allows a single response to take up
to 1 GiB of memory by default, multiplied by your concurrency. Set
:setting:`DOWNLOAD_WARNSIZE` first to find out whether the website actually
serves responses that big.
- Lower the number of :ref:`scheduled requests <topics-scheduler>` held in
memory:
- Increase the :attr:`~scrapy.Request.priority` of requests whose
:attr:`~scrapy.Request.callback` cannot yield additional requests.
For example, the following spider uses a higher priority (1) for book
requests than for pagination requests:
.. code-block:: python
from scrapy import Spider
class BooksToScrapeComSpider(Spider):
name = "books_toscrape_com"
start_urls = [
"http://books.toscrape.com/catalogue/category/books/mystery_3/index.html"
]
def parse(self, response):
next_page_links = response.css(".next a")
yield from response.follow_all(next_page_links)
book_links = response.css("article a")
yield from response.follow_all(book_links, callback=self.parse_book, priority=1)
def parse_book(self, response):
yield {
"name": response.css("h1::text").get(),
"price": response.css(".price_color::text").re_first("£(.*)"),
"url": response.url,
}
.. note:: If the number of request-yielding, low-priority requests
scheduled at any given time is lower than concurrency settings
(:setting:`CONCURRENT_REQUESTS_PER_DOMAIN` or
:setting:`CONCURRENT_REQUESTS`), as in the example above, this can
slow down your crawl by turning those requests into a bottleneck.
- If you have many :ref:`start requests <start-requests>`, consider
:ref:`delaying their iteration <start-requests-lazy>`.
- Set :setting:`JOBDIR` to offload all scheduled requests to disk.
- Be on the lookout for :ref:`memory leaks <topics-leaks>`.
Lowering network usage
----------------------
- Install brotli_ and zstandard_ to support brotli-compressed_ and
zstd-compressed_ responses.
.. _brotli-compressed: https://www.ietf.org/rfc/rfc7932.txt
.. _brotli: https://pypi.org/project/Brotli/
.. _zstd-compressed: https://www.ietf.org/rfc/rfc8478.txt
.. _zstandard: https://pypi.org/project/zstandard/
- Enable :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`
while developing your spider, so that re-runs do not download the same
responses again.
Lowering CPU usage
------------------
- Set :setting:`LOG_LEVEL` to ``"INFO"`` or higher.
- Restrict what you parse. A :ref:`selector <topics-selectors>` over a
smaller part of the response, or a single query whose result you reuse,
beats repeated queries over the whole document.
Other tips
----------
- Try :ref:`using the asyncio reactor <install-asyncio>` with uvloop_ as
:ref:`custom event loop <using-custom-loops>`, i.e. setting
:setting:`ASYNCIO_EVENT_LOOP` to ``"uvloop.Loop"``.
.. _uvloop: https://github.com/MagicStack/uvloop
Alternatively, try :ref:`switching to a non-asyncio reactor
<disable-asyncio>`.
- Disable unused :ref:`components <topics-components>`.
For example, set :setting:`COOKIES_ENABLED` to ``False`` unless you need
cookies.
- Split the crawl across separate processes to use more than one CPU core.
See :ref:`distributed-crawls`.
.. _broad-crawls:
.. _topics-broad-crawls:
Speeding up broad crawls
========================
While Scrapy is well suited for **broad crawls**, i.e. crawls that target many
websites, the default :ref:`settings <topics-settings>` are optimized for
crawls targeting a single website.
For broad crawls, consider these adjustments:
- .. _broad-crawls-concurrency:
Increase the global concurrency:
- Set :setting:`CONCURRENT_REQUESTS` as close to
:setting:`CONCURRENT_REQUESTS_PER_DOMAIN` × [number of target domains]
(e.g. 8 × 10 domains = 80 concurrent requests) as your CPU and memory
allow.
- Increase :setting:`SCRAPER_SLOT_MAX_ACTIVE_SIZE` when increasing
:setting:`CONCURRENT_REQUESTS` stops making a difference.
- .. _broad-crawls-bfo:
If memory is a bottleneck, see if :ref:`crawling in BFO order <bfo>` lowers
memory usage.
- Improve DNS resolution speed:
- Set up your own DNS server, with a local cache and upstream to a `large
DNS server`_, to avoid slowing down your network.
.. _large DNS server: https://en.wikipedia.org/wiki/Public_recursive_name_server#Notable_public_DNS_service_operators
- Increase :setting:`REACTOR_THREADPOOL_MAXSIZE` to the minimum value
that avoids DNS resolution timeouts and makes a noticeable positive
impact in crawl speed.
- Lower the negative impact of some responses:
- Set :setting:`RETRY_ENABLED` to ``False`` or, if you need retries,
consider lowering :setting:`RETRY_TIMES`.
- Lower :setting:`DOWNLOAD_TIMEOUT` to a more reasonable value, to
discard stuck requests more quickly.
- Set :setting:`REDIRECT_ENABLED` to ``False`` unless you want to follow
redirects.

View File

@ -53,65 +53,13 @@ Request objects
``None`` is passed as value, the HTTP header will not be sent at all.
.. caution:: Cookies set via the ``Cookie`` header are not considered by the
:ref:`cookies-mw`. If you need to set cookies for a request, use the
``cookies`` argument. This is a known current limitation that is being
worked on.
:ref:`cookie middleware <cookies>`. If you need to set cookies for a
request, use the ``cookies`` argument.
:type headers: dict
:param cookies: the request cookies. These can be sent in two forms.
.. invisible-code-block: python
from scrapy import Request
1. Using a dict:
.. code-block:: python
request_with_cookies = Request(
url="http://www.example.com",
cookies={"currency": "USD", "country": "UY"},
)
2. Using a list of dicts:
.. code-block:: python
request_with_cookies = Request(
url="https://www.example.com",
cookies=[
{
"name": "currency",
"value": "USD",
"domain": "example.com",
"path": "/currency",
"secure": True,
},
],
)
The latter form allows for customizing the ``domain`` and ``path``
attributes of the cookie. This is only useful if the cookies are saved
for later requests.
.. reqmeta:: dont_merge_cookies
When some site returns cookies (in a response) those are stored in the
cookies for that domain and will be sent again in future requests.
That's the typical behaviour of any regular web browser.
Note that setting the :reqmeta:`dont_merge_cookies` key to ``True`` in
:attr:`request.meta <scrapy.Request.meta>` causes custom cookies to be
ignored.
For more info see :ref:`cookies-mw`.
.. caution:: Cookies set via the ``Cookie`` header are not considered by the
:ref:`cookies-mw`. If you need to set cookies for a request, use the
:class:`scrapy.Request.cookies <scrapy.Request>` parameter. This is a known
current limitation that is being worked on.
:param cookies: the request cookies, as a dict of cookie names and values
or as a list of dicts with a cookie each. See :ref:`cookies`.
:type cookies: dict or list
:param encoding: the encoding of this request (defaults to ``'utf-8'``).

View File

@ -36,6 +36,77 @@ their input in an unsafe way, such as :func:`eval`, :func:`exec`, or
:func:`pickle.loads`, and be careful when writing response data to paths
derived from the response itself.
.. _security-response-size:
Memory use when parsing responses
=================================
Parsing a response with :ref:`selectors <topics-selectors>` builds an in-memory
tree of the whole response body, which takes several times as much memory as
the body itself. Scrapy parses without the size limits that libxml2 applies by
default, so the size of that tree is bound only by the size of the response, as
controlled by :setting:`DOWNLOAD_MAXSIZE` (default: 1 GiB).
XML entities are left unresolved, so the tree stays proportional to the
response body even for input crafted as an `XML bomb
<https://lxml.de/FAQ.html#is-lxml-vulnerable-to-xml-bombs>`_. A server can still
make a crawler allocate a lot of memory by returning a very large response,
though, so if you know the size of the responses you care about, lower the
limit:
.. code-block:: python
DOWNLOAD_MAXSIZE = 32 * 1024 * 1024 # 32 MiB
* **Pro:** a server cannot make the crawler allocate more memory than the limit
allows, whether by returning a large response or by crafting one that is
expensive to parse.
* **Con:** you can no longer scrape sites that legitimately serve responses
above the limit, as those responses are dropped.
.. _security-parser-limits:
Parser limits
-------------
The limits that libxml2 applies by default, such as 256 nesting levels and
10 MB per text node, can be restored by overriding
:attr:`~scrapy.http.TextResponse.selector` in a response subclass and swapping
responses in a :ref:`downloader middleware <topics-downloader-middleware>`:
.. code-block:: python
from functools import cached_property
from scrapy import Selector
from scrapy.http import HtmlResponse
class LimitedHtmlResponse(HtmlResponse):
@cached_property
def selector(self):
return Selector(self, huge_tree=False)
class LimitedParsingMiddleware:
def process_response(self, request, response, spider):
if isinstance(response, HtmlResponse):
return response.replace(cls=LimitedHtmlResponse)
return response
Do the same with :class:`~scrapy.http.XmlResponse` if you also parse XML.
These limits apply per node, so :setting:`DOWNLOAD_MAXSIZE` remains your bound
on total memory: a response made of many small elements is parsed in full and
uses as much memory either way.
* **Pro:** deeply nested responses, and responses with very large individual
nodes, become cheaper to parse.
* **Con:** parsing stops at those limits without raising, so a legitimate page
that exceeds them yields incomplete data and no error.
TLS connections
===============

View File

@ -69,9 +69,10 @@ Example::
precedence and override the project ones.
.. note:: :ref:`Pre-crawler settings <pre-crawler-settings>` cannot be defined
per spider, and :ref:`reactor settings <reactor-settings>` should not have
a different value per spider when :ref:`running multiple spiders in the
same process <run-multiple-spiders>`.
per spider, and :ref:`reactor settings <reactor-settings>` and
:ref:`logging settings <logging-settings>` are subject to restrictions when
:ref:`running multiple spiders in the same process
<run-multiple-spiders>`.
One way to do so is by setting their :attr:`~scrapy.Spider.custom_settings`
attribute:
@ -329,32 +330,41 @@ Reactor settings
**Reactor settings** are settings tied to the :doc:`Twisted reactor
<twisted:core/howto/reactor-basics>`.
These settings can be defined from a spider. However, because only 1 reactor
can be used per process, these settings cannot use a different value per spider
when :ref:`running multiple spiders in the same process
<run-multiple-spiders>`.
Because only 1 reactor can be used per process, these settings cannot use a
different value per spider when :ref:`running multiple spiders in the same
process <run-multiple-spiders>`.
In general, if different spiders define different values, the first defined
value is used. However, if two spiders request a different reactor, an
exception is raised.
These settings are:
These settings are used upon installing the reactor:
- :setting:`ASYNCIO_EVENT_LOOP` (not possible to set per-spider when using
:class:`~scrapy.crawler.AsyncCrawlerProcess`, see below)
- :setting:`TWISTED_REACTOR` (ignored when using
:class:`~scrapy.crawler.AsyncCrawlerProcess`, see below)
They can be :ref:`set from a spider <spider-settings>`, but only the values
from the first spider that runs are used, since that is when the reactor is
installed. If a later spider asks for a different reactor or a different event
loop, an exception is raised. With
:class:`~scrapy.crawler.CrawlerRunner` and
:class:`~scrapy.crawler.AsyncCrawlerRunner` the reactor must be installed
beforehand, so these settings are only used to check that the installed reactor
and event loop match them.
These settings are applied when starting the reactor:
- :setting:`TWISTED_DNS_RESOLVER` and settings used by the corresponding
component, e.g. :setting:`DNSCACHE_ENABLED`, :setting:`DNSCACHE_SIZE`
and :setting:`DNS_TIMEOUT` for the default one.
- :setting:`REACTOR_THREADPOOL_MAXSIZE`
- :setting:`TWISTED_REACTOR` (ignored when using
:class:`~scrapy.crawler.AsyncCrawlerProcess`, see below)
:setting:`ASYNCIO_EVENT_LOOP` and :setting:`TWISTED_REACTOR` are used upon
installing the reactor. The rest of the settings are applied when starting
the reactor.
They are read from the settings of the
:class:`~scrapy.crawler.CrawlerProcess` or
:class:`~scrapy.crawler.AsyncCrawlerProcess` object, so setting them from a
spider or an :ref:`add-on <topics-addons>` has no effect. They are ignored
altogether when using :class:`~scrapy.crawler.CrawlerRunner` or
:class:`~scrapy.crawler.AsyncCrawlerRunner`, which do not start the reactor.
There is an additional restriction for :setting:`TWISTED_REACTOR` and
:setting:`ASYNCIO_EVENT_LOOP` when using
@ -654,9 +664,8 @@ The default headers used for Scrapy HTTP Requests. They're populated in the
:class:`~scrapy.downloadermiddlewares.defaultheaders.DefaultHeadersMiddleware`.
.. caution:: Cookies set via the ``Cookie`` header are not considered by the
:ref:`cookies-mw`. If you need to set cookies for a request, use the
:class:`Request.cookies <scrapy.Request>` parameter. This is a known
current limitation that is being worked on.
:ref:`cookie middleware <cookies>`. If you need to set cookies for a
request, use the :class:`Request.cookies <scrapy.Request>` parameter.
.. caution:: A ``Referer`` header defined here only reaches requests for which
:class:`~scrapy.spidermiddlewares.referer.RefererMiddleware` does not set
@ -1384,7 +1393,7 @@ FEED_TEMPDIR
Default: ``None``
The Feed Temp dir allows you to set a custom folder to save crawler
temporary files before uploading with :ref:`FTP feed storage <topics-feed-storage-ftp>` and
temporary files before uploading with :ref:`FTP feed storage <feed-storage-ftp>` and
:ref:`Amazon S3 <topics-feed-storage-s3>`.
.. setting:: FEED_STORAGE_GCS_ACL
@ -1913,6 +1922,7 @@ Type of in-memory queue used by the scheduler. Other available type is:
.. setting:: SCHEDULER_PRIORITY_QUEUE
.. _broad-crawls-scheduler-priority-queue:
SCHEDULER_PRIORITY_QUEUE
------------------------

View File

@ -290,6 +290,13 @@ spider_opened
reserve per-spider resources, but can be used for any task that needs to be
performed when a spider is opened.
.. versionchanged:: VERSION
Added support for :exc:`~scrapy.exceptions.CloseSpider`.
You may raise a :exc:`~scrapy.exceptions.CloseSpider` exception to close the
spider before it starts crawling, e.g. if a resource that the spider needs
is unavailable.
This signal supports :ref:`asynchronous handlers <signal-deferred>`.
:param spider: the spider which has been opened
@ -338,15 +345,22 @@ spider_error
.. signal:: spider_error
.. function:: spider_error(failure, response, spider)
Sent when a spider callback generates an error (i.e. raises an exception).
Sent when a spider callback or the :meth:`~scrapy.Spider.start` method of a
spider generates an error (i.e. raises an exception).
.. versionchanged:: VERSION
Exceptions from :meth:`~scrapy.Spider.start` are also reported, see
:ref:`start-error`.
This signal does not support :ref:`asynchronous handlers <signal-deferred>`.
:param failure: the exception raised
:type failure: twisted.python.failure.Failure
:param response: the response being processed when the exception was raised
:type response: :class:`~scrapy.http.Response` object
:param response: the response being processed when the exception was
raised, or ``None`` if the exception came from
:meth:`~scrapy.Spider.start`.
:type response: :class:`~scrapy.http.Response` | ``None``
:param spider: the spider which raised the exception
:type spider: :class:`~scrapy.Spider` object

View File

@ -228,25 +228,7 @@ DepthMiddleware
.. module:: scrapy.spidermiddlewares.depth
:synopsis: Depth Spider Middleware
.. class:: DepthMiddleware
DepthMiddleware is used for tracking the depth of each Request inside the
site being scraped. It works by setting ``request.meta['depth'] = 0`` whenever
there is no value previously set (usually just the first Request) and
incrementing it by 1 otherwise.
It can be used to limit the maximum depth to scrape, control Request
priority based on their depth, and things like that.
The :class:`DepthMiddleware` can be configured through the following
settings (see the settings documentation for more info):
* :setting:`DEPTH_LIMIT` - The maximum depth that will be allowed to
crawl for any site. If zero, no limit will be imposed.
* :setting:`DEPTH_STATS_VERBOSE` - Whether to collect the number of
requests for each depth.
* :setting:`DEPTH_PRIORITY` - Whether to prioritize the requests based on
their depth.
.. autoclass:: DepthMiddleware
HttpErrorMiddleware
-------------------

View File

@ -411,6 +411,38 @@ scheduled requests:
await self.crawler.signals.wait_for(signals.scheduler_empty)
yield item_or_request
.. _start-error:
Handling start errors
---------------------
An exception raised by :meth:`~scrapy.Spider.start` ends its iteration, so any
remaining start items and requests are never sent. Scrapy logs the exception,
sends the :signal:`spider_error` signal, and, once the already scheduled
requests are done, closes the spider with the ``start_error``
:stat:`finish_reason`.
.. versionchanged:: VERSION
The close reason used to be ``finished``, and neither the
:signal:`spider_error` signal nor the :stat:`spider_exceptions/count` stat
reported the exception.
To keep the iteration going, catch the exception yourself:
.. code-block:: python
async def start(self):
for url in self.start_urls:
try:
request = Request(url)
except ValueError:
self.logger.exception(f"Skipping start URL {url}")
else:
yield request
To stop the crawl instead, and choose your own :stat:`finish_reason`, raise
:exc:`~scrapy.exceptions.CloseSpider`.
.. _builtin-spiders:
Generic Spiders

View File

@ -301,6 +301,10 @@ one per actual value of the placeholder.
- ``shutdown``: the crawl was interrupted, e.g. by a system signal such
as ``SIGINT`` (:kbd:`Ctrl-C`).
- ``start_error``: :meth:`~scrapy.Spider.start` raised an exception, so
some :ref:`start requests <start-requests>` may never have been sent,
see :ref:`start-error`.
Third-party components and your own code may use any other reason, e.g. by
raising :exc:`~scrapy.exceptions.CloseSpider` with it.
@ -721,18 +725,21 @@ one per actual value of the placeholder.
.. stat:: spider_exceptions/count
``spider_exceptions/count``
Number of unhandled exceptions raised by spider callbacks.
Number of unhandled exceptions raised by spider callbacks or by
:meth:`~scrapy.Spider.start`.
Set by the :ref:`scraper <topics-architecture>`.
Set by the :ref:`engine <topics-architecture>` and the :ref:`scraper
<topics-architecture>`.
.. stat:: spider_exceptions/{exception}
``spider_exceptions/{exception}``
Number of unhandled exceptions raised by spider callbacks, per exception,
where ``{exception}`` is the class name of the exception, e.g.
Same as :stat:`spider_exceptions/count`, per exception, where
``{exception}`` is the class name of the exception, e.g.
``spider_exceptions/ValueError``.
Set by the :ref:`scraper <topics-architecture>`.
Set by the :ref:`engine <topics-architecture>` and the :ref:`scraper
<topics-architecture>`.
.. stat:: start_time

View File

@ -26,6 +26,8 @@ dependencies = [
# Platform-specific dependencies
'PyDispatcher>=2.0.5; platform_python_implementation == "CPython"',
'PyPyDispatcher>=2.1.0; platform_python_implementation == "PyPy"',
'brotli>=1.2.0; implementation_name != "pypy"',
'brotlicffi>=1.2.0.0; implementation_name == "pypy"',
]
classifiers = [
"Development Status :: 5 - Production/Stable",
@ -62,10 +64,6 @@ Tracker = "https://github.com/scrapy/scrapy/issues"
[project.optional-dependencies]
bpython = ["bpython>=0.7.1"]
brotli = [
"brotli>=1.2.0; implementation_name != 'pypy'",
"brotlicffi>=1.2.0.0; implementation_name == 'pypy'",
]
gcs = ["google-cloud-storage>=1.29.0"]
httpx = ["httpx2[http2,socks]>=2.0.0"]
images = ["Pillow>=8.3.2"]
@ -125,23 +123,11 @@ module = [
"tests.test_downloaderslotssettings",
"tests.test_dupefilters",
"tests.test_engine_loop",
"tests.test_exporters",
"tests.test_extension_statsmailer",
"tests.test_extension_throttle",
"tests.test_feedexport",
"tests.test_feedexport_postprocess",
"tests.test_feedexport_storages",
"tests.test_feedexport_uri_params",
"tests.test_item",
"tests.test_linkextractors",
"tests.test_loader",
"tests.test_logformatter",
"tests.test_mail",
"tests.test_pipeline_crawl",
"tests.test_pipeline_files",
"tests.test_pipeline_images",
"tests.test_pipeline_media",
"tests.test_pipelines",
"tests.test_pqueues",
"tests.test_scheduler_base",
"tests.test_settings",
@ -333,6 +319,9 @@ markers = [
]
filterwarnings = [
"ignore::DeprecationWarning:twisted.web.static",
# Jobs that do not report coverage disable it with --no-cov, which pytest-cov
# warns about because the coverage options below stay in place.
"ignore::pytest_cov.CovDisabledWarning",
# Twisted doesn't close failed sockets after CannotListenError: https://github.com/twisted/twisted/issues/6108
"ignore:Exception ignored in. <socket\\.socket.*laddr=..0\\.0\\.0\\.0., 0.:pytest.PytestUnraisableExceptionWarning",
]

View File

@ -281,7 +281,6 @@ class Command(BaseRunSpiderCommand):
) -> list[Any]:
items, requests, opts, depth, spider, callback = args
if opts.pipelines:
assert self.pcrawler.engine
itemproc = self.pcrawler.engine.scraper.itemproc
if hasattr(itemproc, "process_item_async"):
for item in items:

View File

@ -28,6 +28,7 @@ from scrapy.utils.defer import (
maybe_deferred_to_future,
)
from scrapy.utils.httpobj import urlparse_cached
from scrapy.utils.misc import build_from_crawler
if TYPE_CHECKING:
from collections.abc import Generator
@ -99,8 +100,8 @@ class Downloader:
# AUTOTHROTTLE_START_DELAY.
self._delay: float = self.settings.getfloat("DOWNLOAD_DELAY")
self.randomize_delay: bool = self.settings.getbool("RANDOMIZE_DOWNLOAD_DELAY")
self.middleware: DownloaderMiddlewareManager = (
DownloaderMiddlewareManager.from_crawler(crawler)
self.middleware: DownloaderMiddlewareManager = build_from_crawler(
DownloaderMiddlewareManager, crawler
)
self._slot_gc_loop: AsyncioLoopingCall | LoopingCall | None = None
self.per_slot_settings: dict[str, dict[str, Any]] = self.settings.getdict(

View File

@ -17,12 +17,19 @@ from twisted.internet.defer import Deferred, succeed
from twisted.internet.endpoints import TCP4ClientEndpoint
from twisted.internet.protocol import Factory, Protocol, connectionDone
from twisted.python.failure import Failure
from twisted.web._newclient import (
HEADER,
STATUS,
HTTP11ClientProtocol,
HTTPClientParser,
)
from twisted.web.client import (
URI,
Agent,
HTTPConnectionPool,
ResponseDone,
ResponseFailed,
_HTTP11ClientFactory,
)
from twisted.web.client import Response as TxResponse
from twisted.web.http import PotentialDataLoss, _DataLoss
@ -60,7 +67,8 @@ from ._base_http import BaseHttpDownloadHandler
if TYPE_CHECKING:
from twisted.internet.base import ReactorBase
from twisted.internet.interfaces import IConsumer
from twisted.internet.interfaces import IAddress, IConsumer
from twisted.web._newclient import Request as TxRequest
# typing.NotRequired requires Python 3.11
from typing_extensions import NotRequired
@ -95,7 +103,7 @@ class HTTP11DownloadHandler(BaseHttpDownloadHandler):
self._pool.maxPersistentPerHost = crawler.settings.getint(
"CONCURRENT_REQUESTS_PER_DOMAIN"
)
self._pool._factory.noisy = False
self._pool._factory = _LenientHTTP11ClientFactory
self._contextFactory: IPolicyForHTTPS = _load_context_factory_from_settings(
crawler
@ -548,7 +556,8 @@ class _ScrapyAgent:
txresponse._transport._producer.abortConnection()
raise DownloadCancelledError(warning_msg)
if warnsize and expected_size > warnsize:
reached_warnsize = bool(warnsize and expected_size > warnsize)
if reached_warnsize:
logger.warning(
get_warnsize_msg(expected_size, warnsize, request, expected=True)
)
@ -561,6 +570,7 @@ class _ScrapyAgent:
request=request,
maxsize=maxsize,
warnsize=warnsize,
reached_warnsize=reached_warnsize,
fail_on_dataloss=fail_on_dataloss,
crawler=self._crawler,
tls_verbose_logging=self._tls_verbose_logging,
@ -625,6 +635,7 @@ class _ResponseReader(Protocol):
fail_on_dataloss: bool,
crawler: Crawler,
*,
reached_warnsize: bool = False,
tls_verbose_logging: bool = False,
):
self._finished: Deferred[_ResultT] = finished
@ -634,7 +645,7 @@ class _ResponseReader(Protocol):
self._maxsize: int = maxsize
self._warnsize: int = warnsize
self._fail_on_dataloss: bool = fail_on_dataloss
self._reached_warnsize: bool = False
self._reached_warnsize: bool = reached_warnsize
self._bytes_received: int = 0
self._certificate: ssl.Certificate | None = None
self._ip_address: ipaddress.IPv4Address | ipaddress.IPv6Address | None = None
@ -737,3 +748,77 @@ class _ResponseReader(Protocol):
reason = Failure(exc)
self._finished.errback(reason)
class _LenientHTTPClientParser(HTTPClientParser):
"""Response parser that skips bad response header lines, those with no
colon in them, instead of failing to parse the whole response.
Some servers send such lines, and web browsers skip them and keep parsing
the header lines that follow. See
https://github.com/scrapy/scrapy/issues/210.
"""
def lineReceived(self, line: bytes) -> None:
# A copy of twisted.web._newclient.HTTPParser.lineReceived() where the
# header name and value are only extracted from header lines that have
# a colon.
# Handle the normal CR LF case.
if line[-1:] == b"\r":
line = line[:-1]
if self.state == STATUS:
self.statusReceived(line) # type: ignore[no-untyped-call]
self.state = HEADER
return
# HEADER is the only other state in which lines are received, as the
# parser switches to raw mode for the response body.
if not line or line[0] not in b" \t":
if self._partialHeader is not None:
header = b"".join(self._partialHeader)
if b":" in header:
name, value = header.split(b":", 1)
self.headerReceived(name, value.strip()) # type: ignore[no-untyped-call]
else:
logger.debug(
f"Skipping the bad response header line {header!r}, as "
f"it has no colon."
)
if not line:
# Empty line means the header section is over.
self.allHeadersReceived() # type: ignore[no-untyped-call]
else:
# Line not beginning with LWS is another header.
self._partialHeader = [line]
else:
# A line beginning with LWS is a continuation of a header begun on
# a previous line.
self._partialHeader.append(line) # type: ignore[union-attr]
class _LenientHTTP11ClientProtocol(HTTP11ClientProtocol):
"""Protocol that parses responses with :class:`_LenientHTTPClientParser`."""
def request(self, request: TxRequest) -> Deferred[IResponse]:
d: Deferred[IResponse] = super().request(request)
# HTTP11ClientProtocol.request() hardcodes the parser class, so the
# only way to use a different one is to replace the class of the parser
# object that it creates. This is safe because
# _LenientHTTPClientParser defines no additional state. The parser is
# always there because HTTPConnectionPool only reuses connections whose
# protocol is in the QUIESCENT state, for which request() always
# creates a parser.
assert self._parser is not None
self._parser.__class__ = _LenientHTTPClientParser
return d
class _LenientHTTP11ClientFactory(_HTTP11ClientFactory):
"""Factory that builds :class:`_LenientHTTP11ClientProtocol` protocols."""
noisy = False
def buildProtocol(self, addr: IAddress | None) -> HTTP11ClientProtocol:
return _LenientHTTP11ClientProtocol(self._quiescentCallback) # type: ignore[no-untyped-call]

View File

@ -112,7 +112,6 @@ class ExecutionEngine:
self.crawler: Crawler = crawler
self.settings: Settings = crawler.settings
self.signals: SignalManager = crawler.signals
assert crawler.logformatter
self.logformatter: LogFormatter = crawler.logformatter
self._slot: _Slot | None = None
self.spider: Spider | None = None
@ -125,6 +124,9 @@ class ExecutionEngine:
] = spider_closed_callback
self.start_time: float | None = None
self._start: AsyncIterator[Any] | None = None
# Whether Spider.start() raised, i.e. some start items or requests may
# never have reached the engine.
self._start_error: bool = False
self._closewait: Deferred[None] | None = None
self._start_request_processing_awaitable: (
asyncio.Future[None] | Deferred[None] | None
@ -246,7 +248,7 @@ class ExecutionEngine:
)
return deferred_from_coro(self.close_async())
async def close_async(self) -> None:
async def close_async(self, *, reason: str = "shutdown") -> None:
"""
Gracefully close the execution engine.
If it has already been started, stop it. In all cases, close the spider and the downloader.
@ -254,9 +256,7 @@ class ExecutionEngine:
if self.running:
await self.stop_async() # will also close spider and downloader
elif self.spider is not None:
await self.close_spider_async(
reason="shutdown"
) # will also close downloader
await self.close_spider_async(reason=reason) # will also close downloader
elif hasattr(self, "downloader"):
self.downloader.close()
@ -277,13 +277,29 @@ class ExecutionEngine:
item_or_request = await anext(self._start)
except StopAsyncIteration:
self._start = None
except CloseSpider as exception:
self._start = None
_schedule_coro(
self.close_spider_async(reason=exception.reason or "cancelled")
)
except Exception as exception:
self._start = None
self._start_error = True
exception_traceback = format_exc()
logger.error(
f"Error while reading start items and requests: {exception}.\n{exception_traceback}",
exc_info=True,
)
self.signals.send_catch_log(
signal=signals.spider_error,
failure=Failure(),
response=None,
spider=self.spider,
)
self.crawler.stats.inc_value("spider_exceptions/count")
self.crawler.stats.inc_value(
f"spider_exceptions/{type(exception).__name__}"
)
else:
if not self.spider:
return # spider already closed
@ -539,24 +555,39 @@ class ExecutionEngine:
nextcall = CallLaterOnce(self._start_scheduled_requests)
scheduler = build_from_crawler(self.scheduler_cls, self.crawler)
self._slot = _Slot(close_if_idle, nextcall, scheduler)
self._start = await self.scraper.spidermw.process_start()
if hasattr(scheduler, "open") and (d := scheduler.open(self.crawler.spider)):
await maybe_deferred_to_future(d)
await self.scraper.open_spider_async()
assert self.crawler.stats
if argument_is_required(self.crawler.stats.open_spider, "spider"):
# A component that fails to start can ask for the spider to be closed.
# The rest of the startup runs anyway, so that components that are
# started also get stopped, and the request is honored once the spider
# is open.
close_spider_exc: CloseSpider | None = None
try:
self._start = await self.scraper.spidermw.process_start()
if hasattr(scheduler, "open") and (
d := scheduler.open(self.crawler.spider)
):
await maybe_deferred_to_future(d)
await self.scraper.open_spider_async()
except CloseSpider as exc:
close_spider_exc = exc
stats = self.crawler.stats
if argument_is_required(stats.open_spider, "spider"):
warnings.warn(
f"The open_spider() method of {global_object_name(type(self.crawler.stats))} requires a spider argument,"
f"The open_spider() method of {global_object_name(type(stats))} requires a spider argument,"
f" this is deprecated and the argument will not be passed in future Scrapy versions.",
ScrapyDeprecationWarning,
stacklevel=2,
)
self.crawler.stats.open_spider(spider=self.crawler.spider)
stats.open_spider(spider=self.crawler.spider)
else:
self.crawler.stats.open_spider()
await self.signals.send_catch_log_async(
signals.spider_opened, spider=self.crawler.spider
stats.open_spider()
results = await self.signals.send_catch_log_async(
signals.spider_opened, spider=self.crawler.spider, dont_log=CloseSpider
)
for _, result in results:
if isinstance(result, CloseSpider):
close_spider_exc = close_spider_exc or result
if close_spider_exc is not None:
raise close_spider_exc
def _spider_idle(self) -> None:
"""
@ -579,7 +610,8 @@ class ExecutionEngine:
if DontCloseSpider in detected_ex:
return
if self.spider_is_idle():
ex = detected_ex.get(CloseSpider, CloseSpider(reason="finished"))
default_reason = "start_error" if self._start_error else "finished"
ex = detected_ex.get(CloseSpider, CloseSpider(reason=default_reason))
assert isinstance(ex, CloseSpider) # typing
_schedule_coro(self.close_spider_async(reason=ex.reason))
@ -655,20 +687,18 @@ class ExecutionEngine:
extra={"spider": spider},
)
assert self.crawler.stats
try:
if argument_is_required(self.crawler.stats.close_spider, "spider"):
stats = self.crawler.stats
if argument_is_required(stats.close_spider, "spider"):
warnings.warn(
f"The close_spider() method of {global_object_name(type(self.crawler.stats))} requires a spider argument,"
f"The close_spider() method of {global_object_name(type(stats))} requires a spider argument,"
f" this is deprecated and the argument will not be passed in future Scrapy versions.",
ScrapyDeprecationWarning,
stacklevel=2,
)
self.crawler.stats.close_spider(
spider=self.crawler.spider, reason=reason
)
stats.close_spider(spider=self.crawler.spider, reason=reason)
else:
self.crawler.stats.close_spider(reason=reason)
stats.close_spider(reason=reason)
except Exception:
logger.error("Stats close failure")

View File

@ -36,7 +36,11 @@ from scrapy.utils.defer import (
)
from scrapy.utils.deprecate import method_is_overridden
from scrapy.utils.log import failure_to_exc_info, logformatter_adapter
from scrapy.utils.misc import load_object, warn_on_generator_with_return_value
from scrapy.utils.misc import (
build_from_crawler,
load_object,
warn_on_generator_with_return_value,
)
from scrapy.utils.python import global_object_name
from scrapy.utils.spider import iterate_spider_output
@ -102,13 +106,13 @@ class Slot:
class Scraper:
def __init__(self, crawler: Crawler) -> None:
self.slot: Slot | None = None
self.spidermw: SpiderMiddlewareManager = SpiderMiddlewareManager.from_crawler(
crawler
self.spidermw: SpiderMiddlewareManager = build_from_crawler(
SpiderMiddlewareManager, crawler
)
itemproc_cls: type[ItemPipelineManager] = load_object(
crawler.settings["ITEM_PROCESSOR"]
)
self.itemproc: ItemPipelineManager = itemproc_cls.from_crawler(crawler)
self.itemproc: ItemPipelineManager = build_from_crawler(itemproc_cls, crawler)
self._itemproc_has_async: dict[str, bool] = {}
for method in [
"open_spider",
@ -120,7 +124,6 @@ class Scraper:
self.concurrent_items: int = crawler.settings.getint("CONCURRENT_ITEMS")
self.crawler: Crawler = crawler
self.signals: SignalManager = crawler.signals
assert crawler.logformatter
self.logformatter: LogFormatter = crawler.logformatter
def _check_deprecated_itemproc_method(self, method: str) -> None:
@ -355,7 +358,6 @@ class Scraper:
assert self.crawler.spider
exc = _failure.value
if isinstance(exc, CloseSpider):
assert self.crawler.engine is not None # typing
_schedule_coro(
self.crawler.engine.close_spider_async(reason=exc.reason or "cancelled")
)
@ -374,11 +376,9 @@ class Scraper:
response=response,
spider=self.crawler.spider,
)
assert self.crawler.stats
self.crawler.stats.inc_value("spider_exceptions/count")
self.crawler.stats.inc_value(
f"spider_exceptions/{_failure.value.__class__.__name__}"
)
stats = self.crawler.stats
stats.inc_value("spider_exceptions/count")
stats.inc_value(f"spider_exceptions/{_failure.value.__class__.__name__}")
def handle_spider_output(
self,
@ -456,7 +456,6 @@ class Scraper:
Items are sent to the item pipelines, requests are scheduled.
"""
if isinstance(output, Request):
assert self.crawler.engine is not None # typing
self.crawler.engine.crawl(request=output)
return
if output is not None:

View File

@ -8,14 +8,14 @@ import signal
import warnings
from abc import ABC, abstractmethod
from functools import partial
from typing import TYPE_CHECKING, Any, TypeVar
from typing import TYPE_CHECKING, Any, Generic, TypeVar, overload
from twisted.internet.defer import Deferred, DeferredList, inlineCallbacks
from scrapy import Spider
from scrapy.addons import AddonManager
from scrapy.core.engine import ExecutionEngine
from scrapy.exceptions import ScrapyDeprecationWarning
from scrapy.exceptions import CloseSpider, ScrapyDeprecationWarning
from scrapy.extension import ExtensionManager
from scrapy.settings import SETTINGS_PRIORITIES, Settings, overridden_settings
from scrapy.signalmanager import SignalManager
@ -58,7 +58,55 @@ logger = logging.getLogger(__name__)
_T = TypeVar("_T")
class _LateAttribute(Generic[_T]):
"""Descriptor for a :class:`Crawler` attribute that only gets a value once
the crawl starts.
The value is kept in an attribute of the same name prefixed with an
underscore, and reading it before it is set raises :exc:`RuntimeError`.
This way the public attribute can be annotated as always set, and its
users, both in Scrapy and in third-party code, do not need to narrow its
type on every use. Code that runs before the crawl starts reads the
underscore-prefixed attribute instead.
"""
def __set_name__(self, owner: type[Crawler], name: str) -> None:
self._name = name
self._private_name = f"_{name}"
@overload
def __get__(self, instance: None, owner: type[Crawler]) -> _LateAttribute[_T]: ...
@overload
def __get__(self, instance: Crawler, owner: type[Crawler]) -> _T: ...
def __get__(
self, instance: Crawler | None, owner: type[Crawler]
) -> _LateAttribute[_T] | _T:
if instance is None:
return self
value: _T | None = getattr(instance, self._private_name)
if value is None:
raise RuntimeError(
f"Crawler.{self._name} is not set yet. It is set when the "
"crawl starts, so it can only be used from then on, e.g. "
"from the spider_opened signal handler onwards."
)
return value
def __set__(self, instance: Crawler, value: _T) -> None:
setattr(instance, self._private_name, value)
class Crawler:
engine: _LateAttribute[ExecutionEngine] = _LateAttribute()
extensions: _LateAttribute[ExtensionManager] = _LateAttribute()
logformatter: _LateAttribute[LogFormatter] = _LateAttribute()
request_fingerprinter: _LateAttribute[RequestFingerprinterProtocol] = (
_LateAttribute()
)
stats: _LateAttribute[StatsCollector] = _LateAttribute()
def __init__(
self,
spidercls: type[Spider],
@ -83,12 +131,13 @@ class Crawler:
self.crawling: bool = False
self._started: bool = False
self.extensions: ExtensionManager | None = None
self.stats: StatsCollector | None = None
self.logformatter: LogFormatter | None = None
self.request_fingerprinter: RequestFingerprinterProtocol | None = None
self.spider: Spider | None = None
self.engine: ExecutionEngine | None = None
self._engine: ExecutionEngine | None = None
self._extensions: ExtensionManager | None = None
self._logformatter: LogFormatter | None = None
self._request_fingerprinter: RequestFingerprinterProtocol | None = None
self._stats: StatsCollector | None = None
def _update_root_log_handler(self) -> None:
if get_scrapy_root_handler() is not None:
@ -107,7 +156,7 @@ class Crawler:
self.stats = load_object(self.settings["STATS_CLASS"])(self)
lf_cls: type[LogFormatter] = load_object(self.settings["LOG_FORMATTER"])
self.logformatter = lf_cls.from_crawler(self)
self.logformatter = build_from_crawler(lf_cls, self)
self.request_fingerprinter = build_from_crawler(
load_object(self.settings["REQUEST_FINGERPRINTER_CLASS"]),
@ -151,7 +200,7 @@ class Crawler:
logger.debug("Not using a Twisted reactor")
self._apply_reactorless_default_settings()
self.extensions = ExtensionManager.from_crawler(self)
self.extensions = build_from_crawler(ExtensionManager, self)
self.settings.freeze()
d = dict(overridden_settings(self.settings))
@ -221,12 +270,16 @@ class Crawler:
self._apply_settings()
self._update_root_log_handler()
self.engine = self._create_engine()
yield deferred_from_coro(self.engine.open_spider_async())
yield deferred_from_coro(self.engine.start_async())
try:
yield deferred_from_coro(self.engine.open_spider_async())
except CloseSpider as exc:
yield deferred_from_coro(self.engine.close_async(reason=exc.reason))
else:
yield deferred_from_coro(self.engine.start_async())
except Exception:
self.crawling = False
if self.engine is not None:
yield deferred_from_coro(self.engine.close_async())
if self._engine is not None:
yield deferred_from_coro(self._engine.close_async())
raise
async def crawl_async(self, *args: Any, **kwargs: Any) -> None:
@ -251,12 +304,16 @@ class Crawler:
self._apply_settings()
self._update_root_log_handler()
self.engine = self._create_engine()
await self.engine.open_spider_async()
await self.engine.start_async()
try:
await self.engine.open_spider_async()
except CloseSpider as exc:
await self.engine.close_async(reason=exc.reason)
else:
await self.engine.start_async()
except Exception:
self.crawling = False
if self.engine is not None:
await self.engine.close_async()
if self._engine is not None:
await self._engine.close_async()
raise
def _create_spider(self, *args: Any, **kwargs: Any) -> Spider:
@ -282,7 +339,6 @@ class Crawler:
"""
if self.crawling:
self.crawling = False
assert self.engine
if self.engine.running:
await self.engine.stop_async()
@ -313,7 +369,7 @@ class Crawler:
This method can only be called after the crawl engine has been created,
e.g. at signals :signal:`engine_started` or :signal:`spider_opened`.
"""
if not self.engine:
if self._engine is None:
raise RuntimeError(
"Crawler.get_downloader_middleware() can only be called after "
"the crawl engine has been created."
@ -331,7 +387,7 @@ class Crawler:
created, e.g. at signals :signal:`engine_started` or
:signal:`spider_opened`.
"""
if not self.extensions:
if self._extensions is None:
raise RuntimeError(
"Crawler.get_extension() can only be called after the "
"extension manager has been created."
@ -348,7 +404,7 @@ class Crawler:
This method can only be called after the crawl engine has been created,
e.g. at signals :signal:`engine_started` or :signal:`spider_opened`.
"""
if not self.engine:
if self._engine is None:
raise RuntimeError(
"Crawler.get_item_pipeline() can only be called after the "
"crawl engine has been created."
@ -365,7 +421,7 @@ class Crawler:
This method can only be called after the crawl engine has been created,
e.g. at signals :signal:`engine_started` or :signal:`spider_opened`.
"""
if not self.engine:
if self._engine is None:
raise RuntimeError(
"Crawler.get_spider_middleware() can only be called after the "
"crawl engine has been created."

View File

@ -55,7 +55,6 @@ class HttpCacheMiddleware:
@classmethod
def from_crawler(cls, crawler: Crawler) -> Self:
assert crawler.stats
o = cls(crawler.settings, crawler.stats)
crawler.signals.connect(o.spider_opened, signal=signals.spider_opened)
crawler.signals.connect(o.spider_closed, signal=signals.spider_closed)

View File

@ -30,27 +30,7 @@ if TYPE_CHECKING:
logger = getLogger(__name__)
ACCEPTED_ENCODINGS: list[bytes] = [b"gzip", b"deflate"]
try:
try:
import brotli
except ImportError:
import brotlicffi as brotli
except ImportError:
pass
else:
try:
brotli.Decompressor.can_accept_more_data # noqa: B018
except AttributeError: # pragma: no cover
warnings.warn(
"You have brotli installed. But 'br' encoding support now requires "
"brotli's or brotlicffi's version >= 1.2.0. Please upgrade "
"brotli/brotlicffi to make Scrapy decode 'br' encoded responses.",
stacklevel=2,
)
else:
ACCEPTED_ENCODINGS.append(b"br")
ACCEPTED_ENCODINGS: list[bytes] = [b"gzip", b"deflate", b"br"]
if find_spec("zstandard") is not None:
ACCEPTED_ENCODINGS.append(b"zstd")
@ -205,8 +185,6 @@ class HttpCompressionMiddleware:
f"{self.__class__.__name__} cannot decode the response for {response.url} "
f"from unsupported encoding(s) '{encodings_str}'."
)
if b"br" in encodings:
msg += " You need to install brotli or brotlicffi >= 1.2.0 to decode 'br'."
if b"zstd" in encodings:
msg += " You need to install zstandard to decode 'zstd'."
logger.warning(msg)

View File

@ -21,6 +21,36 @@ logger = logging.getLogger(__name__)
class OffsiteMiddleware:
"""Filter out requests for URLs outside the domains covered by the spider.
.. versionadded:: 2.11.2
A request is allowed if its host name is in the
:attr:`~scrapy.Spider.allowed_domains` attribute of the spider, or is a
subdomain of one of those domains. E.g. ``www.example.org`` also allows
``bob.www.example.org``, but neither ``www2.example.org`` nor
``example.org``. See :meth:`should_follow` to use a different policy.
If the spider does not define :attr:`~scrapy.Spider.allowed_domains`, or
the attribute is empty, every request is allowed.
Filtered requests are logged as follows::
DEBUG: Filtered offsite request to 'offsite.example': <GET http://offsite.example/some/page.html>
Only the first request filtered for a given domain is logged, to keep the
log readable.
.. reqmeta:: allow_offsite
allow_offsite
-------------
Requests with the ``allow_offsite`` :attr:`~scrapy.Request.meta` key set to
``True``, or with :attr:`~scrapy.Request.dont_filter` set to ``True``, are
allowed regardless of their host name.
"""
crawler: Crawler
host_regex: re.Pattern[str]
@ -31,7 +61,6 @@ class OffsiteMiddleware:
@classmethod
def from_crawler(cls, crawler: Crawler) -> Self:
assert crawler.stats
o = cls(crawler.stats)
crawler.signals.connect(o.spider_opened, signal=signals.spider_opened)
crawler.signals.connect(o.request_scheduled, signal=signals.request_scheduled)
@ -72,6 +101,23 @@ class OffsiteMiddleware:
raise IgnoreRequest(f"Filtered offsite request to {domain!r}")
def should_follow(self, request: Request, spider: Spider) -> bool:
"""Return ``True`` if *request* is on site, ``False`` if it must be
filtered out.
Override this method to implement a different offsite policy. For
example, to allow the domains in
:attr:`~scrapy.Spider.allowed_domains` but none of their subdomains:
.. code-block:: python
from scrapy.downloadermiddlewares.offsite import OffsiteMiddleware
from scrapy.utils.httpobj import urlparse_cached
class RootOnlyOffsiteMiddleware(OffsiteMiddleware):
def should_follow(self, request, spider):
return urlparse_cached(request).hostname in spider.allowed_domains
"""
self._update_host_regex(spider)
regex = self.host_regex
# hostname can be None for wrong urls (like javascript links)
@ -79,7 +125,6 @@ class OffsiteMiddleware:
return bool(regex.search(host))
def get_host_regex(self, spider: Spider) -> re.Pattern[str]:
"""Override this method to implement a different offsite policy"""
allowed_domains = getattr(spider, "allowed_domains", None)
if not allowed_domains:
return re.compile("") # allow all by default

View File

@ -94,7 +94,6 @@ def get_retry_request(
retry-related job stats
"""
settings = spider.crawler.settings
assert spider.crawler.stats
stats = spider.crawler.stats
retry_times = request.meta.get("retry_times", 0) + 1
if max_retry_times is None:

View File

@ -18,7 +18,7 @@ from scrapy.http.request import NO_CALLBACK
from scrapy.utils.decorators import _warn_spider_arg
from scrapy.utils.defer import maybe_deferred_to_future
from scrapy.utils.httpobj import urlparse_cached
from scrapy.utils.misc import load_object
from scrapy.utils.misc import build_from_crawler, load_object
if TYPE_CHECKING:
# typing.Self requires Python 3.11
@ -27,6 +27,7 @@ if TYPE_CHECKING:
from scrapy import Spider
from scrapy.crawler import Crawler
from scrapy.robotstxt import RobotParser
from scrapy.statscollectors import StatsCollector
logger = logging.getLogger(__name__)
@ -41,13 +42,14 @@ class RobotsTxtMiddleware:
self._default_useragent: str = crawler.settings["USER_AGENT"]
self._robotstxt_useragent: str | None = crawler.settings["ROBOTSTXT_USER_AGENT"]
self.crawler: Crawler = crawler
self._stats: StatsCollector = crawler.stats
self._parsers: dict[str, RobotParser | Deferred[RobotParser | None] | None] = {}
self._parserimpl: RobotParser = load_object(
crawler.settings.get("ROBOTSTXT_PARSER")
)
# check if parser dependencies are met, this should throw an error otherwise.
self._parserimpl.from_crawler(self.crawler, b"")
build_from_crawler(self._parserimpl, self.crawler, b"")
@classmethod
def from_crawler(cls, crawler: Crawler) -> Self:
@ -78,8 +80,7 @@ class RobotsTxtMiddleware:
{"request": request},
extra={"spider": self.crawler.spider},
)
assert self.crawler.stats
self.crawler.stats.inc_value("robotstxt/forbidden")
self._stats.inc_value("robotstxt/forbidden")
raise IgnoreRequest("Forbidden by robots.txt")
async def robot_parser(self, request: Request) -> RobotParser | None:
@ -95,8 +96,6 @@ class RobotsTxtMiddleware:
meta={"dont_obey_robotstxt": True},
callback=NO_CALLBACK,
)
assert self.crawler.engine
assert self.crawler.stats
try:
resp = await self.crawler.engine.download_async(robotsreq)
await self._parse_robots(resp, netloc, request)
@ -109,7 +108,7 @@ class RobotsTxtMiddleware:
extra={"spider": self.crawler.spider},
)
self._robots_error(e, netloc)
self.crawler.stats.inc_value("robotstxt/request_count")
self._stats.inc_value("robotstxt/request_count")
parser = self._parsers[netloc]
if isinstance(parser, Deferred):
@ -119,12 +118,9 @@ class RobotsTxtMiddleware:
async def _parse_robots(
self, response: Response, netloc: str, request: Request
) -> None:
assert self.crawler.stats
self.crawler.stats.inc_value("robotstxt/response_count")
self.crawler.stats.inc_value(
f"robotstxt/response_status_count/{response.status}"
)
rp = self._parserimpl.from_crawler(self.crawler, response.body)
self._stats.inc_value("robotstxt/response_count")
self._stats.inc_value(f"robotstxt/response_status_count/{response.status}")
rp = build_from_crawler(self._parserimpl, self.crawler, response.body)
await self.crawler.signals.send_catch_log_async(
signal=signals.robots_parsed,
robotparser=rp,
@ -138,8 +134,7 @@ class RobotsTxtMiddleware:
def _robots_error(self, exc: Exception, netloc: str) -> None:
if not isinstance(exc, IgnoreRequest):
key = f"robotstxt/exception_count/{type(exc)}"
assert self.crawler.stats
self.crawler.stats.inc_value(key)
self._stats.inc_value(key)
rp_dfd = self._parsers[netloc]
assert isinstance(rp_dfd, Deferred)
self._parsers[netloc] = None

View File

@ -43,7 +43,6 @@ class DownloaderStats:
def from_crawler(cls, crawler: Crawler) -> Self:
if not crawler.settings.getbool("DOWNLOADER_STATS"):
raise NotConfigured
assert crawler.stats
return cls(crawler.stats)
@_warn_spider_arg

View File

@ -95,7 +95,6 @@ class RFPDupeFilter(BaseDupeFilter):
@classmethod
def from_crawler(cls, crawler: Crawler) -> Self:
assert crawler.request_fingerprinter
debug = crawler.settings.getbool("DUPEFILTER_DEBUG")
return cls(
job_dir(crawler.settings),
@ -134,5 +133,4 @@ class RFPDupeFilter(BaseDupeFilter):
self.logger.debug(msg, {"request": request}, extra={"spider": spider})
self.logdupes = False
assert spider.crawler.stats
spider.crawler.stats.inc_value("dupefilter/filtered")

View File

@ -56,8 +56,11 @@ class DontCloseSpider(Exception):
class CloseSpider(Exception):
"""Raised from a :ref:`spider callback <topics-spiders>` to request the
spider to be closed/stopped.
"""Raised from a :ref:`spider callback <topics-spiders>`, or while the
spider is starting, to request the spider to be closed/stopped.
.. versionchanged:: VERSION
Added support for raising it while the spider is starting.
*reason* is a string with the reason for closing.

View File

@ -85,11 +85,16 @@ class BaseItemExporter(ABC):
declared = (name for name in adapter.field_names() if name in populated)
return dict.fromkeys([*declared, *adapter.keys()])
def _get_serialized_fields(
def get_serialized_fields(
self, item: Any, default_value: Any = None, include_empty: bool | None = None
) -> Iterable[tuple[str, Any]]:
"""Return the fields to export as an iterable of tuples
(name, serialized_value)
"""Return the fields of *item* to export, as an iterable of
``(name, serialized_value)`` tuples, taking :attr:`fields_to_export`
into account and applying :meth:`serialize_field` to every value.
Fields missing from *item* are exported with *default_value*.
*include_empty* overrides :attr:`export_empty_fields`.
"""
item = ItemAdapter(item)
@ -136,7 +141,7 @@ class JsonLinesItemExporter(BaseItemExporter):
self.encoder: JSONEncoder = ScrapyJSONEncoder(**self._kwargs)
def export_item(self, item: Any) -> None:
itemdict = dict(self._get_serialized_fields(item))
itemdict = dict(self.get_serialized_fields(item))
data = self.encoder.encode(itemdict) + "\n"
self.file.write(to_bytes(data, self.encoding))
@ -176,7 +181,7 @@ class JsonItemExporter(BaseItemExporter):
self.file.write(b"]")
def export_item(self, item: Any) -> None:
itemdict = dict(self._get_serialized_fields(item))
itemdict = dict(self.get_serialized_fields(item))
data = to_bytes(self.encoder.encode(itemdict), self.encoding)
self._add_comma_after_first()
self.file.write(data)
@ -216,7 +221,7 @@ class XmlItemExporter(BaseItemExporter):
self._beautify_indent(depth=1)
self.xg.startElement(self.item_element, AttributesImpl({}))
self._beautify_newline()
for name, value in self._get_serialized_fields(item, default_value=""):
for name, value in self.get_serialized_fields(item, default_value=""):
self._export_xml_field(name, value, depth=2)
self._beautify_indent(depth=1)
self.xg.endElement(self.item_element)
@ -310,7 +315,7 @@ class CsvItemExporter(BaseItemExporter):
f"See: https://docs.scrapy.org/en/latest/topics/feed-exports.html#feed-export-fields",
)
self._data_loss_warned = True
fields = self._get_serialized_fields(item, default_value="", include_empty=True)
fields = self.get_serialized_fields(item, default_value="", include_empty=True)
values = list(self._build_row(x for _, x in fields))
self.csv_writer.writerow(values)
@ -347,7 +352,7 @@ class PickleItemExporter(BaseItemExporter):
self.protocol: int = protocol
def export_item(self, item: Any) -> None:
d = dict(self._get_serialized_fields(item))
d = dict(self.get_serialized_fields(item))
pickle.dump(d, self.file, self.protocol)
@ -365,7 +370,7 @@ class MarshalItemExporter(BaseItemExporter):
self.file: BytesIO = file
def export_item(self, item: Any) -> None:
marshal.dump(dict(self._get_serialized_fields(item)), self.file)
marshal.dump(dict(self.get_serialized_fields(item)), self.file)
class PprintItemExporter(BaseItemExporter):
@ -374,7 +379,7 @@ class PprintItemExporter(BaseItemExporter):
self.file: BytesIO = file
def export_item(self, item: Any) -> None:
itemdict = dict(self._get_serialized_fields(item))
itemdict = dict(self.get_serialized_fields(item))
self.file.write(to_bytes(pprint.pformat(itemdict) + "\n"))
@ -417,5 +422,5 @@ class PythonItemExporter(BaseItemExporter):
yield key, self._serialize_value(value)
def export_item(self, item: Any) -> dict[str | bytes, Any]: # type: ignore[override]
result: dict[str | bytes, Any] = dict(self._get_serialized_fields(item))
result: dict[str | bytes, Any] = dict(self.get_serialized_fields(item))
return result

View File

@ -102,7 +102,6 @@ class CloseSpider:
self._close_spider("closespider_pagecount_no_item")
def spider_opened(self, spider: Spider) -> None:
assert self.crawler.engine
self.task = call_later(
self.close_on["timeout"], self._close_spider, "closespider_timeout"
)
@ -146,5 +145,4 @@ class CloseSpider:
self._close_spider("closespider_timeout_no_item")
def _close_spider(self, reason: str) -> None:
assert self.crawler.engine
_schedule_coro(self.crawler.engine.close_spider_async(reason=reason))

View File

@ -26,7 +26,6 @@ class CoreStats:
@classmethod
def from_crawler(cls, crawler: Crawler) -> Self:
assert crawler.stats
o = cls(crawler.stats)
crawler.signals.connect(o.spider_opened, signal=signals.spider_opened)
crawler.signals.connect(o.spider_closed, signal=signals.spider_closed)

View File

@ -45,7 +45,6 @@ class StackTraceDump:
return cls(crawler)
def dump_stacktrace(self, signum: int, frame: FrameType | None) -> None:
assert self.crawler.engine
log_args = {
"stackdumps": self._thread_stacks(),
"enginestatus": format_engine_status(self.crawler.engine),

View File

@ -149,7 +149,7 @@ class BlockingFeedStorage(ABC):
return NamedTemporaryFile(prefix="feed-", dir=path)
def store(self, file: IO[bytes]) -> Deferred[None] | None:
def store(self, file: IO[bytes]) -> Deferred[None]:
return deferred_from_coro(run_in_thread(self._store_in_thread, file))
@abstractmethod
@ -363,6 +363,7 @@ class FTPFeedStorage(BlockingFeedStorage):
self.username: str = u.username or ""
self.password: str = unquote(u.password or "")
self.path: str = u.path
self.tls: bool = u.scheme == "ftps"
self.use_active_mode: bool = use_active_mode
self.overwrite: bool = not feed_options or feed_options.get("overwrite", True)
@ -390,6 +391,7 @@ class FTPFeedStorage(BlockingFeedStorage):
password=self.password,
use_active_mode=self.use_active_mode,
overwrite=self.overwrite,
tls=self.tls,
)
@ -611,7 +613,6 @@ class FeedExporter:
logmsg = f"{slot.format} feed ({slot.itemcount} items) in: {slot.uri}"
slot_type = type(slot.storage).__name__
assert self.crawler.stats
try:
await ensure_awaitable(slot.storage.store(self._get_file(slot)))
except Exception:

View File

@ -275,7 +275,6 @@ class DbmCacheStorage:
extra={"spider": spider},
)
assert spider.crawler.request_fingerprinter
self._fingerprinter: RequestFingerprinterProtocol = (
spider.crawler.request_fingerprinter
)
@ -341,7 +340,6 @@ class FilesystemCacheStorage:
extra={"spider": spider},
)
assert spider.crawler.request_fingerprinter
self._fingerprinter = spider.crawler.request_fingerprinter
def close_spider(self, spider: Spider) -> None:

View File

@ -37,7 +37,6 @@ class LogStats:
interval: float = crawler.settings.getfloat("LOGSTATS_INTERVAL")
if not interval:
raise NotConfigured
assert crawler.stats
o = cls(crawler.stats, interval)
crawler.signals.connect(o.spider_opened, signal=signals.spider_opened)
crawler.signals.connect(o.spider_closed, signal=signals.spider_closed)

View File

@ -29,7 +29,6 @@ class MemoryDebugger:
def from_crawler(cls, crawler: Crawler) -> Self:
if not crawler.settings.getbool("MEMDEBUG_ENABLED"):
raise NotConfigured
assert crawler.stats
o = cls(crawler.stats)
crawler.signals.connect(o.spider_closed, signal=signals.spider_closed)
return o

View File

@ -19,6 +19,7 @@ from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning
from scrapy.utils.asyncio import AsyncioLoopingCall, create_looping_call
from scrapy.utils.defer import _schedule_coro
from scrapy.utils.engine import get_engine_status
from scrapy.utils.misc import build_from_crawler
if TYPE_CHECKING:
from twisted.internet.task import LoopingCall
@ -27,6 +28,7 @@ if TYPE_CHECKING:
from typing_extensions import Self
from scrapy.crawler import Crawler
from scrapy.statscollectors import StatsCollector
logger = logging.getLogger(__name__)
@ -43,6 +45,7 @@ class MemoryUsage:
raise NotConfigured from exc
self.crawler: Crawler = crawler
self._stats: StatsCollector = crawler.stats
self.warned: bool = False
self.notify_mails: list[str] = crawler.settings.getlist("MEMUSAGE_NOTIFY_MAIL")
if self.notify_mails: # pragma: no cover
@ -55,7 +58,7 @@ class MemoryUsage:
category=ScrapyDeprecationWarning,
stacklevel=2,
)
self.mail = MailSender.from_crawler(crawler)
self.mail = build_from_crawler(MailSender, crawler)
self.limit: int = crawler.settings.getint("MEMUSAGE_LIMIT_MB") * 1024 * 1024
self.warning: int = crawler.settings.getint("MEMUSAGE_WARNING_MB") * 1024 * 1024
@ -77,8 +80,7 @@ class MemoryUsage:
return size
def engine_started(self) -> None:
assert self.crawler.stats
self.crawler.stats.set_value("memusage/startup", self.get_virtual_size())
self._stats.set_value("memusage/startup", self.get_virtual_size())
self.tasks: list[AsyncioLoopingCall | LoopingCall] = []
tsk = create_looping_call(self.update)
self.tasks.append(tsk)
@ -98,15 +100,12 @@ class MemoryUsage:
tsk.stop()
def update(self) -> None:
assert self.crawler.stats
self.crawler.stats.max_value("memusage/max", self.get_virtual_size())
self._stats.max_value("memusage/max", self.get_virtual_size())
def _check_limit(self) -> None:
assert self.crawler.engine
assert self.crawler.stats
peak_mem_usage = self.get_virtual_size()
if peak_mem_usage > self.limit:
self.crawler.stats.set_value("memusage/limit_reached", 1)
self._stats.set_value("memusage/limit_reached", 1)
mem = self.limit / 1024 / 1024
logger.error(
"Memory usage exceeded %(memusage)dMiB. Shutting down Scrapy...",
@ -119,7 +118,7 @@ class MemoryUsage:
f"memory usage exceeded {mem}MiB at {socket.gethostname()}"
)
self._send_report(self.notify_mails, subj)
self.crawler.stats.set_value("memusage/limit_notified", 1)
self._stats.set_value("memusage/limit_notified", 1)
if self.crawler.engine.spider is not None:
_schedule_coro(
@ -136,9 +135,8 @@ class MemoryUsage:
def _check_warning(self) -> None:
if self.warned: # warn only once
return
assert self.crawler.stats
if self.get_virtual_size() > self.warning:
self.crawler.stats.set_value("memusage/warning_reached", 1)
self._stats.set_value("memusage/warning_reached", 1)
self.crawler.signals.send_catch_log(signal=signals.memusage_warning_reached)
mem = self.warning / 1024 / 1024
logger.warning(
@ -152,16 +150,13 @@ class MemoryUsage:
f"memory usage reached {mem}MiB at {socket.gethostname()}"
)
self._send_report(self.notify_mails, subj)
self.crawler.stats.set_value("memusage/warning_notified", 1)
self._stats.set_value("memusage/warning_notified", 1)
self.warned = True
def _send_report(self, rcpts: list[str], subject: str) -> None: # pragma: no cover
"""send notification mail with some additional useful info"""
assert self.crawler.engine
assert self.crawler.stats
stats = self.crawler.stats
s = f"Memory usage at engine startup : {stats.get_value('memusage/startup') / 1024 / 1024}M\r\n"
s += f"Maximum memory usage : {stats.get_value('memusage/max') / 1024 / 1024}M\r\n"
s = f"Memory usage at engine startup : {self._stats.get_value('memusage/startup') / 1024 / 1024}M\r\n"
s += f"Maximum memory usage : {self._stats.get_value('memusage/max') / 1024 / 1024}M\r\n"
s += f"Current memory usage : {self.get_virtual_size() / 1024 / 1024}M\r\n"
s += (

View File

@ -87,7 +87,6 @@ class PeriodicLog:
)
if not (ext_stats or ext_delta or ext_timing_enabled):
raise NotConfigured
assert crawler.stats
assert ext_stats is not None
assert ext_delta is not None
o = cls(

View File

@ -12,6 +12,7 @@ from typing import TYPE_CHECKING
from scrapy import Spider, signals
from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning
from scrapy.mail import MailSender
from scrapy.utils.misc import build_from_crawler
if TYPE_CHECKING:
from twisted.internet.defer import Deferred
@ -41,8 +42,7 @@ class StatsMailer:
recipients: list[str] = crawler.settings.getlist("STATSMAILER_RCPTS")
if not recipients:
raise NotConfigured
mail: MailSender = MailSender.from_crawler(crawler)
assert crawler.stats
mail: MailSender = build_from_crawler(MailSender, crawler)
o = cls(crawler.stats, recipients, mail)
crawler.signals.connect(o.spider_closed, signal=signals.spider_closed)
return o

View File

@ -108,7 +108,6 @@ class TelnetConsole(protocol.ServerFactory):
def _get_telnet_vars(self) -> dict[str, Any]:
# Note: if you add entries here also update topics/telnetconsole.rst
assert self.crawler.engine
telnet_vars: dict[str, Any] = {
"engine": self.crawler.engine,
"spider": self.crawler.engine.spider,

View File

@ -45,7 +45,6 @@ class AutoThrottle:
def _spider_opened(self, spider: Spider) -> None:
self.mindelay = self._min_delay()
self.maxdelay = self._max_delay()
assert self.crawler.engine
self.crawler.engine.downloader._delay = self._start_delay()
def _min_delay(self) -> float:
@ -98,7 +97,6 @@ class AutoThrottle:
key: str | None = request.meta.get("download_slot")
if key is None:
return None, None
assert self.crawler.engine
return key, self.crawler.engine.downloader.slots.get(key)
def _adjust_delay(self, slot: Slot, latency: float, response: Response) -> None:

View File

@ -305,6 +305,11 @@ class Response(object_ref):
:class:`~.TextResponse` provides a :meth:`~.TextResponse.follow_all`
method which supports selectors in addition to absolute/relative URLs
and Link objects.
.. caution:: Every returned request gets its own *meta* and
*cb_kwargs* dictionaries, but the values within them are shared.
Mutating one of those values, e.g. appending to a list, affects
all the returned requests.
"""
if not hasattr(urls, "__iter__"):
raise TypeError("'urls' argument must be an iterable")

View File

@ -276,6 +276,9 @@ class TextResponse(Response):
using the ``css`` or ``xpath`` parameters, this method will not produce requests for
selectors from which links cannot be obtained (for instance, anchor tags without an
``href`` attribute)
.. seealso:: :meth:`.Response.follow_all`, for a caution about mutable
*meta* and *cb_kwargs* values.
"""
arguments = [x for x in (urls, css, xpath) if x is not None]
if len(arguments) != 1:

View File

@ -332,7 +332,7 @@ class LxmlLinkExtractor:
unique=unique,
process=process_value,
strip=strip,
canonicalized=not canonicalize,
canonicalized=True,
)
self.allow_res: list[re.Pattern[str]] = self._compile_regexes(allow)
self.deny_res: list[re.Pattern[str]] = self._compile_regexes(deny)

View File

@ -27,9 +27,7 @@ from twisted.internet.defer import Deferred, maybeDeferred
from scrapy.exceptions import IgnoreRequest, NotConfigured, ScrapyDeprecationWarning
from scrapy.http import Request, Response
from scrapy.http.request import NO_CALLBACK
from scrapy.pipelines.media import (
FileException as FileException, # noqa: PLC0414 # re-exported for backward compatibility
)
from scrapy.pipelines.media import FileException as _FileException
from scrapy.pipelines.media import (
FileInfo,
FileInfoOrError,
@ -626,7 +624,7 @@ class FilesPipeline(MediaPipeline):
f"{request} referred in <{referer}>: {failure.value}",
extra={"spider": info.spider},
)
raise FileException
raise _FileException
async def media_downloaded(
self,
@ -645,7 +643,7 @@ class FilesPipeline(MediaPipeline):
{"status": response.status, "request": request, "referer": referer},
extra={"spider": info.spider},
)
raise FileException("download-error")
raise _FileException("download-error")
if not response.body:
logger.warning(
@ -654,7 +652,7 @@ class FilesPipeline(MediaPipeline):
{"request": request, "referer": referer},
extra={"spider": info.spider},
)
raise FileException("empty-content")
raise _FileException("empty-content")
status = "cached" if "cached" in response.flags else "downloaded"
logger.debug(
@ -670,7 +668,7 @@ class FilesPipeline(MediaPipeline):
checksum: str = await ensure_awaitable(
self.file_downloaded(response, request, info, item=item)
)
except FileException as exc:
except _FileException as exc:
logger.warning(
"File (error): Error processing file from %(request)s "
"referred in <%(referer)s>: %(errormsg)s",
@ -687,7 +685,7 @@ class FilesPipeline(MediaPipeline):
exc_info=True,
extra={"spider": info.spider},
)
raise FileException(str(exc)) from exc
raise _FileException(str(exc)) from exc
return {
"url": request.url,
@ -697,9 +695,9 @@ class FilesPipeline(MediaPipeline):
}
def inc_stats(self, status: str) -> None:
assert self.crawler.stats
self.crawler.stats.inc_value("file_count")
self.crawler.stats.inc_value(f"file_status_count/{status}")
stats = self.crawler.stats
stats.inc_value("file_count")
stats.inc_value(f"file_status_count/{status}")
async def _file_downloaded(
self,
@ -770,3 +768,15 @@ class FilesPipeline(MediaPipeline):
if media_type:
media_ext = cast("str", mimetypes.guess_extension(media_type))
return f"full/{media_guid}{media_ext}"
def __getattr__(name: str) -> Any:
if name == "FileException":
warnings.warn(
"scrapy.pipelines.files.FileException is deprecated, use "
"scrapy.pipelines.media.FileException instead.",
ScrapyDeprecationWarning,
stacklevel=2,
)
return _FileException
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")

View File

@ -18,18 +18,13 @@ from itemadapter import ItemAdapter
from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning
from scrapy.http import Request, Response
from scrapy.http.request import NO_CALLBACK
from scrapy.pipelines.files import (
FileException,
FilesPipeline,
GCSFilesStore,
S3FilesStore,
_md5sum,
)
from scrapy.pipelines.files import FilesPipeline, GCSFilesStore, S3FilesStore, _md5sum
from scrapy.pipelines.media import FileException
from scrapy.utils.defer import ensure_awaitable
from scrapy.utils.python import to_bytes
if TYPE_CHECKING:
from collections.abc import Iterable
from collections.abc import Iterator
from os import PathLike
from PIL import Image
@ -180,7 +175,7 @@ class ImagesPipeline(FilesPipeline):
info: MediaPipeline.SpiderInfo,
*,
item: Any = None,
) -> Iterable[tuple[str, Image.Image, BytesIO]]:
) -> Iterator[tuple[str, Image.Image, BytesIO]]:
path = self.file_path(request, response=response, info=info, item=item)
orig_image = self._Image.open(BytesIO(response.body))
transposed_image = self._ImageOps.exif_transpose(orig_image)

View File

@ -100,7 +100,6 @@ class MediaPipeline(ABC):
stacklevel=2,
)
self.crawler: Crawler = crawler
assert crawler.request_fingerprinter
self._fingerprinter: RequestFingerprinterProtocol = (
crawler.request_fingerprinter
)
@ -228,7 +227,6 @@ class MediaPipeline(ABC):
) -> FileInfo:
try:
self._modify_media_request(request)
assert self.crawler.engine
response = await self.crawler.engine.download_async(request)
return await ensure_awaitable(
self.media_downloaded(response, request, info, item=item)

View File

@ -265,7 +265,6 @@ class ScrapyPriorityQueue:
class DownloaderInterface:
def __init__(self, crawler: Crawler):
assert crawler.engine
self.downloader: Downloader = crawler.engine.downloader
def stats(self, possible_slots: Iterable[str]) -> list[tuple[int, str]]:

View File

@ -379,6 +379,7 @@ FEED_STORAGES_BASE = {
"": "scrapy.extensions.feedexport.FileFeedStorage",
"file": "scrapy.extensions.feedexport.FileFeedStorage",
"ftp": "scrapy.extensions.feedexport.FTPFeedStorage",
"ftps": "scrapy.extensions.feedexport.FTPFeedStorage",
"gs": "scrapy.extensions.feedexport.GCSFeedStorage",
"s3": "scrapy.extensions.feedexport.S3FeedStorage",
"stdout": "scrapy.extensions.feedexport.StdoutFeedStorage",

View File

@ -193,7 +193,6 @@ class Shell:
"""
if not self.spider:
await self._open_spider(spider)
assert self.crawler.engine is not None
# send the request to the engine
self.crawler.engine.crawl(request)
# this will fire when the request callback runs (via the callback hijacking in _request_deferred())
@ -204,7 +203,6 @@ class Shell:
spider = self.crawler.spider or self.crawler._create_spider()
self.crawler.spider = spider
assert self.crawler.engine
await self.crawler.engine.open_spider_async(close_if_idle=False)
self.spider = spider

View File

@ -28,6 +28,27 @@ logger = logging.getLogger(__name__)
class DepthMiddleware(BaseSpiderMiddleware):
"""Track the depth of each request within the site being scraped, setting
``request.meta["depth"]`` to 0 when there is no value previously set
(usually just the first request) and incrementing it by 1 otherwise.
It can be used to limit the maximum depth to scrape, control request
priority based on their depth, and things like that, through the
:setting:`DEPTH_LIMIT`, :setting:`DEPTH_STATS_VERBOSE` and
:setting:`DEPTH_PRIORITY` settings.
.. reqmeta:: depth_reset
depth_reset
-----------
.. versionadded:: VERSION
:attr:`~scrapy.Request.meta` key that, set to ``True``, gives a request
depth 0 instead of the depth of its source response plus 1, e.g. to keep
:setting:`DEPTH_LIMIT` from applying across a domain change.
"""
crawler: Crawler
def __init__( # pylint: disable=super-init-not-called
@ -49,7 +70,6 @@ class DepthMiddleware(BaseSpiderMiddleware):
maxdepth = settings.getint("DEPTH_LIMIT")
verbose = settings.getbool("DEPTH_STATS_VERBOSE")
prio = settings.getint("DEPTH_PRIORITY")
assert crawler.stats
o = cls(maxdepth, crawler.stats, verbose, prio)
o.crawler = crawler
return o
@ -87,10 +107,13 @@ class DepthMiddleware(BaseSpiderMiddleware):
def get_processed_request(
self, request: Request, response: Response | None
) -> Request | None:
# Consumed here so that it cannot reach response.meta and, from there,
# spread to further requests through a meta copy.
depth_reset = request.meta.pop("depth_reset", False)
if response is None:
# start requests
return request
depth = response.meta["depth"] + 1
depth = 0 if depth_reset else response.meta["depth"] + 1
request.meta["depth"] = depth
if self.prio:
request.priority -= depth * self.prio

View File

@ -78,9 +78,9 @@ class HttpErrorMiddleware:
self, response: Response, exception: Exception, spider: Spider | None = None
) -> Iterable[Any] | None:
if isinstance(exception, HttpError):
assert self.crawler.stats
self.crawler.stats.inc_value("httperror/response_ignored_count")
self.crawler.stats.inc_value(
stats = self.crawler.stats
stats.inc_value("httperror/response_ignored_count")
stats.inc_value(
f"httperror/response_ignored_status_count/{response.status}"
)
logger.info(

View File

@ -15,8 +15,7 @@ logger = logging.getLogger(__name__)
class MetaCopyDetectionMiddleware(BaseSpiderMiddleware):
"""Warn when a spider yields a request with internal meta keys that should
not be copied from response.meta, or when two requests share the same meta
dict object.
not be copied from response.meta.
Each warning is emitted at most once per crawl.
"""

View File

@ -48,6 +48,5 @@ class UrlLengthMiddleware(BaseSpiderMiddleware):
{"maxlength": self.maxlength, "url": request.url},
extra={"spider": self.crawler.spider},
)
assert self.crawler.stats
self.crawler.stats.inc_value("urllength/request_ignored_count")
return None

View File

@ -2,11 +2,10 @@ import contextlib
import zlib
from io import BytesIO
with contextlib.suppress(ImportError):
try:
import brotli
except ImportError:
import brotlicffi as brotli
try:
import brotli
except ImportError:
import brotlicffi as brotli
with contextlib.suppress(ImportError):
import zstandard

View File

@ -1,7 +1,8 @@
import posixpath
from contextlib import closing
from ftplib import FTP, error_perm
from ftplib import FTP, FTP_TLS, error_perm
from posixpath import dirname
from ssl import create_default_context
from typing import IO
@ -29,13 +30,20 @@ def ftp_store_file(
password: str,
use_active_mode: bool = False,
overwrite: bool = True,
tls: bool = False,
) -> None:
"""Opens a FTP connection with passed credentials,sets current directory
to the directory extracted from given path, then uploads the file to server
"""Opens a FTP connection with passed credentials, sets current directory
to the directory extracted from given path, then uploads the file to server.
If *tls* is ``True``, the connection is secured with TLS (FTPS), and the
certificate of the server is verified.
"""
with FTP() as ftp, closing(file):
ftp = FTP_TLS(context=create_default_context()) if tls else FTP()
with ftp, closing(file):
ftp.connect(host, port)
ftp.login(username, password)
if isinstance(ftp, FTP_TLS):
ftp.prot_p()
if use_active_mode:
ftp.set_pasv(False)
file.seek(0)

View File

@ -2,7 +2,9 @@ from __future__ import annotations
import logging
import pprint
import re
import sys
import warnings
from collections.abc import MutableMapping
from logging.config import dictConfig
from typing import TYPE_CHECKING, Any, cast
@ -12,6 +14,7 @@ from twisted.python import log as twisted_log
from twisted.python.failure import Failure
import scrapy
from scrapy.exceptions import ScrapyDeprecationWarning
from scrapy.settings import Settings
from scrapy.utils.versions import get_versions
@ -239,13 +242,15 @@ class LogCounterHandler(logging.Handler):
def emit(self, record: logging.LogRecord) -> None:
sname = f"log_count/{record.levelname}"
assert self.crawler.stats
self.crawler.stats.inc_value(sname)
_MSG_MAPPING_PLACEHOLDER = re.compile(r"%\(\w+\)")
def logformatter_adapter(
logkws: LogFormatterResult,
) -> tuple[int, str, dict[str, Any] | tuple[Any, ...]]:
) -> tuple[Any, ...]:
"""
Helper that takes the dictionary output from the methods in LogFormatter
and adapts it into a tuple of positional arguments for logger.log calls.
@ -253,10 +258,28 @@ def logformatter_adapter(
level = logkws.get("level", logging.INFO)
message = logkws.get("msg") or ""
# NOTE: This also handles 'args' being an empty dict, that case doesn't
# play well in logger.log calls
args = cast("dict[str, Any]", logkws) if not logkws.get("args") else logkws["args"]
args = logkws.get("args")
# logging interpolates the message whenever it receives any positional
# argument, so empty args are left out. Tuple args become one positional
# argument each, while a dict is a single positional argument.
if not args:
if _MSG_MAPPING_PLACEHOLDER.search(message):
# The log formatter method has already returned, so there is no
# frame of it left in the stack to point at. msg is part of the
# warning message instead, so that each offending method gets its
# own warning.
warnings.warn(
f"A log formatter method returned msg {message!r} with "
f"%(name)s placeholders and no args. Interpolating msg with "
f"the returned dict is deprecated, return those values under "
f"args instead.",
ScrapyDeprecationWarning,
stacklevel=1,
)
return (level, message, logkws)
return (level, message)
if isinstance(args, tuple):
return (level, message, *args)
return (level, message, args)

View File

@ -90,6 +90,10 @@ def open_in_browser(
def parse_details(self, response):
if "item name" not in response.text:
open_in_browser(response)
On the Windows Subsystem for Linux, set the ``BROWSER`` environment
variable to `wslview <https://github.com/wslutilities/wslu>`_ to open the
response in a Windows browser, which cannot read Linux paths otherwise.
"""
# circular imports
from scrapy.http import HtmlResponse, TextResponse # noqa: PLC0415

View File

@ -10,18 +10,11 @@ from scrapy.http import Request, Response
class ScrapyJSONEncoder(json.JSONEncoder):
DATE_FORMAT = "%Y-%m-%d"
TIME_FORMAT = "%H:%M:%S"
def default(self, o: Any) -> Any:
if isinstance(o, set):
return list(o)
if isinstance(o, datetime.datetime):
return o.strftime(f"{self.DATE_FORMAT} {self.TIME_FORMAT}")
if isinstance(o, datetime.date):
return o.strftime(self.DATE_FORMAT)
if isinstance(o, datetime.time):
return o.strftime(self.TIME_FORMAT)
if isinstance(o, (datetime.datetime, datetime.date, datetime.time)):
return o.isoformat()
if isinstance(o, decimal.Decimal):
return str(o)
if isinstance(o, defer.Deferred):

View File

@ -14,7 +14,6 @@ This library has a minimal performance impact.
from __future__ import annotations
from collections import defaultdict
from operator import itemgetter
from time import monotonic_ns
from types import NoneType
@ -28,8 +27,8 @@ if TYPE_CHECKING:
from typing_extensions import Self
live_refs: defaultdict[type, WeakKeyDictionary[object, float]] = defaultdict(
WeakKeyDictionary
live_refs: WeakKeyDictionary[type, WeakKeyDictionary[object, float]] = (
WeakKeyDictionary()
)
@ -41,7 +40,11 @@ class object_ref:
def __new__(cls, *args: Any, **kwargs: Any) -> Self:
obj = object.__new__(cls)
live_refs[cls][obj] = monotonic_ns()
try:
refs = live_refs[cls]
except KeyError:
refs = live_refs[cls] = WeakKeyDictionary()
refs[obj] = monotonic_ns()
return obj

View File

@ -1,3 +1,4 @@
scrapy/core/downloader/handlers/http.py
scrapy/extensions/statsmailer.py
scrapy/interfaces.py
scrapy/mail.py

View File

@ -1,4 +1,5 @@
from datetime import datetime, timedelta, timezone
from ipaddress import IPv4Address
from pathlib import Path
from cryptography.hazmat.backends import default_backend
@ -12,6 +13,7 @@ from cryptography.hazmat.primitives.serialization import (
from cryptography.x509 import (
CertificateBuilder,
DNSName,
IPAddress,
Name,
NameAttribute,
SubjectAlternativeName,
@ -53,7 +55,9 @@ def generate_keys():
.not_valid_before(datetime.now(tz=timezone.utc))
.not_valid_after(datetime.now(tz=timezone.utc) + timedelta(days=10))
.add_extension(
SubjectAlternativeName([DNSName("localhost")]),
SubjectAlternativeName(
[DNSName("localhost"), IPAddress(IPv4Address("127.0.0.1"))]
),
critical=False,
)
.sign(key, SHA256(), default_backend())

View File

@ -10,7 +10,7 @@ from tempfile import mkdtemp
from typing import TYPE_CHECKING
from pyftpdlib.authorizers import DummyAuthorizer
from pyftpdlib.handlers import FTPHandler
from pyftpdlib.handlers import FTPHandler, TLS_FTPHandler
from pyftpdlib.servers import FTPServer
from tests.utils import get_script_run_env
@ -25,27 +25,32 @@ if TYPE_CHECKING:
class MockFTPServer:
"""Creates an FTP server on a random port with a default passwordless user
(anonymous) and a temporary root path that you can read from the
:attr:`path` attribute."""
:attr:`path` attribute.
def __init__(self) -> None:
self.proc: Popen[str] | None = None
If *tls* is ``True``, the server requires FTPS, using the test certificate
from :file:`tests/keys`.
"""
proc: Popen[str]
port: int
path: Path
def __init__(self, tls: bool = False) -> None:
self.host: str = "127.0.0.1"
self.port: int | None = None
self.path: Path | None = None
self.tls: bool = tls
def __enter__(self) -> Self:
self.path = Path(mkdtemp())
self.proc = Popen(
[sys.executable, "-u", "-m", "tests.mockserver.ftp", "-d", str(self.path)],
[sys.executable, "-u", "-m", "tests.mockserver.ftp", "-d", str(self.path)]
+ (["--tls"] if self.tls else []),
stderr=PIPE,
env=get_script_run_env(),
text=True,
)
assert self.proc.stderr is not None
for line in self.proc.stderr:
if "starting FTP server" in line and (
m := re.search(r"starting FTP server on ([^ :]+):(\d+),", line)
):
if m := re.search(r"starting FTPS? .*on ([^ :]+):(\d+),", line):
self.port = int(m.group(2))
break
else:
@ -63,23 +68,32 @@ class MockFTPServer:
traceback: TracebackType | None,
) -> None:
rmtree(str(self.path))
assert self.proc is not None
self.proc.kill()
self.proc.communicate()
def url(self, path: str) -> str:
return f"ftp://{self.host}:{self.port}/{path}"
scheme = "ftps" if self.tls else "ftp"
return f"{scheme}://{self.host}:{self.port}/{path}"
def main() -> None:
parser = ArgumentParser()
parser.add_argument("-d", "--directory", required=True)
parser.add_argument("--tls", action="store_true")
args = parser.parse_args()
authorizer = DummyAuthorizer()
full_permissions = "elradfmwMT"
authorizer.add_anonymous(args.directory, perm=full_permissions)
handler = FTPHandler
if args.tls:
keys = Path(__file__).parent.parent / "keys"
handler = TLS_FTPHandler
handler.certfile = str(keys / "localhost.crt")
handler.keyfile = str(keys / "localhost.key")
handler.tls_control_required = True
handler.tls_data_required = True
else:
handler = FTPHandler
handler.authorizer = authorizer
address = ("127.0.0.1", 0)
server = FTPServer(address, handler)

View File

@ -11,6 +11,7 @@ from tests import tests_datadir
from .http_base import BaseMockServer, main_factory
from .http_resources import (
ArbitraryLengthPayloadResource,
BadHeader,
BaseResource,
BrokenChunkedResource,
BrokenDownloadResource,
@ -52,6 +53,7 @@ class Root(BaseResource):
put_child(self, b"partial", Partial())
put_child(self, b"drop", Drop())
put_child(self, b"raw", Raw())
put_child(self, b"bad-header", BadHeader())
put_child(self, b"echo", Echo())
put_child(self, b"payload", PayloadResource())
put_child(self, b"alpayload", ArbitraryLengthPayloadResource())

View File

@ -210,6 +210,39 @@ class Raw(LeafResource):
request.finish()
class BadHeader(LeafResource):
"""Sends a response with a bad header line, one with no colon in it, like
some servers do, between two good ones.
One of the good header lines is split into two lines, so that handling of
such headers is also covered.
"""
response = (
b"HTTP/1.1 200 OK\r\n"
b"Content-Length: 5\r\n"
b"Content-Type: text/html\r\n"
b"X-Folded-Header: one\r\n"
b"\ttwo\r\n"
b'<meta http-equiv="Content-Type" content="text/html; charset=utf-8" />\r\n'
b"X-After-Bad-Header: works\r\n"
b"\r\n"
b"Works"
)
def render_GET(self, request: Request) -> int:
request.startedWriting = 1
self.deferRequest(request, 0, self._delayedRender, request)
return NOT_DONE_YET
def _delayedRender(self, request: Request) -> None:
request.write(self.response)
# Clients that stop parsing headers at the bad one don't get
# Content-Length, so they need the connection to be closed to know that
# the response body is over.
close_connection(request)
class Echo(LeafResource):
def render_GET(self, request: Request) -> bytes:
assert request.content

View File

@ -201,7 +201,7 @@ class TestInteractiveShell:
env = os.environ.copy()
env["SCRAPY_PYTHON_SHELL"] = "python"
logfile = BytesIO()
p = PopenSpawn(args, env=env, timeout=5)
p = PopenSpawn(args, env=env, timeout=60)
p.logfile_read = logfile
p.expect_exact("Available Scrapy objects")
p.sendline(f"fetch('{mockserver.url('/')}')")
@ -235,7 +235,7 @@ class TestInteractiveShell:
def _run_interactive_shell(self, env: dict[str, str]) -> str:
args = (sys.executable, "-m", "scrapy.cmdline", "shell")
logfile = BytesIO()
p = PopenSpawn(args, env=env, timeout=5)
p = PopenSpawn(args, env=env, timeout=60)
p.logfile_read = logfile
p.expect_exact("Available Scrapy objects")
p.sendeof()
@ -256,7 +256,7 @@ class TestInteractiveShell:
self._isolate_config(env, config_home)
args = (sys.executable, "-m", "scrapy.cmdline", "shell")
logfile = BytesIO()
p = PopenSpawn(args, env=env, timeout=10)
p = PopenSpawn(args, env=env, timeout=60)
p.logfile_read = logfile
p.expect_exact("Available Scrapy objects")
# The standard Python shell never imports IPython, whereas the IPython
@ -365,7 +365,7 @@ class TestShell:
crawler.engine = MagicMock()
crawler.engine.open_spider_async = AsyncMock()
shell = Shell(crawler)
spider = Spider("test")
spider = Spider.from_crawler(crawler, "test")
await shell._open_spider(spider)
assert shell.spider is spider
assert crawler.spider is spider

View File

@ -74,6 +74,31 @@ class TestCrawler:
assert not settings.frozen
assert crawler.settings.frozen
@pytest.mark.parametrize(
"attr",
["extensions", "logformatter", "request_fingerprinter", "stats"],
)
def test_late_attr_before_apply_settings(self, attr: str) -> None:
crawler = get_raw_crawler(DefaultSpider)
with pytest.raises(RuntimeError, match=rf"Crawler\.{attr} is not set yet"):
getattr(crawler, attr)
crawler._apply_settings()
assert getattr(crawler, attr) is not None
@pytest.mark.parametrize(
"attr",
["engine", "extensions", "logformatter", "request_fingerprinter", "stats"],
)
def test_late_attr_on_class(self, attr: str) -> None:
# Introspection tools such as help() read these off the class.
assert getattr(Crawler, attr) is getattr(Crawler, attr)
def test_late_attr_engine_before_crawl(self) -> None:
crawler = get_raw_crawler(DefaultSpider)
crawler._apply_settings()
with pytest.raises(RuntimeError, match=r"Crawler\.engine is not set yet"):
_ = crawler.engine
@pytest.mark.parametrize(
("attr", "setting"),
[

View File

@ -400,7 +400,7 @@ class TestAsyncCrawlerProcessSubprocess(TestCrawlerProcessSubprocessBase):
def test_reactorless_import_hook(self) -> None:
log = self.run_script("reactorless_import_hook.py")
assert "Not using a Twisted reactor" in log
assert "Spider closed (finished)" in log
assert "Spider closed (start_error)" in log
assert "ImportError: Import of twisted.internet.reactor is forbidden" in log
def test_reactorless_import_hook_uninstall(self) -> None:

View File

@ -61,6 +61,7 @@ class HttpxDownloadHandlerMixin:
class TestHttp(HttpxDownloadHandlerMixin, TestHttpBase):
handler_supports_bindaddress_meta = False
handler_bad_header_handling = "fail"
@pytest.mark.skipif(
sys.platform == "darwin",
@ -82,6 +83,7 @@ class TestHttp(HttpxDownloadHandlerMixin, TestHttpBase):
class TestHttps(HttpxDownloadHandlerMixin, TestHttpsBase):
handler_supports_bindaddress_meta = False
handler_bad_header_handling = "fail"
tls_log_message = "SSL connection to 127.0.0.1 using protocol TLSv1.3, cipher"
@pytest.mark.skip(reason="The check is Twisted-specific")

View File

@ -212,4 +212,4 @@ class TestAnonymousFTP(TestFTPBase):
def test_not_configured_without_reactor() -> None:
crawler = Crawler(Spider, {"TWISTED_REACTOR_ENABLED": False})
with pytest.raises(NotConfigured):
FTPDownloadHandler.from_crawler(crawler)
build_from_crawler(FTPDownloadHandler, crawler)

View File

@ -11,6 +11,7 @@ from scrapy import Spider
from scrapy.core.downloader.handlers.http11 import HTTP11DownloadHandler
from scrapy.crawler import Crawler
from scrapy.exceptions import NotConfigured
from scrapy.utils.misc import build_from_crawler
from tests.utils.bases.download_handlers_http import (
TestHttpBase,
TestHttpProxyBase,
@ -51,7 +52,7 @@ class HTTP11DownloadHandlerMixin:
def test_not_configured_without_reactor() -> None:
crawler = Crawler(Spider, {"TWISTED_REACTOR_ENABLED": False})
with pytest.raises(NotConfigured):
HTTP11DownloadHandler.from_crawler(crawler)
build_from_crawler(HTTP11DownloadHandler, crawler)
class TestHttp(HTTP11DownloadHandlerMixin, TestHttpBase):

View File

@ -13,6 +13,7 @@ from scrapy import Spider
from scrapy.crawler import Crawler
from scrapy.exceptions import DownloadFailedError, NotConfigured
from scrapy.http import Request
from scrapy.utils.misc import build_from_crawler
from tests.utils.bases.download_handlers_http import (
TestHttpProxyBase,
TestHttpsBase,
@ -66,7 +67,7 @@ def test_not_configured_without_reactor() -> None:
crawler = Crawler(Spider, {"TWISTED_REACTOR_ENABLED": False})
with pytest.raises(NotConfigured):
H2DownloadHandler.from_crawler(crawler)
build_from_crawler(H2DownloadHandler, crawler)
class TestHttp2(H2DownloadHandlerMixin, TestHttpsBase):

View File

@ -14,6 +14,7 @@ from scrapy.exceptions import ScrapyDeprecationWarning, _InvalidOutput
from scrapy.http import Request, Response
from scrapy.spiders import Spider
from scrapy.utils.defer import maybe_deferred_to_future
from scrapy.utils.misc import build_from_crawler
from scrapy.utils.python import to_bytes
from scrapy.utils.test import get_crawler, get_from_asyncio_queue
from tests.utils.decorators import coroutine_test
@ -30,7 +31,7 @@ class TestManagerBase:
async def get_mwman(self) -> AsyncGenerator[DownloaderMiddlewareManager]:
crawler = get_crawler(Spider, self.settings_dict)
crawler.spider = crawler._create_spider("foo")
mwman = DownloaderMiddlewareManager.from_crawler(crawler)
mwman = build_from_crawler(DownloaderMiddlewareManager, crawler)
crawler.engine = crawler._create_engine()
await crawler.engine.open_spider_async()
try:

View File

@ -10,6 +10,7 @@ from scrapy.downloadermiddlewares.redirect import RedirectMiddleware
from scrapy.exceptions import NotConfigured
from scrapy.http import Request, Response
from scrapy.http.request import CookiesT, VerboseCookie
from scrapy.utils.misc import build_from_crawler
from scrapy.utils.python import to_bytes
from scrapy.utils.request import _to_verbose_cookies
from scrapy.utils.spider import DefaultSpider
@ -72,8 +73,8 @@ class TestCookiesMiddleware:
def setup_method(self):
crawler = get_crawler(DefaultSpider)
crawler.spider = crawler._create_spider()
self.mw = CookiesMiddleware.from_crawler(crawler)
self.redirect_middleware = RedirectMiddleware.from_crawler(crawler)
self.mw = build_from_crawler(CookiesMiddleware, crawler)
self.redirect_middleware = build_from_crawler(RedirectMiddleware, crawler)
def teardown_method(self):
del self.mw
@ -94,19 +95,19 @@ class TestCookiesMiddleware:
def test_setting_false_cookies_enabled(self):
with pytest.raises(NotConfigured):
CookiesMiddleware.from_crawler(
get_crawler(settings_dict={"COOKIES_ENABLED": False})
build_from_crawler(
CookiesMiddleware, get_crawler(settings_dict={"COOKIES_ENABLED": False})
)
def test_setting_default_cookies_enabled(self):
assert isinstance(
CookiesMiddleware.from_crawler(get_crawler()), CookiesMiddleware
build_from_crawler(CookiesMiddleware, get_crawler()), CookiesMiddleware
)
def test_setting_true_cookies_enabled(self):
assert isinstance(
CookiesMiddleware.from_crawler(
get_crawler(settings_dict={"COOKIES_ENABLED": True})
build_from_crawler(
CookiesMiddleware, get_crawler(settings_dict={"COOKIES_ENABLED": True})
),
CookiesMiddleware,
)
@ -115,7 +116,7 @@ class TestCookiesMiddleware:
self, caplog: pytest.LogCaptureFixture
) -> None:
crawler = get_crawler(settings_dict={"COOKIES_DEBUG": True})
mw = CookiesMiddleware.from_crawler(crawler)
mw = build_from_crawler(CookiesMiddleware, crawler)
caplog.clear()
with caplog.at_level(
logging.DEBUG, logger="scrapy.downloadermiddlewares.cookies"
@ -145,7 +146,7 @@ class TestCookiesMiddleware:
def test_debug_no_cookies(self, caplog: pytest.LogCaptureFixture) -> None:
crawler = get_crawler(settings_dict={"COOKIES_DEBUG": True})
mw = CookiesMiddleware.from_crawler(crawler)
mw = build_from_crawler(CookiesMiddleware, crawler)
caplog.clear()
with caplog.at_level(
logging.DEBUG, logger="scrapy.downloadermiddlewares.cookies"
@ -161,7 +162,7 @@ class TestCookiesMiddleware:
self, caplog: pytest.LogCaptureFixture
) -> None:
crawler = get_crawler(settings_dict={"COOKIES_DEBUG": False})
mw = CookiesMiddleware.from_crawler(crawler)
mw = build_from_crawler(CookiesMiddleware, crawler)
caplog.clear()
with caplog.at_level(
logging.DEBUG, logger="scrapy.downloadermiddlewares.cookies"

View File

@ -3,6 +3,7 @@ from __future__ import annotations
from scrapy.downloadermiddlewares.defaultheaders import DefaultHeadersMiddleware
from scrapy.http import Request
from scrapy.spiders import Spider
from scrapy.utils.misc import build_from_crawler
from scrapy.utils.python import to_bytes
from scrapy.utils.test import get_crawler
@ -13,7 +14,7 @@ def get_defaults_mw() -> tuple[dict[bytes, list[bytes]], DefaultHeadersMiddlewar
to_bytes(k): [to_bytes(v)]
for k, v in crawler.settings.get("DEFAULT_REQUEST_HEADERS").items()
}
return defaults, DefaultHeadersMiddleware.from_crawler(crawler)
return defaults, build_from_crawler(DefaultHeadersMiddleware, crawler)
def test_process_request():

View File

@ -5,6 +5,7 @@ from typing import Any
from scrapy.downloadermiddlewares.downloadtimeout import DownloadTimeoutMiddleware
from scrapy.http import Request
from scrapy.spiders import Spider
from scrapy.utils.misc import build_from_crawler
from scrapy.utils.test import get_crawler
@ -12,7 +13,7 @@ def get_request_spider_mw(settings: dict[str, Any] | None = None):
crawler = get_crawler(Spider, settings)
spider = crawler._create_spider("foo")
request = Request("http://scrapytest.org/")
return request, spider, DownloadTimeoutMiddleware.from_crawler(crawler)
return request, spider, build_from_crawler(DownloadTimeoutMiddleware, crawler)
def test_default_download_timeout():

View File

@ -7,6 +7,7 @@ from scrapy.downloadermiddlewares.httpauth import HttpAuthMiddleware
from scrapy.exceptions import ScrapyDeprecationWarning
from scrapy.http import Request
from scrapy.spiders import Spider
from scrapy.utils.misc import build_from_crawler
from scrapy.utils.test import get_crawler
_DOMAIN_NOT_SET = object()
@ -21,7 +22,7 @@ def make_mw(
}
if domain is not _DOMAIN_NOT_SET:
settings["HTTPAUTH_DOMAIN"] = domain
return HttpAuthMiddleware.from_crawler(get_crawler(settings_dict=settings))
return build_from_crawler(HttpAuthMiddleware, get_crawler(settings_dict=settings))
# --- Spider attribute tests (deprecated) ---

View File

@ -18,7 +18,7 @@ from scrapy.exceptions import IgnoreRequest
from scrapy.extensions.httpcache import DummyPolicy
from scrapy.http import HtmlResponse, Request, Response
from scrapy.spiders import Spider
from scrapy.utils.misc import load_object
from scrapy.utils.misc import build_from_crawler, load_object
from scrapy.utils.test import get_crawler
if TYPE_CHECKING:
@ -89,7 +89,7 @@ class TestBase:
def _middleware(self, **new_settings: Any) -> Generator[HttpCacheMiddleware]:
with self._get_crawler(**new_settings) as crawler:
assert crawler.spider
mw = HttpCacheMiddleware.from_crawler(crawler)
mw = build_from_crawler(HttpCacheMiddleware, crawler)
mw.spider_opened(crawler.spider)
try:
yield mw

View File

@ -18,6 +18,7 @@ from scrapy.responsetypes import responsetypes
from scrapy.spiders import Spider
from scrapy.utils._compression import _DecompressionMaxSizeExceeded
from scrapy.utils.gz import gunzip
from scrapy.utils.misc import build_from_crawler
from scrapy.utils.test import get_crawler
from tests import tests_datadir
@ -52,20 +53,6 @@ FORMAT = {
}
def _skip_if_no_br() -> None:
try:
try:
import brotli # noqa: PLC0415
brotli.Decompressor.can_accept_more_data
except (ImportError, AttributeError):
import brotlicffi # noqa: PLC0415
brotlicffi.Decompressor.can_accept_more_data
except (ImportError, AttributeError):
pytest.skip("no brotli support")
def _skip_if_no_zstd() -> None:
pytest.importorskip("zstandard")
@ -73,7 +60,7 @@ def _skip_if_no_zstd() -> None:
class TestHttpCompression:
def setup_method(self):
self.crawler = get_crawler(Spider)
self.mw = HttpCompressionMiddleware.from_crawler(self.crawler)
self.mw = build_from_crawler(HttpCompressionMiddleware, self.crawler)
assert self.crawler.stats
self.crawler.stats.open_spider()
@ -107,20 +94,22 @@ class TestHttpCompression:
def test_setting_false_compression_enabled(self):
with pytest.raises(NotConfigured):
HttpCompressionMiddleware.from_crawler(
get_crawler(settings_dict={"COMPRESSION_ENABLED": False})
build_from_crawler(
HttpCompressionMiddleware,
get_crawler(settings_dict={"COMPRESSION_ENABLED": False}),
)
def test_setting_default_compression_enabled(self):
assert isinstance(
HttpCompressionMiddleware.from_crawler(get_crawler()),
build_from_crawler(HttpCompressionMiddleware, get_crawler()),
HttpCompressionMiddleware,
)
def test_setting_true_compression_enabled(self):
assert isinstance(
HttpCompressionMiddleware.from_crawler(
get_crawler(settings_dict={"COMPRESSION_ENABLED": True})
build_from_crawler(
HttpCompressionMiddleware,
get_crawler(settings_dict={"COMPRESSION_ENABLED": True}),
),
HttpCompressionMiddleware,
)
@ -161,8 +150,6 @@ class TestHttpCompression:
self.assertStatsEqual("httpcompression/response_bytes", 74837)
def test_process_response_br(self):
_skip_if_no_br()
response = self._getresponse("br")
assert response.request
request = response.request
@ -174,32 +161,6 @@ class TestHttpCompression:
self.assertStatsEqual("httpcompression/response_count", 1)
self.assertStatsEqual("httpcompression/response_bytes", 74837)
def test_process_response_br_unsupported(self, caplog: pytest.LogCaptureFixture):
if find_spec("brotli") is not None or find_spec("brotlicffi") is not None:
pytest.skip("Requires not having brotli support")
response = self._getresponse("br")
assert response.request
request = response.request
assert response.headers["Content-Encoding"] == b"br"
caplog.clear()
with caplog.at_level(
WARNING, logger="scrapy.downloadermiddlewares.httpcompression"
):
newresponse = self.mw.process_response(request, response)
assert caplog.record_tuples == [
(
"scrapy.downloadermiddlewares.httpcompression",
WARNING,
(
"HttpCompressionMiddleware cannot decode the response for "
"http://scrapytest.org/ from unsupported encoding(s) 'br'. "
"You need to install brotli or brotlicffi >= 1.2.0 to decode 'br'."
),
),
]
assert newresponse is not response
assert newresponse.headers.getlist("Content-Encoding") == [b"br"]
def test_process_response_zstd(self):
_skip_if_no_zstd()
@ -538,7 +499,7 @@ class TestHttpCompression:
settings = {"DOWNLOAD_MAXSIZE": 1_000_000}
crawler = get_crawler(Spider, settings_dict=settings)
spider = crawler._create_spider("scrapytest.org")
mw = HttpCompressionMiddleware.from_crawler(crawler)
mw = build_from_crawler(HttpCompressionMiddleware, crawler)
mw.open_spider(spider)
response = self._getresponse(f"bomb-{compression_id}") # 11_511_612 B
@ -550,8 +511,6 @@ class TestHttpCompression:
assert cause.decompressed_size < 1_100_000
def test_compression_bomb_setting_br(self):
_skip_if_no_br()
self._test_compression_bomb_setting("br")
def test_compression_bomb_setting_deflate(self):
@ -569,7 +528,7 @@ class TestHttpCompression:
settings = {"DOWNLOAD_MAXSIZE": 1_000_000}
crawler = get_crawler(Spider, settings_dict=settings)
spider = crawler._create_spider("scrapytest.org")
mw = HttpCompressionMiddleware.from_crawler(crawler)
mw = build_from_crawler(HttpCompressionMiddleware, crawler)
mw.open_spider(spider)
response = self._getresponse("bomb-gzip") # 11_511_612 B
@ -596,7 +555,7 @@ class TestHttpCompression:
crawler = get_crawler(DownloadMaxSizeSpider)
spider = crawler._create_spider("scrapytest.org")
mw = HttpCompressionMiddleware.from_crawler(crawler)
mw = build_from_crawler(HttpCompressionMiddleware, crawler)
mw.open_spider(spider)
response = self._getresponse(f"bomb-{compression_id}")
@ -609,8 +568,6 @@ class TestHttpCompression:
@pytest.mark.filterwarnings("ignore::scrapy.exceptions.ScrapyDeprecationWarning")
def test_compression_bomb_spider_attr_br(self):
_skip_if_no_br()
self._test_compression_bomb_spider_attr("br")
@pytest.mark.filterwarnings("ignore::scrapy.exceptions.ScrapyDeprecationWarning")
@ -630,7 +587,7 @@ class TestHttpCompression:
def _test_compression_bomb_request_meta(self, compression_id: str) -> None:
crawler = get_crawler(Spider)
spider = crawler._create_spider("scrapytest.org")
mw = HttpCompressionMiddleware.from_crawler(crawler)
mw = build_from_crawler(HttpCompressionMiddleware, crawler)
mw.open_spider(spider)
response = self._getresponse(f"bomb-{compression_id}")
@ -643,8 +600,6 @@ class TestHttpCompression:
assert cause.decompressed_size < 1_100_000
def test_compression_bomb_request_meta_br(self):
_skip_if_no_br()
self._test_compression_bomb_request_meta("br")
def test_compression_bomb_request_meta_deflate(self):
@ -664,7 +619,7 @@ class TestHttpCompression:
settings = {"DOWNLOAD_WARNSIZE": 10_000_000}
crawler = get_crawler(Spider, settings_dict=settings)
spider = crawler._create_spider("scrapytest.org")
mw = HttpCompressionMiddleware.from_crawler(crawler)
mw = build_from_crawler(HttpCompressionMiddleware, crawler)
mw.open_spider(spider)
response = self._getresponse(f"bomb-{compression_id}")
@ -689,8 +644,6 @@ class TestHttpCompression:
def test_download_warnsize_setting_br(
self, caplog: pytest.LogCaptureFixture
) -> None:
_skip_if_no_br()
self._test_download_warnsize_setting(caplog, "br")
def test_download_warnsize_setting_deflate(
@ -718,7 +671,7 @@ class TestHttpCompression:
crawler = get_crawler(DownloadWarnSizeSpider)
spider = crawler._create_spider("scrapytest.org")
mw = HttpCompressionMiddleware.from_crawler(crawler)
mw = build_from_crawler(HttpCompressionMiddleware, crawler)
mw.open_spider(spider)
response = self._getresponse(f"bomb-{compression_id}")
@ -744,8 +697,6 @@ class TestHttpCompression:
def test_download_warnsize_spider_attr_br(
self, caplog: pytest.LogCaptureFixture
) -> None:
_skip_if_no_br()
self._test_download_warnsize_spider_attr(caplog, "br")
@pytest.mark.filterwarnings("ignore::scrapy.exceptions.ScrapyDeprecationWarning")
@ -773,7 +724,7 @@ class TestHttpCompression:
) -> None:
crawler = get_crawler(Spider)
spider = crawler._create_spider("scrapytest.org")
mw = HttpCompressionMiddleware.from_crawler(crawler)
mw = build_from_crawler(HttpCompressionMiddleware, crawler)
mw.open_spider(spider)
response = self._getresponse(f"bomb-{compression_id}")
response.meta["download_warnsize"] = 10_000_000
@ -799,8 +750,6 @@ class TestHttpCompression:
def test_download_warnsize_request_meta_br(
self, caplog: pytest.LogCaptureFixture
) -> None:
_skip_if_no_br()
self._test_download_warnsize_request_meta(caplog, "br")
def test_download_warnsize_request_meta_deflate(
@ -823,7 +772,7 @@ class TestHttpCompression:
def _get_truncated_response(self, compression_id: str) -> Response:
crawler = get_crawler(Spider)
spider = crawler._create_spider("scrapytest.org")
mw = HttpCompressionMiddleware.from_crawler(crawler)
mw = build_from_crawler(HttpCompressionMiddleware, crawler)
mw.open_spider(spider)
response = self._getresponse(compression_id)
truncated_body = response.body[: len(response.body) // 2]
@ -834,7 +783,6 @@ class TestHttpCompression:
return new_response
def test_process_truncated_response_br(self):
_skip_if_no_br()
resp = self._get_truncated_response("br")
assert resp.body.startswith(b"<!DOCTYPE")

View File

@ -6,6 +6,7 @@ from scrapy.downloadermiddlewares.httpproxy import HttpProxyMiddleware
from scrapy.exceptions import NotConfigured
from scrapy.http import Request
from scrapy.spiders import Spider
from scrapy.utils.misc import build_from_crawler
from scrapy.utils.test import get_crawler
@ -20,7 +21,7 @@ class TestHttpProxyMiddleware:
def test_not_enabled(self):
crawler = get_crawler(Spider, {"HTTPPROXY_ENABLED": False})
with pytest.raises(NotConfigured):
HttpProxyMiddleware.from_crawler(crawler)
build_from_crawler(HttpProxyMiddleware, crawler)
def test_no_environment_proxies(self):
os.environ.clear()

View File

@ -6,6 +6,8 @@ import pytest
from scrapy import Request, Spider
from scrapy.downloadermiddlewares.offsite import OffsiteMiddleware
from scrapy.exceptions import IgnoreRequest
from scrapy.utils.httpobj import urlparse_cached
from scrapy.utils.misc import build_from_crawler
from scrapy.utils.test import get_crawler
UNSET = object()
@ -30,7 +32,7 @@ UNSET = object()
def test_process_request_domain_filtering(allowed_domain, url, allowed):
crawler = get_crawler(Spider)
crawler.spider = crawler._create_spider(name="a", allowed_domains=[allowed_domain])
mw = OffsiteMiddleware.from_crawler(crawler)
mw = build_from_crawler(OffsiteMiddleware, crawler)
mw.spider_opened(crawler.spider)
request = Request(url)
if allowed:
@ -52,7 +54,7 @@ def test_process_request_domain_filtering(allowed_domain, url, allowed):
def test_process_request_dont_filter(value, filtered):
crawler = get_crawler(Spider)
crawler.spider = crawler._create_spider(name="a", allowed_domains=["a.example"])
mw = OffsiteMiddleware.from_crawler(crawler)
mw = build_from_crawler(OffsiteMiddleware, crawler)
mw.spider_opened(crawler.spider)
kwargs: dict[str, Any] = {}
if value is not UNSET:
@ -81,7 +83,7 @@ def test_process_request_dont_filter(value, filtered):
def test_process_request_allow_offsite(allow_offsite, dont_filter, filtered):
crawler = get_crawler(Spider)
crawler.spider = crawler._create_spider(name="a", allowed_domains=["a.example"])
mw = OffsiteMiddleware.from_crawler(crawler)
mw = build_from_crawler(OffsiteMiddleware, crawler)
mw.spider_opened(crawler.spider)
kwargs: dict[str, Any] = {"meta": {}}
if allow_offsite is not UNSET:
@ -110,7 +112,7 @@ def test_process_request_no_allowed_domains(value):
if value is not UNSET:
kwargs["allowed_domains"] = value
crawler.spider = crawler._create_spider(name="a", **kwargs)
mw = OffsiteMiddleware.from_crawler(crawler)
mw = build_from_crawler(OffsiteMiddleware, crawler)
mw.spider_opened(crawler.spider)
request = Request("https://example.com")
assert mw.process_request(request) is None
@ -120,7 +122,7 @@ def test_process_request_invalid_domains():
crawler = get_crawler(Spider)
allowed_domains = ["a.example", None, "http:////b.example", "//c.example"]
crawler.spider = crawler._create_spider(name="a", allowed_domains=allowed_domains)
mw = OffsiteMiddleware.from_crawler(crawler)
mw = build_from_crawler(OffsiteMiddleware, crawler)
mw.spider_opened(crawler.spider)
request = Request("https://a.example")
assert mw.process_request(request) is None
@ -149,7 +151,7 @@ def test_process_request_invalid_domains():
def test_request_scheduled_domain_filtering(allowed_domain, url, allowed):
crawler = get_crawler(Spider)
crawler.spider = crawler._create_spider(name="a", allowed_domains=[allowed_domain])
mw = OffsiteMiddleware.from_crawler(crawler)
mw = build_from_crawler(OffsiteMiddleware, crawler)
mw.spider_opened(crawler.spider)
request = Request(url)
if allowed:
@ -171,7 +173,7 @@ def test_request_scheduled_domain_filtering(allowed_domain, url, allowed):
def test_request_scheduled_dont_filter(value, filtered):
crawler = get_crawler(Spider)
crawler.spider = crawler._create_spider(name="a", allowed_domains=["a.example"])
mw = OffsiteMiddleware.from_crawler(crawler)
mw = build_from_crawler(OffsiteMiddleware, crawler)
mw.spider_opened(crawler.spider)
kwargs: dict[str, Any] = {}
if value is not UNSET:
@ -198,7 +200,7 @@ def test_request_scheduled_no_allowed_domains(value):
if value is not UNSET:
kwargs["allowed_domains"] = value
crawler.spider = crawler._create_spider(name="a", **kwargs)
mw = OffsiteMiddleware.from_crawler(crawler)
mw = build_from_crawler(OffsiteMiddleware, crawler)
mw.spider_opened(crawler.spider)
request = Request("https://example.com")
mw.request_scheduled(request, crawler.spider)
@ -208,7 +210,7 @@ def test_request_scheduled_invalid_domains():
crawler = get_crawler(Spider)
allowed_domains = ["a.example", None, "http:////b.example", "//c.example"]
crawler.spider = crawler._create_spider(name="a", allowed_domains=allowed_domains)
mw = OffsiteMiddleware.from_crawler(crawler)
mw = build_from_crawler(OffsiteMiddleware, crawler)
mw.spider_opened(crawler.spider)
request = Request("https://a.example")
mw.request_scheduled(request, crawler.spider)
@ -221,7 +223,7 @@ def test_request_scheduled_invalid_domains():
def test_repeated_offsite_domain():
crawler = get_crawler(Spider)
crawler.spider = crawler._create_spider(name="a", allowed_domains=["example.com"])
mw = OffsiteMiddleware.from_crawler(crawler)
mw = build_from_crawler(OffsiteMiddleware, crawler)
mw.spider_opened(crawler.spider)
req1 = Request("http://other.org/1")
req2 = Request("http://other.org/2")
@ -237,10 +239,25 @@ def test_repeated_offsite_domain():
assert crawler.stats.get_value("offsite/filtered") == 2
def test_should_follow_override():
class RootOnlyOffsiteMiddleware(OffsiteMiddleware):
def should_follow(self, request: Request, spider: Spider) -> bool:
allowed_domains: list[str] = getattr(spider, "allowed_domains", [])
return urlparse_cached(request).hostname in allowed_domains
crawler = get_crawler(Spider)
crawler.spider = crawler._create_spider(name="a", allowed_domains=["example.com"])
mw = build_from_crawler(RootOnlyOffsiteMiddleware, crawler)
mw.spider_opened(crawler.spider)
assert mw.process_request(Request("https://example.com/1")) is None
with pytest.raises(IgnoreRequest):
mw.process_request(Request("https://www.example.com/1"))
def test_ignore_request_reason():
crawler = get_crawler(Spider)
crawler.spider = crawler._create_spider(name="a", allowed_domains=["example.com"])
mw = OffsiteMiddleware.from_crawler(crawler)
mw = build_from_crawler(OffsiteMiddleware, crawler)
mw.spider_opened(crawler.spider)
request = Request("http://other.org/1")
with pytest.raises(
@ -258,7 +275,7 @@ def test_dynamic_allowed_domains():
crawler = get_crawler(DomainSpider)
spider = DomainSpider.from_crawler(crawler, allowed_domains=["a.example"])
crawler.spider = spider
mw = OffsiteMiddleware.from_crawler(crawler)
mw = build_from_crawler(OffsiteMiddleware, crawler)
mw.spider_opened(spider)
with pytest.raises(IgnoreRequest):
@ -284,7 +301,7 @@ def test_dynamic_allowed_domains_caching():
crawler = get_crawler(DomainSpider)
spider = DomainSpider.from_crawler(crawler, allowed_domains=["a.example"])
crawler.spider = spider
mw = TrackingMiddleware.from_crawler(crawler)
mw = build_from_crawler(TrackingMiddleware, crawler)
mw.spider_opened(spider)
for _ in range(3):

View File

@ -27,7 +27,7 @@ class TestRedirectMiddleware(TestRedirectBase):
def setup_method(self):
crawler = get_crawler(DefaultSpider)
crawler.spider = crawler._create_spider()
self.mw = self.mwcls.from_crawler(crawler)
self.mw = build_from_crawler(self.mwcls, crawler)
def get_response(self, request, location, status=302):
headers = {"Location": location}
@ -208,7 +208,7 @@ class TestRedirectMiddleware(TestRedirectBase):
response = Response(source_url, headers=resp_headers, status=302)
crawler = get_crawler()
referer_mw = build_from_crawler(RefererMiddleware, crawler)
redirect_mw = self.mwcls.from_crawler(crawler)
redirect_mw = build_from_crawler(self.mwcls, crawler)
redirect_mw._referer_spider_middleware = referer_mw
redirect_request = redirect_mw.process_response(source_request, response)
if expected_referer:
@ -223,7 +223,7 @@ class TestRedirectMiddleware(TestRedirectBase):
source_url, headers={"Referer": "http://example.com/old"}
)
response = Response(source_url, headers={"Location": redirect_url}, status=302)
redirect_mw = self.mwcls.from_crawler(get_crawler())
redirect_mw = build_from_crawler(self.mwcls, get_crawler())
redirect_mw._referer_spider_middleware = None
redirect_request = redirect_mw.process_response(source_request, response)
assert "Referer" not in redirect_request.headers
@ -352,7 +352,7 @@ class TestRedirectMiddleware(TestRedirectBase):
@pytest.mark.parametrize(SCHEME_PARAMS, REDIRECT_SCHEME_CASES)
def test_redirect_schemes(url, location, target):
crawler = get_crawler(Spider)
mw = RedirectMiddleware.from_crawler(crawler)
mw = build_from_crawler(RedirectMiddleware, crawler)
request = Request(url)
response = Response(url, headers={"Location": location}, status=301)
redirect = mw.process_response(request, response)
@ -477,4 +477,4 @@ def test_warning_subclass(caplog):
def test_not_configured():
crawler = get_crawler(DefaultSpider, {"REDIRECT_ENABLED": False})
with pytest.raises(NotConfigured):
RedirectMiddleware.from_crawler(crawler)
build_from_crawler(RedirectMiddleware, crawler)

View File

@ -32,7 +32,7 @@ class TestMetaRefreshMiddleware(TestRedirectBase):
def setup_method(self):
crawler = get_crawler(Spider)
self.mw = self.mwcls.from_crawler(crawler)
self.mw = build_from_crawler(self.mwcls, crawler)
def _body(
self, interval: int = 5, url: str = "http://example.org/newpage"
@ -95,7 +95,7 @@ class TestMetaRefreshMiddleware(TestRedirectBase):
"""Test that Scrapy 1.x behavior remains possible"""
settings = {"METAREFRESH_IGNORE_TAGS": ["script", "noscript"]}
crawler = get_crawler(Spider, settings)
mw = MetaRefreshMiddleware.from_crawler(crawler)
mw = build_from_crawler(MetaRefreshMiddleware, crawler)
req = Request(url="http://example.org")
body = (
"""<noscript><meta http-equiv="refresh" """
@ -134,7 +134,7 @@ class TestMetaRefreshMiddleware(TestRedirectBase):
)
def test_meta_refresh_schemes(url, location, target):
crawler = get_crawler(Spider)
mw = MetaRefreshMiddleware.from_crawler(crawler)
mw = build_from_crawler(MetaRefreshMiddleware, crawler)
request = Request(url)
response = HtmlResponse(url, body=meta_refresh_body(location))
redirect = mw.process_response(request, response)
@ -169,4 +169,4 @@ def test_warning_meta_refresh_middleware(caplog):
def test_not_configured():
crawler = get_crawler(Spider, {"METAREFRESH_ENABLED": False})
with pytest.raises(NotConfigured):
MetaRefreshMiddleware.from_crawler(crawler)
build_from_crawler(MetaRefreshMiddleware, crawler)

View File

@ -16,6 +16,7 @@ from scrapy.exceptions import (
from scrapy.http import Request, Response
from scrapy.settings.default_settings import RETRY_EXCEPTIONS
from scrapy.spiders import Spider
from scrapy.utils.misc import build_from_crawler
from scrapy.utils.spider import DefaultSpider
from scrapy.utils.test import get_crawler
@ -24,7 +25,7 @@ class TestRetry:
def setup_method(self):
self.crawler = get_crawler(DefaultSpider)
self.crawler.spider = self.crawler._create_spider()
self.mw = RetryMiddleware.from_crawler(self.crawler)
self.mw = build_from_crawler(RetryMiddleware, self.crawler)
self.mw.max_retry_times = 2
def test_priority_adjust(self):
@ -94,7 +95,7 @@ class TestRetry:
DefaultSpider, settings_dict={"RETRY_GIVE_UP_LOG_LEVEL": "WARNING"}
)
crawler.spider = crawler._create_spider()
mw = RetryMiddleware.from_crawler(crawler)
mw = build_from_crawler(RetryMiddleware, crawler)
mw.max_retry_times = 0
req = Request("http://example.com/503")
rsp = Response("http://example.com/503", body=b"", status=503)
@ -148,7 +149,7 @@ class TestRetry:
}
crawler = get_crawler(DefaultSpider, settings_dict=settings_dict)
crawler.spider = crawler._create_spider()
mw = RetryMiddleware.from_crawler(crawler)
mw = build_from_crawler(RetryMiddleware, crawler)
req = Request(f"http://www.scrapytest.org/{exc.__name__}")
self._test_retry_exception(req, exc("foo"), mw)
@ -178,7 +179,7 @@ class TestMaxRetryTimes:
def get_middleware(self, settings: dict[str, Any] | None = None) -> RetryMiddleware:
crawler = get_crawler(DefaultSpider, settings or {})
crawler.spider = crawler._create_spider()
return RetryMiddleware.from_crawler(crawler)
return build_from_crawler(RetryMiddleware, crawler)
def test_with_settings_zero(self):
max_retry_times = 0

Some files were not shown because too many files have changed in this diff Show More