Merge remote-tracking branch 'origin/master' into no-warnings

This commit is contained in:
Adrian Chaves 2026-07-31 14:44:48 +02:00
commit b8cb7d2d6e
38 changed files with 1382 additions and 70 deletions

45
.github/workflows/codspeed.yml vendored Normal file
View File

@ -0,0 +1,45 @@
---
name: codspeed
on:
push:
branches:
- master
pull_request:
paths:
- scrapy/**
- tests/benchmarks/**
- .github/workflows/codspeed.yml
- tox.ini
workflow_dispatch:
concurrency:
group: ${{ github.workflow }}-${{ github.head_ref || github.run_id }}
cancel-in-progress: true
permissions: {}
jobs:
benchmark:
runs-on: ubuntu-latest
permissions:
contents: read
id-token: write # OIDC authentication with CodSpeed
steps:
- uses: actions/checkout@v6
- name: Set up Python 3.14
uses: actions/setup-python@v6
with:
python-version: '3.14'
- name: Install dependencies
run: |
pip install --upgrade pip
pip install --upgrade tox
tox -n -e benchmark
- name: Run benchmarks
uses: CodSpeedHQ/action@v4
with:
mode: simulation
run: tox -e benchmark

View File

@ -14,14 +14,18 @@ jobs:
tests:
runs-on: macos-latest
env:
PYTEST_ADDOPTS: -n auto
PYTEST_ADDOPTS: ${{ matrix.coverage && '-n auto' || '-n auto --no-cov' }}
strategy:
fail-fast: false
matrix:
python-version: ["3.10", "3.11", "3.12", "3.13", "3.14"]
python-version: ["3.10", "3.11", "3.12", "3.13"]
env:
- TOXENV: py
include:
- python-version: '3.14'
env:
TOXENV: py
coverage: true
- python-version: '3.14'
env:
TOXENV: no-reactor
@ -41,6 +45,7 @@ jobs:
tox
- name: Upload coverage report
if: ${{ matrix.coverage }}
uses: codecov/codecov-action@v5
- name: Upload test results

View File

@ -14,7 +14,7 @@ jobs:
tests:
runs-on: ubuntu-latest
env:
PYTEST_ADDOPTS: -n auto
PYTEST_ADDOPTS: ${{ matrix.coverage && '-n auto' || '-n auto --no-cov' }}
strategy:
fail-fast: false
matrix:
@ -34,12 +34,15 @@ jobs:
- python-version: "3.14"
env:
TOXENV: py
coverage: true
- python-version: "3.14"
env:
TOXENV: default-reactor
coverage: true
- python-version: "3.14"
env:
TOXENV: no-reactor
coverage: true
# pinned due to https://github.com/pypy/pypy/issues/5388
- python-version: pypy3.11-7.3.20
env:
@ -49,12 +52,15 @@ jobs:
- python-version: "3.10.19"
env:
TOXENV: min
coverage: true
- python-version: "3.10.19"
env:
TOXENV: min-default-reactor
coverage: true
- python-version: "3.10.19"
env:
TOXENV: min-no-reactor
coverage: true
# pinned due to https://github.com/pypy/pypy/issues/5388
- python-version: pypy3.11-7.3.20
env:
@ -62,16 +68,20 @@ jobs:
- python-version: "3.10.19"
env:
TOXENV: min-extra-deps
coverage: true
- python-version: "3.10.19"
env:
TOXENV: min-botocore
coverage: true
- python-version: "3.14"
env:
TOXENV: extra-deps
coverage: true
- python-version: "3.14"
env:
TOXENV: no-reactor-extra-deps
coverage: true
# pinned due to https://github.com/pypy/pypy/issues/5388
- python-version: pypy3.11-7.3.20
env:
@ -79,6 +89,7 @@ jobs:
- python-version: "3.14"
env:
TOXENV: botocore
coverage: true
steps:
- uses: actions/checkout@v6
@ -104,6 +115,7 @@ jobs:
tox
- name: Upload coverage report
if: ${{ matrix.coverage }}
uses: codecov/codecov-action@v5
- name: Upload test results

View File

@ -14,7 +14,7 @@ jobs:
tests:
runs-on: windows-latest
env:
PYTEST_ADDOPTS: -n auto
PYTEST_ADDOPTS: ${{ matrix.coverage && '-n auto' || '-n auto --no-cov' }}
strategy:
fail-fast: false
matrix:
@ -34,6 +34,7 @@ jobs:
- python-version: "3.14"
env:
TOXENV: py
coverage: true
- python-version: "3.14"
env:
TOXENV: default-reactor
@ -68,6 +69,7 @@ jobs:
tox
- name: Upload coverage report
if: ${{ matrix.coverage }}
uses: codecov/codecov-action@v5
- name: Upload test results

View File

@ -27,6 +27,6 @@ repos:
hooks:
- id: sphinx-lint
- repo: https://github.com/scrapy/sphinx-scrapy
rev: 0.8.8
rev: 0.8.9
hooks:
- id: sphinx-scrapy

View File

@ -5,7 +5,7 @@
:alt: Scrapy
:width: 480px
|version| |python_version| |ubuntu| |macos| |windows| |coverage| |conda| |deepwiki|
|version| |python_version| |tests| |coverage| |conda| |deepwiki|
.. |version| image:: https://img.shields.io/pypi/v/Scrapy.svg
:target: https://pypi.org/pypi/Scrapy
@ -15,17 +15,9 @@
:target: https://pypi.org/pypi/Scrapy
:alt: Supported Python Versions
.. |ubuntu| image:: https://github.com/scrapy/scrapy/workflows/Ubuntu/badge.svg
:target: https://github.com/scrapy/scrapy/actions?query=workflow%3AUbuntu
:alt: Ubuntu
.. |macos| image:: https://github.com/scrapy/scrapy/workflows/macOS/badge.svg
:target: https://github.com/scrapy/scrapy/actions?query=workflow%3AmacOS
:alt: macOS
.. |windows| image:: https://github.com/scrapy/scrapy/workflows/Windows/badge.svg
:target: https://github.com/scrapy/scrapy/actions?query=workflow%3AWindows
:alt: Windows
.. |tests| image:: https://img.shields.io/github/check-runs/scrapy/scrapy/master?label=tests
:target: https://github.com/scrapy/scrapy/actions?query=branch%3Amaster
:alt: Tests
.. |coverage| image:: https://img.shields.io/codecov/c/github/scrapy/scrapy/master.svg
:target: https://codecov.io/github/scrapy/scrapy?branch=master

View File

@ -54,6 +54,9 @@ if not H2_ENABLED:
if find_spec("httpx2") is None and find_spec("httpx") is None:
collect_ignore.append("scrapy/core/downloader/handlers/_httpx.py")
if find_spec("pytest_codspeed") is None:
collect_ignore.append("tests/benchmarks")
def pytest_addoption(parser, pluginmanager):
if pluginmanager.hasplugin("twisted"):

View File

@ -5,4 +5,4 @@ sphinx
sphinx-notfound-page
sphinx-rtd-theme
sphinx-rtd-dark-mode
sphinx-scrapy @ git+https://github.com/scrapy/sphinx-scrapy.git@0.8.8
sphinx-scrapy @ git+https://github.com/scrapy/sphinx-scrapy.git@0.8.9

View File

@ -153,7 +153,7 @@ sphinx-rtd-theme==3.1.0
# via
# -r docs/requirements.in
# sphinx-rtd-dark-mode
sphinx-scrapy @ git+https://github.com/scrapy/sphinx-scrapy.git@c0b2ac815afc3cb8857d575cecb5d55c05e6b737
sphinx-scrapy @ git+https://github.com/scrapy/sphinx-scrapy.git@912ed0507405e16ac60a47dd08195a1cd0ced984
# via -r docs/requirements.in
sphinx-sitemap==2.9.0
# via sphinx-scrapy

View File

@ -114,8 +114,8 @@ some usage help and the available commands::
scrapy <command> [options] [args]
Available commands:
crawl Run a spider
fetch Fetch a URL using the Scrapy downloader
runspider Run a spider from a Python file, no project required
[...]
The first line will print the currently active project if you're inside a
@ -263,7 +263,9 @@ crawl
* Syntax: ``scrapy crawl <spider>``
* Requires project: *yes*
Start crawling using a spider.
Start crawling using the spider with the given :attr:`~scrapy.Spider.name`,
which must be one of those that :command:`list` reports. To run a spider from a
file instead, use :command:`runspider`.
Supported options:
@ -571,8 +573,9 @@ runspider
* Syntax: ``scrapy runspider <spider_file.py>``
* Requires project: *no*
Run a spider self-contained in a Python file, without having to create a
project.
Run the spider defined in the given Python file, without requiring a project.
Supported options: the same as :command:`crawl`.
Example usage::
@ -665,6 +668,8 @@ Example:
COMMANDS_MODULE = "mybot.commands"
.. note:: This is a :ref:`pre-crawler setting <pre-crawler-settings>`.
.. _Deploying your project: https://scrapyd.readthedocs.io/en/latest/deploy.html
Register commands via setup.py entry points

View File

@ -136,18 +136,10 @@ Core Stats extension
Enable the collection of core statistics, provided the stats collection is
enabled (see :ref:`topics-stats`).
The following stats are collected:
* ``start_time``: start date/time of the crawl (:class:`~datetime.datetime`).
* ``finish_time``: end date/time of the crawl (:class:`~datetime.datetime`).
* ``elapsed_time_seconds``: total crawl duration in seconds (:class:`float`).
* ``finish_reason``: the closing reason string (e.g. ``"finished"``,
``"closespider_timeout"``).
* ``item_scraped_count``: total number of items that passed all pipelines.
* ``item_dropped_count``: total number of items dropped by a pipeline.
* ``item_dropped_reasons_count/<ExceptionName>``: per-exception drop count
(e.g. ``item_dropped_reasons_count/DropItem``).
* ``response_received_count``: total number of HTTP responses received.
The following stats are collected: :stat:`elapsed_time_seconds`,
:stat:`finish_reason`, :stat:`finish_time`, :stat:`item_dropped_count`,
:stat:`item_dropped_reasons_count/{exception}`, :stat:`item_scraped_count`,
:stat:`response_received_count`, :stat:`start_time`.
Log Count extension
~~~~~~~~~~~~~~~~~~~
@ -190,7 +182,7 @@ Monitors the memory used by the Scrapy process that runs the spider and:
1. sends a :signal:`memusage_warning_reached` signal when it exceeds
:setting:`MEMUSAGE_WARNING_MB`
2. closes the spider with the `"memusage_exceeded"` reason when it exceeds
2. closes the spider with the ``"memusage_exceeded"`` reason when it exceeds
:setting:`MEMUSAGE_LIMIT_MB`
This extension is enabled by the :setting:`MEMUSAGE_ENABLED` setting and
@ -214,7 +206,8 @@ An extension for debugging memory usage. It collects information about:
* objects left alive that shouldn't. For more info, see :ref:`topics-leaks-trackrefs`
To enable this extension, turn on the :setting:`MEMDEBUG_ENABLED` setting. The
info will be stored in the stats.
info will be stored in the :stat:`memdebug/gc_garbage_count` and
:stat:`memdebug/live_refs/{cls}` stats.
.. _topics-extensions-ref-spiderstate:

View File

@ -305,10 +305,21 @@ These settings cannot be :ref:`set from a spider <spider-settings>`.
These settings are:
- :setting:`TWISTED_REACTOR_ENABLED`
- :setting:`ADDONS`
- :setting:`COMMANDS_MODULE`
- :setting:`FORCE_CRAWLER_PROCESS`
- :setting:`SPIDER_LOADER_CLASS` and settings used by the corresponding
spider loader class, e.g. :setting:`SPIDER_MODULES` and
:setting:`SPIDER_LOADER_WARN_ONLY` for the default spider loader class.
- :setting:`TWISTED_REACTOR_ENABLED`
:setting:`ADDONS` is a special case: it can be set from a spider, but the
``update_pre_crawler_settings()`` method of :ref:`add-ons <topics-addons>`
enabled that way is not called.
:setting:`TWISTED_REACTOR` also acts as a pre-crawler setting when running a
:ref:`command that needs a CrawlerProcess <topics-commands-crawlerprocess>`,
since its project-level value determines the crawler process class.
.. _reactor-settings:
@ -409,6 +420,9 @@ Default: ``{}``
A dict containing paths to the add-ons enabled in your project and their
priorities. For more information, see :ref:`topics-addons`.
.. note:: This is a :ref:`pre-crawler setting <pre-crawler-settings>`, with a
caveat described in that section.
.. setting:: ASYNCIO_EVENT_LOOP
ASYNCIO_EVENT_LOOP
@ -1402,6 +1416,8 @@ When :setting:`TWISTED_REACTOR_ENABLED` is set to ``False``,
Set this to ``True`` if you want to set :setting:`TWISTED_REACTOR` to a
non-default value in :ref:`per-spider settings <spider-settings>`.
.. note:: This is a :ref:`pre-crawler setting <pre-crawler-settings>`.
.. setting:: FTP_PASSIVE_MODE
FTP_PASSIVE_MODE
@ -1855,7 +1871,8 @@ Default: ``False``
Setting to ``True`` will log debug information about the requests scheduler.
This currently logs (only once) if the requests cannot be serialized to disk.
Stats counter (``scheduler/unserializable``) tracks the number of times this happens.
The :stat:`scheduler/unserializable` stat tracks the number of times this
happens.
Example entry in logs::

View File

@ -504,6 +504,27 @@ headers_received
:param spider: the spider associated with the response
:type spider: :class:`~scrapy.Spider` object
robots_parsed
~~~~~~~~~~~~~
.. signal:: robots_parsed
.. function:: robots_parsed(robotparser, request)
.. versionadded:: VERSION
Sent by
:class:`~scrapy.downloadermiddlewares.robotstxt.RobotsTxtMiddleware` after it
downloads and parses a :file:`robots.txt` file, for the host that *request*
targets.
This signal supports :ref:`asynchronous handlers <signal-deferred>`.
:param robotparser: the parser holding the parsed :file:`robots.txt` contents
:type robotparser: :class:`~scrapy.robotstxt.RobotParser` object
:param request: the request that triggered the :file:`robots.txt` download
:type request: :class:`~scrapy.Request` object
Response signals
----------------

View File

@ -21,6 +21,8 @@ using the Stats Collector from.
Another feature of the Stats Collector is that it's very efficient (when
enabled) and extremely efficient (almost unnoticeable) when disabled.
See :ref:`topics-stats-reference` below for the stats that Scrapy sets.
.. _topics-stats-usecases:
Common Stats Collector uses
@ -101,3 +103,642 @@ DummyStatsCollector
-------------------
.. autoclass:: DummyStatsCollector
.. _topics-stats-reference:
Built-in stats reference
========================
Scrapy sets the following :ref:`stats <topics-stats>`. Components other than
those built into Scrapy may set additional stats; see their documentation.
Stat keys that contain a ``{placeholder}`` below stand for a family of stats,
one per actual value of the placeholder.
.. note:: Most stats are set by a specific :ref:`component
<topics-components>`, and are only present if that component is enabled and
its code path is reached. A stat that is missing from
:meth:`~scrapy.statscollectors.StatsCollector.get_stats` output is
equivalent to a counter of 0.
.. stat:: downloader/exception_count
``downloader/exception_count``
Number of exceptions raised while downloading requests.
Set by :class:`~scrapy.downloadermiddlewares.stats.DownloaderStats`.
.. stat:: downloader/exception_type_count/{exception_type}
``downloader/exception_type_count/{exception_type}``
Number of exceptions raised while downloading requests, per exception type,
where ``{exception_type}`` is the import path of the exception class, e.g.
``twisted.internet.error.DNSLookupError``.
Set by :class:`~scrapy.downloadermiddlewares.stats.DownloaderStats`.
.. stat:: downloader/request_bytes
``downloader/request_bytes``
Total size, in bytes, of the requests sent, counting the request line, the
headers and the body. As with :stat:`downloader/request_count`, requests
served from the cache are also counted.
It is an approximation, reconstructed from each :class:`~scrapy.Request`
object instead of measured on the wire, so it does not account for the
actual bytes that the :ref:`download handler
<topics-download-handlers>` sends, e.g. transport-level overhead.
Set by :class:`~scrapy.downloadermiddlewares.stats.DownloaderStats`.
.. stat:: downloader/request_count
``downloader/request_count``
Number of requests sent.
Requests that :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`
serves from the cache are also counted, even though they are never sent,
because it handles requests after
:class:`~scrapy.downloadermiddlewares.stats.DownloaderStats`.
Set by :class:`~scrapy.downloadermiddlewares.stats.DownloaderStats`.
.. stat:: downloader/request_method_count/{method}
``downloader/request_method_count/{method}``
Number of requests sent, per HTTP method, e.g. ``GET`` or ``POST``. As with
:stat:`downloader/request_count`, requests served from the cache are also
counted.
Set by :class:`~scrapy.downloadermiddlewares.stats.DownloaderStats`.
.. stat:: downloader/response_bytes
``downloader/response_bytes``
Total size, in bytes, of the responses received, counting the status line,
the headers and the body. It covers the same responses as
:stat:`downloader/response_count`.
The body is counted as received, i.e. still compressed for responses that
used ``Content-Encoding``, because
:class:`~scrapy.downloadermiddlewares.stats.DownloaderStats` handles
responses before
:class:`~scrapy.downloadermiddlewares.httpcompression.HttpCompressionMiddleware`
decompresses them. See :stat:`httpcompression/response_bytes` for
decompressed sizes.
Set by :class:`~scrapy.downloadermiddlewares.stats.DownloaderStats`.
.. stat:: downloader/response_count
``downloader/response_count``
Number of responses received.
It counts responses that :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`
serves from the cache, even though they do not come from the network, and
responses that a downloader middleware consumes before they reach your
spider, e.g. redirect responses that :class:`~scrapy.downloadermiddlewares.redirect.RedirectMiddleware`
turns into new requests. Compare with :stat:`response_received_count`.
Set by :class:`~scrapy.downloadermiddlewares.stats.DownloaderStats`.
.. stat:: downloader/response_status_count/{status_code}
``downloader/response_status_count/{status_code}``
Number of responses received, per HTTP status code, e.g. ``200`` or
``404``. It covers the same responses as :stat:`downloader/response_count`.
Set by :class:`~scrapy.downloadermiddlewares.stats.DownloaderStats`.
.. stat:: dupefilter/filtered
``dupefilter/filtered``
Number of requests dropped as duplicates.
Set by :class:`~scrapy.dupefilters.RFPDupeFilter`.
.. stat:: elapsed_time_seconds
``elapsed_time_seconds``
Time, as a :class:`float`, in seconds, between the :signal:`spider_opened`
and the :signal:`spider_closed` signals.
Set by :class:`~scrapy.extensions.corestats.CoreStats`.
.. stat:: feedexport/failed_count/{storage}
``feedexport/failed_count/{storage}``
Number of :ref:`feeds <topics-feed-exports>` that could not be stored, per
:ref:`storage backend <topics-feed-storage-backends>`, where ``{storage}``
is the class name of the storage backend, e.g. ``FileFeedStorage``.
.. stat:: feedexport/success_count/{storage}
``feedexport/success_count/{storage}``
Number of :ref:`feeds <topics-feed-exports>` stored successfully, per
:ref:`storage backend <topics-feed-storage-backends>`, where ``{storage}``
is the class name of the storage backend, e.g. ``FileFeedStorage``.
.. stat:: file_count
``file_count``
Number of files handled by the :ref:`media pipelines
<topics-media-pipeline>`.
.. stat:: file_status_count/{status}
``file_status_count/{status}``
Number of files handled by the :ref:`media pipelines
<topics-media-pipeline>`, per status, where ``{status}`` is one of:
- ``downloaded``: the file was downloaded.
- ``cached``: the file came from the
:class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`
cache.
- ``uptodate``: the file was already in the storage backend and had not
:ref:`expired <file-expiration>`, so it was not downloaded again.
.. stat:: finish_reason
``finish_reason``
String indicating why the crawl finished. It matches the *reason* argument
of the :signal:`spider_closed` signal.
Scrapy uses the following reasons:
- ``cancelled``: the spider was closed without a more specific reason,
e.g. because :exc:`~scrapy.exceptions.CloseSpider` was raised without
one.
- ``closespider_errorcount``: see :setting:`CLOSESPIDER_ERRORCOUNT`.
- ``closespider_itemcount``: see :setting:`CLOSESPIDER_ITEMCOUNT`.
- ``closespider_pagecount``: see :setting:`CLOSESPIDER_PAGECOUNT`.
- ``closespider_pagecount_no_item``: see
:setting:`CLOSESPIDER_PAGECOUNT_NO_ITEM`.
- ``closespider_timeout``: see :setting:`CLOSESPIDER_TIMEOUT`.
- ``closespider_timeout_no_item``: see
:setting:`CLOSESPIDER_TIMEOUT_NO_ITEM`.
- ``finished``: the spider became idle with no pending requests, i.e. it
finished normally.
- ``memusage_exceeded``: see :setting:`MEMUSAGE_LIMIT_MB`.
- ``shutdown``: the crawl was interrupted, e.g. by a system signal such
as ``SIGINT`` (:kbd:`Ctrl-C`).
Third-party components and your own code may use any other reason, e.g. by
raising :exc:`~scrapy.exceptions.CloseSpider` with it.
Set by :class:`~scrapy.extensions.corestats.CoreStats`.
.. stat:: finish_time
``finish_time``
Timezone-aware :class:`~datetime.datetime` object, in UTC, indicating when
the :signal:`spider_closed` signal was sent.
Set by :class:`~scrapy.extensions.corestats.CoreStats`.
.. stat:: httpcache/errorrecovery
``httpcache/errorrecovery``
Number of times that a stale cached response was used because downloading a
fresh response raised an exception.
Set by :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`.
.. stat:: httpcache/firsthand
``httpcache/firsthand``
Number of responses that were downloaded without a matching cache entry to
validate against, i.e. responses for requests counted in
:stat:`httpcache/miss`.
It is lower than :stat:`httpcache/miss` when some of those requests yield
no response, either because they are dropped (see
:stat:`httpcache/ignore`) or because their download fails.
Set by :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`.
.. stat:: httpcache/hit
``httpcache/hit``
Number of requests served from the cache.
Set by :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`.
.. stat:: httpcache/ignore
``httpcache/ignore``
Number of requests dropped because they were not in the cache and
:setting:`HTTPCACHE_IGNORE_MISSING` is ``True``.
Set by :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`.
.. stat:: httpcache/invalidate
``httpcache/invalidate``
Number of times that a cached response failed validation and was replaced
with a freshly downloaded response.
Set by :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`.
.. stat:: httpcache/miss
``httpcache/miss``
Number of requests for which no cache entry could be read, either because
there was none or because reading it failed, in which case the request is
also counted in :stat:`httpcache/retrieve_error`. Those requests are
downloaded (see :stat:`httpcache/firsthand`), or dropped if
:setting:`HTTPCACHE_IGNORE_MISSING` is ``True`` (see
:stat:`httpcache/ignore`).
Requests with a stale cache entry are not counted here; see
:stat:`httpcache/revalidate` and :stat:`httpcache/invalidate`.
Set by :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`.
.. stat:: httpcache/retrieve_error
``httpcache/retrieve_error``
Number of cache entries that could not be read, and hence were treated as
cache misses. Those requests are also counted in :stat:`httpcache/miss`.
Set by :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`.
.. stat:: httpcache/revalidate
``httpcache/revalidate``
Number of times that a cached response was successfully validated against
the target server, and hence used instead of the fresh response.
Set by :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`.
.. stat:: httpcache/store
``httpcache/store``
Number of responses stored in the cache.
Set by :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`.
.. stat:: httpcache/uncacheable
``httpcache/uncacheable``
Number of responses not stored in the cache because the
:setting:`HTTPCACHE_POLICY` did not allow it.
Every response considered for caching is counted either here or in
:stat:`httpcache/store`, so ``httpcache/store + httpcache/uncacheable``
equals ``httpcache/firsthand + httpcache/invalidate``.
Set by :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`.
.. stat:: httpcompression/response_bytes
``httpcompression/response_bytes``
Total size, in bytes, of decompressed response bodies, counting only the
body and only responses that were actually decompressed. Compare with
:stat:`downloader/response_bytes`.
Set by
:class:`~scrapy.downloadermiddlewares.httpcompression.HttpCompressionMiddleware`.
.. stat:: httpcompression/response_count
``httpcompression/response_count``
Number of decompressed responses.
Set by
:class:`~scrapy.downloadermiddlewares.httpcompression.HttpCompressionMiddleware`.
.. stat:: httperror/response_ignored_count
``httperror/response_ignored_count``
Number of responses dropped because of their HTTP status code.
Set by :class:`~scrapy.spidermiddlewares.httperror.HttpErrorMiddleware`.
.. stat:: httperror/response_ignored_status_count/{status_code}
``httperror/response_ignored_status_count/{status_code}``
Number of responses dropped because of their HTTP status code, per HTTP
status code, e.g. ``404``.
Set by :class:`~scrapy.spidermiddlewares.httperror.HttpErrorMiddleware`.
.. stat:: item_dropped_count
``item_dropped_count``
Number of items dropped by an :ref:`item pipeline
<topics-item-pipeline>`, i.e. number of times that the
:signal:`item_dropped` signal was sent.
Set by :class:`~scrapy.extensions.corestats.CoreStats`.
.. stat:: item_dropped_reasons_count/{exception}
``item_dropped_reasons_count/{exception}``
Number of items dropped, per exception, where ``{exception}`` is the class
name of the exception that caused the item to be dropped.
Only :exc:`~scrapy.exceptions.DropItem` and its subclasses drop items, and
each one is counted under its own class name, e.g.
``item_dropped_reasons_count/DropItem`` for
:exc:`~scrapy.exceptions.DropItem` itself and
``item_dropped_reasons_count/MyDropItem`` for a ``MyDropItem`` subclass of
it. Any other exception raised by an :ref:`item pipeline
<topics-item-pipeline>` triggers the :signal:`item_error` signal instead of
:signal:`item_dropped`, and is not counted here or in
:stat:`item_dropped_count`.
Set by :class:`~scrapy.extensions.corestats.CoreStats`.
.. stat:: item_scraped_count
``item_scraped_count``
Number of items that passed all :ref:`item pipelines
<topics-item-pipeline>`, i.e. number of times that the
:signal:`item_scraped` signal was sent.
Set by :class:`~scrapy.extensions.corestats.CoreStats`.
.. stat:: items_per_minute
``items_per_minute``
Average number of items scraped per minute during the crawl.
It is ``None`` if the crawl took less than a minute.
Set by :class:`~scrapy.extensions.logstats.LogStats`.
.. stat:: log_count/{level}
``log_count/{level}``
Number of log messages, per logging level name, e.g. ``INFO`` or
``WARNING``.
Only messages that the :setting:`LOG_LEVEL` setting allows are counted.
Set by :class:`~scrapy.extensions.logcount.LogCount`.
.. stat:: memdebug/gc_garbage_count
``memdebug/gc_garbage_count``
Number of objects in :data:`gc.garbage` when the spider is closed.
Set by :class:`~scrapy.extensions.memdebug.MemoryDebugger`, which requires
:setting:`MEMDEBUG_ENABLED` to be ``True``.
.. stat:: memdebug/live_refs/{cls}
``memdebug/live_refs/{cls}``
Number of live objects of class ``{cls}`` when the spider is closed, as
reported by :ref:`trackref <topics-leaks-trackrefs>`, e.g.
``memdebug/live_refs/HtmlResponse``.
Only set for classes with at least 1 live object.
Set by :class:`~scrapy.extensions.memdebug.MemoryDebugger`, which requires
:setting:`MEMDEBUG_ENABLED` to be ``True``.
.. stat:: memusage/limit_reached
``memusage/limit_reached``
``1`` if memory usage exceeded :setting:`MEMUSAGE_LIMIT_MB`, which also
stops the crawl.
Set by :class:`~scrapy.extensions.memusage.MemoryUsage`.
.. stat:: memusage/max
``memusage/max``
Maximum peak memory usage, in bytes, observed during the crawl.
Set by :class:`~scrapy.extensions.memusage.MemoryUsage`.
.. stat:: memusage/startup
``memusage/startup``
Peak memory usage, in bytes, when the engine started.
Set by :class:`~scrapy.extensions.memusage.MemoryUsage`.
.. stat:: memusage/warning_reached
``memusage/warning_reached``
``1`` if memory usage exceeded :setting:`MEMUSAGE_WARNING_MB`.
Set by :class:`~scrapy.extensions.memusage.MemoryUsage`.
.. stat:: offsite/domains
``offsite/domains``
Number of distinct domains for which at least 1 request was dropped for
being offsite.
Set by :class:`~scrapy.downloadermiddlewares.offsite.OffsiteMiddleware`.
.. stat:: offsite/filtered
``offsite/filtered``
Number of requests dropped for being offsite.
Set by :class:`~scrapy.downloadermiddlewares.offsite.OffsiteMiddleware`.
.. stat:: request_depth_count/{depth}
``request_depth_count/{depth}``
Number of requests scheduled at depth ``{depth}``, e.g.
``request_depth_count/2``.
Set by :class:`~scrapy.spidermiddlewares.depth.DepthMiddleware`, which
requires :setting:`DEPTH_STATS_VERBOSE` to be ``True`` for this stat.
.. stat:: request_depth_max
``request_depth_max``
Maximum depth reached.
Set by :class:`~scrapy.spidermiddlewares.depth.DepthMiddleware`.
.. stat:: response_received_count
``response_received_count``
Number of responses received, i.e. number of times that the
:signal:`response_received` signal was sent.
Unlike :stat:`downloader/response_count`, it does not count responses that
a downloader middleware consumes before they reach the engine, e.g.
redirect responses that :class:`~scrapy.downloadermiddlewares.redirect.RedirectMiddleware`
turns into new requests. Both count responses that :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`
serves from the cache.
Set by :class:`~scrapy.extensions.corestats.CoreStats`.
.. stat:: responses_per_minute
``responses_per_minute``
Average number of responses received per minute during the crawl.
It is ``None`` if the crawl took less than a minute.
Set by :class:`~scrapy.extensions.logstats.LogStats`.
.. stat:: retry/count
``retry/count``
Number of requests retried.
Set by :func:`~scrapy.downloadermiddlewares.retry.get_retry_request`, which
:class:`~scrapy.downloadermiddlewares.retry.RetryMiddleware` uses.
.. stat:: retry/max_reached
``retry/max_reached``
Number of requests that were not retried because they had already been
retried :setting:`RETRY_TIMES` times.
Set by :func:`~scrapy.downloadermiddlewares.retry.get_retry_request`, which
:class:`~scrapy.downloadermiddlewares.retry.RetryMiddleware` uses.
.. stat:: retry/reason_count/{reason}
``retry/reason_count/{reason}``
Number of requests retried, per reason, e.g.
``retry/reason_count/twisted.internet.error.TimeoutError`` or
``retry/reason_count/504 Gateway Time-out``.
Set by :func:`~scrapy.downloadermiddlewares.retry.get_retry_request`, which
:class:`~scrapy.downloadermiddlewares.retry.RetryMiddleware` uses.
.. note:: Code calling
:func:`~scrapy.downloadermiddlewares.retry.get_retry_request` may pass a
custom *stats_base_key*, in which case ``retry`` is replaced with that key
in the 3 stats above.
.. stat:: robotstxt/exception_count/{exception_type}
``robotstxt/exception_count/{exception_type}``
Number of exceptions raised while downloading ``robots.txt`` files, per
exception type, where ``{exception_type}`` is the string representation of
the exception class, e.g. ``<class
'twisted.internet.error.DNSLookupError'>``.
Set by
:class:`~scrapy.downloadermiddlewares.robotstxt.RobotsTxtMiddleware`.
.. stat:: robotstxt/forbidden
``robotstxt/forbidden``
Number of requests dropped for being disallowed by ``robots.txt``.
Set by
:class:`~scrapy.downloadermiddlewares.robotstxt.RobotsTxtMiddleware`.
.. stat:: robotstxt/request_count
``robotstxt/request_count``
Number of ``robots.txt`` files requested, i.e. 1 per network location for
which at least 1 request was sent.
Set by
:class:`~scrapy.downloadermiddlewares.robotstxt.RobotsTxtMiddleware`.
.. stat:: robotstxt/response_count
``robotstxt/response_count``
Number of ``robots.txt`` responses received.
Set by
:class:`~scrapy.downloadermiddlewares.robotstxt.RobotsTxtMiddleware`.
.. stat:: robotstxt/response_status_count/{status_code}
``robotstxt/response_status_count/{status_code}``
Number of ``robots.txt`` responses received, per HTTP status code, e.g.
``404``.
Set by
:class:`~scrapy.downloadermiddlewares.robotstxt.RobotsTxtMiddleware`.
.. stat:: scheduler/dequeued
``scheduler/dequeued``
Number of requests read from the :ref:`scheduler <topics-scheduler>`.
.. stat:: scheduler/dequeued/disk
``scheduler/dequeued/disk``
Number of requests read from the disk queue of the :ref:`scheduler
<topics-scheduler>`.
.. stat:: scheduler/dequeued/memory
``scheduler/dequeued/memory``
Number of requests read from the memory queue of the :ref:`scheduler
<topics-scheduler>`.
.. stat:: scheduler/enqueued
``scheduler/enqueued``
Number of requests stored into the :ref:`scheduler <topics-scheduler>`.
.. stat:: scheduler/enqueued/disk
``scheduler/enqueued/disk``
Number of requests stored into the disk queue of the :ref:`scheduler
<topics-scheduler>`.
.. stat:: scheduler/enqueued/memory
``scheduler/enqueued/memory``
Number of requests stored into the memory queue of the :ref:`scheduler
<topics-scheduler>`.
.. stat:: scheduler/unserializable
``scheduler/unserializable``
Number of requests that could not be stored into the disk queue of the
:ref:`scheduler <topics-scheduler>` because they could not be
:ref:`serialized <request-serialization>`, and hence were stored into the
memory queue instead.
.. stat:: spider_exceptions/count
``spider_exceptions/count``
Number of unhandled exceptions raised by spider callbacks.
Set by the :ref:`scraper <topics-architecture>`.
.. stat:: spider_exceptions/{exception}
``spider_exceptions/{exception}``
Number of unhandled exceptions raised by spider callbacks, per exception,
where ``{exception}`` is the class name of the exception, e.g.
``spider_exceptions/ValueError``.
Set by the :ref:`scraper <topics-architecture>`.
.. stat:: start_time
``start_time``
Timezone-aware :class:`~datetime.datetime` object, in UTC, indicating when
the :signal:`spider_opened` signal was sent.
Set by :class:`~scrapy.extensions.corestats.CoreStats`.
.. stat:: urllength/request_ignored_count
``urllength/request_ignored_count``
Number of requests dropped for having a URL longer than
:setting:`URLLENGTH_LIMIT`.
Set by :class:`~scrapy.spidermiddlewares.urllength.UrlLengthMiddleware`.

View File

@ -16,7 +16,7 @@ class Command(BaseRunSpiderCommand):
return "[options] <spider>"
def short_desc(self) -> str:
return "Run a spider"
return "Run a spider of the current project, by name"
def run(self, args: list[str], opts: argparse.Namespace) -> None:
if len(args) < 1:

View File

@ -32,10 +32,7 @@ def sanitize_module_name(module_name: str) -> str:
def extract_domain(url: str) -> str:
"""Extract domain name from URL string"""
o = urlparse(url)
if o.scheme == "" and o.netloc == "":
o = urlparse("//" + url.lstrip("/"))
return o.netloc
return urlparse(url).netloc
def verify_url_scheme(url: str) -> str:

View File

@ -41,7 +41,7 @@ class Command(BaseRunSpiderCommand):
spider: Spider | None = None
items: ClassVar[dict[int, list[Any]]] = {}
requests: ClassVar[dict[int, list[Request]]] = {}
spidercls: type[Spider] | None
spidercls: type[Spider] | None = None
first_response = None

View File

@ -38,7 +38,7 @@ class Command(BaseRunSpiderCommand):
return "[options] <spider_file>"
def short_desc(self) -> str:
return "Run a self-contained spider (without creating a project)"
return "Run a spider from a Python file, no project required"
def long_desc(self) -> str:
return "Run the spider defined in the given file"

View File

@ -366,8 +366,8 @@ class Scheduler(BaseScheduler):
Unless the received request is filtered out by the Dupefilter, attempt to push
it into the disk queue, falling back to pushing it into the memory queue.
Increment the appropriate stats, such as: ``scheduler/enqueued``,
``scheduler/enqueued/disk``, ``scheduler/enqueued/memory``.
Increment the appropriate stats, such as: :stat:`scheduler/enqueued`,
:stat:`scheduler/enqueued/disk`, :stat:`scheduler/enqueued/memory`.
Return ``True`` if the request was stored successfully, ``False`` otherwise.
"""
@ -390,8 +390,8 @@ class Scheduler(BaseScheduler):
falling back to the disk queue if the memory queue is empty.
Return ``None`` if there are no more enqueued requests.
Increment the appropriate stats, such as: ``scheduler/dequeued``,
``scheduler/dequeued/disk``, ``scheduler/dequeued/memory``.
Increment the appropriate stats, such as: :stat:`scheduler/dequeued`,
:stat:`scheduler/dequeued/disk`, :stat:`scheduler/dequeued/memory`.
"""
request: Request | None = self.mqs.pop()
assert self.stats is not None

View File

@ -11,6 +11,7 @@ from typing import TYPE_CHECKING
from twisted.internet.defer import Deferred
from scrapy import signals
from scrapy.exceptions import IgnoreRequest, NotConfigured
from scrapy.http import Request, Response
from scrapy.http.request import NO_CALLBACK
@ -98,7 +99,7 @@ class RobotsTxtMiddleware:
assert self.crawler.stats
try:
resp = await self.crawler.engine.download_async(robotsreq)
self._parse_robots(resp, netloc)
await self._parse_robots(resp, netloc, request)
except Exception as e:
if not isinstance(e, IgnoreRequest):
logger.error(
@ -115,13 +116,20 @@ class RobotsTxtMiddleware:
return await maybe_deferred_to_future(parser)
return parser
def _parse_robots(self, response: Response, netloc: str) -> None:
async def _parse_robots(
self, response: Response, netloc: str, request: Request
) -> None:
assert self.crawler.stats
self.crawler.stats.inc_value("robotstxt/response_count")
self.crawler.stats.inc_value(
f"robotstxt/response_status_count/{response.status}"
)
rp = self._parserimpl.from_crawler(self.crawler, response.body)
await self.crawler.signals.send_catch_log_async(
signal=signals.robots_parsed,
robotparser=rp,
request=request,
)
rp_dfd = self._parsers[netloc]
assert isinstance(rp_dfd, Deferred)
self._parsers[netloc] = rp

View File

@ -20,7 +20,7 @@ class LogCount:
"""Install a log handler that counts log messages by level.
The handler installed is :class:`scrapy.utils.log.LogCounterHandler`.
The counts are stored in stats as ``log_count/<level>``.
The counts are stored in the :stat:`log_count/{level}` stat.
.. versionadded:: 2.14
"""

View File

@ -67,6 +67,15 @@ class RobotParser(metaclass=ABCMeta):
:type user_agent: str or bytes
"""
def crawl_delay(self, user_agent: str | bytes) -> float | None:
"""Return the ``Crawl-delay`` directive for ``user_agent`` as a number
of seconds, or ``None`` if it is not set or the backend does not support
it.
.. versionadded:: VERSION
"""
return None
class PythonRobotParser(RobotParser):
def __init__(self, robotstxt_body: bytes, spider: Spider | None):
@ -85,6 +94,10 @@ class PythonRobotParser(RobotParser):
url = to_unicode(url)
return self.rp.can_fetch(user_agent, url)
def crawl_delay(self, user_agent: str | bytes) -> float | None:
delay = self.rp.crawl_delay(to_unicode(user_agent))
return None if delay is None else float(delay)
class RerpRobotParser(RobotParser):
def __init__(self, robotstxt_body: bytes, spider: Spider | None):
@ -105,6 +118,10 @@ class RerpRobotParser(RobotParser):
url = to_unicode(url)
return cast("bool", self.rp.is_allowed(user_agent, url))
def crawl_delay(self, user_agent: str | bytes) -> float | None:
delay = self.rp.get_crawl_delay(to_unicode(user_agent))
return None if delay is None else float(delay)
class ProtegoRobotParser(RobotParser):
def __init__(self, robotstxt_body: bytes, spider: Spider | None):
@ -121,3 +138,7 @@ class ProtegoRobotParser(RobotParser):
user_agent = to_unicode(user_agent)
url = to_unicode(url)
return self.rp.can_fetch(url, user_agent)
def crawl_delay(self, user_agent: str | bytes) -> float | None:
delay = self.rp.crawl_delay(to_unicode(user_agent))
return None if delay is None else float(delay)

View File

@ -21,6 +21,7 @@ response_received = object()
response_downloaded = object()
headers_received = object()
bytes_received = object()
robots_parsed = object()
item_scraped = object()
item_dropped = object()
item_error = object()

View File

@ -0,0 +1,28 @@
from __future__ import annotations
from typing import TYPE_CHECKING, Any
from scrapy.utils.test import get_crawler
if TYPE_CHECKING:
from scrapy import Spider
from scrapy.crawler import Crawler
def crawl(spidercls: type[Spider], settings: dict[str, Any], **kwargs: Any) -> Crawler:
"""Run a crawl to completion and return its crawler.
Unlike the rest of the test suite, benchmarks run without ``pytest-twisted``
and drive the reactor themselves, since the code being measured must be
callable synchronously by ``pytest-codspeed``.
"""
from twisted.internet import reactor
crawler = get_crawler(spidercls, settings)
result: list[Any] = []
crawler.crawl(**kwargs).addBoth(result.append)
while not result:
reactor.iterate(0.001)
if isinstance(result[0], BaseException):
raise result[0]
return crawler

View File

@ -0,0 +1,27 @@
from __future__ import annotations
from typing import TYPE_CHECKING
import pytest
from scrapy.utils.reactor import install_reactor
if TYPE_CHECKING:
from collections.abc import Generator
@pytest.fixture(scope="session", autouse=True)
def running_reactor() -> Generator[None]:
install_reactor("twisted.internet.asyncioreactor.AsyncioSelectorReactor")
from twisted.internet import reactor
# Marks the reactor as running without blocking, so that crawls can be
# driven with reactor.iterate(), see tests.benchmarks.crawl().
reactor.startRunning(installSignalHandlers=False)
yield
reactor.stop()
# Lets the shutdown event triggers run, e.g. to join the thread pool.
reactor.iterate(0)

View File

@ -0,0 +1,69 @@
from __future__ import annotations
from typing import TYPE_CHECKING, Any
from urllib.parse import urlencode
import pytest
from scrapy import Field, Item, Request, Spider
from scrapy.linkextractors import LinkExtractor
from tests.benchmarks import crawl
if TYPE_CHECKING:
from collections.abc import AsyncIterator
from pytest_codspeed import BenchmarkFixture # type: ignore[import-not-found]
from scrapy.http import Response
from tests.mockserver.http import MockServer
pytest.importorskip("pytest_codspeed", reason="Benchmarks require pytest-codspeed")
PAGES = 100
LINKS_PER_PAGE = 5
class _Page(Item):
url = Field()
anchors = Field()
class _FollowSpider(Spider):
name = "benchmark"
url: str
link_extractor = LinkExtractor()
async def start(self) -> AsyncIterator[Any]:
yield Request(self.url, dont_filter=True)
def parse(self, response: Response) -> Any:
yield _Page(
url=response.url,
anchors=response.css("a::text").getall(),
)
for link in self.link_extractor.extract_links(response): # type: ignore[arg-type]
yield Request(link.url)
class _Pipeline:
def process_item(self, item: Any) -> Any:
return item
def test_overhead_http(benchmark: BenchmarkFixture, mockserver: MockServer) -> None:
"""Per-request overhead of a crawl over HTTP.
The pages are small on purpose, so that the cost of parsing them stays
negligible next to the cost of moving requests and responses through the
engine, the middlewares and the download handler.
"""
query = urlencode({"total": PAGES, "show": LINKS_PER_PAGE, "order": "desc"})
url = mockserver.url(f"/follow?{query}")
settings = {"ITEM_PIPELINES": {_Pipeline: 100}, "LOG_ENABLED": False}
def run() -> None:
crawler = crawl(_FollowSpider, settings, url=url)
assert crawler.stats
assert crawler.stats.get_value("item_scraped_count") == PAGES + 1
benchmark(run)

View File

@ -23,6 +23,18 @@ class TestCrawlCommand(TestProjectBase):
_, _, stderr = self.crawl(code, proj_path, args=args)
return stderr
def test_no_spider(self, proj_path: Path) -> None:
returncode, out, _ = proc("crawl", cwd=proj_path)
assert returncode == 2
assert "Usage" in out
def test_multiple_spiders(self, proj_path: Path) -> None:
returncode, _, err = proc("crawl", "myspider", "myspider2", cwd=proj_path)
assert returncode == 2
assert (
"running 'scrapy crawl' with more than one spider is not supported" in err
)
def test_no_output(self, proj_path: Path) -> None:
spider_code = """
import scrapy

View File

@ -2,13 +2,24 @@ from __future__ import annotations
from typing import TYPE_CHECKING
import pytest
from tests.utils.bases.commands import TestProjectBase
from tests.utils.cmdline import proc
if TYPE_CHECKING:
from pathlib import Path
from tests.mockserver.http import MockServer
class TestFetchCommand:
@pytest.mark.parametrize("args", [(), ("not-a-url",), ("a:b", "c:d")])
def test_bad_arguments(self, args: tuple[str, ...]) -> None:
returncode, out, _ = proc("fetch", *args)
assert returncode == 2
assert "Usage" in out
def test_output(self, mockserver: MockServer) -> None:
_, out, _ = proc("fetch", mockserver.url("/text"))
assert out.strip() == "Works"
@ -36,3 +47,24 @@ class TestFetchCommand:
"fetch", "-s", "TWISTED_REACTOR_ENABLED=False", mockserver.url("/text")
)
assert out.strip() == "Works"
class TestFetchCommandWithSpider(TestProjectBase):
@pytest.fixture(autouse=True)
def create_files(self, proj_path: Path) -> None:
(proj_path / self.project_name / "spiders" / "myspider.py").write_text(
"""
import scrapy
class MySpider(scrapy.Spider):
name = "myspider"
custom_settings = {"USER_AGENT": "myspider-user-agent"}
""",
encoding="utf-8",
)
def test_spider(self, proj_path: Path, mockserver: MockServer) -> None:
_, out, err = proc(
"fetch", "--spider", "myspider", mockserver.url("/echo"), cwd=proj_path
)
assert "myspider-user-agent" in out, err

View File

@ -64,6 +64,24 @@ class TestGenspiderCommand(TestProjectBase):
assert call("genspider", "--dump=basic", cwd=proj_path) == 0
assert call("genspider", "-d", "basic", cwd=proj_path) == 0
@pytest.mark.parametrize(
"args",
[("--dump=nonexistent",), ("-t", "nonexistent", "test_name", "test.com")],
)
def test_unknown_template(self, args: tuple[str, ...], proj_path: Path) -> None:
returncode, out, err = proc("genspider", *args, cwd=proj_path)
assert returncode == 0, err
assert "Unable to find template: nonexistent" in out
assert not (proj_path / self.project_name / "spiders" / "test_name.py").exists()
def test_name_not_starting_with_a_letter(self, proj_path: Path) -> None:
"""The module name, unlike the spider name, is prefixed with a letter."""
_, out, err = proc("genspider", "1st_spider", "test.com", cwd=proj_path)
assert "Created spider '1st_spider'" in out, err
spider = proj_path / self.project_name / "spiders" / "a1st_spider.py"
assert spider.exists()
assert find_in_file(spider, r'name\s*=\s*"1st_spider"') is not None
@pytest.mark.skipif(
sys.platform == "win32", reason="requires a POSIX shell editor script"
)
@ -87,7 +105,8 @@ class TestGenspiderCommand(TestProjectBase):
)
def test_same_name_as_project(self, proj_path: Path) -> None:
assert call("genspider", self.project_name, cwd=proj_path) == 2
_, out, err = proc("genspider", self.project_name, "test.com", cwd=proj_path)
assert "Cannot create a spider with the same name as your project" in out, err
assert not (
proj_path / self.project_name / "spiders" / f"{self.project_name}.py"
).exists()

View File

@ -3,6 +3,7 @@ from __future__ import annotations
import argparse
import re
from typing import TYPE_CHECKING
from urllib.parse import urlparse
import pytest
@ -552,6 +553,130 @@ ITEM_PIPELINES = {{'{self.project_name}.pipelines.MyPipeline': 1}}
content = '[\n{},\n{"foo": "bar"}\n]'
assert file_path.read_text(encoding="utf-8") == content
@pytest.mark.parametrize("args", [(), ("not-a-url",), ("a:b", "c:d")])
def test_bad_arguments(self, args: tuple[str, ...], proj_path: Path) -> None:
returncode, out, _ = proc("parse", *args, cwd=proj_path)
assert returncode == 2
assert "Usage" in out
@pytest.mark.parametrize(
("option", "message"),
[
("--meta", "Invalid -m/--meta value"),
("-m", "Invalid -m/--meta value"),
("--cbkwargs", "Invalid --cbkwargs value"),
],
)
def test_invalid_json(
self, option: str, message: str, proj_path: Path, mockserver: MockServer
) -> None:
returncode, _, err = proc(
"parse",
"--spider",
self.spider_name,
option,
"{invalid",
mockserver.url("/html"),
cwd=proj_path,
)
assert returncode == 2
assert message in err
def test_unknown_spider(self, proj_path: Path, mockserver: MockServer) -> None:
returncode, _, err = proc(
"parse",
"--spider",
"nonexistent",
mockserver.url("/html"),
cwd=proj_path,
)
assert returncode == 0, err
assert "Unable to find spider: nonexistent" in err
def test_spider_found_by_url(self, proj_path: Path, mockserver: MockServer) -> None:
"""Without --spider, the spider is chosen based on the URL."""
url = mockserver.url("/html")
# The spider name doubles as a domain of the spider, and it is matched
# against the netloc of the URL, hence the port.
(proj_path / self.project_name / "spiders" / "urlspider.py").write_text(
f"""
import scrapy
class UrlSpider(scrapy.Spider):
name = "{urlparse(url).netloc}"
def parse(self, response):
return [{{"found_by_url": True}}]
""",
encoding="utf-8",
)
returncode, out, err = proc("parse", url, cwd=proj_path)
assert returncode == 0, err
assert "Unable to find spider for" not in err
assert "{'found_by_url': True}" in out
def test_legacy_item_processor(
self, proj_path: Path, mockserver: MockServer
) -> None:
"""--pipelines supports an ITEM_PROCESSOR without process_item_async()."""
(proj_path / self.project_name / "legacy.py").write_text(
"""
import logging
from twisted.internet.defer import succeed
class LegacyItemProcessor:
@classmethod
def from_crawler(cls, crawler):
return cls()
def open_spider(self, spider):
return succeed(None)
def close_spider(self, spider):
return succeed(None)
def process_item(self, item, spider):
logging.info("Legacy item processor!")
return succeed(item)
""",
encoding="utf-8",
)
_, _, stderr = proc(
"parse",
"--spider",
self.spider_name,
"--pipelines",
"-c",
"parse",
"-s",
f"ITEM_PROCESSOR={self.project_name}.legacy.LegacyItemProcessor",
mockserver.url("/html"),
cwd=proj_path,
)
assert "INFO: Legacy item processor!" in stderr
@pytest.mark.parametrize("verbose", [True, False])
def test_no_items_no_links(
self, verbose: bool, proj_path: Path, mockserver: MockServer
) -> None:
args = ["--verbose"] if verbose else []
_, out, err = proc(
"parse",
"--spider",
self.spider_name,
"-c",
"parse",
"--noitems",
"--nolinks",
*args,
mockserver.url("/html"),
cwd=proj_path,
)
assert "# Scraped Items" not in out, err
assert "# Requests" not in out
def test_parse_add_options(self):
command = parse.Command()
command.settings = Settings()

View File

@ -136,6 +136,12 @@ class MySpider(scrapy.Spider):
log = self.get_log(tmp_path, "from scrapy.spiders import Spider\n")
assert "No spider found in file" in log
@pytest.mark.parametrize("args", [(), ("a.py", "b.py")])
def test_runspider_bad_arguments(self, args: tuple[str, ...]) -> None:
returncode, out, _ = proc("runspider", *args)
assert returncode == 2
assert "Usage" in out
def test_runspider_file_not_found(self) -> None:
_, _, log = proc("runspider", "some_non_existent_file")
assert "File not found: some_non_existent_file" in log

View File

@ -18,6 +18,7 @@ from scrapy.shell import Shell, inspect_response
from scrapy.utils.reactor import _asyncio_reactor_path
from scrapy.utils.test import get_crawler
from tests import NON_EXISTING_RESOLVABLE, tests_datadir
from tests.utils.bases.commands import TestProjectBase
from tests.utils.cmdline import proc
from tests.utils.decorators import coroutine_test
@ -172,6 +173,33 @@ def _stop(p: PopenSpawn[str]) -> None:
pipe.close()
class TestShellCommandWithSpider(TestProjectBase):
@pytest.fixture(autouse=True)
def create_files(self, proj_path: Path) -> None:
(proj_path / self.project_name / "spiders" / "myspider.py").write_text(
"""
import scrapy
class MySpider(scrapy.Spider):
name = "myspider"
""",
encoding="utf-8",
)
def test_spider(self, proj_path: Path, mockserver: MockServer) -> None:
ret, out, err = proc(
"shell",
"--spider",
"myspider",
mockserver.url("/text"),
"-c",
"spider.name",
cwd=proj_path,
)
assert ret == 0, err
assert out.strip() == "myspider"
class TestInteractiveShell:
def test_fetch(self, mockserver: MockServer) -> None:
args = (

View File

@ -3,21 +3,27 @@ from __future__ import annotations
import argparse
import json
import sys
from pathlib import Path
from typing import TYPE_CHECKING
import pytest
import scrapy
from scrapy.cmdline import _pop_command_name, execute
from scrapy.commands import ScrapyCommand, ScrapyHelpFormatter, view
from scrapy.commands import ScrapyCommand, ScrapyHelpFormatter
from scrapy.exceptions import ScrapyDeprecationWarning
from scrapy.settings import Settings
from scrapy.utils.reactor import _asyncio_reactor_path
from tests.utils.bases.commands import TestProjectBase
from tests.utils.cmdline import call, proc, write_recording_editor
from tests.utils.cmdline import (
call,
proc,
write_recording_browser,
write_recording_editor,
)
if TYPE_CHECKING:
from pathlib import Path
from tests.mockserver.http import MockServer
class EmptyCommand(ScrapyCommand):
@ -107,6 +113,93 @@ class TestCommandSettings:
)
class TestGlobalOptions:
"""Tests for the options that every command supports."""
spider_code = """
import scrapy
class MySpider(scrapy.Spider):
name = "myspider"
async def start(self):
self.logger.debug("It works!")
return
yield
"""
@pytest.fixture
def spider_path(self, tmp_path: Path) -> Path:
path = tmp_path / "myspider.py"
path.write_text(self.spider_code, encoding="utf-8")
return path
def test_invalid_set(self, spider_path: Path) -> None:
returncode, _, err = proc("runspider", str(spider_path), "-s", "FOO")
assert returncode == 2
assert "Invalid -s value, use -s NAME=VALUE" in err
def test_invalid_spider_argument(self, spider_path: Path) -> None:
returncode, _, err = proc("runspider", str(spider_path), "-a", "FOO")
assert returncode == 2
assert "Invalid -a value, use -a NAME=VALUE" in err
def test_logfile(self, tmp_path: Path, spider_path: Path) -> None:
logfile = tmp_path / "scrapy.log"
returncode, _, err = proc(
"runspider", str(spider_path), "--logfile", str(logfile)
)
assert returncode == 0, err
assert "It works!" in logfile.read_text(encoding="utf-8")
assert "It works!" not in err
def test_loglevel(self, spider_path: Path) -> None:
returncode, _, err = proc("runspider", str(spider_path), "--loglevel", "INFO")
assert returncode == 0, err
assert "It works!" not in err
assert "Spider closed (finished)" in err
def test_nolog(self, spider_path: Path) -> None:
returncode, _, err = proc("runspider", str(spider_path), "--nolog")
assert returncode == 0, err
assert not err
def test_pidfile(self, tmp_path: Path, spider_path: Path) -> None:
pidfile = tmp_path / "scrapy.pid"
returncode, _, err = proc(
"runspider", str(spider_path), "--pidfile", str(pidfile)
)
assert returncode == 0, err
assert pidfile.read_text(encoding="utf-8").strip().isdigit()
def test_pdb(self, spider_path: Path) -> None:
returncode, _, err = proc("runspider", str(spider_path), "--pdb")
assert returncode == 0, err
assert "It works!" in err
class TestSettingsCommand:
@pytest.mark.parametrize(
("option", "setting", "expected"),
[
("--get", "BOT_NAME", "scrapybot"),
("--getbool", "COOKIES_ENABLED", "True"),
("--getint", "CONCURRENT_REQUESTS", "16"),
("--getfloat", "DOWNLOAD_DELAY", "0.0"),
("--getlist", "SPIDER_MODULES", "[]"),
],
)
def test_get(self, option: str, setting: str, expected: str) -> None:
returncode, out, err = proc("settings", option, setting)
assert returncode == 0, err
assert out.startswith(expected)
def test_no_option(self) -> None:
returncode, out, err = proc("settings")
assert returncode == 0, err
assert not out
class TestCommandCrawlerProcess(TestProjectBase):
"""Test that the command uses the expected kind of *CrawlerProcess
and produces expected errors when needed."""
@ -577,18 +670,31 @@ class TestBenchCommand:
class TestViewCommand:
def test_methods(self) -> None:
command = view.Command()
command.settings = Settings()
parser = argparse.ArgumentParser(
prog="scrapy",
prefix_chars="-",
formatter_class=ScrapyHelpFormatter,
conflict_handler="resolve",
@pytest.mark.skipif(
sys.platform == "win32", reason="requires a POSIX shell browser script"
)
def test_view(
self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, mockserver: MockServer
) -> None:
opened = tmp_path / "opened.txt"
browser = tmp_path / "fake-browser.sh"
write_recording_browser(browser, opened)
monkeypatch.setenv("BROWSER", str(browser))
returncode, _, err = proc("view", mockserver.url("/html"), cwd=tmp_path)
assert returncode == 0, err
url = opened.read_text(encoding="utf-8")
assert url.startswith("file://")
body = Path(url.removeprefix("file://")).read_text(encoding="utf-8")
assert "<p class='one'>Works</p>" in body
def test_non_text_response(self, mockserver: MockServer) -> None:
returncode, _, err = proc(
"view", mockserver.url("/static/files/images/scrapy.png")
)
command.add_options(parser)
assert command.short_desc() == "Open URL in browser, as seen by Scrapy"
assert "URL using the Scrapy downloader and show its" in command.long_desc()
assert returncode == 0, err
assert "Cannot view a non-text response." in err
class TestEditCommand(TestProjectBase):
@ -615,6 +721,11 @@ class TestEditCommand(TestProjectBase):
assert returncode == 1
assert "Spider not found: nonexistent" in err
def test_edit_no_spider(self, proj_path: Path) -> None:
returncode, out, _ = proc("edit", cwd=proj_path)
assert returncode == 2
assert "Usage" in out
class TestHelpMessage(TestProjectBase):
@pytest.mark.parametrize(

View File

@ -8,6 +8,7 @@ import pytest
from twisted.internet.defer import Deferred, DeferredList
from twisted.python import failure
from scrapy import signals
from scrapy.downloadermiddlewares.robotstxt import RobotsTxtMiddleware
from scrapy.exceptions import CannotResolveHostError, IgnoreRequest, NotConfigured
from scrapy.http import Request, Response, TextResponse
@ -27,6 +28,7 @@ class TestRobotsTxtMiddleware:
self.crawler: mock.MagicMock = mock.MagicMock()
self.crawler.settings = Settings()
self.crawler.engine.download_async = mock.AsyncMock()
self.crawler.signals.send_catch_log_async = mock.AsyncMock(return_value=[])
def teardown_method(self):
del self.crawler
@ -74,6 +76,21 @@ Disallow: /some/randome/page.html
Request("http://site.local/wiki/Käyttäjä:"), middleware
)
@coroutine_test
async def test_robotstxt_emits_robots_parsed_signal(self):
crawler = self._get_successful_crawler()
middleware = RobotsTxtMiddleware(crawler)
request = Request("http://site.local/allowed")
await self.assertNotIgnored(request, middleware)
calls = [
kwargs
for _, kwargs in crawler.signals.send_catch_log_async.call_args_list
if kwargs.get("signal") is signals.robots_parsed
]
assert len(calls) == 1
assert calls[0]["request"] is request
assert calls[0]["robotparser"] is not None
@coroutine_test
async def test_robotstxt_multiple_reqs(self) -> None:
middleware = RobotsTxtMiddleware(self._get_successful_crawler())

View File

@ -4,6 +4,7 @@ from scrapy.robotstxt import (
ProtegoRobotParser,
PythonRobotParser,
RerpRobotParser,
RobotParser,
decode_robotstxt,
)
from scrapy.utils._deps_compat import STDLIB_IMPROVED_ROBOTFILEPARSER
@ -78,6 +79,16 @@ class BaseRobotParserTest:
assert rp.allowed("https://site.local/index.html", "*")
assert rp.allowed("https://site.local/disallowed", "*")
def test_crawl_delay(self):
robotstxt_body = b"User-agent: *\nDisallow: /private\nCrawl-delay: 10\n"
rp = self.parser_cls.from_crawler(crawler=None, robotstxt_body=robotstxt_body)
assert rp.crawl_delay("*") == 10.0
def test_crawl_delay_unset(self):
robotstxt_body = b"User-agent: *\nDisallow: /private\n"
rp = self.parser_cls.from_crawler(crawler=None, robotstxt_body=robotstxt_body)
assert rp.crawl_delay("*") is None
def test_unicode_url_and_useragent(self):
robotstxt_robotstxt_body = """
User-Agent: *
@ -102,6 +113,22 @@ class BaseRobotParserTest:
assert not rp.allowed("https://site.local/some/randome/page.html", "UnicödeBöt")
class TestRobotParser:
def test_crawl_delay_unsupported(self):
class AllowAllRobotParser(RobotParser):
@classmethod
def from_crawler(cls, crawler, robotstxt_body):
return cls()
def allowed(self, url, user_agent):
return True
rp = AllowAllRobotParser.from_crawler(
crawler=None, robotstxt_body=b"User-agent: *\nCrawl-delay: 10\n"
)
assert rp.crawl_delay("*") is None
class TestDecodeRobotsTxt:
def test_native_string_conversion(self):
robotstxt_body = b"User-agent: *\nDisallow: /\n"

View File

@ -4,7 +4,7 @@ from importlib.util import find_spec
import pytest
from scrapy.utils.console import get_shell_embed_func
from scrapy.utils.console import get_shell_embed_func, start_python_console
def test_get_shell_embed_func():
@ -59,3 +59,20 @@ def test_get_shell_embed_func_default():
else:
expected = "_embed_standard_shell"
assert shell.__name__ == expected
def test_start_python_console_exit(monkeypatch: pytest.MonkeyPatch) -> None:
def embed(namespace: dict[str, object], banner: str) -> None:
raise SystemExit
monkeypatch.setattr(
"scrapy.utils.console.get_shell_embed_func", lambda shells: embed
)
start_python_console()
def test_start_python_console_no_shell(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setattr(
"scrapy.utils.console.get_shell_embed_func", lambda shells: None
)
start_python_console()

View File

@ -46,3 +46,16 @@ def write_recording_editor(editor: Path) -> None:
open (its last argument) into the file given as its first argument."""
editor.write_text('#!/bin/sh\nprintf "%s" "$2" > "$1"\n', encoding="utf-8")
editor.chmod(0o755)
def write_recording_browser(browser: Path, recorded: Path) -> None:
"""Create an executable browser script that writes the URL it is asked to
open into *recorded*.
``webbrowser`` only passes the URL to the command from the ``BROWSER``
environment variable, hence the hardcoded output path.
"""
browser.write_text(
f'#!/bin/sh\nprintf "%s" "$1" > "{recorded}"\n', encoding="utf-8"
)
browser.chmod(0o755)

20
tox.ini
View File

@ -5,7 +5,7 @@
[tox]
requires =
sphinx-scrapy[tox] @ git+https://github.com/scrapy/sphinx-scrapy.git@0.8.8
sphinx-scrapy[tox] @ git+https://github.com/scrapy/sphinx-scrapy.git@0.8.9
envlist =
pre-commit
pylint
@ -30,6 +30,7 @@ envlist =
botocore
pypy3
pypy3-extra-deps
benchmark
minversion = 1.7.0
[test-requirements]
@ -320,3 +321,20 @@ setenv =
{[min]setenv}
commands =
pytest {posargs:--cov-config=pyproject.toml --cov=scrapy --cov-report=xml --cov-report= tests --junitxml=min-botocore.junit.xml -o junit_family=legacy} -m requires_botocore
# CPU benchmarks, tracked on CodSpeed.
#
# pytest-twisted is left out on purpose: benchmarked code must be callable
# synchronously, so tests/benchmarks drives the reactor itself.
[testenv:benchmark]
basepython = python3.14
deps =
pytest >= 8.4.1
pytest-codspeed
passenv =
*codspeed*
*ci*
commands =
pytest {posargs:tests/benchmarks} --codspeed --codspeed-mode=simulation