mirror of https://github.com/scrapy/scrapy.git
Merge remote-tracking branch 'origin/master' into no-warnings
This commit is contained in:
commit
b8cb7d2d6e
|
|
@ -0,0 +1,45 @@
|
|||
---
|
||||
name: codspeed
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
pull_request:
|
||||
paths:
|
||||
- scrapy/**
|
||||
- tests/benchmarks/**
|
||||
- .github/workflows/codspeed.yml
|
||||
- tox.ini
|
||||
workflow_dispatch:
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
permissions: {}
|
||||
|
||||
jobs:
|
||||
benchmark:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write # OIDC authentication with CodSpeed
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
|
||||
- name: Set up Python 3.14
|
||||
uses: actions/setup-python@v6
|
||||
with:
|
||||
python-version: '3.14'
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
pip install --upgrade pip
|
||||
pip install --upgrade tox
|
||||
tox -n -e benchmark
|
||||
- name: Run benchmarks
|
||||
uses: CodSpeedHQ/action@v4
|
||||
with:
|
||||
mode: simulation
|
||||
run: tox -e benchmark
|
||||
|
|
@ -14,14 +14,18 @@ jobs:
|
|||
tests:
|
||||
runs-on: macos-latest
|
||||
env:
|
||||
PYTEST_ADDOPTS: -n auto
|
||||
PYTEST_ADDOPTS: ${{ matrix.coverage && '-n auto' || '-n auto --no-cov' }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
python-version: ["3.10", "3.11", "3.12", "3.13", "3.14"]
|
||||
python-version: ["3.10", "3.11", "3.12", "3.13"]
|
||||
env:
|
||||
- TOXENV: py
|
||||
include:
|
||||
- python-version: '3.14'
|
||||
env:
|
||||
TOXENV: py
|
||||
coverage: true
|
||||
- python-version: '3.14'
|
||||
env:
|
||||
TOXENV: no-reactor
|
||||
|
|
@ -41,6 +45,7 @@ jobs:
|
|||
tox
|
||||
|
||||
- name: Upload coverage report
|
||||
if: ${{ matrix.coverage }}
|
||||
uses: codecov/codecov-action@v5
|
||||
|
||||
- name: Upload test results
|
||||
|
|
|
|||
|
|
@ -14,7 +14,7 @@ jobs:
|
|||
tests:
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
PYTEST_ADDOPTS: -n auto
|
||||
PYTEST_ADDOPTS: ${{ matrix.coverage && '-n auto' || '-n auto --no-cov' }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
|
|
@ -34,12 +34,15 @@ jobs:
|
|||
- python-version: "3.14"
|
||||
env:
|
||||
TOXENV: py
|
||||
coverage: true
|
||||
- python-version: "3.14"
|
||||
env:
|
||||
TOXENV: default-reactor
|
||||
coverage: true
|
||||
- python-version: "3.14"
|
||||
env:
|
||||
TOXENV: no-reactor
|
||||
coverage: true
|
||||
# pinned due to https://github.com/pypy/pypy/issues/5388
|
||||
- python-version: pypy3.11-7.3.20
|
||||
env:
|
||||
|
|
@ -49,12 +52,15 @@ jobs:
|
|||
- python-version: "3.10.19"
|
||||
env:
|
||||
TOXENV: min
|
||||
coverage: true
|
||||
- python-version: "3.10.19"
|
||||
env:
|
||||
TOXENV: min-default-reactor
|
||||
coverage: true
|
||||
- python-version: "3.10.19"
|
||||
env:
|
||||
TOXENV: min-no-reactor
|
||||
coverage: true
|
||||
# pinned due to https://github.com/pypy/pypy/issues/5388
|
||||
- python-version: pypy3.11-7.3.20
|
||||
env:
|
||||
|
|
@ -62,16 +68,20 @@ jobs:
|
|||
- python-version: "3.10.19"
|
||||
env:
|
||||
TOXENV: min-extra-deps
|
||||
coverage: true
|
||||
- python-version: "3.10.19"
|
||||
env:
|
||||
TOXENV: min-botocore
|
||||
coverage: true
|
||||
|
||||
- python-version: "3.14"
|
||||
env:
|
||||
TOXENV: extra-deps
|
||||
coverage: true
|
||||
- python-version: "3.14"
|
||||
env:
|
||||
TOXENV: no-reactor-extra-deps
|
||||
coverage: true
|
||||
# pinned due to https://github.com/pypy/pypy/issues/5388
|
||||
- python-version: pypy3.11-7.3.20
|
||||
env:
|
||||
|
|
@ -79,6 +89,7 @@ jobs:
|
|||
- python-version: "3.14"
|
||||
env:
|
||||
TOXENV: botocore
|
||||
coverage: true
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
|
|
@ -104,6 +115,7 @@ jobs:
|
|||
tox
|
||||
|
||||
- name: Upload coverage report
|
||||
if: ${{ matrix.coverage }}
|
||||
uses: codecov/codecov-action@v5
|
||||
|
||||
- name: Upload test results
|
||||
|
|
|
|||
|
|
@ -14,7 +14,7 @@ jobs:
|
|||
tests:
|
||||
runs-on: windows-latest
|
||||
env:
|
||||
PYTEST_ADDOPTS: -n auto
|
||||
PYTEST_ADDOPTS: ${{ matrix.coverage && '-n auto' || '-n auto --no-cov' }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
|
|
@ -34,6 +34,7 @@ jobs:
|
|||
- python-version: "3.14"
|
||||
env:
|
||||
TOXENV: py
|
||||
coverage: true
|
||||
- python-version: "3.14"
|
||||
env:
|
||||
TOXENV: default-reactor
|
||||
|
|
@ -68,6 +69,7 @@ jobs:
|
|||
tox
|
||||
|
||||
- name: Upload coverage report
|
||||
if: ${{ matrix.coverage }}
|
||||
uses: codecov/codecov-action@v5
|
||||
|
||||
- name: Upload test results
|
||||
|
|
|
|||
|
|
@ -27,6 +27,6 @@ repos:
|
|||
hooks:
|
||||
- id: sphinx-lint
|
||||
- repo: https://github.com/scrapy/sphinx-scrapy
|
||||
rev: 0.8.8
|
||||
rev: 0.8.9
|
||||
hooks:
|
||||
- id: sphinx-scrapy
|
||||
|
|
|
|||
16
README.rst
16
README.rst
|
|
@ -5,7 +5,7 @@
|
|||
:alt: Scrapy
|
||||
:width: 480px
|
||||
|
||||
|version| |python_version| |ubuntu| |macos| |windows| |coverage| |conda| |deepwiki|
|
||||
|version| |python_version| |tests| |coverage| |conda| |deepwiki|
|
||||
|
||||
.. |version| image:: https://img.shields.io/pypi/v/Scrapy.svg
|
||||
:target: https://pypi.org/pypi/Scrapy
|
||||
|
|
@ -15,17 +15,9 @@
|
|||
:target: https://pypi.org/pypi/Scrapy
|
||||
:alt: Supported Python Versions
|
||||
|
||||
.. |ubuntu| image:: https://github.com/scrapy/scrapy/workflows/Ubuntu/badge.svg
|
||||
:target: https://github.com/scrapy/scrapy/actions?query=workflow%3AUbuntu
|
||||
:alt: Ubuntu
|
||||
|
||||
.. |macos| image:: https://github.com/scrapy/scrapy/workflows/macOS/badge.svg
|
||||
:target: https://github.com/scrapy/scrapy/actions?query=workflow%3AmacOS
|
||||
:alt: macOS
|
||||
|
||||
.. |windows| image:: https://github.com/scrapy/scrapy/workflows/Windows/badge.svg
|
||||
:target: https://github.com/scrapy/scrapy/actions?query=workflow%3AWindows
|
||||
:alt: Windows
|
||||
.. |tests| image:: https://img.shields.io/github/check-runs/scrapy/scrapy/master?label=tests
|
||||
:target: https://github.com/scrapy/scrapy/actions?query=branch%3Amaster
|
||||
:alt: Tests
|
||||
|
||||
.. |coverage| image:: https://img.shields.io/codecov/c/github/scrapy/scrapy/master.svg
|
||||
:target: https://codecov.io/github/scrapy/scrapy?branch=master
|
||||
|
|
|
|||
|
|
@ -54,6 +54,9 @@ if not H2_ENABLED:
|
|||
if find_spec("httpx2") is None and find_spec("httpx") is None:
|
||||
collect_ignore.append("scrapy/core/downloader/handlers/_httpx.py")
|
||||
|
||||
if find_spec("pytest_codspeed") is None:
|
||||
collect_ignore.append("tests/benchmarks")
|
||||
|
||||
|
||||
def pytest_addoption(parser, pluginmanager):
|
||||
if pluginmanager.hasplugin("twisted"):
|
||||
|
|
|
|||
|
|
@ -5,4 +5,4 @@ sphinx
|
|||
sphinx-notfound-page
|
||||
sphinx-rtd-theme
|
||||
sphinx-rtd-dark-mode
|
||||
sphinx-scrapy @ git+https://github.com/scrapy/sphinx-scrapy.git@0.8.8
|
||||
sphinx-scrapy @ git+https://github.com/scrapy/sphinx-scrapy.git@0.8.9
|
||||
|
|
|
|||
|
|
@ -153,7 +153,7 @@ sphinx-rtd-theme==3.1.0
|
|||
# via
|
||||
# -r docs/requirements.in
|
||||
# sphinx-rtd-dark-mode
|
||||
sphinx-scrapy @ git+https://github.com/scrapy/sphinx-scrapy.git@c0b2ac815afc3cb8857d575cecb5d55c05e6b737
|
||||
sphinx-scrapy @ git+https://github.com/scrapy/sphinx-scrapy.git@912ed0507405e16ac60a47dd08195a1cd0ced984
|
||||
# via -r docs/requirements.in
|
||||
sphinx-sitemap==2.9.0
|
||||
# via sphinx-scrapy
|
||||
|
|
|
|||
|
|
@ -114,8 +114,8 @@ some usage help and the available commands::
|
|||
scrapy <command> [options] [args]
|
||||
|
||||
Available commands:
|
||||
crawl Run a spider
|
||||
fetch Fetch a URL using the Scrapy downloader
|
||||
runspider Run a spider from a Python file, no project required
|
||||
[...]
|
||||
|
||||
The first line will print the currently active project if you're inside a
|
||||
|
|
@ -263,7 +263,9 @@ crawl
|
|||
* Syntax: ``scrapy crawl <spider>``
|
||||
* Requires project: *yes*
|
||||
|
||||
Start crawling using a spider.
|
||||
Start crawling using the spider with the given :attr:`~scrapy.Spider.name`,
|
||||
which must be one of those that :command:`list` reports. To run a spider from a
|
||||
file instead, use :command:`runspider`.
|
||||
|
||||
Supported options:
|
||||
|
||||
|
|
@ -571,8 +573,9 @@ runspider
|
|||
* Syntax: ``scrapy runspider <spider_file.py>``
|
||||
* Requires project: *no*
|
||||
|
||||
Run a spider self-contained in a Python file, without having to create a
|
||||
project.
|
||||
Run the spider defined in the given Python file, without requiring a project.
|
||||
|
||||
Supported options: the same as :command:`crawl`.
|
||||
|
||||
Example usage::
|
||||
|
||||
|
|
@ -665,6 +668,8 @@ Example:
|
|||
|
||||
COMMANDS_MODULE = "mybot.commands"
|
||||
|
||||
.. note:: This is a :ref:`pre-crawler setting <pre-crawler-settings>`.
|
||||
|
||||
.. _Deploying your project: https://scrapyd.readthedocs.io/en/latest/deploy.html
|
||||
|
||||
Register commands via setup.py entry points
|
||||
|
|
|
|||
|
|
@ -136,18 +136,10 @@ Core Stats extension
|
|||
Enable the collection of core statistics, provided the stats collection is
|
||||
enabled (see :ref:`topics-stats`).
|
||||
|
||||
The following stats are collected:
|
||||
|
||||
* ``start_time``: start date/time of the crawl (:class:`~datetime.datetime`).
|
||||
* ``finish_time``: end date/time of the crawl (:class:`~datetime.datetime`).
|
||||
* ``elapsed_time_seconds``: total crawl duration in seconds (:class:`float`).
|
||||
* ``finish_reason``: the closing reason string (e.g. ``"finished"``,
|
||||
``"closespider_timeout"``).
|
||||
* ``item_scraped_count``: total number of items that passed all pipelines.
|
||||
* ``item_dropped_count``: total number of items dropped by a pipeline.
|
||||
* ``item_dropped_reasons_count/<ExceptionName>``: per-exception drop count
|
||||
(e.g. ``item_dropped_reasons_count/DropItem``).
|
||||
* ``response_received_count``: total number of HTTP responses received.
|
||||
The following stats are collected: :stat:`elapsed_time_seconds`,
|
||||
:stat:`finish_reason`, :stat:`finish_time`, :stat:`item_dropped_count`,
|
||||
:stat:`item_dropped_reasons_count/{exception}`, :stat:`item_scraped_count`,
|
||||
:stat:`response_received_count`, :stat:`start_time`.
|
||||
|
||||
Log Count extension
|
||||
~~~~~~~~~~~~~~~~~~~
|
||||
|
|
@ -190,7 +182,7 @@ Monitors the memory used by the Scrapy process that runs the spider and:
|
|||
|
||||
1. sends a :signal:`memusage_warning_reached` signal when it exceeds
|
||||
:setting:`MEMUSAGE_WARNING_MB`
|
||||
2. closes the spider with the `"memusage_exceeded"` reason when it exceeds
|
||||
2. closes the spider with the ``"memusage_exceeded"`` reason when it exceeds
|
||||
:setting:`MEMUSAGE_LIMIT_MB`
|
||||
|
||||
This extension is enabled by the :setting:`MEMUSAGE_ENABLED` setting and
|
||||
|
|
@ -214,7 +206,8 @@ An extension for debugging memory usage. It collects information about:
|
|||
* objects left alive that shouldn't. For more info, see :ref:`topics-leaks-trackrefs`
|
||||
|
||||
To enable this extension, turn on the :setting:`MEMDEBUG_ENABLED` setting. The
|
||||
info will be stored in the stats.
|
||||
info will be stored in the :stat:`memdebug/gc_garbage_count` and
|
||||
:stat:`memdebug/live_refs/{cls}` stats.
|
||||
|
||||
.. _topics-extensions-ref-spiderstate:
|
||||
|
||||
|
|
|
|||
|
|
@ -305,10 +305,21 @@ These settings cannot be :ref:`set from a spider <spider-settings>`.
|
|||
|
||||
These settings are:
|
||||
|
||||
- :setting:`TWISTED_REACTOR_ENABLED`
|
||||
- :setting:`ADDONS`
|
||||
- :setting:`COMMANDS_MODULE`
|
||||
- :setting:`FORCE_CRAWLER_PROCESS`
|
||||
- :setting:`SPIDER_LOADER_CLASS` and settings used by the corresponding
|
||||
spider loader class, e.g. :setting:`SPIDER_MODULES` and
|
||||
:setting:`SPIDER_LOADER_WARN_ONLY` for the default spider loader class.
|
||||
- :setting:`TWISTED_REACTOR_ENABLED`
|
||||
|
||||
:setting:`ADDONS` is a special case: it can be set from a spider, but the
|
||||
``update_pre_crawler_settings()`` method of :ref:`add-ons <topics-addons>`
|
||||
enabled that way is not called.
|
||||
|
||||
:setting:`TWISTED_REACTOR` also acts as a pre-crawler setting when running a
|
||||
:ref:`command that needs a CrawlerProcess <topics-commands-crawlerprocess>`,
|
||||
since its project-level value determines the crawler process class.
|
||||
|
||||
.. _reactor-settings:
|
||||
|
||||
|
|
@ -409,6 +420,9 @@ Default: ``{}``
|
|||
A dict containing paths to the add-ons enabled in your project and their
|
||||
priorities. For more information, see :ref:`topics-addons`.
|
||||
|
||||
.. note:: This is a :ref:`pre-crawler setting <pre-crawler-settings>`, with a
|
||||
caveat described in that section.
|
||||
|
||||
.. setting:: ASYNCIO_EVENT_LOOP
|
||||
|
||||
ASYNCIO_EVENT_LOOP
|
||||
|
|
@ -1402,6 +1416,8 @@ When :setting:`TWISTED_REACTOR_ENABLED` is set to ``False``,
|
|||
Set this to ``True`` if you want to set :setting:`TWISTED_REACTOR` to a
|
||||
non-default value in :ref:`per-spider settings <spider-settings>`.
|
||||
|
||||
.. note:: This is a :ref:`pre-crawler setting <pre-crawler-settings>`.
|
||||
|
||||
.. setting:: FTP_PASSIVE_MODE
|
||||
|
||||
FTP_PASSIVE_MODE
|
||||
|
|
@ -1855,7 +1871,8 @@ Default: ``False``
|
|||
|
||||
Setting to ``True`` will log debug information about the requests scheduler.
|
||||
This currently logs (only once) if the requests cannot be serialized to disk.
|
||||
Stats counter (``scheduler/unserializable``) tracks the number of times this happens.
|
||||
The :stat:`scheduler/unserializable` stat tracks the number of times this
|
||||
happens.
|
||||
|
||||
Example entry in logs::
|
||||
|
||||
|
|
|
|||
|
|
@ -504,6 +504,27 @@ headers_received
|
|||
:param spider: the spider associated with the response
|
||||
:type spider: :class:`~scrapy.Spider` object
|
||||
|
||||
robots_parsed
|
||||
~~~~~~~~~~~~~
|
||||
|
||||
.. signal:: robots_parsed
|
||||
.. function:: robots_parsed(robotparser, request)
|
||||
|
||||
.. versionadded:: VERSION
|
||||
|
||||
Sent by
|
||||
:class:`~scrapy.downloadermiddlewares.robotstxt.RobotsTxtMiddleware` after it
|
||||
downloads and parses a :file:`robots.txt` file, for the host that *request*
|
||||
targets.
|
||||
|
||||
This signal supports :ref:`asynchronous handlers <signal-deferred>`.
|
||||
|
||||
:param robotparser: the parser holding the parsed :file:`robots.txt` contents
|
||||
:type robotparser: :class:`~scrapy.robotstxt.RobotParser` object
|
||||
|
||||
:param request: the request that triggered the :file:`robots.txt` download
|
||||
:type request: :class:`~scrapy.Request` object
|
||||
|
||||
|
||||
Response signals
|
||||
----------------
|
||||
|
|
|
|||
|
|
@ -21,6 +21,8 @@ using the Stats Collector from.
|
|||
Another feature of the Stats Collector is that it's very efficient (when
|
||||
enabled) and extremely efficient (almost unnoticeable) when disabled.
|
||||
|
||||
See :ref:`topics-stats-reference` below for the stats that Scrapy sets.
|
||||
|
||||
.. _topics-stats-usecases:
|
||||
|
||||
Common Stats Collector uses
|
||||
|
|
@ -101,3 +103,642 @@ DummyStatsCollector
|
|||
-------------------
|
||||
|
||||
.. autoclass:: DummyStatsCollector
|
||||
|
||||
.. _topics-stats-reference:
|
||||
|
||||
Built-in stats reference
|
||||
========================
|
||||
|
||||
Scrapy sets the following :ref:`stats <topics-stats>`. Components other than
|
||||
those built into Scrapy may set additional stats; see their documentation.
|
||||
|
||||
Stat keys that contain a ``{placeholder}`` below stand for a family of stats,
|
||||
one per actual value of the placeholder.
|
||||
|
||||
.. note:: Most stats are set by a specific :ref:`component
|
||||
<topics-components>`, and are only present if that component is enabled and
|
||||
its code path is reached. A stat that is missing from
|
||||
:meth:`~scrapy.statscollectors.StatsCollector.get_stats` output is
|
||||
equivalent to a counter of 0.
|
||||
|
||||
.. stat:: downloader/exception_count
|
||||
|
||||
``downloader/exception_count``
|
||||
Number of exceptions raised while downloading requests.
|
||||
|
||||
Set by :class:`~scrapy.downloadermiddlewares.stats.DownloaderStats`.
|
||||
|
||||
.. stat:: downloader/exception_type_count/{exception_type}
|
||||
|
||||
``downloader/exception_type_count/{exception_type}``
|
||||
Number of exceptions raised while downloading requests, per exception type,
|
||||
where ``{exception_type}`` is the import path of the exception class, e.g.
|
||||
``twisted.internet.error.DNSLookupError``.
|
||||
|
||||
Set by :class:`~scrapy.downloadermiddlewares.stats.DownloaderStats`.
|
||||
|
||||
.. stat:: downloader/request_bytes
|
||||
|
||||
``downloader/request_bytes``
|
||||
Total size, in bytes, of the requests sent, counting the request line, the
|
||||
headers and the body. As with :stat:`downloader/request_count`, requests
|
||||
served from the cache are also counted.
|
||||
|
||||
It is an approximation, reconstructed from each :class:`~scrapy.Request`
|
||||
object instead of measured on the wire, so it does not account for the
|
||||
actual bytes that the :ref:`download handler
|
||||
<topics-download-handlers>` sends, e.g. transport-level overhead.
|
||||
|
||||
Set by :class:`~scrapy.downloadermiddlewares.stats.DownloaderStats`.
|
||||
|
||||
.. stat:: downloader/request_count
|
||||
|
||||
``downloader/request_count``
|
||||
Number of requests sent.
|
||||
|
||||
Requests that :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`
|
||||
serves from the cache are also counted, even though they are never sent,
|
||||
because it handles requests after
|
||||
:class:`~scrapy.downloadermiddlewares.stats.DownloaderStats`.
|
||||
|
||||
Set by :class:`~scrapy.downloadermiddlewares.stats.DownloaderStats`.
|
||||
|
||||
.. stat:: downloader/request_method_count/{method}
|
||||
|
||||
``downloader/request_method_count/{method}``
|
||||
Number of requests sent, per HTTP method, e.g. ``GET`` or ``POST``. As with
|
||||
:stat:`downloader/request_count`, requests served from the cache are also
|
||||
counted.
|
||||
|
||||
Set by :class:`~scrapy.downloadermiddlewares.stats.DownloaderStats`.
|
||||
|
||||
.. stat:: downloader/response_bytes
|
||||
|
||||
``downloader/response_bytes``
|
||||
Total size, in bytes, of the responses received, counting the status line,
|
||||
the headers and the body. It covers the same responses as
|
||||
:stat:`downloader/response_count`.
|
||||
|
||||
The body is counted as received, i.e. still compressed for responses that
|
||||
used ``Content-Encoding``, because
|
||||
:class:`~scrapy.downloadermiddlewares.stats.DownloaderStats` handles
|
||||
responses before
|
||||
:class:`~scrapy.downloadermiddlewares.httpcompression.HttpCompressionMiddleware`
|
||||
decompresses them. See :stat:`httpcompression/response_bytes` for
|
||||
decompressed sizes.
|
||||
|
||||
Set by :class:`~scrapy.downloadermiddlewares.stats.DownloaderStats`.
|
||||
|
||||
.. stat:: downloader/response_count
|
||||
|
||||
``downloader/response_count``
|
||||
Number of responses received.
|
||||
|
||||
It counts responses that :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`
|
||||
serves from the cache, even though they do not come from the network, and
|
||||
responses that a downloader middleware consumes before they reach your
|
||||
spider, e.g. redirect responses that :class:`~scrapy.downloadermiddlewares.redirect.RedirectMiddleware`
|
||||
turns into new requests. Compare with :stat:`response_received_count`.
|
||||
|
||||
Set by :class:`~scrapy.downloadermiddlewares.stats.DownloaderStats`.
|
||||
|
||||
.. stat:: downloader/response_status_count/{status_code}
|
||||
|
||||
``downloader/response_status_count/{status_code}``
|
||||
Number of responses received, per HTTP status code, e.g. ``200`` or
|
||||
``404``. It covers the same responses as :stat:`downloader/response_count`.
|
||||
|
||||
Set by :class:`~scrapy.downloadermiddlewares.stats.DownloaderStats`.
|
||||
|
||||
.. stat:: dupefilter/filtered
|
||||
|
||||
``dupefilter/filtered``
|
||||
Number of requests dropped as duplicates.
|
||||
|
||||
Set by :class:`~scrapy.dupefilters.RFPDupeFilter`.
|
||||
|
||||
.. stat:: elapsed_time_seconds
|
||||
|
||||
``elapsed_time_seconds``
|
||||
Time, as a :class:`float`, in seconds, between the :signal:`spider_opened`
|
||||
and the :signal:`spider_closed` signals.
|
||||
|
||||
Set by :class:`~scrapy.extensions.corestats.CoreStats`.
|
||||
|
||||
.. stat:: feedexport/failed_count/{storage}
|
||||
|
||||
``feedexport/failed_count/{storage}``
|
||||
Number of :ref:`feeds <topics-feed-exports>` that could not be stored, per
|
||||
:ref:`storage backend <topics-feed-storage-backends>`, where ``{storage}``
|
||||
is the class name of the storage backend, e.g. ``FileFeedStorage``.
|
||||
|
||||
.. stat:: feedexport/success_count/{storage}
|
||||
|
||||
``feedexport/success_count/{storage}``
|
||||
Number of :ref:`feeds <topics-feed-exports>` stored successfully, per
|
||||
:ref:`storage backend <topics-feed-storage-backends>`, where ``{storage}``
|
||||
is the class name of the storage backend, e.g. ``FileFeedStorage``.
|
||||
|
||||
.. stat:: file_count
|
||||
|
||||
``file_count``
|
||||
Number of files handled by the :ref:`media pipelines
|
||||
<topics-media-pipeline>`.
|
||||
|
||||
.. stat:: file_status_count/{status}
|
||||
|
||||
``file_status_count/{status}``
|
||||
Number of files handled by the :ref:`media pipelines
|
||||
<topics-media-pipeline>`, per status, where ``{status}`` is one of:
|
||||
|
||||
- ``downloaded``: the file was downloaded.
|
||||
|
||||
- ``cached``: the file came from the
|
||||
:class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`
|
||||
cache.
|
||||
|
||||
- ``uptodate``: the file was already in the storage backend and had not
|
||||
:ref:`expired <file-expiration>`, so it was not downloaded again.
|
||||
|
||||
.. stat:: finish_reason
|
||||
|
||||
``finish_reason``
|
||||
String indicating why the crawl finished. It matches the *reason* argument
|
||||
of the :signal:`spider_closed` signal.
|
||||
|
||||
Scrapy uses the following reasons:
|
||||
|
||||
- ``cancelled``: the spider was closed without a more specific reason,
|
||||
e.g. because :exc:`~scrapy.exceptions.CloseSpider` was raised without
|
||||
one.
|
||||
|
||||
- ``closespider_errorcount``: see :setting:`CLOSESPIDER_ERRORCOUNT`.
|
||||
|
||||
- ``closespider_itemcount``: see :setting:`CLOSESPIDER_ITEMCOUNT`.
|
||||
|
||||
- ``closespider_pagecount``: see :setting:`CLOSESPIDER_PAGECOUNT`.
|
||||
|
||||
- ``closespider_pagecount_no_item``: see
|
||||
:setting:`CLOSESPIDER_PAGECOUNT_NO_ITEM`.
|
||||
|
||||
- ``closespider_timeout``: see :setting:`CLOSESPIDER_TIMEOUT`.
|
||||
|
||||
- ``closespider_timeout_no_item``: see
|
||||
:setting:`CLOSESPIDER_TIMEOUT_NO_ITEM`.
|
||||
|
||||
- ``finished``: the spider became idle with no pending requests, i.e. it
|
||||
finished normally.
|
||||
|
||||
- ``memusage_exceeded``: see :setting:`MEMUSAGE_LIMIT_MB`.
|
||||
|
||||
- ``shutdown``: the crawl was interrupted, e.g. by a system signal such
|
||||
as ``SIGINT`` (:kbd:`Ctrl-C`).
|
||||
|
||||
Third-party components and your own code may use any other reason, e.g. by
|
||||
raising :exc:`~scrapy.exceptions.CloseSpider` with it.
|
||||
|
||||
Set by :class:`~scrapy.extensions.corestats.CoreStats`.
|
||||
|
||||
.. stat:: finish_time
|
||||
|
||||
``finish_time``
|
||||
Timezone-aware :class:`~datetime.datetime` object, in UTC, indicating when
|
||||
the :signal:`spider_closed` signal was sent.
|
||||
|
||||
Set by :class:`~scrapy.extensions.corestats.CoreStats`.
|
||||
|
||||
.. stat:: httpcache/errorrecovery
|
||||
|
||||
``httpcache/errorrecovery``
|
||||
Number of times that a stale cached response was used because downloading a
|
||||
fresh response raised an exception.
|
||||
|
||||
Set by :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`.
|
||||
|
||||
.. stat:: httpcache/firsthand
|
||||
|
||||
``httpcache/firsthand``
|
||||
Number of responses that were downloaded without a matching cache entry to
|
||||
validate against, i.e. responses for requests counted in
|
||||
:stat:`httpcache/miss`.
|
||||
|
||||
It is lower than :stat:`httpcache/miss` when some of those requests yield
|
||||
no response, either because they are dropped (see
|
||||
:stat:`httpcache/ignore`) or because their download fails.
|
||||
|
||||
Set by :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`.
|
||||
|
||||
.. stat:: httpcache/hit
|
||||
|
||||
``httpcache/hit``
|
||||
Number of requests served from the cache.
|
||||
|
||||
Set by :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`.
|
||||
|
||||
.. stat:: httpcache/ignore
|
||||
|
||||
``httpcache/ignore``
|
||||
Number of requests dropped because they were not in the cache and
|
||||
:setting:`HTTPCACHE_IGNORE_MISSING` is ``True``.
|
||||
|
||||
Set by :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`.
|
||||
|
||||
.. stat:: httpcache/invalidate
|
||||
|
||||
``httpcache/invalidate``
|
||||
Number of times that a cached response failed validation and was replaced
|
||||
with a freshly downloaded response.
|
||||
|
||||
Set by :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`.
|
||||
|
||||
.. stat:: httpcache/miss
|
||||
|
||||
``httpcache/miss``
|
||||
Number of requests for which no cache entry could be read, either because
|
||||
there was none or because reading it failed, in which case the request is
|
||||
also counted in :stat:`httpcache/retrieve_error`. Those requests are
|
||||
downloaded (see :stat:`httpcache/firsthand`), or dropped if
|
||||
:setting:`HTTPCACHE_IGNORE_MISSING` is ``True`` (see
|
||||
:stat:`httpcache/ignore`).
|
||||
|
||||
Requests with a stale cache entry are not counted here; see
|
||||
:stat:`httpcache/revalidate` and :stat:`httpcache/invalidate`.
|
||||
|
||||
Set by :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`.
|
||||
|
||||
.. stat:: httpcache/retrieve_error
|
||||
|
||||
``httpcache/retrieve_error``
|
||||
Number of cache entries that could not be read, and hence were treated as
|
||||
cache misses. Those requests are also counted in :stat:`httpcache/miss`.
|
||||
|
||||
Set by :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`.
|
||||
|
||||
.. stat:: httpcache/revalidate
|
||||
|
||||
``httpcache/revalidate``
|
||||
Number of times that a cached response was successfully validated against
|
||||
the target server, and hence used instead of the fresh response.
|
||||
|
||||
Set by :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`.
|
||||
|
||||
.. stat:: httpcache/store
|
||||
|
||||
``httpcache/store``
|
||||
Number of responses stored in the cache.
|
||||
|
||||
Set by :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`.
|
||||
|
||||
.. stat:: httpcache/uncacheable
|
||||
|
||||
``httpcache/uncacheable``
|
||||
Number of responses not stored in the cache because the
|
||||
:setting:`HTTPCACHE_POLICY` did not allow it.
|
||||
|
||||
Every response considered for caching is counted either here or in
|
||||
:stat:`httpcache/store`, so ``httpcache/store + httpcache/uncacheable``
|
||||
equals ``httpcache/firsthand + httpcache/invalidate``.
|
||||
|
||||
Set by :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`.
|
||||
|
||||
.. stat:: httpcompression/response_bytes
|
||||
|
||||
``httpcompression/response_bytes``
|
||||
Total size, in bytes, of decompressed response bodies, counting only the
|
||||
body and only responses that were actually decompressed. Compare with
|
||||
:stat:`downloader/response_bytes`.
|
||||
|
||||
Set by
|
||||
:class:`~scrapy.downloadermiddlewares.httpcompression.HttpCompressionMiddleware`.
|
||||
|
||||
.. stat:: httpcompression/response_count
|
||||
|
||||
``httpcompression/response_count``
|
||||
Number of decompressed responses.
|
||||
|
||||
Set by
|
||||
:class:`~scrapy.downloadermiddlewares.httpcompression.HttpCompressionMiddleware`.
|
||||
|
||||
.. stat:: httperror/response_ignored_count
|
||||
|
||||
``httperror/response_ignored_count``
|
||||
Number of responses dropped because of their HTTP status code.
|
||||
|
||||
Set by :class:`~scrapy.spidermiddlewares.httperror.HttpErrorMiddleware`.
|
||||
|
||||
.. stat:: httperror/response_ignored_status_count/{status_code}
|
||||
|
||||
``httperror/response_ignored_status_count/{status_code}``
|
||||
Number of responses dropped because of their HTTP status code, per HTTP
|
||||
status code, e.g. ``404``.
|
||||
|
||||
Set by :class:`~scrapy.spidermiddlewares.httperror.HttpErrorMiddleware`.
|
||||
|
||||
.. stat:: item_dropped_count
|
||||
|
||||
``item_dropped_count``
|
||||
Number of items dropped by an :ref:`item pipeline
|
||||
<topics-item-pipeline>`, i.e. number of times that the
|
||||
:signal:`item_dropped` signal was sent.
|
||||
|
||||
Set by :class:`~scrapy.extensions.corestats.CoreStats`.
|
||||
|
||||
.. stat:: item_dropped_reasons_count/{exception}
|
||||
|
||||
``item_dropped_reasons_count/{exception}``
|
||||
Number of items dropped, per exception, where ``{exception}`` is the class
|
||||
name of the exception that caused the item to be dropped.
|
||||
|
||||
Only :exc:`~scrapy.exceptions.DropItem` and its subclasses drop items, and
|
||||
each one is counted under its own class name, e.g.
|
||||
``item_dropped_reasons_count/DropItem`` for
|
||||
:exc:`~scrapy.exceptions.DropItem` itself and
|
||||
``item_dropped_reasons_count/MyDropItem`` for a ``MyDropItem`` subclass of
|
||||
it. Any other exception raised by an :ref:`item pipeline
|
||||
<topics-item-pipeline>` triggers the :signal:`item_error` signal instead of
|
||||
:signal:`item_dropped`, and is not counted here or in
|
||||
:stat:`item_dropped_count`.
|
||||
|
||||
Set by :class:`~scrapy.extensions.corestats.CoreStats`.
|
||||
|
||||
.. stat:: item_scraped_count
|
||||
|
||||
``item_scraped_count``
|
||||
Number of items that passed all :ref:`item pipelines
|
||||
<topics-item-pipeline>`, i.e. number of times that the
|
||||
:signal:`item_scraped` signal was sent.
|
||||
|
||||
Set by :class:`~scrapy.extensions.corestats.CoreStats`.
|
||||
|
||||
.. stat:: items_per_minute
|
||||
|
||||
``items_per_minute``
|
||||
Average number of items scraped per minute during the crawl.
|
||||
|
||||
It is ``None`` if the crawl took less than a minute.
|
||||
|
||||
Set by :class:`~scrapy.extensions.logstats.LogStats`.
|
||||
|
||||
.. stat:: log_count/{level}
|
||||
|
||||
``log_count/{level}``
|
||||
Number of log messages, per logging level name, e.g. ``INFO`` or
|
||||
``WARNING``.
|
||||
|
||||
Only messages that the :setting:`LOG_LEVEL` setting allows are counted.
|
||||
|
||||
Set by :class:`~scrapy.extensions.logcount.LogCount`.
|
||||
|
||||
.. stat:: memdebug/gc_garbage_count
|
||||
|
||||
``memdebug/gc_garbage_count``
|
||||
Number of objects in :data:`gc.garbage` when the spider is closed.
|
||||
|
||||
Set by :class:`~scrapy.extensions.memdebug.MemoryDebugger`, which requires
|
||||
:setting:`MEMDEBUG_ENABLED` to be ``True``.
|
||||
|
||||
.. stat:: memdebug/live_refs/{cls}
|
||||
|
||||
``memdebug/live_refs/{cls}``
|
||||
Number of live objects of class ``{cls}`` when the spider is closed, as
|
||||
reported by :ref:`trackref <topics-leaks-trackrefs>`, e.g.
|
||||
``memdebug/live_refs/HtmlResponse``.
|
||||
|
||||
Only set for classes with at least 1 live object.
|
||||
|
||||
Set by :class:`~scrapy.extensions.memdebug.MemoryDebugger`, which requires
|
||||
:setting:`MEMDEBUG_ENABLED` to be ``True``.
|
||||
|
||||
.. stat:: memusage/limit_reached
|
||||
|
||||
``memusage/limit_reached``
|
||||
``1`` if memory usage exceeded :setting:`MEMUSAGE_LIMIT_MB`, which also
|
||||
stops the crawl.
|
||||
|
||||
Set by :class:`~scrapy.extensions.memusage.MemoryUsage`.
|
||||
|
||||
.. stat:: memusage/max
|
||||
|
||||
``memusage/max``
|
||||
Maximum peak memory usage, in bytes, observed during the crawl.
|
||||
|
||||
Set by :class:`~scrapy.extensions.memusage.MemoryUsage`.
|
||||
|
||||
.. stat:: memusage/startup
|
||||
|
||||
``memusage/startup``
|
||||
Peak memory usage, in bytes, when the engine started.
|
||||
|
||||
Set by :class:`~scrapy.extensions.memusage.MemoryUsage`.
|
||||
|
||||
.. stat:: memusage/warning_reached
|
||||
|
||||
``memusage/warning_reached``
|
||||
``1`` if memory usage exceeded :setting:`MEMUSAGE_WARNING_MB`.
|
||||
|
||||
Set by :class:`~scrapy.extensions.memusage.MemoryUsage`.
|
||||
|
||||
.. stat:: offsite/domains
|
||||
|
||||
``offsite/domains``
|
||||
Number of distinct domains for which at least 1 request was dropped for
|
||||
being offsite.
|
||||
|
||||
Set by :class:`~scrapy.downloadermiddlewares.offsite.OffsiteMiddleware`.
|
||||
|
||||
.. stat:: offsite/filtered
|
||||
|
||||
``offsite/filtered``
|
||||
Number of requests dropped for being offsite.
|
||||
|
||||
Set by :class:`~scrapy.downloadermiddlewares.offsite.OffsiteMiddleware`.
|
||||
|
||||
.. stat:: request_depth_count/{depth}
|
||||
|
||||
``request_depth_count/{depth}``
|
||||
Number of requests scheduled at depth ``{depth}``, e.g.
|
||||
``request_depth_count/2``.
|
||||
|
||||
Set by :class:`~scrapy.spidermiddlewares.depth.DepthMiddleware`, which
|
||||
requires :setting:`DEPTH_STATS_VERBOSE` to be ``True`` for this stat.
|
||||
|
||||
.. stat:: request_depth_max
|
||||
|
||||
``request_depth_max``
|
||||
Maximum depth reached.
|
||||
|
||||
Set by :class:`~scrapy.spidermiddlewares.depth.DepthMiddleware`.
|
||||
|
||||
.. stat:: response_received_count
|
||||
|
||||
``response_received_count``
|
||||
Number of responses received, i.e. number of times that the
|
||||
:signal:`response_received` signal was sent.
|
||||
|
||||
Unlike :stat:`downloader/response_count`, it does not count responses that
|
||||
a downloader middleware consumes before they reach the engine, e.g.
|
||||
redirect responses that :class:`~scrapy.downloadermiddlewares.redirect.RedirectMiddleware`
|
||||
turns into new requests. Both count responses that :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`
|
||||
serves from the cache.
|
||||
|
||||
Set by :class:`~scrapy.extensions.corestats.CoreStats`.
|
||||
|
||||
.. stat:: responses_per_minute
|
||||
|
||||
``responses_per_minute``
|
||||
Average number of responses received per minute during the crawl.
|
||||
|
||||
It is ``None`` if the crawl took less than a minute.
|
||||
|
||||
Set by :class:`~scrapy.extensions.logstats.LogStats`.
|
||||
|
||||
.. stat:: retry/count
|
||||
|
||||
``retry/count``
|
||||
Number of requests retried.
|
||||
|
||||
Set by :func:`~scrapy.downloadermiddlewares.retry.get_retry_request`, which
|
||||
:class:`~scrapy.downloadermiddlewares.retry.RetryMiddleware` uses.
|
||||
|
||||
.. stat:: retry/max_reached
|
||||
|
||||
``retry/max_reached``
|
||||
Number of requests that were not retried because they had already been
|
||||
retried :setting:`RETRY_TIMES` times.
|
||||
|
||||
Set by :func:`~scrapy.downloadermiddlewares.retry.get_retry_request`, which
|
||||
:class:`~scrapy.downloadermiddlewares.retry.RetryMiddleware` uses.
|
||||
|
||||
.. stat:: retry/reason_count/{reason}
|
||||
|
||||
``retry/reason_count/{reason}``
|
||||
Number of requests retried, per reason, e.g.
|
||||
``retry/reason_count/twisted.internet.error.TimeoutError`` or
|
||||
``retry/reason_count/504 Gateway Time-out``.
|
||||
|
||||
Set by :func:`~scrapy.downloadermiddlewares.retry.get_retry_request`, which
|
||||
:class:`~scrapy.downloadermiddlewares.retry.RetryMiddleware` uses.
|
||||
|
||||
.. note:: Code calling
|
||||
:func:`~scrapy.downloadermiddlewares.retry.get_retry_request` may pass a
|
||||
custom *stats_base_key*, in which case ``retry`` is replaced with that key
|
||||
in the 3 stats above.
|
||||
|
||||
.. stat:: robotstxt/exception_count/{exception_type}
|
||||
|
||||
``robotstxt/exception_count/{exception_type}``
|
||||
Number of exceptions raised while downloading ``robots.txt`` files, per
|
||||
exception type, where ``{exception_type}`` is the string representation of
|
||||
the exception class, e.g. ``<class
|
||||
'twisted.internet.error.DNSLookupError'>``.
|
||||
|
||||
Set by
|
||||
:class:`~scrapy.downloadermiddlewares.robotstxt.RobotsTxtMiddleware`.
|
||||
|
||||
.. stat:: robotstxt/forbidden
|
||||
|
||||
``robotstxt/forbidden``
|
||||
Number of requests dropped for being disallowed by ``robots.txt``.
|
||||
|
||||
Set by
|
||||
:class:`~scrapy.downloadermiddlewares.robotstxt.RobotsTxtMiddleware`.
|
||||
|
||||
.. stat:: robotstxt/request_count
|
||||
|
||||
``robotstxt/request_count``
|
||||
Number of ``robots.txt`` files requested, i.e. 1 per network location for
|
||||
which at least 1 request was sent.
|
||||
|
||||
Set by
|
||||
:class:`~scrapy.downloadermiddlewares.robotstxt.RobotsTxtMiddleware`.
|
||||
|
||||
.. stat:: robotstxt/response_count
|
||||
|
||||
``robotstxt/response_count``
|
||||
Number of ``robots.txt`` responses received.
|
||||
|
||||
Set by
|
||||
:class:`~scrapy.downloadermiddlewares.robotstxt.RobotsTxtMiddleware`.
|
||||
|
||||
.. stat:: robotstxt/response_status_count/{status_code}
|
||||
|
||||
``robotstxt/response_status_count/{status_code}``
|
||||
Number of ``robots.txt`` responses received, per HTTP status code, e.g.
|
||||
``404``.
|
||||
|
||||
Set by
|
||||
:class:`~scrapy.downloadermiddlewares.robotstxt.RobotsTxtMiddleware`.
|
||||
|
||||
.. stat:: scheduler/dequeued
|
||||
|
||||
``scheduler/dequeued``
|
||||
Number of requests read from the :ref:`scheduler <topics-scheduler>`.
|
||||
|
||||
.. stat:: scheduler/dequeued/disk
|
||||
|
||||
``scheduler/dequeued/disk``
|
||||
Number of requests read from the disk queue of the :ref:`scheduler
|
||||
<topics-scheduler>`.
|
||||
|
||||
.. stat:: scheduler/dequeued/memory
|
||||
|
||||
``scheduler/dequeued/memory``
|
||||
Number of requests read from the memory queue of the :ref:`scheduler
|
||||
<topics-scheduler>`.
|
||||
|
||||
.. stat:: scheduler/enqueued
|
||||
|
||||
``scheduler/enqueued``
|
||||
Number of requests stored into the :ref:`scheduler <topics-scheduler>`.
|
||||
|
||||
.. stat:: scheduler/enqueued/disk
|
||||
|
||||
``scheduler/enqueued/disk``
|
||||
Number of requests stored into the disk queue of the :ref:`scheduler
|
||||
<topics-scheduler>`.
|
||||
|
||||
.. stat:: scheduler/enqueued/memory
|
||||
|
||||
``scheduler/enqueued/memory``
|
||||
Number of requests stored into the memory queue of the :ref:`scheduler
|
||||
<topics-scheduler>`.
|
||||
|
||||
.. stat:: scheduler/unserializable
|
||||
|
||||
``scheduler/unserializable``
|
||||
Number of requests that could not be stored into the disk queue of the
|
||||
:ref:`scheduler <topics-scheduler>` because they could not be
|
||||
:ref:`serialized <request-serialization>`, and hence were stored into the
|
||||
memory queue instead.
|
||||
|
||||
.. stat:: spider_exceptions/count
|
||||
|
||||
``spider_exceptions/count``
|
||||
Number of unhandled exceptions raised by spider callbacks.
|
||||
|
||||
Set by the :ref:`scraper <topics-architecture>`.
|
||||
|
||||
.. stat:: spider_exceptions/{exception}
|
||||
|
||||
``spider_exceptions/{exception}``
|
||||
Number of unhandled exceptions raised by spider callbacks, per exception,
|
||||
where ``{exception}`` is the class name of the exception, e.g.
|
||||
``spider_exceptions/ValueError``.
|
||||
|
||||
Set by the :ref:`scraper <topics-architecture>`.
|
||||
|
||||
.. stat:: start_time
|
||||
|
||||
``start_time``
|
||||
Timezone-aware :class:`~datetime.datetime` object, in UTC, indicating when
|
||||
the :signal:`spider_opened` signal was sent.
|
||||
|
||||
Set by :class:`~scrapy.extensions.corestats.CoreStats`.
|
||||
|
||||
.. stat:: urllength/request_ignored_count
|
||||
|
||||
``urllength/request_ignored_count``
|
||||
Number of requests dropped for having a URL longer than
|
||||
:setting:`URLLENGTH_LIMIT`.
|
||||
|
||||
Set by :class:`~scrapy.spidermiddlewares.urllength.UrlLengthMiddleware`.
|
||||
|
|
|
|||
|
|
@ -16,7 +16,7 @@ class Command(BaseRunSpiderCommand):
|
|||
return "[options] <spider>"
|
||||
|
||||
def short_desc(self) -> str:
|
||||
return "Run a spider"
|
||||
return "Run a spider of the current project, by name"
|
||||
|
||||
def run(self, args: list[str], opts: argparse.Namespace) -> None:
|
||||
if len(args) < 1:
|
||||
|
|
|
|||
|
|
@ -32,10 +32,7 @@ def sanitize_module_name(module_name: str) -> str:
|
|||
|
||||
def extract_domain(url: str) -> str:
|
||||
"""Extract domain name from URL string"""
|
||||
o = urlparse(url)
|
||||
if o.scheme == "" and o.netloc == "":
|
||||
o = urlparse("//" + url.lstrip("/"))
|
||||
return o.netloc
|
||||
return urlparse(url).netloc
|
||||
|
||||
|
||||
def verify_url_scheme(url: str) -> str:
|
||||
|
|
|
|||
|
|
@ -41,7 +41,7 @@ class Command(BaseRunSpiderCommand):
|
|||
spider: Spider | None = None
|
||||
items: ClassVar[dict[int, list[Any]]] = {}
|
||||
requests: ClassVar[dict[int, list[Request]]] = {}
|
||||
spidercls: type[Spider] | None
|
||||
spidercls: type[Spider] | None = None
|
||||
|
||||
first_response = None
|
||||
|
||||
|
|
|
|||
|
|
@ -38,7 +38,7 @@ class Command(BaseRunSpiderCommand):
|
|||
return "[options] <spider_file>"
|
||||
|
||||
def short_desc(self) -> str:
|
||||
return "Run a self-contained spider (without creating a project)"
|
||||
return "Run a spider from a Python file, no project required"
|
||||
|
||||
def long_desc(self) -> str:
|
||||
return "Run the spider defined in the given file"
|
||||
|
|
|
|||
|
|
@ -366,8 +366,8 @@ class Scheduler(BaseScheduler):
|
|||
Unless the received request is filtered out by the Dupefilter, attempt to push
|
||||
it into the disk queue, falling back to pushing it into the memory queue.
|
||||
|
||||
Increment the appropriate stats, such as: ``scheduler/enqueued``,
|
||||
``scheduler/enqueued/disk``, ``scheduler/enqueued/memory``.
|
||||
Increment the appropriate stats, such as: :stat:`scheduler/enqueued`,
|
||||
:stat:`scheduler/enqueued/disk`, :stat:`scheduler/enqueued/memory`.
|
||||
|
||||
Return ``True`` if the request was stored successfully, ``False`` otherwise.
|
||||
"""
|
||||
|
|
@ -390,8 +390,8 @@ class Scheduler(BaseScheduler):
|
|||
falling back to the disk queue if the memory queue is empty.
|
||||
Return ``None`` if there are no more enqueued requests.
|
||||
|
||||
Increment the appropriate stats, such as: ``scheduler/dequeued``,
|
||||
``scheduler/dequeued/disk``, ``scheduler/dequeued/memory``.
|
||||
Increment the appropriate stats, such as: :stat:`scheduler/dequeued`,
|
||||
:stat:`scheduler/dequeued/disk`, :stat:`scheduler/dequeued/memory`.
|
||||
"""
|
||||
request: Request | None = self.mqs.pop()
|
||||
assert self.stats is not None
|
||||
|
|
|
|||
|
|
@ -11,6 +11,7 @@ from typing import TYPE_CHECKING
|
|||
|
||||
from twisted.internet.defer import Deferred
|
||||
|
||||
from scrapy import signals
|
||||
from scrapy.exceptions import IgnoreRequest, NotConfigured
|
||||
from scrapy.http import Request, Response
|
||||
from scrapy.http.request import NO_CALLBACK
|
||||
|
|
@ -98,7 +99,7 @@ class RobotsTxtMiddleware:
|
|||
assert self.crawler.stats
|
||||
try:
|
||||
resp = await self.crawler.engine.download_async(robotsreq)
|
||||
self._parse_robots(resp, netloc)
|
||||
await self._parse_robots(resp, netloc, request)
|
||||
except Exception as e:
|
||||
if not isinstance(e, IgnoreRequest):
|
||||
logger.error(
|
||||
|
|
@ -115,13 +116,20 @@ class RobotsTxtMiddleware:
|
|||
return await maybe_deferred_to_future(parser)
|
||||
return parser
|
||||
|
||||
def _parse_robots(self, response: Response, netloc: str) -> None:
|
||||
async def _parse_robots(
|
||||
self, response: Response, netloc: str, request: Request
|
||||
) -> None:
|
||||
assert self.crawler.stats
|
||||
self.crawler.stats.inc_value("robotstxt/response_count")
|
||||
self.crawler.stats.inc_value(
|
||||
f"robotstxt/response_status_count/{response.status}"
|
||||
)
|
||||
rp = self._parserimpl.from_crawler(self.crawler, response.body)
|
||||
await self.crawler.signals.send_catch_log_async(
|
||||
signal=signals.robots_parsed,
|
||||
robotparser=rp,
|
||||
request=request,
|
||||
)
|
||||
rp_dfd = self._parsers[netloc]
|
||||
assert isinstance(rp_dfd, Deferred)
|
||||
self._parsers[netloc] = rp
|
||||
|
|
|
|||
|
|
@ -20,7 +20,7 @@ class LogCount:
|
|||
"""Install a log handler that counts log messages by level.
|
||||
|
||||
The handler installed is :class:`scrapy.utils.log.LogCounterHandler`.
|
||||
The counts are stored in stats as ``log_count/<level>``.
|
||||
The counts are stored in the :stat:`log_count/{level}` stat.
|
||||
|
||||
.. versionadded:: 2.14
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -67,6 +67,15 @@ class RobotParser(metaclass=ABCMeta):
|
|||
:type user_agent: str or bytes
|
||||
"""
|
||||
|
||||
def crawl_delay(self, user_agent: str | bytes) -> float | None:
|
||||
"""Return the ``Crawl-delay`` directive for ``user_agent`` as a number
|
||||
of seconds, or ``None`` if it is not set or the backend does not support
|
||||
it.
|
||||
|
||||
.. versionadded:: VERSION
|
||||
"""
|
||||
return None
|
||||
|
||||
|
||||
class PythonRobotParser(RobotParser):
|
||||
def __init__(self, robotstxt_body: bytes, spider: Spider | None):
|
||||
|
|
@ -85,6 +94,10 @@ class PythonRobotParser(RobotParser):
|
|||
url = to_unicode(url)
|
||||
return self.rp.can_fetch(user_agent, url)
|
||||
|
||||
def crawl_delay(self, user_agent: str | bytes) -> float | None:
|
||||
delay = self.rp.crawl_delay(to_unicode(user_agent))
|
||||
return None if delay is None else float(delay)
|
||||
|
||||
|
||||
class RerpRobotParser(RobotParser):
|
||||
def __init__(self, robotstxt_body: bytes, spider: Spider | None):
|
||||
|
|
@ -105,6 +118,10 @@ class RerpRobotParser(RobotParser):
|
|||
url = to_unicode(url)
|
||||
return cast("bool", self.rp.is_allowed(user_agent, url))
|
||||
|
||||
def crawl_delay(self, user_agent: str | bytes) -> float | None:
|
||||
delay = self.rp.get_crawl_delay(to_unicode(user_agent))
|
||||
return None if delay is None else float(delay)
|
||||
|
||||
|
||||
class ProtegoRobotParser(RobotParser):
|
||||
def __init__(self, robotstxt_body: bytes, spider: Spider | None):
|
||||
|
|
@ -121,3 +138,7 @@ class ProtegoRobotParser(RobotParser):
|
|||
user_agent = to_unicode(user_agent)
|
||||
url = to_unicode(url)
|
||||
return self.rp.can_fetch(url, user_agent)
|
||||
|
||||
def crawl_delay(self, user_agent: str | bytes) -> float | None:
|
||||
delay = self.rp.crawl_delay(to_unicode(user_agent))
|
||||
return None if delay is None else float(delay)
|
||||
|
|
|
|||
|
|
@ -21,6 +21,7 @@ response_received = object()
|
|||
response_downloaded = object()
|
||||
headers_received = object()
|
||||
bytes_received = object()
|
||||
robots_parsed = object()
|
||||
item_scraped = object()
|
||||
item_dropped = object()
|
||||
item_error = object()
|
||||
|
|
|
|||
|
|
@ -0,0 +1,28 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
from scrapy.utils.test import get_crawler
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from scrapy import Spider
|
||||
from scrapy.crawler import Crawler
|
||||
|
||||
|
||||
def crawl(spidercls: type[Spider], settings: dict[str, Any], **kwargs: Any) -> Crawler:
|
||||
"""Run a crawl to completion and return its crawler.
|
||||
|
||||
Unlike the rest of the test suite, benchmarks run without ``pytest-twisted``
|
||||
and drive the reactor themselves, since the code being measured must be
|
||||
callable synchronously by ``pytest-codspeed``.
|
||||
"""
|
||||
from twisted.internet import reactor
|
||||
|
||||
crawler = get_crawler(spidercls, settings)
|
||||
result: list[Any] = []
|
||||
crawler.crawl(**kwargs).addBoth(result.append)
|
||||
while not result:
|
||||
reactor.iterate(0.001)
|
||||
if isinstance(result[0], BaseException):
|
||||
raise result[0]
|
||||
return crawler
|
||||
|
|
@ -0,0 +1,27 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
|
||||
from scrapy.utils.reactor import install_reactor
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Generator
|
||||
|
||||
|
||||
@pytest.fixture(scope="session", autouse=True)
|
||||
def running_reactor() -> Generator[None]:
|
||||
install_reactor("twisted.internet.asyncioreactor.AsyncioSelectorReactor")
|
||||
|
||||
from twisted.internet import reactor
|
||||
|
||||
# Marks the reactor as running without blocking, so that crawls can be
|
||||
# driven with reactor.iterate(), see tests.benchmarks.crawl().
|
||||
reactor.startRunning(installSignalHandlers=False)
|
||||
|
||||
yield
|
||||
|
||||
reactor.stop()
|
||||
# Lets the shutdown event triggers run, e.g. to join the thread pool.
|
||||
reactor.iterate(0)
|
||||
|
|
@ -0,0 +1,69 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING, Any
|
||||
from urllib.parse import urlencode
|
||||
|
||||
import pytest
|
||||
|
||||
from scrapy import Field, Item, Request, Spider
|
||||
from scrapy.linkextractors import LinkExtractor
|
||||
from tests.benchmarks import crawl
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import AsyncIterator
|
||||
|
||||
from pytest_codspeed import BenchmarkFixture # type: ignore[import-not-found]
|
||||
|
||||
from scrapy.http import Response
|
||||
from tests.mockserver.http import MockServer
|
||||
|
||||
pytest.importorskip("pytest_codspeed", reason="Benchmarks require pytest-codspeed")
|
||||
|
||||
PAGES = 100
|
||||
LINKS_PER_PAGE = 5
|
||||
|
||||
|
||||
class _Page(Item):
|
||||
url = Field()
|
||||
anchors = Field()
|
||||
|
||||
|
||||
class _FollowSpider(Spider):
|
||||
name = "benchmark"
|
||||
url: str
|
||||
link_extractor = LinkExtractor()
|
||||
|
||||
async def start(self) -> AsyncIterator[Any]:
|
||||
yield Request(self.url, dont_filter=True)
|
||||
|
||||
def parse(self, response: Response) -> Any:
|
||||
yield _Page(
|
||||
url=response.url,
|
||||
anchors=response.css("a::text").getall(),
|
||||
)
|
||||
for link in self.link_extractor.extract_links(response): # type: ignore[arg-type]
|
||||
yield Request(link.url)
|
||||
|
||||
|
||||
class _Pipeline:
|
||||
def process_item(self, item: Any) -> Any:
|
||||
return item
|
||||
|
||||
|
||||
def test_overhead_http(benchmark: BenchmarkFixture, mockserver: MockServer) -> None:
|
||||
"""Per-request overhead of a crawl over HTTP.
|
||||
|
||||
The pages are small on purpose, so that the cost of parsing them stays
|
||||
negligible next to the cost of moving requests and responses through the
|
||||
engine, the middlewares and the download handler.
|
||||
"""
|
||||
query = urlencode({"total": PAGES, "show": LINKS_PER_PAGE, "order": "desc"})
|
||||
url = mockserver.url(f"/follow?{query}")
|
||||
settings = {"ITEM_PIPELINES": {_Pipeline: 100}, "LOG_ENABLED": False}
|
||||
|
||||
def run() -> None:
|
||||
crawler = crawl(_FollowSpider, settings, url=url)
|
||||
assert crawler.stats
|
||||
assert crawler.stats.get_value("item_scraped_count") == PAGES + 1
|
||||
|
||||
benchmark(run)
|
||||
|
|
@ -23,6 +23,18 @@ class TestCrawlCommand(TestProjectBase):
|
|||
_, _, stderr = self.crawl(code, proj_path, args=args)
|
||||
return stderr
|
||||
|
||||
def test_no_spider(self, proj_path: Path) -> None:
|
||||
returncode, out, _ = proc("crawl", cwd=proj_path)
|
||||
assert returncode == 2
|
||||
assert "Usage" in out
|
||||
|
||||
def test_multiple_spiders(self, proj_path: Path) -> None:
|
||||
returncode, _, err = proc("crawl", "myspider", "myspider2", cwd=proj_path)
|
||||
assert returncode == 2
|
||||
assert (
|
||||
"running 'scrapy crawl' with more than one spider is not supported" in err
|
||||
)
|
||||
|
||||
def test_no_output(self, proj_path: Path) -> None:
|
||||
spider_code = """
|
||||
import scrapy
|
||||
|
|
|
|||
|
|
@ -2,13 +2,24 @@ from __future__ import annotations
|
|||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.utils.bases.commands import TestProjectBase
|
||||
from tests.utils.cmdline import proc
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from pathlib import Path
|
||||
|
||||
from tests.mockserver.http import MockServer
|
||||
|
||||
|
||||
class TestFetchCommand:
|
||||
@pytest.mark.parametrize("args", [(), ("not-a-url",), ("a:b", "c:d")])
|
||||
def test_bad_arguments(self, args: tuple[str, ...]) -> None:
|
||||
returncode, out, _ = proc("fetch", *args)
|
||||
assert returncode == 2
|
||||
assert "Usage" in out
|
||||
|
||||
def test_output(self, mockserver: MockServer) -> None:
|
||||
_, out, _ = proc("fetch", mockserver.url("/text"))
|
||||
assert out.strip() == "Works"
|
||||
|
|
@ -36,3 +47,24 @@ class TestFetchCommand:
|
|||
"fetch", "-s", "TWISTED_REACTOR_ENABLED=False", mockserver.url("/text")
|
||||
)
|
||||
assert out.strip() == "Works"
|
||||
|
||||
|
||||
class TestFetchCommandWithSpider(TestProjectBase):
|
||||
@pytest.fixture(autouse=True)
|
||||
def create_files(self, proj_path: Path) -> None:
|
||||
(proj_path / self.project_name / "spiders" / "myspider.py").write_text(
|
||||
"""
|
||||
import scrapy
|
||||
|
||||
class MySpider(scrapy.Spider):
|
||||
name = "myspider"
|
||||
custom_settings = {"USER_AGENT": "myspider-user-agent"}
|
||||
""",
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
def test_spider(self, proj_path: Path, mockserver: MockServer) -> None:
|
||||
_, out, err = proc(
|
||||
"fetch", "--spider", "myspider", mockserver.url("/echo"), cwd=proj_path
|
||||
)
|
||||
assert "myspider-user-agent" in out, err
|
||||
|
|
|
|||
|
|
@ -64,6 +64,24 @@ class TestGenspiderCommand(TestProjectBase):
|
|||
assert call("genspider", "--dump=basic", cwd=proj_path) == 0
|
||||
assert call("genspider", "-d", "basic", cwd=proj_path) == 0
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"args",
|
||||
[("--dump=nonexistent",), ("-t", "nonexistent", "test_name", "test.com")],
|
||||
)
|
||||
def test_unknown_template(self, args: tuple[str, ...], proj_path: Path) -> None:
|
||||
returncode, out, err = proc("genspider", *args, cwd=proj_path)
|
||||
assert returncode == 0, err
|
||||
assert "Unable to find template: nonexistent" in out
|
||||
assert not (proj_path / self.project_name / "spiders" / "test_name.py").exists()
|
||||
|
||||
def test_name_not_starting_with_a_letter(self, proj_path: Path) -> None:
|
||||
"""The module name, unlike the spider name, is prefixed with a letter."""
|
||||
_, out, err = proc("genspider", "1st_spider", "test.com", cwd=proj_path)
|
||||
assert "Created spider '1st_spider'" in out, err
|
||||
spider = proj_path / self.project_name / "spiders" / "a1st_spider.py"
|
||||
assert spider.exists()
|
||||
assert find_in_file(spider, r'name\s*=\s*"1st_spider"') is not None
|
||||
|
||||
@pytest.mark.skipif(
|
||||
sys.platform == "win32", reason="requires a POSIX shell editor script"
|
||||
)
|
||||
|
|
@ -87,7 +105,8 @@ class TestGenspiderCommand(TestProjectBase):
|
|||
)
|
||||
|
||||
def test_same_name_as_project(self, proj_path: Path) -> None:
|
||||
assert call("genspider", self.project_name, cwd=proj_path) == 2
|
||||
_, out, err = proc("genspider", self.project_name, "test.com", cwd=proj_path)
|
||||
assert "Cannot create a spider with the same name as your project" in out, err
|
||||
assert not (
|
||||
proj_path / self.project_name / "spiders" / f"{self.project_name}.py"
|
||||
).exists()
|
||||
|
|
|
|||
|
|
@ -3,6 +3,7 @@ from __future__ import annotations
|
|||
import argparse
|
||||
import re
|
||||
from typing import TYPE_CHECKING
|
||||
from urllib.parse import urlparse
|
||||
|
||||
import pytest
|
||||
|
||||
|
|
@ -552,6 +553,130 @@ ITEM_PIPELINES = {{'{self.project_name}.pipelines.MyPipeline': 1}}
|
|||
content = '[\n{},\n{"foo": "bar"}\n]'
|
||||
assert file_path.read_text(encoding="utf-8") == content
|
||||
|
||||
@pytest.mark.parametrize("args", [(), ("not-a-url",), ("a:b", "c:d")])
|
||||
def test_bad_arguments(self, args: tuple[str, ...], proj_path: Path) -> None:
|
||||
returncode, out, _ = proc("parse", *args, cwd=proj_path)
|
||||
assert returncode == 2
|
||||
assert "Usage" in out
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("option", "message"),
|
||||
[
|
||||
("--meta", "Invalid -m/--meta value"),
|
||||
("-m", "Invalid -m/--meta value"),
|
||||
("--cbkwargs", "Invalid --cbkwargs value"),
|
||||
],
|
||||
)
|
||||
def test_invalid_json(
|
||||
self, option: str, message: str, proj_path: Path, mockserver: MockServer
|
||||
) -> None:
|
||||
returncode, _, err = proc(
|
||||
"parse",
|
||||
"--spider",
|
||||
self.spider_name,
|
||||
option,
|
||||
"{invalid",
|
||||
mockserver.url("/html"),
|
||||
cwd=proj_path,
|
||||
)
|
||||
assert returncode == 2
|
||||
assert message in err
|
||||
|
||||
def test_unknown_spider(self, proj_path: Path, mockserver: MockServer) -> None:
|
||||
returncode, _, err = proc(
|
||||
"parse",
|
||||
"--spider",
|
||||
"nonexistent",
|
||||
mockserver.url("/html"),
|
||||
cwd=proj_path,
|
||||
)
|
||||
assert returncode == 0, err
|
||||
assert "Unable to find spider: nonexistent" in err
|
||||
|
||||
def test_spider_found_by_url(self, proj_path: Path, mockserver: MockServer) -> None:
|
||||
"""Without --spider, the spider is chosen based on the URL."""
|
||||
url = mockserver.url("/html")
|
||||
# The spider name doubles as a domain of the spider, and it is matched
|
||||
# against the netloc of the URL, hence the port.
|
||||
(proj_path / self.project_name / "spiders" / "urlspider.py").write_text(
|
||||
f"""
|
||||
import scrapy
|
||||
|
||||
class UrlSpider(scrapy.Spider):
|
||||
name = "{urlparse(url).netloc}"
|
||||
|
||||
def parse(self, response):
|
||||
return [{{"found_by_url": True}}]
|
||||
""",
|
||||
encoding="utf-8",
|
||||
)
|
||||
returncode, out, err = proc("parse", url, cwd=proj_path)
|
||||
assert returncode == 0, err
|
||||
assert "Unable to find spider for" not in err
|
||||
assert "{'found_by_url': True}" in out
|
||||
|
||||
def test_legacy_item_processor(
|
||||
self, proj_path: Path, mockserver: MockServer
|
||||
) -> None:
|
||||
"""--pipelines supports an ITEM_PROCESSOR without process_item_async()."""
|
||||
(proj_path / self.project_name / "legacy.py").write_text(
|
||||
"""
|
||||
import logging
|
||||
|
||||
from twisted.internet.defer import succeed
|
||||
|
||||
|
||||
class LegacyItemProcessor:
|
||||
@classmethod
|
||||
def from_crawler(cls, crawler):
|
||||
return cls()
|
||||
|
||||
def open_spider(self, spider):
|
||||
return succeed(None)
|
||||
|
||||
def close_spider(self, spider):
|
||||
return succeed(None)
|
||||
|
||||
def process_item(self, item, spider):
|
||||
logging.info("Legacy item processor!")
|
||||
return succeed(item)
|
||||
""",
|
||||
encoding="utf-8",
|
||||
)
|
||||
_, _, stderr = proc(
|
||||
"parse",
|
||||
"--spider",
|
||||
self.spider_name,
|
||||
"--pipelines",
|
||||
"-c",
|
||||
"parse",
|
||||
"-s",
|
||||
f"ITEM_PROCESSOR={self.project_name}.legacy.LegacyItemProcessor",
|
||||
mockserver.url("/html"),
|
||||
cwd=proj_path,
|
||||
)
|
||||
assert "INFO: Legacy item processor!" in stderr
|
||||
|
||||
@pytest.mark.parametrize("verbose", [True, False])
|
||||
def test_no_items_no_links(
|
||||
self, verbose: bool, proj_path: Path, mockserver: MockServer
|
||||
) -> None:
|
||||
args = ["--verbose"] if verbose else []
|
||||
_, out, err = proc(
|
||||
"parse",
|
||||
"--spider",
|
||||
self.spider_name,
|
||||
"-c",
|
||||
"parse",
|
||||
"--noitems",
|
||||
"--nolinks",
|
||||
*args,
|
||||
mockserver.url("/html"),
|
||||
cwd=proj_path,
|
||||
)
|
||||
assert "# Scraped Items" not in out, err
|
||||
assert "# Requests" not in out
|
||||
|
||||
def test_parse_add_options(self):
|
||||
command = parse.Command()
|
||||
command.settings = Settings()
|
||||
|
|
|
|||
|
|
@ -136,6 +136,12 @@ class MySpider(scrapy.Spider):
|
|||
log = self.get_log(tmp_path, "from scrapy.spiders import Spider\n")
|
||||
assert "No spider found in file" in log
|
||||
|
||||
@pytest.mark.parametrize("args", [(), ("a.py", "b.py")])
|
||||
def test_runspider_bad_arguments(self, args: tuple[str, ...]) -> None:
|
||||
returncode, out, _ = proc("runspider", *args)
|
||||
assert returncode == 2
|
||||
assert "Usage" in out
|
||||
|
||||
def test_runspider_file_not_found(self) -> None:
|
||||
_, _, log = proc("runspider", "some_non_existent_file")
|
||||
assert "File not found: some_non_existent_file" in log
|
||||
|
|
|
|||
|
|
@ -18,6 +18,7 @@ from scrapy.shell import Shell, inspect_response
|
|||
from scrapy.utils.reactor import _asyncio_reactor_path
|
||||
from scrapy.utils.test import get_crawler
|
||||
from tests import NON_EXISTING_RESOLVABLE, tests_datadir
|
||||
from tests.utils.bases.commands import TestProjectBase
|
||||
from tests.utils.cmdline import proc
|
||||
from tests.utils.decorators import coroutine_test
|
||||
|
||||
|
|
@ -172,6 +173,33 @@ def _stop(p: PopenSpawn[str]) -> None:
|
|||
pipe.close()
|
||||
|
||||
|
||||
class TestShellCommandWithSpider(TestProjectBase):
|
||||
@pytest.fixture(autouse=True)
|
||||
def create_files(self, proj_path: Path) -> None:
|
||||
(proj_path / self.project_name / "spiders" / "myspider.py").write_text(
|
||||
"""
|
||||
import scrapy
|
||||
|
||||
class MySpider(scrapy.Spider):
|
||||
name = "myspider"
|
||||
""",
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
def test_spider(self, proj_path: Path, mockserver: MockServer) -> None:
|
||||
ret, out, err = proc(
|
||||
"shell",
|
||||
"--spider",
|
||||
"myspider",
|
||||
mockserver.url("/text"),
|
||||
"-c",
|
||||
"spider.name",
|
||||
cwd=proj_path,
|
||||
)
|
||||
assert ret == 0, err
|
||||
assert out.strip() == "myspider"
|
||||
|
||||
|
||||
class TestInteractiveShell:
|
||||
def test_fetch(self, mockserver: MockServer) -> None:
|
||||
args = (
|
||||
|
|
|
|||
|
|
@ -3,21 +3,27 @@ from __future__ import annotations
|
|||
import argparse
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
|
||||
import scrapy
|
||||
from scrapy.cmdline import _pop_command_name, execute
|
||||
from scrapy.commands import ScrapyCommand, ScrapyHelpFormatter, view
|
||||
from scrapy.commands import ScrapyCommand, ScrapyHelpFormatter
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
from scrapy.settings import Settings
|
||||
from scrapy.utils.reactor import _asyncio_reactor_path
|
||||
from tests.utils.bases.commands import TestProjectBase
|
||||
from tests.utils.cmdline import call, proc, write_recording_editor
|
||||
from tests.utils.cmdline import (
|
||||
call,
|
||||
proc,
|
||||
write_recording_browser,
|
||||
write_recording_editor,
|
||||
)
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from pathlib import Path
|
||||
from tests.mockserver.http import MockServer
|
||||
|
||||
|
||||
class EmptyCommand(ScrapyCommand):
|
||||
|
|
@ -107,6 +113,93 @@ class TestCommandSettings:
|
|||
)
|
||||
|
||||
|
||||
class TestGlobalOptions:
|
||||
"""Tests for the options that every command supports."""
|
||||
|
||||
spider_code = """
|
||||
import scrapy
|
||||
|
||||
class MySpider(scrapy.Spider):
|
||||
name = "myspider"
|
||||
|
||||
async def start(self):
|
||||
self.logger.debug("It works!")
|
||||
return
|
||||
yield
|
||||
"""
|
||||
|
||||
@pytest.fixture
|
||||
def spider_path(self, tmp_path: Path) -> Path:
|
||||
path = tmp_path / "myspider.py"
|
||||
path.write_text(self.spider_code, encoding="utf-8")
|
||||
return path
|
||||
|
||||
def test_invalid_set(self, spider_path: Path) -> None:
|
||||
returncode, _, err = proc("runspider", str(spider_path), "-s", "FOO")
|
||||
assert returncode == 2
|
||||
assert "Invalid -s value, use -s NAME=VALUE" in err
|
||||
|
||||
def test_invalid_spider_argument(self, spider_path: Path) -> None:
|
||||
returncode, _, err = proc("runspider", str(spider_path), "-a", "FOO")
|
||||
assert returncode == 2
|
||||
assert "Invalid -a value, use -a NAME=VALUE" in err
|
||||
|
||||
def test_logfile(self, tmp_path: Path, spider_path: Path) -> None:
|
||||
logfile = tmp_path / "scrapy.log"
|
||||
returncode, _, err = proc(
|
||||
"runspider", str(spider_path), "--logfile", str(logfile)
|
||||
)
|
||||
assert returncode == 0, err
|
||||
assert "It works!" in logfile.read_text(encoding="utf-8")
|
||||
assert "It works!" not in err
|
||||
|
||||
def test_loglevel(self, spider_path: Path) -> None:
|
||||
returncode, _, err = proc("runspider", str(spider_path), "--loglevel", "INFO")
|
||||
assert returncode == 0, err
|
||||
assert "It works!" not in err
|
||||
assert "Spider closed (finished)" in err
|
||||
|
||||
def test_nolog(self, spider_path: Path) -> None:
|
||||
returncode, _, err = proc("runspider", str(spider_path), "--nolog")
|
||||
assert returncode == 0, err
|
||||
assert not err
|
||||
|
||||
def test_pidfile(self, tmp_path: Path, spider_path: Path) -> None:
|
||||
pidfile = tmp_path / "scrapy.pid"
|
||||
returncode, _, err = proc(
|
||||
"runspider", str(spider_path), "--pidfile", str(pidfile)
|
||||
)
|
||||
assert returncode == 0, err
|
||||
assert pidfile.read_text(encoding="utf-8").strip().isdigit()
|
||||
|
||||
def test_pdb(self, spider_path: Path) -> None:
|
||||
returncode, _, err = proc("runspider", str(spider_path), "--pdb")
|
||||
assert returncode == 0, err
|
||||
assert "It works!" in err
|
||||
|
||||
|
||||
class TestSettingsCommand:
|
||||
@pytest.mark.parametrize(
|
||||
("option", "setting", "expected"),
|
||||
[
|
||||
("--get", "BOT_NAME", "scrapybot"),
|
||||
("--getbool", "COOKIES_ENABLED", "True"),
|
||||
("--getint", "CONCURRENT_REQUESTS", "16"),
|
||||
("--getfloat", "DOWNLOAD_DELAY", "0.0"),
|
||||
("--getlist", "SPIDER_MODULES", "[]"),
|
||||
],
|
||||
)
|
||||
def test_get(self, option: str, setting: str, expected: str) -> None:
|
||||
returncode, out, err = proc("settings", option, setting)
|
||||
assert returncode == 0, err
|
||||
assert out.startswith(expected)
|
||||
|
||||
def test_no_option(self) -> None:
|
||||
returncode, out, err = proc("settings")
|
||||
assert returncode == 0, err
|
||||
assert not out
|
||||
|
||||
|
||||
class TestCommandCrawlerProcess(TestProjectBase):
|
||||
"""Test that the command uses the expected kind of *CrawlerProcess
|
||||
and produces expected errors when needed."""
|
||||
|
|
@ -577,18 +670,31 @@ class TestBenchCommand:
|
|||
|
||||
|
||||
class TestViewCommand:
|
||||
def test_methods(self) -> None:
|
||||
command = view.Command()
|
||||
command.settings = Settings()
|
||||
parser = argparse.ArgumentParser(
|
||||
prog="scrapy",
|
||||
prefix_chars="-",
|
||||
formatter_class=ScrapyHelpFormatter,
|
||||
conflict_handler="resolve",
|
||||
@pytest.mark.skipif(
|
||||
sys.platform == "win32", reason="requires a POSIX shell browser script"
|
||||
)
|
||||
def test_view(
|
||||
self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, mockserver: MockServer
|
||||
) -> None:
|
||||
opened = tmp_path / "opened.txt"
|
||||
browser = tmp_path / "fake-browser.sh"
|
||||
write_recording_browser(browser, opened)
|
||||
monkeypatch.setenv("BROWSER", str(browser))
|
||||
|
||||
returncode, _, err = proc("view", mockserver.url("/html"), cwd=tmp_path)
|
||||
|
||||
assert returncode == 0, err
|
||||
url = opened.read_text(encoding="utf-8")
|
||||
assert url.startswith("file://")
|
||||
body = Path(url.removeprefix("file://")).read_text(encoding="utf-8")
|
||||
assert "<p class='one'>Works</p>" in body
|
||||
|
||||
def test_non_text_response(self, mockserver: MockServer) -> None:
|
||||
returncode, _, err = proc(
|
||||
"view", mockserver.url("/static/files/images/scrapy.png")
|
||||
)
|
||||
command.add_options(parser)
|
||||
assert command.short_desc() == "Open URL in browser, as seen by Scrapy"
|
||||
assert "URL using the Scrapy downloader and show its" in command.long_desc()
|
||||
assert returncode == 0, err
|
||||
assert "Cannot view a non-text response." in err
|
||||
|
||||
|
||||
class TestEditCommand(TestProjectBase):
|
||||
|
|
@ -615,6 +721,11 @@ class TestEditCommand(TestProjectBase):
|
|||
assert returncode == 1
|
||||
assert "Spider not found: nonexistent" in err
|
||||
|
||||
def test_edit_no_spider(self, proj_path: Path) -> None:
|
||||
returncode, out, _ = proc("edit", cwd=proj_path)
|
||||
assert returncode == 2
|
||||
assert "Usage" in out
|
||||
|
||||
|
||||
class TestHelpMessage(TestProjectBase):
|
||||
@pytest.mark.parametrize(
|
||||
|
|
|
|||
|
|
@ -8,6 +8,7 @@ import pytest
|
|||
from twisted.internet.defer import Deferred, DeferredList
|
||||
from twisted.python import failure
|
||||
|
||||
from scrapy import signals
|
||||
from scrapy.downloadermiddlewares.robotstxt import RobotsTxtMiddleware
|
||||
from scrapy.exceptions import CannotResolveHostError, IgnoreRequest, NotConfigured
|
||||
from scrapy.http import Request, Response, TextResponse
|
||||
|
|
@ -27,6 +28,7 @@ class TestRobotsTxtMiddleware:
|
|||
self.crawler: mock.MagicMock = mock.MagicMock()
|
||||
self.crawler.settings = Settings()
|
||||
self.crawler.engine.download_async = mock.AsyncMock()
|
||||
self.crawler.signals.send_catch_log_async = mock.AsyncMock(return_value=[])
|
||||
|
||||
def teardown_method(self):
|
||||
del self.crawler
|
||||
|
|
@ -74,6 +76,21 @@ Disallow: /some/randome/page.html
|
|||
Request("http://site.local/wiki/Käyttäjä:"), middleware
|
||||
)
|
||||
|
||||
@coroutine_test
|
||||
async def test_robotstxt_emits_robots_parsed_signal(self):
|
||||
crawler = self._get_successful_crawler()
|
||||
middleware = RobotsTxtMiddleware(crawler)
|
||||
request = Request("http://site.local/allowed")
|
||||
await self.assertNotIgnored(request, middleware)
|
||||
calls = [
|
||||
kwargs
|
||||
for _, kwargs in crawler.signals.send_catch_log_async.call_args_list
|
||||
if kwargs.get("signal") is signals.robots_parsed
|
||||
]
|
||||
assert len(calls) == 1
|
||||
assert calls[0]["request"] is request
|
||||
assert calls[0]["robotparser"] is not None
|
||||
|
||||
@coroutine_test
|
||||
async def test_robotstxt_multiple_reqs(self) -> None:
|
||||
middleware = RobotsTxtMiddleware(self._get_successful_crawler())
|
||||
|
|
|
|||
|
|
@ -4,6 +4,7 @@ from scrapy.robotstxt import (
|
|||
ProtegoRobotParser,
|
||||
PythonRobotParser,
|
||||
RerpRobotParser,
|
||||
RobotParser,
|
||||
decode_robotstxt,
|
||||
)
|
||||
from scrapy.utils._deps_compat import STDLIB_IMPROVED_ROBOTFILEPARSER
|
||||
|
|
@ -78,6 +79,16 @@ class BaseRobotParserTest:
|
|||
assert rp.allowed("https://site.local/index.html", "*")
|
||||
assert rp.allowed("https://site.local/disallowed", "*")
|
||||
|
||||
def test_crawl_delay(self):
|
||||
robotstxt_body = b"User-agent: *\nDisallow: /private\nCrawl-delay: 10\n"
|
||||
rp = self.parser_cls.from_crawler(crawler=None, robotstxt_body=robotstxt_body)
|
||||
assert rp.crawl_delay("*") == 10.0
|
||||
|
||||
def test_crawl_delay_unset(self):
|
||||
robotstxt_body = b"User-agent: *\nDisallow: /private\n"
|
||||
rp = self.parser_cls.from_crawler(crawler=None, robotstxt_body=robotstxt_body)
|
||||
assert rp.crawl_delay("*") is None
|
||||
|
||||
def test_unicode_url_and_useragent(self):
|
||||
robotstxt_robotstxt_body = """
|
||||
User-Agent: *
|
||||
|
|
@ -102,6 +113,22 @@ class BaseRobotParserTest:
|
|||
assert not rp.allowed("https://site.local/some/randome/page.html", "UnicödeBöt")
|
||||
|
||||
|
||||
class TestRobotParser:
|
||||
def test_crawl_delay_unsupported(self):
|
||||
class AllowAllRobotParser(RobotParser):
|
||||
@classmethod
|
||||
def from_crawler(cls, crawler, robotstxt_body):
|
||||
return cls()
|
||||
|
||||
def allowed(self, url, user_agent):
|
||||
return True
|
||||
|
||||
rp = AllowAllRobotParser.from_crawler(
|
||||
crawler=None, robotstxt_body=b"User-agent: *\nCrawl-delay: 10\n"
|
||||
)
|
||||
assert rp.crawl_delay("*") is None
|
||||
|
||||
|
||||
class TestDecodeRobotsTxt:
|
||||
def test_native_string_conversion(self):
|
||||
robotstxt_body = b"User-agent: *\nDisallow: /\n"
|
||||
|
|
|
|||
|
|
@ -4,7 +4,7 @@ from importlib.util import find_spec
|
|||
|
||||
import pytest
|
||||
|
||||
from scrapy.utils.console import get_shell_embed_func
|
||||
from scrapy.utils.console import get_shell_embed_func, start_python_console
|
||||
|
||||
|
||||
def test_get_shell_embed_func():
|
||||
|
|
@ -59,3 +59,20 @@ def test_get_shell_embed_func_default():
|
|||
else:
|
||||
expected = "_embed_standard_shell"
|
||||
assert shell.__name__ == expected
|
||||
|
||||
|
||||
def test_start_python_console_exit(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
def embed(namespace: dict[str, object], banner: str) -> None:
|
||||
raise SystemExit
|
||||
|
||||
monkeypatch.setattr(
|
||||
"scrapy.utils.console.get_shell_embed_func", lambda shells: embed
|
||||
)
|
||||
start_python_console()
|
||||
|
||||
|
||||
def test_start_python_console_no_shell(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setattr(
|
||||
"scrapy.utils.console.get_shell_embed_func", lambda shells: None
|
||||
)
|
||||
start_python_console()
|
||||
|
|
|
|||
|
|
@ -46,3 +46,16 @@ def write_recording_editor(editor: Path) -> None:
|
|||
open (its last argument) into the file given as its first argument."""
|
||||
editor.write_text('#!/bin/sh\nprintf "%s" "$2" > "$1"\n', encoding="utf-8")
|
||||
editor.chmod(0o755)
|
||||
|
||||
|
||||
def write_recording_browser(browser: Path, recorded: Path) -> None:
|
||||
"""Create an executable browser script that writes the URL it is asked to
|
||||
open into *recorded*.
|
||||
|
||||
``webbrowser`` only passes the URL to the command from the ``BROWSER``
|
||||
environment variable, hence the hardcoded output path.
|
||||
"""
|
||||
browser.write_text(
|
||||
f'#!/bin/sh\nprintf "%s" "$1" > "{recorded}"\n', encoding="utf-8"
|
||||
)
|
||||
browser.chmod(0o755)
|
||||
|
|
|
|||
20
tox.ini
20
tox.ini
|
|
@ -5,7 +5,7 @@
|
|||
|
||||
[tox]
|
||||
requires =
|
||||
sphinx-scrapy[tox] @ git+https://github.com/scrapy/sphinx-scrapy.git@0.8.8
|
||||
sphinx-scrapy[tox] @ git+https://github.com/scrapy/sphinx-scrapy.git@0.8.9
|
||||
envlist =
|
||||
pre-commit
|
||||
pylint
|
||||
|
|
@ -30,6 +30,7 @@ envlist =
|
|||
botocore
|
||||
pypy3
|
||||
pypy3-extra-deps
|
||||
benchmark
|
||||
minversion = 1.7.0
|
||||
|
||||
[test-requirements]
|
||||
|
|
@ -320,3 +321,20 @@ setenv =
|
|||
{[min]setenv}
|
||||
commands =
|
||||
pytest {posargs:--cov-config=pyproject.toml --cov=scrapy --cov-report=xml --cov-report= tests --junitxml=min-botocore.junit.xml -o junit_family=legacy} -m requires_botocore
|
||||
|
||||
|
||||
# CPU benchmarks, tracked on CodSpeed.
|
||||
#
|
||||
# pytest-twisted is left out on purpose: benchmarked code must be callable
|
||||
# synchronously, so tests/benchmarks drives the reactor itself.
|
||||
|
||||
[testenv:benchmark]
|
||||
basepython = python3.14
|
||||
deps =
|
||||
pytest >= 8.4.1
|
||||
pytest-codspeed
|
||||
passenv =
|
||||
*codspeed*
|
||||
*ci*
|
||||
commands =
|
||||
pytest {posargs:tests/benchmarks} --codspeed --codspeed-mode=simulation
|
||||
|
|
|
|||
Loading…
Reference in New Issue