mirror of https://github.com/scrapy/scrapy.git
Merge remote-tracking branch 'upstream/master' into periodic_log_2
This commit is contained in:
commit
1f03cb1419
|
|
@ -1,5 +1,5 @@
|
|||
[bumpversion]
|
||||
current_version = 2.8.0
|
||||
current_version = 2.10.0
|
||||
commit = True
|
||||
tag = True
|
||||
tag_name = {new_version}
|
||||
|
|
|
|||
|
|
@ -7,7 +7,7 @@ jobs:
|
|||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
python-version: ["3.7", "3.8", "3.9", "3.10", "3.11"]
|
||||
python-version: ["3.8", "3.9", "3.10", "3.11"]
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
|
|
|
|||
|
|
@ -8,9 +8,6 @@ jobs:
|
|||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- python-version: 3.8
|
||||
env:
|
||||
TOXENV: py
|
||||
- python-version: 3.9
|
||||
env:
|
||||
TOXENV: py
|
||||
|
|
@ -28,19 +25,38 @@ jobs:
|
|||
TOXENV: pypy3
|
||||
|
||||
# pinned deps
|
||||
- python-version: 3.7.13
|
||||
- python-version: 3.8.17
|
||||
env:
|
||||
TOXENV: pinned
|
||||
- python-version: 3.7.13
|
||||
- python-version: 3.8.17
|
||||
env:
|
||||
TOXENV: asyncio-pinned
|
||||
- python-version: pypy3.7
|
||||
- python-version: pypy3.8
|
||||
env:
|
||||
TOXENV: pypy3-pinned
|
||||
- python-version: 3.8.17
|
||||
env:
|
||||
TOXENV: extra-deps-pinned
|
||||
- python-version: 3.8.17
|
||||
env:
|
||||
TOXENV: botocore-pinned
|
||||
|
||||
- python-version: "3.11"
|
||||
env:
|
||||
TOXENV: extra-deps
|
||||
- python-version: "3.11"
|
||||
env:
|
||||
TOXENV: botocore
|
||||
|
||||
- python-version: "3.12.0-rc.1"
|
||||
env:
|
||||
TOXENV: py
|
||||
- python-version: "3.12.0-rc.1"
|
||||
env:
|
||||
TOXENV: asyncio
|
||||
- python-version: "3.12.0-rc.1"
|
||||
env:
|
||||
TOXENV: extra-deps
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
|
|
@ -51,7 +67,7 @@ jobs:
|
|||
python-version: ${{ matrix.python-version }}
|
||||
|
||||
- name: Install system libraries
|
||||
if: matrix.python-version == 'pypy3.9' || contains(matrix.env.TOXENV, 'pinned')
|
||||
if: matrix.python-version == 'pypy3.9' || contains(matrix.env.TOXENV, 'pinned') || contains(matrix.python-version, '3.12.0')
|
||||
run: |
|
||||
sudo apt-get update
|
||||
sudo apt-get install libxml2-dev libxslt-dev
|
||||
|
|
|
|||
|
|
@ -8,12 +8,9 @@ jobs:
|
|||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- python-version: 3.7
|
||||
env:
|
||||
TOXENV: windows-pinned
|
||||
- python-version: 3.8
|
||||
env:
|
||||
TOXENV: py
|
||||
TOXENV: windows-pinned
|
||||
- python-version: 3.9
|
||||
env:
|
||||
TOXENV: py
|
||||
|
|
@ -23,13 +20,12 @@ jobs:
|
|||
- python-version: "3.10"
|
||||
env:
|
||||
TOXENV: asyncio
|
||||
# no binary package for lxml for 3.11 yet
|
||||
# - python-version: "3.11"
|
||||
# env:
|
||||
# TOXENV: py
|
||||
# - python-version: "3.11"
|
||||
# env:
|
||||
# TOXENV: asyncio
|
||||
- python-version: "3.11"
|
||||
env:
|
||||
TOXENV: py
|
||||
- python-version: "3.11"
|
||||
env:
|
||||
TOXENV: asyncio
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
|
|
|
|||
|
|
@ -5,20 +5,20 @@ repos:
|
|||
- id: bandit
|
||||
args: [-r, -c, .bandit.yml]
|
||||
- repo: https://github.com/PyCQA/flake8
|
||||
rev: 5.0.4 # 6.0.0 drops Python 3.7 support
|
||||
rev: 6.1.0
|
||||
hooks:
|
||||
- id: flake8
|
||||
- repo: https://github.com/psf/black.git
|
||||
rev: 23.3.0
|
||||
rev: 23.7.0
|
||||
hooks:
|
||||
- id: black
|
||||
- repo: https://github.com/pycqa/isort
|
||||
rev: 5.11.5 # 5.12 drops Python 3.7 support
|
||||
rev: 5.12.0
|
||||
hooks:
|
||||
- id: isort
|
||||
- repo: https://github.com/adamchainz/blacken-docs
|
||||
rev: 1.13.0
|
||||
rev: 1.15.0
|
||||
hooks:
|
||||
- id: blacken-docs
|
||||
additional_dependencies:
|
||||
- black==23.3.0
|
||||
- black==23.7.0
|
||||
|
|
|
|||
|
|
@ -58,7 +58,7 @@ including a list of features.
|
|||
Requirements
|
||||
============
|
||||
|
||||
* Python 3.7+
|
||||
* Python 3.8+
|
||||
* Works on Linux, Windows, macOS, BSD
|
||||
|
||||
Install
|
||||
|
|
|
|||
22
conftest.py
22
conftest.py
|
|
@ -1,6 +1,10 @@
|
|||
import platform
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
from twisted import version as twisted_version
|
||||
from twisted.python.versions import Version
|
||||
from twisted.web.http import H2_ENABLED
|
||||
|
||||
from scrapy.utils.reactor import install_reactor
|
||||
|
|
@ -14,6 +18,10 @@ def _py_files(folder):
|
|||
collect_ignore = [
|
||||
# not a test, but looks like a test
|
||||
"scrapy/utils/testsite.py",
|
||||
"tests/ftpserver.py",
|
||||
"tests/mockserver.py",
|
||||
"tests/pipelines.py",
|
||||
"tests/spiders.py",
|
||||
# contains scripts to be run by tests/test_crawler.py::CrawlerProcessSubprocess
|
||||
*_py_files("tests/CrawlerProcess"),
|
||||
# contains scripts to be run by tests/test_crawler.py::CrawlerRunnerSubprocess
|
||||
|
|
@ -73,6 +81,20 @@ def only_not_asyncio(request, reactor_pytest):
|
|||
pytest.skip("This test is only run without --reactor=asyncio")
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def requires_uvloop(request):
|
||||
if not request.node.get_closest_marker("requires_uvloop"):
|
||||
return
|
||||
if sys.implementation.name == "pypy":
|
||||
pytest.skip("uvloop does not support pypy properly")
|
||||
if platform.system() == "Windows":
|
||||
pytest.skip("uvloop does not support Windows")
|
||||
if twisted_version == Version("twisted", 21, 2, 0):
|
||||
pytest.skip("https://twistedmatrix.com/trac/ticket/10106")
|
||||
if sys.version_info >= (3, 12):
|
||||
pytest.skip("uvloop doesn't support Python 3.12 yet")
|
||||
|
||||
|
||||
def pytest_configure(config):
|
||||
if config.getoption("--reactor") == "asyncio":
|
||||
install_reactor("twisted.internet.asyncioreactor.AsyncioSelectorReactor")
|
||||
|
|
|
|||
|
|
@ -265,15 +265,15 @@ To run a specific test (say ``tests/test_loader.py``) use:
|
|||
|
||||
To run the tests on a specific :doc:`tox <tox:index>` environment, use
|
||||
``-e <name>`` with an environment name from ``tox.ini``. For example, to run
|
||||
the tests with Python 3.7 use::
|
||||
the tests with Python 3.10 use::
|
||||
|
||||
tox -e py37
|
||||
tox -e py310
|
||||
|
||||
You can also specify a comma-separated list of environments, and use :ref:`tox’s
|
||||
parallel mode <tox:parallel_mode>` to run the tests on multiple environments in
|
||||
parallel::
|
||||
|
||||
tox -e py37,py38 -p auto
|
||||
tox -e py39,py310 -p auto
|
||||
|
||||
To pass command-line options to :doc:`pytest <pytest:index>`, add them after
|
||||
``--`` in your call to :doc:`tox <tox:index>`. Using ``--`` overrides the
|
||||
|
|
@ -283,9 +283,9 @@ default positional arguments (``scrapy tests``) after ``--`` as well::
|
|||
tox -- scrapy tests -x # stop after first failure
|
||||
|
||||
You can also use the `pytest-xdist`_ plugin. For example, to run all tests on
|
||||
the Python 3.7 :doc:`tox <tox:index>` environment using all your CPU cores::
|
||||
the Python 3.10 :doc:`tox <tox:index>` environment using all your CPU cores::
|
||||
|
||||
tox -e py37 -- scrapy tests -n auto
|
||||
tox -e py310 -- scrapy tests -n auto
|
||||
|
||||
To see coverage report install :doc:`coverage <coverage:index>`
|
||||
(``pip install coverage``) and run:
|
||||
|
|
@ -322,4 +322,4 @@ And their unit-tests are in::
|
|||
.. _pytest-xdist: https://github.com/pytest-dev/pytest-xdist
|
||||
.. _good first issues: https://github.com/scrapy/scrapy/issues?q=is%3Aissue+is%3Aopen+label%3A%22good+first+issue%22
|
||||
.. _help wanted issues: https://github.com/scrapy/scrapy/issues?q=is%3Aissue+is%3Aopen+label%3A%22help+wanted%22
|
||||
.. _test coverage: https://app.codecov.io/gh/scrapy/scrapy
|
||||
.. _test coverage: https://app.codecov.io/gh/scrapy/scrapy
|
||||
|
|
|
|||
|
|
@ -222,6 +222,7 @@ Extending Scrapy
|
|||
:hidden:
|
||||
|
||||
topics/architecture
|
||||
topics/addons
|
||||
topics/downloader-middleware
|
||||
topics/spider-middleware
|
||||
topics/extensions
|
||||
|
|
@ -235,6 +236,9 @@ Extending Scrapy
|
|||
:doc:`topics/architecture`
|
||||
Understand the Scrapy architecture.
|
||||
|
||||
:doc:`topics/addons`
|
||||
Enable and configure third-party extensions.
|
||||
|
||||
:doc:`topics/downloader-middleware`
|
||||
Customize how pages get requested and downloaded.
|
||||
|
||||
|
|
|
|||
|
|
@ -9,7 +9,7 @@ Installation guide
|
|||
Supported Python versions
|
||||
=========================
|
||||
|
||||
Scrapy requires Python 3.7+, either the CPython implementation (default) or
|
||||
Scrapy requires Python 3.8+, either the CPython implementation (default) or
|
||||
the PyPy implementation (see :ref:`python:implementations`).
|
||||
|
||||
.. _intro-install-scrapy:
|
||||
|
|
|
|||
297
docs/news.rst
297
docs/news.rst
|
|
@ -3,6 +3,299 @@
|
|||
Release notes
|
||||
=============
|
||||
|
||||
.. _release-2.10.0:
|
||||
|
||||
Scrapy 2.10.0 (2023-08-04)
|
||||
--------------------------
|
||||
|
||||
Highlights:
|
||||
|
||||
- Added Python 3.12 support, dropped Python 3.7 support.
|
||||
|
||||
- The new add-ons framework simplifies configuring 3rd-party components that
|
||||
support it.
|
||||
|
||||
- Exceptions to retry can now be configured.
|
||||
|
||||
- Many fixes and improvements for feed exports.
|
||||
|
||||
Modified requirements
|
||||
~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
- Dropped support for Python 3.7. (:issue:`5953`)
|
||||
|
||||
- Added support for the upcoming Python 3.12. (:issue:`5984`)
|
||||
|
||||
- Minimum versions increased for these dependencies:
|
||||
|
||||
- lxml_: 4.3.0 → 4.4.1
|
||||
|
||||
- cryptography_: 3.4.6 → 36.0.0
|
||||
|
||||
- ``pkg_resources`` is no longer used. (:issue:`5956`, :issue:`5958`)
|
||||
|
||||
- boto3_ is now recommended instead of botocore_ for exporting to S3.
|
||||
(:issue:`5833`).
|
||||
|
||||
Backward-incompatible changes
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
- The value of the :setting:`FEED_STORE_EMPTY` setting is now ``True``
|
||||
instead of ``False``. In earlier Scrapy versions empty files were created
|
||||
even when this setting was ``False`` (which was a bug that is now fixed),
|
||||
so the new default should keep the old behavior. (:issue:`872`,
|
||||
:issue:`5847`)
|
||||
|
||||
Deprecation removals
|
||||
~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
- When a function is assigned to the :setting:`FEED_URI_PARAMS` setting,
|
||||
returning ``None`` or modifying the ``params`` input parameter, deprecated
|
||||
in Scrapy 2.6, is no longer supported. (:issue:`5994`, :issue:`5996`)
|
||||
|
||||
- The ``scrapy.utils.reqser`` module, deprecated in Scrapy 2.6, is removed.
|
||||
(:issue:`5994`, :issue:`5996`)
|
||||
|
||||
- The ``scrapy.squeues`` classes ``PickleFifoDiskQueueNonRequest``,
|
||||
``PickleLifoDiskQueueNonRequest``, ``MarshalFifoDiskQueueNonRequest``,
|
||||
and ``MarshalLifoDiskQueueNonRequest``, deprecated in
|
||||
Scrapy 2.6, are removed. (:issue:`5994`, :issue:`5996`)
|
||||
|
||||
- The property ``open_spiders`` and the methods ``has_capacity`` and
|
||||
``schedule`` of :class:`scrapy.core.engine.ExecutionEngine`,
|
||||
deprecated in Scrapy 2.6, are removed. (:issue:`5994`, :issue:`5998`)
|
||||
|
||||
- Passing a ``spider`` argument to the
|
||||
:meth:`~scrapy.core.engine.ExecutionEngine.spider_is_idle`,
|
||||
:meth:`~scrapy.core.engine.ExecutionEngine.crawl` and
|
||||
:meth:`~scrapy.core.engine.ExecutionEngine.download` methods of
|
||||
:class:`scrapy.core.engine.ExecutionEngine`, deprecated in Scrapy 2.6, is
|
||||
no longer supported. (:issue:`5994`, :issue:`5998`)
|
||||
|
||||
Deprecations
|
||||
~~~~~~~~~~~~
|
||||
|
||||
- :class:`scrapy.utils.datatypes.CaselessDict` is deprecated, use
|
||||
:class:`scrapy.utils.datatypes.CaseInsensitiveDict` instead.
|
||||
(:issue:`5146`)
|
||||
|
||||
- Passing the ``custom`` argument to
|
||||
:func:`scrapy.utils.conf.build_component_list` is deprecated, it was used
|
||||
in the past to merge ``FOO`` and ``FOO_BASE`` setting values but now Scrapy
|
||||
uses :func:`scrapy.settings.BaseSettings.getwithbase` to do the same.
|
||||
Code that uses this argument and cannot be switched to ``getwithbase()``
|
||||
can be switched to merging the values explicitly. (:issue:`5726`,
|
||||
:issue:`5923`)
|
||||
|
||||
New features
|
||||
~~~~~~~~~~~~
|
||||
|
||||
- Added support for :ref:`Scrapy add-ons <topics-addons>`. (:issue:`5950`)
|
||||
|
||||
- Added the :setting:`RETRY_EXCEPTIONS` setting that configures which
|
||||
exceptions will be retried by
|
||||
:class:`~scrapy.downloadermiddlewares.retry.RetryMiddleware`.
|
||||
(:issue:`2701`, :issue:`5929`)
|
||||
|
||||
- Added the possiiblity to close the spider if no items were produced in the
|
||||
specified time, configured by :setting:`CLOSESPIDER_TIMEOUT_NO_ITEM`.
|
||||
(:issue:`5979`)
|
||||
|
||||
- Added support for the :setting:`AWS_REGION_NAME` setting to feed exports.
|
||||
(:issue:`5980`)
|
||||
|
||||
- Added support for using :class:`pathlib.Path` objects that refer to
|
||||
absolute Windows paths in the :setting:`FEEDS` setting. (:issue:`5939`)
|
||||
|
||||
Bug fixes
|
||||
~~~~~~~~~
|
||||
|
||||
- Fixed creating empty feeds even with ``FEED_STORE_EMPTY=False``.
|
||||
(:issue:`872`, :issue:`5847`)
|
||||
|
||||
- Fixed using absolute Windows paths when specifying output files.
|
||||
(:issue:`5969`, :issue:`5971`)
|
||||
|
||||
- Fixed problems with uploading large files to S3 by switching to multipart
|
||||
uploads (requires boto3_). (:issue:`960`, :issue:`5735`, :issue:`5833`)
|
||||
|
||||
- Fixed the JSON exporter writing extra commas when some exceptions occur.
|
||||
(:issue:`3090`, :issue:`5952`)
|
||||
|
||||
- Fixed the "read of closed file" error in the CSV exporter. (:issue:`5043`,
|
||||
:issue:`5705`)
|
||||
|
||||
- Fixed an error when a component added by the class object throws
|
||||
:exc:`~scrapy.exceptions.NotConfigured` with a message. (:issue:`5950`,
|
||||
:issue:`5992`)
|
||||
|
||||
- Added the missing :meth:`scrapy.settings.BaseSettings.pop` method.
|
||||
(:issue:`5959`, :issue:`5960`, :issue:`5963`)
|
||||
|
||||
- Added :class:`~scrapy.utils.datatypes.CaseInsensitiveDict` as a replacement
|
||||
for :class:`~scrapy.utils.datatypes.CaselessDict` that fixes some API
|
||||
inconsistencies. (:issue:`5146`)
|
||||
|
||||
Documentation
|
||||
~~~~~~~~~~~~~
|
||||
|
||||
- Documented :meth:`scrapy.Spider.update_settings`. (:issue:`5745`,
|
||||
:issue:`5846`)
|
||||
|
||||
- Documented possible problems with early Twisted reactor installation and
|
||||
their solutions. (:issue:`5981`, :issue:`6000`)
|
||||
|
||||
- Added examples of making additional requests in callbacks. (:issue:`5927`)
|
||||
|
||||
- Improved the feed export docs. (:issue:`5579`, :issue:`5931`)
|
||||
|
||||
- Clarified the docs about request objects on redirection. (:issue:`5707`,
|
||||
:issue:`5937`)
|
||||
|
||||
Quality assurance
|
||||
~~~~~~~~~~~~~~~~~
|
||||
|
||||
- Added support for running tests against the installed Scrapy version.
|
||||
(:issue:`4914`, :issue:`5949`)
|
||||
|
||||
- Extended typing hints. (:issue:`5925`, :issue:`5977`)
|
||||
|
||||
- Fixed the ``test_utils_asyncio.AsyncioTest.test_set_asyncio_event_loop``
|
||||
test. (:issue:`5951`)
|
||||
|
||||
- Fixed the ``test_feedexport.BatchDeliveriesTest.test_batch_path_differ``
|
||||
test on Windows. (:issue:`5847`)
|
||||
|
||||
- Enabled CI runs for Python 3.11 on Windows. (:issue:`5999`)
|
||||
|
||||
- Simplified skipping tests that depend on ``uvloop``. (:issue:`5984`)
|
||||
|
||||
- Fixed the ``extra-deps-pinned`` tox env. (:issue:`5948`)
|
||||
|
||||
- Implemented cleanups. (:issue:`5965`, :issue:`5986`)
|
||||
|
||||
.. _release-2.9.0:
|
||||
|
||||
Scrapy 2.9.0 (2023-05-08)
|
||||
-------------------------
|
||||
|
||||
Highlights:
|
||||
|
||||
- Per-domain download settings.
|
||||
- Compatibility with new cryptography_ and new parsel_.
|
||||
- JMESPath selectors from the new parsel_.
|
||||
- Bug fixes.
|
||||
|
||||
Deprecations
|
||||
~~~~~~~~~~~~
|
||||
|
||||
- :class:`scrapy.extensions.feedexport._FeedSlot` is renamed to
|
||||
:class:`scrapy.extensions.feedexport.FeedSlot` and the old name is
|
||||
deprecated. (:issue:`5876`)
|
||||
|
||||
New features
|
||||
~~~~~~~~~~~~
|
||||
|
||||
- Settings correponding to :setting:`DOWNLOAD_DELAY`,
|
||||
:setting:`CONCURRENT_REQUESTS_PER_DOMAIN` and
|
||||
:setting:`RANDOMIZE_DOWNLOAD_DELAY` can now be set on a per-domain basis
|
||||
via the new :setting:`DOWNLOAD_SLOTS` setting. (:issue:`5328`)
|
||||
|
||||
- Added :meth:`TextResponse.jmespath`, a shortcut for JMESPath selectors
|
||||
available since parsel_ 1.8.1. (:issue:`5894`, :issue:`5915`)
|
||||
|
||||
- Added :signal:`feed_slot_closed` and :signal:`feed_exporter_closed`
|
||||
signals. (:issue:`5876`)
|
||||
|
||||
- Added :func:`scrapy.utils.request.request_to_curl`, a function to produce a
|
||||
curl command from a :class:`~scrapy.Request` object. (:issue:`5892`)
|
||||
|
||||
- Values of :setting:`FILES_STORE` and :setting:`IMAGES_STORE` can now be
|
||||
:class:`pathlib.Path` instances. (:issue:`5801`)
|
||||
|
||||
Bug fixes
|
||||
~~~~~~~~~
|
||||
|
||||
- Fixed a warning with Parsel 1.8.1+. (:issue:`5903`, :issue:`5918`)
|
||||
|
||||
- Fixed an error when using feed postprocessing with S3 storage.
|
||||
(:issue:`5500`, :issue:`5581`)
|
||||
|
||||
- Added the missing :meth:`scrapy.settings.BaseSettings.setdefault` method.
|
||||
(:issue:`5811`, :issue:`5821`)
|
||||
|
||||
- Fixed an error when using cryptography_ 40.0.0+ and
|
||||
:setting:`DOWNLOADER_CLIENT_TLS_VERBOSE_LOGGING` is enabled.
|
||||
(:issue:`5857`, :issue:`5858`)
|
||||
|
||||
- The checksums returned by :class:`~scrapy.pipelines.files.FilesPipeline`
|
||||
for files on Google Cloud Storage are no longer Base64-encoded.
|
||||
(:issue:`5874`, :issue:`5891`)
|
||||
|
||||
- :func:`scrapy.utils.request.request_from_curl` now supports $-prefixed
|
||||
string values for the curl ``--data-raw`` argument, which are produced by
|
||||
browsers for data that includes certain symbols. (:issue:`5899`,
|
||||
:issue:`5901`)
|
||||
|
||||
- The :command:`parse` command now also works with async generator callbacks.
|
||||
(:issue:`5819`, :issue:`5824`)
|
||||
|
||||
- The :command:`genspider` command now properly works with HTTPS URLs.
|
||||
(:issue:`3553`, :issue:`5808`)
|
||||
|
||||
- Improved handling of asyncio loops. (:issue:`5831`, :issue:`5832`)
|
||||
|
||||
- :class:`LinkExtractor <scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor>`
|
||||
now skips certain malformed URLs instead of raising an exception.
|
||||
(:issue:`5881`)
|
||||
|
||||
- :func:`scrapy.utils.python.get_func_args` now supports more types of
|
||||
callables. (:issue:`5872`, :issue:`5885`)
|
||||
|
||||
- Fixed an error when processing non-UTF8 values of ``Content-Type`` headers.
|
||||
(:issue:`5914`, :issue:`5917`)
|
||||
|
||||
- Fixed an error breaking user handling of send failures in
|
||||
:meth:`scrapy.mail.MailSender.send()`. (:issue:`1611`, :issue:`5880`)
|
||||
|
||||
Documentation
|
||||
~~~~~~~~~~~~~
|
||||
|
||||
- Expanded contributing docs. (:issue:`5109`, :issue:`5851`)
|
||||
|
||||
- Added blacken-docs_ to pre-commit and reformatted the docs with it.
|
||||
(:issue:`5813`, :issue:`5816`)
|
||||
|
||||
- Fixed a JS issue. (:issue:`5875`, :issue:`5877`)
|
||||
|
||||
- Fixed ``make htmlview``. (:issue:`5878`, :issue:`5879`)
|
||||
|
||||
- Fixed typos and other small errors. (:issue:`5827`, :issue:`5839`,
|
||||
:issue:`5883`, :issue:`5890`, :issue:`5895`, :issue:`5904`)
|
||||
|
||||
Quality assurance
|
||||
~~~~~~~~~~~~~~~~~
|
||||
|
||||
- Extended typing hints. (:issue:`5805`, :issue:`5889`, :issue:`5896`)
|
||||
|
||||
- Tests for most of the examples in the docs are now run as a part of CI,
|
||||
found problems were fixed. (:issue:`5816`, :issue:`5826`, :issue:`5919`)
|
||||
|
||||
- Removed usage of deprecated Python classes. (:issue:`5849`)
|
||||
|
||||
- Silenced ``include-ignored`` warnings from coverage. (:issue:`5820`)
|
||||
|
||||
- Fixed a random failure of the ``test_feedexport.test_batch_path_differ``
|
||||
test. (:issue:`5855`, :issue:`5898`)
|
||||
|
||||
- Updated docstrings to match output produced by parsel_ 1.8.1 so that they
|
||||
don't cause test failures. (:issue:`5902`, :issue:`5919`)
|
||||
|
||||
- Other CI and pre-commit improvements. (:issue:`5802`, :issue:`5823`,
|
||||
:issue:`5908`)
|
||||
|
||||
.. _blacken-docs: https://github.com/adamchainz/blacken-docs
|
||||
|
||||
.. _release-2.8.0:
|
||||
|
||||
Scrapy 2.8.0 (2023-02-02)
|
||||
|
|
@ -4207,8 +4500,6 @@ Relocations
|
|||
+ Note: telnet is not enabled on Python 3
|
||||
(https://github.com/scrapy/scrapy/pull/1524#issuecomment-146985595)
|
||||
|
||||
.. _parsel: https://github.com/scrapy/parsel
|
||||
|
||||
|
||||
Bugfixes
|
||||
~~~~~~~~
|
||||
|
|
@ -5628,6 +5919,7 @@ First release of Scrapy.
|
|||
|
||||
|
||||
.. _AJAX crawlable urls: https://developers.google.com/search/docs/ajax-crawling/docs/getting-started?csw=1
|
||||
.. _boto3: https://github.com/boto/boto3
|
||||
.. _botocore: https://github.com/boto/botocore
|
||||
.. _chunked transfer encoding: https://en.wikipedia.org/wiki/Chunked_transfer_encoding
|
||||
.. _ClientForm: http://wwwsearch.sourceforge.net/old/ClientForm/
|
||||
|
|
@ -5638,6 +5930,7 @@ First release of Scrapy.
|
|||
.. _LevelDB: https://github.com/google/leveldb
|
||||
.. _lxml: https://lxml.de/
|
||||
.. _marshal: https://docs.python.org/2/library/marshal.html
|
||||
.. _parsel: https://github.com/scrapy/parsel
|
||||
.. _parsel.csstranslator.GenericTranslator: https://parsel.readthedocs.io/en/latest/parsel.html#parsel.csstranslator.GenericTranslator
|
||||
.. _parsel.csstranslator.HTMLTranslator: https://parsel.readthedocs.io/en/latest/parsel.html#parsel.csstranslator.HTMLTranslator
|
||||
.. _parsel.csstranslator.XPathExpr: https://parsel.readthedocs.io/en/latest/parsel.html#parsel.csstranslator.XPathExpr
|
||||
|
|
|
|||
|
|
@ -0,0 +1,193 @@
|
|||
.. _topics-addons:
|
||||
|
||||
=======
|
||||
Add-ons
|
||||
=======
|
||||
|
||||
Scrapy's add-on system is a framework which unifies managing and configuring
|
||||
components that extend Scrapy's core functionality, such as middlewares,
|
||||
extensions, or pipelines. It provides users with a plug-and-play experience in
|
||||
Scrapy extension management, and grants extensive configuration control to
|
||||
developers.
|
||||
|
||||
|
||||
Activating and configuring add-ons
|
||||
==================================
|
||||
|
||||
During :class:`~scrapy.crawler.Crawler` initialization, the list of enabled
|
||||
add-ons is read from your ``ADDONS`` setting.
|
||||
|
||||
The ``ADDONS`` setting is a dict in which every key is an add-on class or its
|
||||
import path and the value is its priority.
|
||||
|
||||
This is an example where two add-ons are enabled in a project's
|
||||
``settings.py``::
|
||||
|
||||
ADDONS = {
|
||||
'path.to.someaddon': 0,
|
||||
SomeAddonClass: 1,
|
||||
}
|
||||
|
||||
|
||||
Writing your own add-ons
|
||||
========================
|
||||
|
||||
Add-ons are Python classes that include the following method:
|
||||
|
||||
.. method:: update_settings(settings)
|
||||
|
||||
This method is called during the initialization of the
|
||||
:class:`~scrapy.crawler.Crawler`. Here, you should perform dependency checks
|
||||
(e.g. for external Python libraries) and update the
|
||||
:class:`~scrapy.settings.Settings` object as wished, e.g. enable components
|
||||
for this add-on or set required configuration of other extensions.
|
||||
|
||||
:param settings: The settings object storing Scrapy/component configuration
|
||||
:type settings: :class:`~scrapy.settings.Settings`
|
||||
|
||||
They can also have the following method:
|
||||
|
||||
.. classmethod:: from_crawler(cls, crawler)
|
||||
:noindex:
|
||||
|
||||
If present, this class method is called to create an add-on instance
|
||||
from a :class:`~scrapy.crawler.Crawler`. It must return a new instance
|
||||
of the add-on. The crawler object provides access to all Scrapy core
|
||||
components like settings and signals; it is a way for the add-on to access
|
||||
them and hook its functionality into Scrapy.
|
||||
|
||||
:param crawler: The crawler that uses this add-on
|
||||
:type crawler: :class:`~scrapy.crawler.Crawler`
|
||||
|
||||
The settings set by the add-on should use the ``addon`` priority (see
|
||||
:ref:`populating-settings` and :func:`scrapy.settings.BaseSettings.set`)::
|
||||
|
||||
class MyAddon:
|
||||
def update_settings(self, settings):
|
||||
settings.set("DNSCACHE_ENABLED", True, "addon")
|
||||
|
||||
This allows users to override these settings in the project or spider
|
||||
configuration. This is not possible with settings that are mutable objects,
|
||||
such as the dict that is a value of :setting:`ITEM_PIPELINES`. In these cases
|
||||
you can provide an add-on-specific setting that governs whether the add-on will
|
||||
modify :setting:`ITEM_PIPELINES`::
|
||||
|
||||
class MyAddon:
|
||||
def update_settings(self, settings):
|
||||
if settings.getbool("MYADDON_ENABLE_PIPELINE"):
|
||||
settings["ITEM_PIPELINES"]["path.to.mypipeline"] = 200
|
||||
|
||||
If the ``update_settings`` method raises
|
||||
:exc:`scrapy.exceptions.NotConfigured`, the add-on will be skipped. This makes
|
||||
it easy to enable an add-on only when some conditions are met.
|
||||
|
||||
Fallbacks
|
||||
---------
|
||||
|
||||
Some components provided by add-ons need to fall back to "default"
|
||||
implementations, e.g. a custom download handler needs to send the request that
|
||||
it doesn't handle via the default download handler, or a stats collector that
|
||||
includes some additional processing but otherwise uses the default stats
|
||||
collector. And it's possible that a project needs to use several custom
|
||||
components of the same type, e.g. two custom download handlers that support
|
||||
different kinds of custom requests and still need to use the default download
|
||||
handler for other requests. To make such use cases easier to configure, we
|
||||
recommend that such custom components should be written in the following way:
|
||||
|
||||
1. The custom component (e.g. ``MyDownloadHandler``) shouldn't inherit from the
|
||||
default Scrapy one (e.g.
|
||||
``scrapy.core.downloader.handlers.http.HTTPDownloadHandler``), but instead
|
||||
be able to load the class of the fallback component from a special setting
|
||||
(e.g. ``MY_FALLBACK_DOWNLOAD_HANDLER``), create an instance of it and use
|
||||
it.
|
||||
2. The add-ons that include these components should read the current value of
|
||||
the default setting (e.g. ``DOWNLOAD_HANDLERS``) in their
|
||||
``update_settings()`` methods, save that value into the fallback setting
|
||||
(``MY_FALLBACK_DOWNLOAD_HANDLER`` mentioned earlier) and set the default
|
||||
setting to the component provided by the add-on (e.g.
|
||||
``MyDownloadHandler``). If the fallback setting is already set by the user,
|
||||
they shouldn't change it.
|
||||
3. This way, if there are several add-ons that want to modify the same setting,
|
||||
all of them will fallback to the component from the previous one and then to
|
||||
the Scrapy default. The order of that depends on the priority order in the
|
||||
``ADDONS`` setting.
|
||||
|
||||
|
||||
Add-on examples
|
||||
===============
|
||||
|
||||
Set some basic configuration:
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
class MyAddon:
|
||||
def update_settings(self, settings):
|
||||
settings["ITEM_PIPELINES"]["path.to.mypipeline"] = 200
|
||||
settings.set("DNSCACHE_ENABLED", True, "addon")
|
||||
|
||||
Check dependencies:
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
class MyAddon:
|
||||
def update_settings(self, settings):
|
||||
try:
|
||||
import boto
|
||||
except ImportError:
|
||||
raise NotConfigured("MyAddon requires the boto library")
|
||||
...
|
||||
|
||||
Access the crawler instance:
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
class MyAddon:
|
||||
def __init__(self, crawler) -> None:
|
||||
super().__init__()
|
||||
self.crawler = crawler
|
||||
|
||||
@classmethod
|
||||
def from_crawler(cls, crawler):
|
||||
return cls(crawler)
|
||||
|
||||
def update_settings(self, settings):
|
||||
...
|
||||
|
||||
Use a fallback component:
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
from scrapy.core.downloader.handlers.http import HTTPDownloadHandler
|
||||
|
||||
|
||||
FALLBACK_SETTING = "MY_FALLBACK_DOWNLOAD_HANDLER"
|
||||
|
||||
|
||||
class MyHandler:
|
||||
lazy = False
|
||||
|
||||
def __init__(self, settings, crawler):
|
||||
dhcls = load_object(settings.get(FALLBACK_SETTING))
|
||||
self._fallback_handler = create_instance(
|
||||
dhcls,
|
||||
settings=None,
|
||||
crawler=crawler,
|
||||
)
|
||||
|
||||
def download_request(self, request, spider):
|
||||
if request.meta.get("my_params"):
|
||||
# handle the request
|
||||
...
|
||||
else:
|
||||
return self._fallback_handler.download_request(request, spider)
|
||||
|
||||
|
||||
class MyAddon:
|
||||
def update_settings(self, settings):
|
||||
if not settings.get(FALLBACK_SETTING):
|
||||
settings.set(
|
||||
FALLBACK_SETTING,
|
||||
settings.getwithbase("DOWNLOAD_HANDLERS")["https"],
|
||||
"addon",
|
||||
)
|
||||
settings["DOWNLOAD_HANDLERS"]["https"] = MyHandler
|
||||
|
|
@ -137,6 +137,7 @@ Settings API
|
|||
SETTINGS_PRIORITIES = {
|
||||
"default": 0,
|
||||
"command": 10,
|
||||
"addon": 15,
|
||||
"project": 20,
|
||||
"spider": 30,
|
||||
"cmdline": 40,
|
||||
|
|
|
|||
|
|
@ -27,54 +27,43 @@ reactor manually. You can do that using
|
|||
install_reactor('twisted.internet.asyncioreactor.AsyncioSelectorReactor')
|
||||
|
||||
|
||||
.. _using-custom-loops:
|
||||
.. _asyncio-preinstalled-reactor:
|
||||
|
||||
Using custom asyncio loops
|
||||
==========================
|
||||
Handling a pre-installed reactor
|
||||
================================
|
||||
|
||||
You can also use custom asyncio event loops with the asyncio reactor. Set the
|
||||
:setting:`ASYNCIO_EVENT_LOOP` setting to the import path of the desired event loop class to
|
||||
use it instead of the default asyncio event loop.
|
||||
``twisted.internet.reactor`` and some other Twisted imports install the default
|
||||
Twisted reactor as a side effect. Once a Twisted reactor is installed, it is
|
||||
not possible to switch to a different reactor at run time.
|
||||
|
||||
If you :ref:`configure the asyncio Twisted reactor <install-asyncio>` and, at
|
||||
run time, Scrapy complains that a different reactor is already installed,
|
||||
chances are you have some such imports in your code.
|
||||
|
||||
You can usually fix the issue by moving those offending module-level Twisted
|
||||
imports to the method or function definitions where they are used. For example,
|
||||
if you have something like:
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
from twisted.internet import reactor
|
||||
|
||||
|
||||
.. _asyncio-windows:
|
||||
def my_function():
|
||||
reactor.callLater(...)
|
||||
|
||||
Windows-specific notes
|
||||
======================
|
||||
Switch to something like:
|
||||
|
||||
The Windows implementation of :mod:`asyncio` can use two event loop
|
||||
implementations:
|
||||
.. code-block:: python
|
||||
|
||||
- :class:`~asyncio.SelectorEventLoop`, default before Python 3.8, required
|
||||
when using Twisted.
|
||||
def my_function():
|
||||
from twisted.internet import reactor
|
||||
|
||||
- :class:`~asyncio.ProactorEventLoop`, default since Python 3.8, cannot work
|
||||
with Twisted.
|
||||
reactor.callLater(...)
|
||||
|
||||
So on Python 3.8+ the event loop class needs to be changed.
|
||||
|
||||
.. versionchanged:: 2.6.0
|
||||
The event loop class is changed automatically when you change the
|
||||
:setting:`TWISTED_REACTOR` setting or call
|
||||
:func:`~scrapy.utils.reactor.install_reactor`.
|
||||
|
||||
To change the event loop class manually, call the following code before
|
||||
installing the reactor::
|
||||
|
||||
import asyncio
|
||||
asyncio.set_event_loop_policy(asyncio.WindowsSelectorEventLoopPolicy())
|
||||
|
||||
You can put this in the same function that installs the reactor, if you do that
|
||||
yourself, or in some code that runs before the reactor is installed, e.g.
|
||||
``settings.py``.
|
||||
|
||||
.. note:: Other libraries you use may require
|
||||
:class:`~asyncio.ProactorEventLoop`, e.g. because it supports
|
||||
subprocesses (this is the case with `playwright`_), so you cannot use
|
||||
them together with Scrapy on Windows (but you should be able to use
|
||||
them on WSL or native Linux).
|
||||
|
||||
.. _playwright: https://github.com/microsoft/playwright-python
|
||||
Alternatively, you can try to :ref:`manually install the asyncio reactor
|
||||
<install-asyncio>`, with :func:`~scrapy.utils.reactor.install_reactor`, before
|
||||
those imports happen.
|
||||
|
||||
|
||||
.. _asyncio-await-dfd:
|
||||
|
|
@ -122,3 +111,36 @@ example:
|
|||
f"TWISTED_REACTOR setting. See the asyncio documentation "
|
||||
f"of Scrapy for more information."
|
||||
)
|
||||
|
||||
|
||||
.. _asyncio-windows:
|
||||
|
||||
Windows-specific notes
|
||||
======================
|
||||
|
||||
The Windows implementation of :mod:`asyncio` can use two event loop
|
||||
implementations, :class:`~asyncio.ProactorEventLoop` (default) and
|
||||
:class:`~asyncio.SelectorEventLoop`. However, only
|
||||
:class:`~asyncio.SelectorEventLoop` works with Twisted.
|
||||
|
||||
Scrapy changes the event loop class to :class:`~asyncio.SelectorEventLoop`
|
||||
automatically when you change the :setting:`TWISTED_REACTOR` setting or call
|
||||
:func:`~scrapy.utils.reactor.install_reactor`.
|
||||
|
||||
.. note:: Other libraries you use may require
|
||||
:class:`~asyncio.ProactorEventLoop`, e.g. because it supports
|
||||
subprocesses (this is the case with `playwright`_), so you cannot use
|
||||
them together with Scrapy on Windows (but you should be able to use
|
||||
them on WSL or native Linux).
|
||||
|
||||
.. _playwright: https://github.com/microsoft/playwright-python
|
||||
|
||||
|
||||
.. _using-custom-loops:
|
||||
|
||||
Using custom asyncio loops
|
||||
==========================
|
||||
|
||||
You can also use custom asyncio event loops with the asyncio reactor. Set the
|
||||
:setting:`ASYNCIO_EVENT_LOOP` setting to the import path of the desired event
|
||||
loop class to use it instead of the default asyncio event loop.
|
||||
|
|
|
|||
|
|
@ -70,7 +70,7 @@ If your requirement is a minimum Scrapy version, you may use
|
|||
|
||||
.. code-block:: python
|
||||
|
||||
from pkg_resources import parse_version
|
||||
from packaging.version import parse as parse_version
|
||||
|
||||
import scrapy
|
||||
|
||||
|
|
|
|||
|
|
@ -134,6 +134,63 @@ Common use cases for asynchronous code include:
|
|||
.. _aio-libs: https://github.com/aio-libs
|
||||
|
||||
|
||||
.. _inline-requests:
|
||||
|
||||
Inline requests
|
||||
===============
|
||||
|
||||
The spider below shows how to send a request and await its response all from
|
||||
within a spider callback:
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
from scrapy import Spider, Request
|
||||
from scrapy.utils.defer import maybe_deferred_to_future
|
||||
|
||||
|
||||
class SingleRequestSpider(Spider):
|
||||
name = "single"
|
||||
start_urls = ["https://example.org/product"]
|
||||
|
||||
async def parse(self, response, **kwargs):
|
||||
additional_request = Request("https://example.org/price")
|
||||
deferred = self.crawler.engine.download(additional_request)
|
||||
additional_response = await maybe_deferred_to_future(deferred)
|
||||
yield {
|
||||
"h1": response.css("h1").get(),
|
||||
"price": additional_response.css("#price").get(),
|
||||
}
|
||||
|
||||
You can also send multiple requests in parallel:
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
from scrapy import Spider, Request
|
||||
from scrapy.utils.defer import maybe_deferred_to_future
|
||||
from twisted.internet.defer import DeferredList
|
||||
|
||||
|
||||
class MultipleRequestsSpider(Spider):
|
||||
name = "multiple"
|
||||
start_urls = ["https://example.com/product"]
|
||||
|
||||
async def parse(self, response, **kwargs):
|
||||
additional_requests = [
|
||||
Request("https://example.com/price"),
|
||||
Request("https://example.com/color"),
|
||||
]
|
||||
deferreds = []
|
||||
for r in additional_requests:
|
||||
deferred = self.crawler.engine.download(r)
|
||||
deferreds.append(deferred)
|
||||
responses = await maybe_deferred_to_future(DeferredList(deferreds))
|
||||
yield {
|
||||
"h1": response.css("h1::text").get(),
|
||||
"price": responses[0][1].css(".price::text").get(),
|
||||
"price2": responses[1][1].css(".color::text").get(),
|
||||
}
|
||||
|
||||
|
||||
.. _sync-async-spider-middleware:
|
||||
|
||||
Mixing synchronous and asynchronous spider middlewares
|
||||
|
|
|
|||
|
|
@ -915,6 +915,7 @@ settings (see the settings documentation for more info):
|
|||
* :setting:`RETRY_ENABLED`
|
||||
* :setting:`RETRY_TIMES`
|
||||
* :setting:`RETRY_HTTP_CODES`
|
||||
* :setting:`RETRY_EXCEPTIONS`
|
||||
|
||||
.. reqmeta:: dont_retry
|
||||
|
||||
|
|
@ -966,6 +967,37 @@ In some cases you may want to add 400 to :setting:`RETRY_HTTP_CODES` because
|
|||
it is a common code used to indicate server overload. It is not included by
|
||||
default because HTTP specs say so.
|
||||
|
||||
.. setting:: RETRY_EXCEPTIONS
|
||||
|
||||
RETRY_EXCEPTIONS
|
||||
^^^^^^^^^^^^^^^^
|
||||
|
||||
Default::
|
||||
|
||||
[
|
||||
'twisted.internet.defer.TimeoutError',
|
||||
'twisted.internet.error.TimeoutError',
|
||||
'twisted.internet.error.DNSLookupError',
|
||||
'twisted.internet.error.ConnectionRefusedError',
|
||||
'twisted.internet.error.ConnectionDone',
|
||||
'twisted.internet.error.ConnectError',
|
||||
'twisted.internet.error.ConnectionLost',
|
||||
'twisted.internet.error.TCPTimedOutError',
|
||||
'twisted.web.client.ResponseFailed',
|
||||
IOError,
|
||||
'scrapy.core.downloader.handlers.http11.TunnelError',
|
||||
]
|
||||
|
||||
List of exceptions to retry.
|
||||
|
||||
Each list entry may be an exception type or its import path as a string.
|
||||
|
||||
An exception will not be caught when the exception type is not in
|
||||
:setting:`RETRY_EXCEPTIONS` or when the maximum number of retries for a request
|
||||
has been exceeded (see :setting:`RETRY_TIMES`). To learn about uncaught
|
||||
exception propagation, see
|
||||
:meth:`~scrapy.downloadermiddlewares.DownloaderMiddleware.process_exception`.
|
||||
|
||||
.. setting:: RETRY_PRIORITY_ADJUST
|
||||
|
||||
RETRY_PRIORITY_ADJUST
|
||||
|
|
|
|||
|
|
@ -258,6 +258,7 @@ The conditions for closing a spider can be configured through the following
|
|||
settings:
|
||||
|
||||
* :setting:`CLOSESPIDER_TIMEOUT`
|
||||
* :setting:`CLOSESPIDER_TIMEOUT_NO_ITEM`
|
||||
* :setting:`CLOSESPIDER_ITEMCOUNT`
|
||||
* :setting:`CLOSESPIDER_PAGECOUNT`
|
||||
* :setting:`CLOSESPIDER_ERRORCOUNT`
|
||||
|
|
@ -280,6 +281,18 @@ more than that number of second, it will be automatically closed with the
|
|||
reason ``closespider_timeout``. If zero (or non set), spiders won't be closed by
|
||||
timeout.
|
||||
|
||||
.. setting:: CLOSESPIDER_TIMEOUT_NO_ITEM
|
||||
|
||||
CLOSESPIDER_TIMEOUT_NO_ITEM
|
||||
"""""""""""""""""""""""""""
|
||||
|
||||
Default: ``0``
|
||||
|
||||
An integer which specifies a number of seconds. If the spider has not produced
|
||||
any items in the last number of seconds, it will be closed with the reason
|
||||
``closespider_timeout_no_item``. If zero (or non set), spiders won't be closed
|
||||
regardless if it hasn't produced any items.
|
||||
|
||||
.. setting:: CLOSESPIDER_ITEMCOUNT
|
||||
|
||||
CLOSESPIDER_ITEMCOUNT
|
||||
|
|
|
|||
|
|
@ -101,12 +101,12 @@ The storages backends supported out of the box are:
|
|||
|
||||
- :ref:`topics-feed-storage-fs`
|
||||
- :ref:`topics-feed-storage-ftp`
|
||||
- :ref:`topics-feed-storage-s3` (requires botocore_)
|
||||
- :ref:`topics-feed-storage-s3` (requires boto3_)
|
||||
- :ref:`topics-feed-storage-gcs` (requires `google-cloud-storage`_)
|
||||
- :ref:`topics-feed-storage-stdout`
|
||||
|
||||
Some storage backends may be unavailable if the required external libraries are
|
||||
not available. For example, the S3 backend is only available if the botocore_
|
||||
not available. For example, the S3 backend is only available if the boto3_
|
||||
library is installed.
|
||||
|
||||
|
||||
|
|
@ -156,8 +156,8 @@ The feeds are stored in the local filesystem.
|
|||
- Required external libraries: none
|
||||
|
||||
Note that for the local filesystem storage (only) you can omit the scheme if
|
||||
you specify an absolute path like ``/tmp/export.csv``. This only works on Unix
|
||||
systems though.
|
||||
you specify an absolute path like ``/tmp/export.csv`` (Unix systems only).
|
||||
Alternatively you can also use a :class:`pathlib.Path` object.
|
||||
|
||||
.. _topics-feed-storage-ftp:
|
||||
|
||||
|
|
@ -175,6 +175,12 @@ FTP supports two different connection modes: `active or passive
|
|||
mode by default. To use the active connection mode instead, set the
|
||||
:setting:`FEED_STORAGE_FTP_ACTIVE` setting to ``True``.
|
||||
|
||||
The default value for the ``overwrite`` key in the :setting:`FEEDS` for this
|
||||
storage backend is: ``True``.
|
||||
|
||||
.. caution:: The value ``True`` in ``overwrite`` will cause you to lose the
|
||||
previous version of your data.
|
||||
|
||||
This storage backend uses :ref:`delayed file delivery <delayed-file-delivery>`.
|
||||
|
||||
|
||||
|
|
@ -193,7 +199,7 @@ The feeds are stored on `Amazon S3`_.
|
|||
|
||||
- ``s3://aws_key:aws_secret@mybucket/path/to/export.csv``
|
||||
|
||||
- Required external libraries: `botocore`_ >= 1.4.87
|
||||
- Required external libraries: `boto3`_ >= 1.20.0
|
||||
|
||||
The AWS credentials can be passed as user/password in the URI, or they can be
|
||||
passed through the following settings:
|
||||
|
|
@ -204,10 +210,18 @@ passed through the following settings:
|
|||
|
||||
.. _temporary security credentials: https://docs.aws.amazon.com/general/latest/gr/aws-sec-cred-types.html#temporary-access-keys
|
||||
|
||||
You can also define a custom ACL and custom endpoint for exported feeds using this setting:
|
||||
You can also define a custom ACL, custom endpoint, and region name for exported
|
||||
feeds using these settings:
|
||||
|
||||
- :setting:`FEED_STORAGE_S3_ACL`
|
||||
- :setting:`AWS_ENDPOINT_URL`
|
||||
- :setting:`AWS_REGION_NAME`
|
||||
|
||||
The default value for the ``overwrite`` key in the :setting:`FEEDS` for this
|
||||
storage backend is: ``True``.
|
||||
|
||||
.. caution:: The value ``True`` in ``overwrite`` will cause you to lose the
|
||||
previous version of your data.
|
||||
|
||||
This storage backend uses :ref:`delayed file delivery <delayed-file-delivery>`.
|
||||
|
||||
|
|
@ -236,6 +250,12 @@ You can set a *Project ID* and *Access Control List (ACL)* through the following
|
|||
- :setting:`FEED_STORAGE_GCS_ACL`
|
||||
- :setting:`GCS_PROJECT_ID`
|
||||
|
||||
The default value for the ``overwrite`` key in the :setting:`FEEDS` for this
|
||||
storage backend is: ``True``.
|
||||
|
||||
.. caution:: The value ``True`` in ``overwrite`` will cause you to lose the
|
||||
previous version of your data.
|
||||
|
||||
This storage backend uses :ref:`delayed file delivery <delayed-file-delivery>`.
|
||||
|
||||
.. _google-cloud-storage: https://cloud.google.com/storage/docs/reference/libraries#client-libraries-install-python
|
||||
|
|
@ -488,6 +508,8 @@ as a fallback value if that key is not provided for a specific feed definition:
|
|||
- :ref:`topics-feed-storage-s3`: ``True`` (appending `is not supported
|
||||
<https://forums.aws.amazon.com/message.jspa?messageID=540395>`_)
|
||||
|
||||
- :ref:`topics-feed-storage-gcs`: ``True`` (appending is not supported)
|
||||
|
||||
- :ref:`topics-feed-storage-stdout`: ``False`` (overwriting is not supported)
|
||||
|
||||
.. versionadded:: 2.4.0
|
||||
|
|
@ -552,9 +574,12 @@ to ``.json`` or ``.xml``.
|
|||
FEED_STORE_EMPTY
|
||||
----------------
|
||||
|
||||
Default: ``False``
|
||||
Default: ``True``
|
||||
|
||||
Whether to export empty feeds (i.e. feeds with no items).
|
||||
If ``False``, and there are no items to export, no new files are created and
|
||||
existing files are not modified, even if the :ref:`overwrite feed option
|
||||
<feed-options>` is enabled.
|
||||
|
||||
.. setting:: FEED_STORAGES
|
||||
|
||||
|
|
@ -779,6 +804,6 @@ source spider in the feed URI:
|
|||
|
||||
.. _URIs: https://en.wikipedia.org/wiki/Uniform_Resource_Identifier
|
||||
.. _Amazon S3: https://aws.amazon.com/s3/
|
||||
.. _botocore: https://github.com/boto/botocore
|
||||
.. _boto3: https://github.com/boto/boto3
|
||||
.. _Canned ACL: https://docs.aws.amazon.com/AmazonS3/latest/dev/acl-overview.html#canned-acl
|
||||
.. _Google Cloud Storage: https://cloud.google.com/storage/
|
||||
|
|
|
|||
|
|
@ -215,7 +215,7 @@ item.
|
|||
screenshot_url = self.SPLASH_URL.format(encoded_item_url)
|
||||
request = scrapy.Request(screenshot_url, callback=NO_CALLBACK)
|
||||
response = await maybe_deferred_to_future(
|
||||
spider.crawler.engine.download(request, spider)
|
||||
spider.crawler.engine.download(request)
|
||||
)
|
||||
|
||||
if response.status != 200:
|
||||
|
|
|
|||
|
|
@ -1103,9 +1103,10 @@ Response objects
|
|||
through all :ref:`Downloader Middlewares <topics-downloader-middleware>`.
|
||||
In particular, this means that:
|
||||
|
||||
- HTTP redirections will cause the original request (to the URL before
|
||||
redirection) to be assigned to the redirected response (with the final
|
||||
URL after redirection).
|
||||
- HTTP redirections will create a new request from the request before
|
||||
redirection. It has the majority of the same metadata and original
|
||||
request attributes and gets assigned to the redirected response
|
||||
instead of the propagation of the original request.
|
||||
|
||||
- Response.request.url doesn't always equal Response.url
|
||||
|
||||
|
|
|
|||
|
|
@ -40,8 +40,9 @@ precedence:
|
|||
1. Command line options (most precedence)
|
||||
2. Settings per-spider
|
||||
3. Project settings module
|
||||
4. Default settings per-command
|
||||
5. Default global settings (less precedence)
|
||||
4. Settings set by add-ons
|
||||
5. Default settings per-command
|
||||
6. Default global settings (less precedence)
|
||||
|
||||
The population of these settings sources is taken care of internally, but a
|
||||
manual handling is possible using API calls. See the
|
||||
|
|
@ -66,8 +67,8 @@ Example::
|
|||
----------------------
|
||||
|
||||
Spiders (See the :ref:`topics-spiders` chapter for reference) can define their
|
||||
own settings that will take precedence and override the project ones. They can
|
||||
do so by setting their :attr:`~scrapy.Spider.custom_settings` attribute:
|
||||
own settings that will take precedence and override the project ones. One way
|
||||
to do so is by setting their :attr:`~scrapy.Spider.custom_settings` attribute:
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
|
|
@ -81,6 +82,22 @@ do so by setting their :attr:`~scrapy.Spider.custom_settings` attribute:
|
|||
"SOME_SETTING": "some value",
|
||||
}
|
||||
|
||||
It's often better to implement :meth:`~scrapy.Spider.update_settings` instead,
|
||||
and settings set there should use the "spider" priority explicitly:
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
import scrapy
|
||||
|
||||
|
||||
class MySpider(scrapy.Spider):
|
||||
name = "myspider"
|
||||
|
||||
@classmethod
|
||||
def update_settings(cls, settings):
|
||||
super().update_settings(settings)
|
||||
settings.set("SOME_SETTING", "some value", priority="spider")
|
||||
|
||||
3. Project settings module
|
||||
--------------------------
|
||||
|
||||
|
|
@ -89,7 +106,13 @@ project, it's where most of your custom settings will be populated. For a
|
|||
standard Scrapy project, this means you'll be adding or changing the settings
|
||||
in the ``settings.py`` file created for your project.
|
||||
|
||||
4. Default settings per-command
|
||||
4. Settings set by add-ons
|
||||
--------------------------
|
||||
|
||||
:ref:`Add-ons <topics-addons>` can modify settings. They should do this with
|
||||
this priority, though this is not enforced.
|
||||
|
||||
5. Default settings per-command
|
||||
-------------------------------
|
||||
|
||||
Each :doc:`Scrapy tool </topics/commands>` command can have its own default
|
||||
|
|
@ -97,7 +120,7 @@ settings, which override the global default settings. Those custom command
|
|||
settings are specified in the ``default_settings`` attribute of the command
|
||||
class.
|
||||
|
||||
5. Default global settings
|
||||
6. Default global settings
|
||||
--------------------------
|
||||
|
||||
The global defaults are located in the ``scrapy.settings.default_settings``
|
||||
|
|
@ -201,6 +224,16 @@ to any particular component. In that case the module of that component will be
|
|||
shown, typically an extension, middleware or pipeline. It also means that the
|
||||
component must be enabled in order for the setting to have any effect.
|
||||
|
||||
.. setting:: ADDONS
|
||||
|
||||
ADDONS
|
||||
------
|
||||
|
||||
Default: ``{}``
|
||||
|
||||
A dict containing paths to the add-ons enabled in your project and their
|
||||
priorities. For more information, see :ref:`topics-addons`.
|
||||
|
||||
.. setting:: AWS_ACCESS_KEY_ID
|
||||
|
||||
AWS_ACCESS_KEY_ID
|
||||
|
|
@ -783,7 +816,7 @@ DOWNLOAD_SLOTS
|
|||
|
||||
Default: ``{}``
|
||||
|
||||
Allows to define concurrency/delay parameters on per slot(domain) basis:
|
||||
Allows to define concurrency/delay parameters on per slot (domain) basis:
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
|
|
@ -964,7 +997,6 @@ some of them need to be enabled through a setting.
|
|||
For more information See the :ref:`extensions user guide <topics-extensions>`
|
||||
and the :ref:`list of available extensions <topics-extensions-ref>`.
|
||||
|
||||
|
||||
.. setting:: FEED_TEMPDIR
|
||||
|
||||
FEED_TEMPDIR
|
||||
|
|
|
|||
|
|
@ -145,6 +145,46 @@ scrapy.Spider
|
|||
:param kwargs: keyword arguments passed to the :meth:`__init__` method
|
||||
:type kwargs: dict
|
||||
|
||||
.. classmethod:: update_settings(settings)
|
||||
|
||||
The ``update_settings()`` method is used to modify the spider's settings
|
||||
and is called during initialization of a spider instance.
|
||||
|
||||
It takes a :class:`~scrapy.settings.Settings` object as a parameter and
|
||||
can add or update the spider's configuration values. This method is a
|
||||
class method, meaning that it is called on the :class:`~scrapy.Spider`
|
||||
class and allows all instances of the spider to share the same
|
||||
configuration.
|
||||
|
||||
While per-spider settings can be set in
|
||||
:attr:`~scrapy.Spider.custom_settings`, using ``update_settings()``
|
||||
allows you to dynamically add, remove or change settings based on other
|
||||
settings, spider attributes or other factors and use setting priorities
|
||||
other than ``'spider'``. Also, it's easy to extend ``update_settings()``
|
||||
in a subclass by overriding it, while doing the same with
|
||||
:attr:`~scrapy.Spider.custom_settings` can be hard.
|
||||
|
||||
For example, suppose a spider needs to modify :setting:`FEEDS`:
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
import scrapy
|
||||
|
||||
|
||||
class MySpider(scrapy.Spider):
|
||||
name = "myspider"
|
||||
custom_feed = {
|
||||
"/home/user/documents/items.json": {
|
||||
"format": "json",
|
||||
"indent": 4,
|
||||
}
|
||||
}
|
||||
|
||||
@classmethod
|
||||
def update_settings(cls, settings):
|
||||
super().update_settings(settings)
|
||||
settings.setdefault("FEEDS", {}).update(cls.custom_feed)
|
||||
|
||||
.. method:: start_requests()
|
||||
|
||||
This method must return an iterable with the first Requests to crawl for
|
||||
|
|
|
|||
|
|
@ -30,7 +30,7 @@ def main():
|
|||
try:
|
||||
with Path("build/linkcheck/output.txt").open(encoding="utf-8") as out:
|
||||
output_lines = out.readlines()
|
||||
except IOError:
|
||||
except OSError:
|
||||
print("linkcheck output not found; please run linkcheck first.")
|
||||
sys.exit(1)
|
||||
|
||||
|
|
|
|||
|
|
@ -20,6 +20,7 @@ addopts =
|
|||
markers =
|
||||
only_asyncio: marks tests as only enabled when --reactor=asyncio is passed
|
||||
only_not_asyncio: marks tests as only enabled when --reactor=asyncio is not passed
|
||||
requires_uvloop: marks tests as only enabled when uvloop is known to be working
|
||||
filterwarnings =
|
||||
ignore:scrapy.downloadermiddlewares.decompression is deprecated
|
||||
ignore:Module scrapy.utils.reqser is deprecated
|
||||
|
|
|
|||
|
|
@ -1 +1 @@
|
|||
2.8.0
|
||||
2.10.0
|
||||
|
|
|
|||
|
|
@ -34,8 +34,8 @@ twisted_version = (_txv.major, _txv.minor, _txv.micro)
|
|||
|
||||
|
||||
# Check minimum required Python version
|
||||
if sys.version_info < (3, 7):
|
||||
print(f"Scrapy {__version__} requires Python 3.7+")
|
||||
if sys.version_info < (3, 8):
|
||||
print(f"Scrapy {__version__} requires Python 3.8+")
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -0,0 +1,54 @@
|
|||
import logging
|
||||
from typing import TYPE_CHECKING, Any, List
|
||||
|
||||
from scrapy.exceptions import NotConfigured
|
||||
from scrapy.settings import Settings
|
||||
from scrapy.utils.conf import build_component_list
|
||||
from scrapy.utils.misc import create_instance, load_object
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from scrapy.crawler import Crawler
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class AddonManager:
|
||||
"""This class facilitates loading and storing :ref:`topics-addons`."""
|
||||
|
||||
def __init__(self, crawler: "Crawler") -> None:
|
||||
self.crawler: "Crawler" = crawler
|
||||
self.addons: List[Any] = []
|
||||
|
||||
def load_settings(self, settings: Settings) -> None:
|
||||
"""Load add-ons and configurations from a settings object.
|
||||
|
||||
This will load the add-on for every add-on path in the
|
||||
``ADDONS`` setting and execute their ``update_settings`` methods.
|
||||
|
||||
:param settings: The :class:`~scrapy.settings.Settings` object from \
|
||||
which to read the add-on configuration
|
||||
:type settings: :class:`~scrapy.settings.Settings`
|
||||
"""
|
||||
enabled: List[Any] = []
|
||||
for clspath in build_component_list(settings["ADDONS"]):
|
||||
try:
|
||||
addoncls = load_object(clspath)
|
||||
addon = create_instance(
|
||||
addoncls, settings=settings, crawler=self.crawler
|
||||
)
|
||||
addon.update_settings(settings)
|
||||
self.addons.append(addon)
|
||||
except NotConfigured as e:
|
||||
if e.args:
|
||||
logger.warning(
|
||||
"Disabled %(clspath)s: %(eargs)s",
|
||||
{"clspath": clspath, "eargs": e.args[0]},
|
||||
extra={"crawler": self.crawler},
|
||||
)
|
||||
logger.info(
|
||||
"Enabled addons:\n%(addons)s",
|
||||
{
|
||||
"addons": enabled,
|
||||
},
|
||||
extra={"crawler": self.crawler},
|
||||
)
|
||||
|
|
@ -3,8 +3,7 @@ import cProfile
|
|||
import inspect
|
||||
import os
|
||||
import sys
|
||||
|
||||
import pkg_resources
|
||||
from importlib.metadata import entry_points
|
||||
|
||||
import scrapy
|
||||
from scrapy.commands import BaseRunSpiderCommand, ScrapyCommand, ScrapyHelpFormatter
|
||||
|
|
@ -49,7 +48,11 @@ def _get_commands_from_module(module, inproject):
|
|||
|
||||
def _get_commands_from_entry_points(inproject, group="scrapy.commands"):
|
||||
cmds = {}
|
||||
for entry_point in pkg_resources.iter_entry_points(group):
|
||||
if sys.version_info >= (3, 10):
|
||||
eps = entry_points(group=group)
|
||||
else:
|
||||
eps = entry_points().get(group, ())
|
||||
for entry_point in eps:
|
||||
obj = entry_point.load()
|
||||
if inspect.isclass(obj):
|
||||
cmds[entry_point.name] = obj()
|
||||
|
|
|
|||
|
|
@ -4,7 +4,7 @@ Base class for Scrapy commands
|
|||
import argparse
|
||||
import os
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, Optional
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from twisted.python import failure
|
||||
|
||||
|
|
@ -116,7 +116,7 @@ class ScrapyCommand:
|
|||
if opts.pdb:
|
||||
failure.startDebugMode()
|
||||
|
||||
def run(self, args, opts):
|
||||
def run(self, args: List[str], opts: argparse.Namespace) -> None:
|
||||
"""
|
||||
Entry point for running commands
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -1,7 +1,10 @@
|
|||
import sys
|
||||
from argparse import Namespace
|
||||
from typing import List, Type
|
||||
|
||||
from w3lib.url import is_url
|
||||
|
||||
from scrapy import Spider
|
||||
from scrapy.commands import ScrapyCommand
|
||||
from scrapy.exceptions import UsageError
|
||||
from scrapy.http import Request
|
||||
|
|
@ -57,7 +60,7 @@ class Command(ScrapyCommand):
|
|||
def _print_bytes(self, bytes_):
|
||||
sys.stdout.buffer.write(bytes_ + b"\n")
|
||||
|
||||
def run(self, args, opts):
|
||||
def run(self, args: List[str], opts: Namespace) -> None:
|
||||
if len(args) != 1 or not is_url(args[0]):
|
||||
raise UsageError()
|
||||
request = Request(
|
||||
|
|
@ -73,7 +76,8 @@ class Command(ScrapyCommand):
|
|||
else:
|
||||
request.meta["handle_httpstatus_all"] = True
|
||||
|
||||
spidercls = DefaultSpider
|
||||
spidercls: Type[Spider] = DefaultSpider
|
||||
assert self.crawler_process
|
||||
spider_loader = self.crawler_process.spider_loader
|
||||
if opts.spider:
|
||||
spidercls = spider_loader.load(opts.spider)
|
||||
|
|
|
|||
|
|
@ -3,8 +3,11 @@ Scrapy Shell
|
|||
|
||||
See documentation in docs/topics/shell.rst
|
||||
"""
|
||||
from argparse import Namespace
|
||||
from threading import Thread
|
||||
from typing import List, Type
|
||||
|
||||
from scrapy import Spider
|
||||
from scrapy.commands import ScrapyCommand
|
||||
from scrapy.http import Request
|
||||
from scrapy.shell import Shell
|
||||
|
|
@ -54,15 +57,16 @@ class Command(ScrapyCommand):
|
|||
"""
|
||||
pass
|
||||
|
||||
def run(self, args, opts):
|
||||
def run(self, args: List[str], opts: Namespace) -> None:
|
||||
url = args[0] if args else None
|
||||
if url:
|
||||
# first argument may be a local file
|
||||
url = guess_scheme(url)
|
||||
|
||||
assert self.crawler_process
|
||||
spider_loader = self.crawler_process.spider_loader
|
||||
|
||||
spidercls = DefaultSpider
|
||||
spidercls: Type[Spider] = DefaultSpider
|
||||
if opts.spider:
|
||||
spidercls = spider_loader.load(opts.spider)
|
||||
elif url:
|
||||
|
|
|
|||
|
|
@ -2,7 +2,8 @@ import re
|
|||
import sys
|
||||
from functools import wraps
|
||||
from inspect import getmembers
|
||||
from typing import Dict
|
||||
from types import CoroutineType
|
||||
from typing import AsyncGenerator, Dict
|
||||
from unittest import TestCase
|
||||
|
||||
from scrapy.http import Request
|
||||
|
|
@ -37,7 +38,10 @@ class Contract:
|
|||
else:
|
||||
results.addSuccess(self.testcase_pre)
|
||||
finally:
|
||||
return list(iterate_spider_output(cb(response, **cb_kwargs)))
|
||||
cb_result = cb(response, **cb_kwargs)
|
||||
if isinstance(cb_result, (AsyncGenerator, CoroutineType)):
|
||||
raise TypeError("Contracts don't support async callbacks")
|
||||
return list(iterate_spider_output(cb_result))
|
||||
|
||||
request.callback = wrapper
|
||||
|
||||
|
|
@ -49,7 +53,10 @@ class Contract:
|
|||
|
||||
@wraps(cb)
|
||||
def wrapper(response, **cb_kwargs):
|
||||
output = list(iterate_spider_output(cb(response, **cb_kwargs)))
|
||||
cb_result = cb(response, **cb_kwargs)
|
||||
if isinstance(cb_result, (AsyncGenerator, CoroutineType)):
|
||||
raise TypeError("Contracts don't support async callbacks")
|
||||
output = list(iterate_spider_output(cb_result))
|
||||
try:
|
||||
results.startTest(self.testcase_post)
|
||||
self.post_process(output)
|
||||
|
|
|
|||
|
|
@ -5,7 +5,6 @@ For more information see docs/topics/architecture.rst
|
|||
|
||||
"""
|
||||
import logging
|
||||
import warnings
|
||||
from time import time
|
||||
from typing import (
|
||||
TYPE_CHECKING,
|
||||
|
|
@ -14,7 +13,6 @@ from typing import (
|
|||
Generator,
|
||||
Iterable,
|
||||
Iterator,
|
||||
List,
|
||||
Optional,
|
||||
Set,
|
||||
Type,
|
||||
|
|
@ -29,7 +27,7 @@ from twisted.python.failure import Failure
|
|||
from scrapy import signals
|
||||
from scrapy.core.downloader import Downloader
|
||||
from scrapy.core.scraper import Scraper
|
||||
from scrapy.exceptions import CloseSpider, DontCloseSpider, ScrapyDeprecationWarning
|
||||
from scrapy.exceptions import CloseSpider, DontCloseSpider
|
||||
from scrapy.http import Request, Response
|
||||
from scrapy.logformatter import LogFormatter
|
||||
from scrapy.settings import BaseSettings, Settings
|
||||
|
|
@ -213,7 +211,7 @@ class ExecutionEngine:
|
|||
if request is None:
|
||||
return None
|
||||
|
||||
d = self._download(request, self.spider)
|
||||
d = self._download(request)
|
||||
d.addBoth(self._handle_downloader_output, request)
|
||||
d.addErrback(
|
||||
lambda f: logger.info(
|
||||
|
|
@ -266,13 +264,7 @@ class ExecutionEngine:
|
|||
)
|
||||
return d
|
||||
|
||||
def spider_is_idle(self, spider: Optional[Spider] = None) -> bool:
|
||||
if spider is not None:
|
||||
warnings.warn(
|
||||
"Passing a 'spider' argument to ExecutionEngine.spider_is_idle is deprecated",
|
||||
category=ScrapyDeprecationWarning,
|
||||
stacklevel=2,
|
||||
)
|
||||
def spider_is_idle(self) -> bool:
|
||||
if self.slot is None:
|
||||
raise RuntimeError("Engine slot not assigned")
|
||||
if not self.scraper.slot.is_idle(): # type: ignore[union-attr]
|
||||
|
|
@ -285,18 +277,8 @@ class ExecutionEngine:
|
|||
return False
|
||||
return True
|
||||
|
||||
def crawl(self, request: Request, spider: Optional[Spider] = None) -> None:
|
||||
def crawl(self, request: Request) -> None:
|
||||
"""Inject the request into the spider <-> downloader pipeline"""
|
||||
if spider is not None:
|
||||
warnings.warn(
|
||||
"Passing a 'spider' argument to ExecutionEngine.crawl is deprecated",
|
||||
category=ScrapyDeprecationWarning,
|
||||
stacklevel=2,
|
||||
)
|
||||
if spider is not self.spider:
|
||||
raise RuntimeError(
|
||||
f"The spider {spider.name!r} does not match the open spider"
|
||||
)
|
||||
if self.spider is None:
|
||||
raise RuntimeError(f"No open spider to crawl: {request}")
|
||||
self._schedule_request(request, self.spider)
|
||||
|
|
@ -311,39 +293,24 @@ class ExecutionEngine:
|
|||
signals.request_dropped, request=request, spider=spider
|
||||
)
|
||||
|
||||
def download(self, request: Request, spider: Optional[Spider] = None) -> Deferred:
|
||||
def download(self, request: Request) -> Deferred:
|
||||
"""Return a Deferred which fires with a Response as result, only downloader middlewares are applied"""
|
||||
if spider is not None:
|
||||
warnings.warn(
|
||||
"Passing a 'spider' argument to ExecutionEngine.download is deprecated",
|
||||
category=ScrapyDeprecationWarning,
|
||||
stacklevel=2,
|
||||
)
|
||||
if spider is not self.spider:
|
||||
logger.warning(
|
||||
"The spider '%s' does not match the open spider", spider.name
|
||||
)
|
||||
if self.spider is None:
|
||||
raise RuntimeError(f"No open spider to crawl: {request}")
|
||||
return self._download(request, spider).addBoth(
|
||||
self._downloaded, request, spider
|
||||
)
|
||||
return self._download(request).addBoth(self._downloaded, request)
|
||||
|
||||
def _downloaded(
|
||||
self, result: Union[Response, Request], request: Request, spider: Spider
|
||||
self, result: Union[Response, Request], request: Request
|
||||
) -> Union[Deferred, Response]:
|
||||
assert self.slot is not None # typing
|
||||
self.slot.remove_request(request)
|
||||
return self.download(result, spider) if isinstance(result, Request) else result
|
||||
return self.download(result) if isinstance(result, Request) else result
|
||||
|
||||
def _download(self, request: Request, spider: Optional[Spider]) -> Deferred:
|
||||
def _download(self, request: Request) -> Deferred:
|
||||
assert self.slot is not None # typing
|
||||
|
||||
self.slot.add_request(request)
|
||||
|
||||
if spider is None:
|
||||
spider = self.spider
|
||||
|
||||
def _on_success(result: Union[Response, Request]) -> Union[Response, Request]:
|
||||
if not isinstance(result, (Response, Request)):
|
||||
raise TypeError(
|
||||
|
|
@ -352,15 +319,17 @@ class ExecutionEngine:
|
|||
if isinstance(result, Response):
|
||||
if result.request is None:
|
||||
result.request = request
|
||||
assert spider is not None
|
||||
logkws = self.logformatter.crawled(result.request, result, spider)
|
||||
assert self.spider is not None
|
||||
logkws = self.logformatter.crawled(result.request, result, self.spider)
|
||||
if logkws is not None:
|
||||
logger.log(*logformatter_adapter(logkws), extra={"spider": spider})
|
||||
logger.log(
|
||||
*logformatter_adapter(logkws), extra={"spider": self.spider}
|
||||
)
|
||||
self.signals.send_catch_log(
|
||||
signal=signals.response_received,
|
||||
response=result,
|
||||
request=result.request,
|
||||
spider=spider,
|
||||
spider=self.spider,
|
||||
)
|
||||
return result
|
||||
|
||||
|
|
@ -369,8 +338,8 @@ class ExecutionEngine:
|
|||
self.slot.nextcall.schedule()
|
||||
return _
|
||||
|
||||
assert spider is not None
|
||||
dwld = self.downloader.fetch(request, spider)
|
||||
assert self.spider is not None
|
||||
dwld = self.downloader.fetch(request, self.spider)
|
||||
dwld.addCallbacks(_on_success)
|
||||
dwld.addBoth(_on_complete)
|
||||
return dwld
|
||||
|
|
@ -485,31 +454,3 @@ class ExecutionEngine:
|
|||
dfd.addBoth(lambda _: self._spider_closed_callback(spider))
|
||||
|
||||
return dfd
|
||||
|
||||
@property
|
||||
def open_spiders(self) -> List[Spider]:
|
||||
warnings.warn(
|
||||
"ExecutionEngine.open_spiders is deprecated, please use ExecutionEngine.spider instead",
|
||||
category=ScrapyDeprecationWarning,
|
||||
stacklevel=2,
|
||||
)
|
||||
return [self.spider] if self.spider is not None else []
|
||||
|
||||
def has_capacity(self) -> bool:
|
||||
warnings.warn(
|
||||
"ExecutionEngine.has_capacity is deprecated",
|
||||
ScrapyDeprecationWarning,
|
||||
stacklevel=2,
|
||||
)
|
||||
return not bool(self.slot)
|
||||
|
||||
def schedule(self, request: Request, spider: Spider) -> None:
|
||||
warnings.warn(
|
||||
"ExecutionEngine.schedule is deprecated, please use "
|
||||
"ExecutionEngine.crawl or ExecutionEngine.download instead",
|
||||
category=ScrapyDeprecationWarning,
|
||||
stacklevel=2,
|
||||
)
|
||||
if self.slot is None:
|
||||
raise RuntimeError("Engine slot not assigned")
|
||||
self._schedule_request(request, spider)
|
||||
|
|
|
|||
|
|
@ -1,8 +1,10 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
from abc import abstractmethod
|
||||
from pathlib import Path
|
||||
from typing import Any, Optional, Type, TypeVar, cast
|
||||
from typing import TYPE_CHECKING, Any, Optional, Type, TypeVar, cast
|
||||
|
||||
from twisted.internet.defer import Deferred
|
||||
|
||||
|
|
@ -14,6 +16,11 @@ from scrapy.statscollectors import StatsCollector
|
|||
from scrapy.utils.job import job_dir
|
||||
from scrapy.utils.misc import create_instance, load_object
|
||||
|
||||
if TYPE_CHECKING:
|
||||
# typing.Self requires Python 3.11
|
||||
from typing_extensions import Self
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
|
|
@ -54,7 +61,7 @@ class BaseScheduler(metaclass=BaseSchedulerMeta):
|
|||
"""
|
||||
|
||||
@classmethod
|
||||
def from_crawler(cls, crawler: Crawler):
|
||||
def from_crawler(cls, crawler: Crawler) -> Self:
|
||||
"""
|
||||
Factory method which receives the current :class:`~scrapy.crawler.Crawler` object as argument.
|
||||
"""
|
||||
|
|
@ -325,6 +332,7 @@ class Scheduler(BaseScheduler):
|
|||
|
||||
def _dq(self):
|
||||
"""Create a new priority queue instance, with disk storage"""
|
||||
assert self.dqdir
|
||||
state = self._read_dqs_state(self.dqdir)
|
||||
q = create_instance(
|
||||
self.pqclass,
|
||||
|
|
|
|||
|
|
@ -364,6 +364,7 @@ class Scraper:
|
|||
spider=spider,
|
||||
exception=output.value,
|
||||
)
|
||||
assert ex
|
||||
logkws = self.logformatter.item_error(item, ex, response, spider)
|
||||
logger.log(
|
||||
*logformatter_adapter(logkws),
|
||||
|
|
|
|||
|
|
@ -4,9 +4,14 @@ import logging
|
|||
import pprint
|
||||
import signal
|
||||
import warnings
|
||||
from typing import TYPE_CHECKING, Optional, Type, Union
|
||||
from typing import TYPE_CHECKING, Any, Dict, Generator, Optional, Set, Type, Union, cast
|
||||
|
||||
from twisted.internet import defer
|
||||
from twisted.internet.defer import (
|
||||
Deferred,
|
||||
DeferredList,
|
||||
inlineCallbacks,
|
||||
maybeDeferred,
|
||||
)
|
||||
from zope.interface.exceptions import DoesNotImplement
|
||||
|
||||
try:
|
||||
|
|
@ -18,12 +23,13 @@ except ImportError:
|
|||
from zope.interface.verify import verifyClass
|
||||
|
||||
from scrapy import Spider, signals
|
||||
from scrapy.addons import AddonManager
|
||||
from scrapy.core.engine import ExecutionEngine
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
from scrapy.extension import ExtensionManager
|
||||
from scrapy.interfaces import ISpiderLoader
|
||||
from scrapy.logformatter import LogFormatter
|
||||
from scrapy.settings import Settings, overridden_settings
|
||||
from scrapy.settings import BaseSettings, Settings, overridden_settings
|
||||
from scrapy.signalmanager import SignalManager
|
||||
from scrapy.statscollectors import StatsCollector
|
||||
from scrapy.utils.log import (
|
||||
|
|
@ -54,7 +60,7 @@ class Crawler:
|
|||
def __init__(
|
||||
self,
|
||||
spidercls: Type[Spider],
|
||||
settings: Union[None, dict, Settings] = None,
|
||||
settings: Union[None, Dict[str, Any], Settings] = None,
|
||||
init_reactor: bool = False,
|
||||
):
|
||||
if isinstance(spidercls, Spider):
|
||||
|
|
@ -67,6 +73,9 @@ class Crawler:
|
|||
self.settings: Settings = settings.copy()
|
||||
self.spidercls.update_settings(self.settings)
|
||||
|
||||
self.addons: AddonManager = AddonManager(self)
|
||||
self.addons.load_settings(self.settings)
|
||||
|
||||
self.signals: SignalManager = SignalManager(self)
|
||||
|
||||
self.stats: StatsCollector = load_object(self.settings["STATS_CLASS"])(self)
|
||||
|
|
@ -118,8 +127,8 @@ class Crawler:
|
|||
self.spider: Optional[Spider] = None
|
||||
self.engine: Optional[ExecutionEngine] = None
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def crawl(self, *args, **kwargs):
|
||||
@inlineCallbacks
|
||||
def crawl(self, *args: Any, **kwargs: Any) -> Generator[Deferred, Any, None]:
|
||||
if self.crawling:
|
||||
raise RuntimeError("Crawling already taking place")
|
||||
self.crawling = True
|
||||
|
|
@ -129,26 +138,27 @@ class Crawler:
|
|||
self.engine = self._create_engine()
|
||||
start_requests = iter(self.spider.start_requests())
|
||||
yield self.engine.open_spider(self.spider, start_requests)
|
||||
yield defer.maybeDeferred(self.engine.start)
|
||||
yield maybeDeferred(self.engine.start)
|
||||
except Exception:
|
||||
self.crawling = False
|
||||
if self.engine is not None:
|
||||
yield self.engine.close()
|
||||
raise
|
||||
|
||||
def _create_spider(self, *args, **kwargs):
|
||||
def _create_spider(self, *args: Any, **kwargs: Any) -> Spider:
|
||||
return self.spidercls.from_crawler(self, *args, **kwargs)
|
||||
|
||||
def _create_engine(self):
|
||||
def _create_engine(self) -> ExecutionEngine:
|
||||
return ExecutionEngine(self, lambda _: self.stop())
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def stop(self):
|
||||
@inlineCallbacks
|
||||
def stop(self) -> Generator[Deferred, Any, None]:
|
||||
"""Starts a graceful stop of the crawler and returns a deferred that is
|
||||
fired when the crawler is stopped."""
|
||||
if self.crawling:
|
||||
self.crawling = False
|
||||
yield defer.maybeDeferred(self.engine.stop)
|
||||
assert self.engine
|
||||
yield maybeDeferred(self.engine.stop)
|
||||
|
||||
|
||||
class CrawlerRunner:
|
||||
|
|
@ -171,7 +181,7 @@ class CrawlerRunner:
|
|||
)
|
||||
|
||||
@staticmethod
|
||||
def _get_spider_loader(settings):
|
||||
def _get_spider_loader(settings: BaseSettings):
|
||||
"""Get SpiderLoader instance from settings"""
|
||||
cls_path = settings.get("SPIDER_LOADER_CLASS")
|
||||
loader_cls = load_object(cls_path)
|
||||
|
|
@ -190,26 +200,21 @@ class CrawlerRunner:
|
|||
)
|
||||
return loader_cls.from_settings(settings.frozencopy())
|
||||
|
||||
def __init__(self, settings=None):
|
||||
def __init__(self, settings: Union[Dict[str, Any], Settings, None] = None):
|
||||
if isinstance(settings, dict) or settings is None:
|
||||
settings = Settings(settings)
|
||||
self.settings = settings
|
||||
self.spider_loader = self._get_spider_loader(settings)
|
||||
self._crawlers = set()
|
||||
self._active = set()
|
||||
self._crawlers: Set[Crawler] = set()
|
||||
self._active: Set[Deferred] = set()
|
||||
self.bootstrap_failed = False
|
||||
|
||||
@property
|
||||
def spiders(self):
|
||||
warnings.warn(
|
||||
"CrawlerRunner.spiders attribute is renamed to "
|
||||
"CrawlerRunner.spider_loader.",
|
||||
category=ScrapyDeprecationWarning,
|
||||
stacklevel=2,
|
||||
)
|
||||
return self.spider_loader
|
||||
|
||||
def crawl(self, crawler_or_spidercls, *args, **kwargs):
|
||||
def crawl(
|
||||
self,
|
||||
crawler_or_spidercls: Union[Type[Spider], str, Crawler],
|
||||
*args: Any,
|
||||
**kwargs: Any,
|
||||
) -> Deferred:
|
||||
"""
|
||||
Run a crawler with the provided arguments.
|
||||
|
||||
|
|
@ -239,12 +244,12 @@ class CrawlerRunner:
|
|||
crawler = self.create_crawler(crawler_or_spidercls)
|
||||
return self._crawl(crawler, *args, **kwargs)
|
||||
|
||||
def _crawl(self, crawler, *args, **kwargs):
|
||||
def _crawl(self, crawler: Crawler, *args: Any, **kwargs: Any) -> Deferred:
|
||||
self.crawlers.add(crawler)
|
||||
d = crawler.crawl(*args, **kwargs)
|
||||
self._active.add(d)
|
||||
|
||||
def _done(result):
|
||||
def _done(result: Any) -> Any:
|
||||
self.crawlers.discard(crawler)
|
||||
self._active.discard(d)
|
||||
self.bootstrap_failed |= not getattr(crawler, "spider", None)
|
||||
|
|
@ -252,7 +257,9 @@ class CrawlerRunner:
|
|||
|
||||
return d.addBoth(_done)
|
||||
|
||||
def create_crawler(self, crawler_or_spidercls):
|
||||
def create_crawler(
|
||||
self, crawler_or_spidercls: Union[Type[Spider], str, Crawler]
|
||||
) -> Crawler:
|
||||
"""
|
||||
Return a :class:`~scrapy.crawler.Crawler` object.
|
||||
|
||||
|
|
@ -272,21 +279,22 @@ class CrawlerRunner:
|
|||
return crawler_or_spidercls
|
||||
return self._create_crawler(crawler_or_spidercls)
|
||||
|
||||
def _create_crawler(self, spidercls):
|
||||
def _create_crawler(self, spidercls: Union[str, Type[Spider]]) -> Crawler:
|
||||
if isinstance(spidercls, str):
|
||||
spidercls = self.spider_loader.load(spidercls)
|
||||
return Crawler(spidercls, self.settings)
|
||||
# temporary cast until self.spider_loader is typed
|
||||
return Crawler(cast(Type[Spider], spidercls), self.settings)
|
||||
|
||||
def stop(self):
|
||||
def stop(self) -> Deferred:
|
||||
"""
|
||||
Stops simultaneously all the crawling jobs taking place.
|
||||
|
||||
Returns a deferred that is fired when they all have ended.
|
||||
"""
|
||||
return defer.DeferredList([c.stop() for c in list(self.crawlers)])
|
||||
return DeferredList([c.stop() for c in list(self.crawlers)])
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def join(self):
|
||||
@inlineCallbacks
|
||||
def join(self) -> Generator[Deferred, Any, None]:
|
||||
"""
|
||||
join()
|
||||
|
||||
|
|
@ -294,7 +302,7 @@ class CrawlerRunner:
|
|||
completed their executions.
|
||||
"""
|
||||
while self._active:
|
||||
yield defer.DeferredList(self._active)
|
||||
yield DeferredList(self._active)
|
||||
|
||||
|
||||
class CrawlerProcess(CrawlerRunner):
|
||||
|
|
@ -321,13 +329,17 @@ class CrawlerProcess(CrawlerRunner):
|
|||
process. See :ref:`run-from-script` for an example.
|
||||
"""
|
||||
|
||||
def __init__(self, settings=None, install_root_handler=True):
|
||||
def __init__(
|
||||
self,
|
||||
settings: Union[Dict[str, Any], Settings, None] = None,
|
||||
install_root_handler: bool = True,
|
||||
):
|
||||
super().__init__(settings)
|
||||
configure_logging(self.settings, install_root_handler)
|
||||
log_scrapy_info(self.settings)
|
||||
self._initialized_reactor = False
|
||||
|
||||
def _signal_shutdown(self, signum, _):
|
||||
def _signal_shutdown(self, signum: int, _: Any) -> None:
|
||||
from twisted.internet import reactor
|
||||
|
||||
install_shutdown_handlers(self._signal_kill)
|
||||
|
|
@ -338,7 +350,7 @@ class CrawlerProcess(CrawlerRunner):
|
|||
)
|
||||
reactor.callFromThread(self._graceful_stop_reactor)
|
||||
|
||||
def _signal_kill(self, signum, _):
|
||||
def _signal_kill(self, signum: int, _: Any) -> None:
|
||||
from twisted.internet import reactor
|
||||
|
||||
install_shutdown_handlers(signal.SIG_IGN)
|
||||
|
|
@ -348,14 +360,19 @@ class CrawlerProcess(CrawlerRunner):
|
|||
)
|
||||
reactor.callFromThread(self._stop_reactor)
|
||||
|
||||
def _create_crawler(self, spidercls):
|
||||
def _create_crawler(self, spidercls: Union[Type[Spider], str]) -> Crawler:
|
||||
if isinstance(spidercls, str):
|
||||
spidercls = self.spider_loader.load(spidercls)
|
||||
init_reactor = not self._initialized_reactor
|
||||
self._initialized_reactor = True
|
||||
return Crawler(spidercls, self.settings, init_reactor=init_reactor)
|
||||
# temporary cast until self.spider_loader is typed
|
||||
return Crawler(
|
||||
cast(Type[Spider], spidercls), self.settings, init_reactor=init_reactor
|
||||
)
|
||||
|
||||
def start(self, stop_after_crawl=True, install_signal_handlers=True):
|
||||
def start(
|
||||
self, stop_after_crawl: bool = True, install_signal_handlers: bool = True
|
||||
) -> None:
|
||||
"""
|
||||
This method starts a :mod:`~twisted.internet.reactor`, adjusts its pool
|
||||
size to :setting:`REACTOR_THREADPOOL_MAXSIZE`, and installs a DNS cache
|
||||
|
|
@ -389,12 +406,12 @@ class CrawlerProcess(CrawlerRunner):
|
|||
reactor.addSystemEventTrigger("before", "shutdown", self.stop)
|
||||
reactor.run(installSignalHandlers=False) # blocking call
|
||||
|
||||
def _graceful_stop_reactor(self):
|
||||
def _graceful_stop_reactor(self) -> Deferred:
|
||||
d = self.stop()
|
||||
d.addBoth(self._stop_reactor)
|
||||
return d
|
||||
|
||||
def _stop_reactor(self, _=None):
|
||||
def _stop_reactor(self, _: Any = None) -> None:
|
||||
from twisted.internet import reactor
|
||||
|
||||
try:
|
||||
|
|
|
|||
|
|
@ -63,7 +63,7 @@ class DecompressionMiddleware:
|
|||
archive = BytesIO(response.body)
|
||||
try:
|
||||
body = gzip.GzipFile(fileobj=archive).read()
|
||||
except IOError:
|
||||
except OSError:
|
||||
return
|
||||
|
||||
respcls = responsetypes.from_args(body=body)
|
||||
|
|
@ -72,7 +72,7 @@ class DecompressionMiddleware:
|
|||
def _is_bzip2(self, response):
|
||||
try:
|
||||
body = bz2.decompress(response.body)
|
||||
except IOError:
|
||||
except OSError:
|
||||
return
|
||||
|
||||
respcls = responsetypes.from_args(body=body)
|
||||
|
|
|
|||
|
|
@ -37,7 +37,7 @@ class HttpCacheMiddleware:
|
|||
ConnectionLost,
|
||||
TCPTimedOutError,
|
||||
ResponseFailed,
|
||||
IOError,
|
||||
OSError,
|
||||
)
|
||||
|
||||
def __init__(self, settings: Settings, stats: StatsCollector) -> None:
|
||||
|
|
|
|||
|
|
@ -9,31 +9,36 @@ RETRY_HTTP_CODES - which HTTP response codes to retry
|
|||
Failed pages are collected on the scraping process and rescheduled at the end,
|
||||
once the spider has finished crawling all regular (non failed) pages.
|
||||
"""
|
||||
import warnings
|
||||
from logging import Logger, getLogger
|
||||
from typing import Optional, Union
|
||||
|
||||
from twisted.internet import defer
|
||||
from twisted.internet.error import (
|
||||
ConnectError,
|
||||
ConnectionDone,
|
||||
ConnectionLost,
|
||||
ConnectionRefusedError,
|
||||
DNSLookupError,
|
||||
TCPTimedOutError,
|
||||
TimeoutError,
|
||||
)
|
||||
from twisted.web.client import ResponseFailed
|
||||
|
||||
from scrapy.core.downloader.handlers.http11 import TunnelError
|
||||
from scrapy.exceptions import NotConfigured
|
||||
from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning
|
||||
from scrapy.http.request import Request
|
||||
from scrapy.settings import Settings
|
||||
from scrapy.spiders import Spider
|
||||
from scrapy.utils.misc import load_object
|
||||
from scrapy.utils.python import global_object_name
|
||||
from scrapy.utils.response import response_status_message
|
||||
|
||||
retry_logger = getLogger(__name__)
|
||||
|
||||
|
||||
class BackwardsCompatibilityMetaclass(type):
|
||||
@property
|
||||
def EXCEPTIONS_TO_RETRY(cls):
|
||||
warnings.warn(
|
||||
"Attribute RetryMiddleware.EXCEPTIONS_TO_RETRY is deprecated. "
|
||||
"Use the RETRY_EXCEPTIONS setting instead.",
|
||||
ScrapyDeprecationWarning,
|
||||
stacklevel=2,
|
||||
)
|
||||
return tuple(
|
||||
load_object(x) if isinstance(x, str) else x
|
||||
for x in Settings().getlist("RETRY_EXCEPTIONS")
|
||||
)
|
||||
|
||||
|
||||
def get_retry_request(
|
||||
request: Request,
|
||||
*,
|
||||
|
|
@ -121,23 +126,7 @@ def get_retry_request(
|
|||
return None
|
||||
|
||||
|
||||
class RetryMiddleware:
|
||||
# IOError is raised by the HttpCompression middleware when trying to
|
||||
# decompress an empty response
|
||||
EXCEPTIONS_TO_RETRY = (
|
||||
defer.TimeoutError,
|
||||
TimeoutError,
|
||||
DNSLookupError,
|
||||
ConnectionRefusedError,
|
||||
ConnectionDone,
|
||||
ConnectError,
|
||||
ConnectionLost,
|
||||
TCPTimedOutError,
|
||||
ResponseFailed,
|
||||
IOError,
|
||||
TunnelError,
|
||||
)
|
||||
|
||||
class RetryMiddleware(metaclass=BackwardsCompatibilityMetaclass):
|
||||
def __init__(self, settings):
|
||||
if not settings.getbool("RETRY_ENABLED"):
|
||||
raise NotConfigured
|
||||
|
|
@ -147,6 +136,16 @@ class RetryMiddleware:
|
|||
)
|
||||
self.priority_adjust = settings.getint("RETRY_PRIORITY_ADJUST")
|
||||
|
||||
if not hasattr(
|
||||
self, "EXCEPTIONS_TO_RETRY"
|
||||
): # If EXCEPTIONS_TO_RETRY is not "overriden"
|
||||
self.exceptions_to_retry = tuple(
|
||||
load_object(x) if isinstance(x, str) else x
|
||||
for x in settings.getlist("RETRY_EXCEPTIONS")
|
||||
)
|
||||
else:
|
||||
self.exceptions_to_retry = self.EXCEPTIONS_TO_RETRY
|
||||
|
||||
@classmethod
|
||||
def from_crawler(cls, crawler):
|
||||
return cls(crawler.settings)
|
||||
|
|
@ -160,7 +159,7 @@ class RetryMiddleware:
|
|||
return response
|
||||
|
||||
def process_exception(self, request, exception, spider):
|
||||
if isinstance(exception, self.EXCEPTIONS_TO_RETRY) and not request.meta.get(
|
||||
if isinstance(exception, self.exceptions_to_retry) and not request.meta.get(
|
||||
"dont_retry", False
|
||||
):
|
||||
return self._retry(request, exception, spider)
|
||||
|
|
|
|||
|
|
@ -1,6 +1,8 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from pathlib import Path
|
||||
from typing import Optional, Set, Type, TypeVar
|
||||
from typing import TYPE_CHECKING, Optional, Set
|
||||
from warnings import warn
|
||||
|
||||
from twisted.internet.defer import Deferred
|
||||
|
|
@ -10,16 +12,22 @@ from scrapy.settings import BaseSettings
|
|||
from scrapy.spiders import Spider
|
||||
from scrapy.utils.deprecate import ScrapyDeprecationWarning
|
||||
from scrapy.utils.job import job_dir
|
||||
from scrapy.utils.request import RequestFingerprinter, referer_str
|
||||
from scrapy.utils.request import (
|
||||
RequestFingerprinter,
|
||||
RequestFingerprinterProtocol,
|
||||
referer_str,
|
||||
)
|
||||
|
||||
BaseDupeFilterTV = TypeVar("BaseDupeFilterTV", bound="BaseDupeFilter")
|
||||
if TYPE_CHECKING:
|
||||
# typing.Self requires Python 3.11
|
||||
from typing_extensions import Self
|
||||
|
||||
from scrapy.crawler import Crawler
|
||||
|
||||
|
||||
class BaseDupeFilter:
|
||||
@classmethod
|
||||
def from_settings(
|
||||
cls: Type[BaseDupeFilterTV], settings: BaseSettings
|
||||
) -> BaseDupeFilterTV:
|
||||
def from_settings(cls, settings: BaseSettings) -> Self:
|
||||
return cls()
|
||||
|
||||
def request_seen(self, request: Request) -> bool:
|
||||
|
|
@ -36,9 +44,6 @@ class BaseDupeFilter:
|
|||
pass
|
||||
|
||||
|
||||
RFPDupeFilterTV = TypeVar("RFPDupeFilterTV", bound="RFPDupeFilter")
|
||||
|
||||
|
||||
class RFPDupeFilter(BaseDupeFilter):
|
||||
"""Request Fingerprint duplicates filter"""
|
||||
|
||||
|
|
@ -47,10 +52,12 @@ class RFPDupeFilter(BaseDupeFilter):
|
|||
path: Optional[str] = None,
|
||||
debug: bool = False,
|
||||
*,
|
||||
fingerprinter=None,
|
||||
fingerprinter: Optional[RequestFingerprinterProtocol] = None,
|
||||
) -> None:
|
||||
self.file = None
|
||||
self.fingerprinter = fingerprinter or RequestFingerprinter()
|
||||
self.fingerprinter: RequestFingerprinterProtocol = (
|
||||
fingerprinter or RequestFingerprinter()
|
||||
)
|
||||
self.fingerprints: Set[str] = set()
|
||||
self.logdupes = True
|
||||
self.debug = debug
|
||||
|
|
@ -62,8 +69,11 @@ class RFPDupeFilter(BaseDupeFilter):
|
|||
|
||||
@classmethod
|
||||
def from_settings(
|
||||
cls: Type[RFPDupeFilterTV], settings: BaseSettings, *, fingerprinter=None
|
||||
) -> RFPDupeFilterTV:
|
||||
cls,
|
||||
settings: BaseSettings,
|
||||
*,
|
||||
fingerprinter: Optional[RequestFingerprinterProtocol] = None,
|
||||
) -> Self:
|
||||
debug = settings.getbool("DUPEFILTER_DEBUG")
|
||||
try:
|
||||
return cls(job_dir(settings), debug, fingerprinter=fingerprinter)
|
||||
|
|
@ -75,11 +85,11 @@ class RFPDupeFilter(BaseDupeFilter):
|
|||
ScrapyDeprecationWarning,
|
||||
)
|
||||
result = cls(job_dir(settings), debug)
|
||||
result.fingerprinter = fingerprinter
|
||||
result.fingerprinter = fingerprinter or RequestFingerprinter()
|
||||
return result
|
||||
|
||||
@classmethod
|
||||
def from_crawler(cls, crawler):
|
||||
def from_crawler(cls, crawler: Crawler) -> Self:
|
||||
try:
|
||||
return cls.from_settings(
|
||||
crawler.settings,
|
||||
|
|
|
|||
|
|
@ -4,6 +4,7 @@ Scrapy core exceptions
|
|||
These exceptions are documented in docs/topics/exceptions.rst. Please don't add
|
||||
new exceptions here without documenting them there.
|
||||
"""
|
||||
from typing import Any
|
||||
|
||||
# Internal
|
||||
|
||||
|
|
@ -51,7 +52,7 @@ class StopDownload(Exception):
|
|||
should be handled by the request errback. Note that 'fail' is a keyword-only argument.
|
||||
"""
|
||||
|
||||
def __init__(self, *, fail=True):
|
||||
def __init__(self, *, fail: bool = True):
|
||||
super().__init__()
|
||||
self.fail = fail
|
||||
|
||||
|
|
@ -77,7 +78,7 @@ class NotSupported(Exception):
|
|||
class UsageError(Exception):
|
||||
"""To indicate a command-line usage error"""
|
||||
|
||||
def __init__(self, *a, **kw):
|
||||
def __init__(self, *a: Any, **kw: Any):
|
||||
self.print_help = kw.pop("print_help", True)
|
||||
super().__init__(*a, **kw)
|
||||
|
||||
|
|
|
|||
|
|
@ -7,13 +7,11 @@ import io
|
|||
import marshal
|
||||
import pickle
|
||||
import pprint
|
||||
import warnings
|
||||
from collections.abc import Mapping
|
||||
from xml.sax.saxutils import XMLGenerator
|
||||
|
||||
from itemadapter import ItemAdapter, is_item
|
||||
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
from scrapy.item import Item
|
||||
from scrapy.utils.python import is_listlike, to_bytes, to_unicode
|
||||
from scrapy.utils.serialize import ScrapyJSONEncoder
|
||||
|
|
@ -133,6 +131,13 @@ class JsonItemExporter(BaseItemExporter):
|
|||
if self.indent is not None:
|
||||
self.file.write(b"\n")
|
||||
|
||||
def _add_comma_after_first(self):
|
||||
if self.first_item:
|
||||
self.first_item = False
|
||||
else:
|
||||
self.file.write(b",")
|
||||
self._beautify_newline()
|
||||
|
||||
def start_exporting(self):
|
||||
self.file.write(b"[")
|
||||
self._beautify_newline()
|
||||
|
|
@ -142,14 +147,10 @@ class JsonItemExporter(BaseItemExporter):
|
|||
self.file.write(b"]")
|
||||
|
||||
def export_item(self, item):
|
||||
if self.first_item:
|
||||
self.first_item = False
|
||||
else:
|
||||
self.file.write(b",")
|
||||
self._beautify_newline()
|
||||
itemdict = dict(self._get_serialized_fields(item))
|
||||
data = self.encoder.encode(itemdict)
|
||||
self.file.write(to_bytes(data, self.encoding))
|
||||
data = to_bytes(self.encoder.encode(itemdict), self.encoding)
|
||||
self._add_comma_after_first()
|
||||
self.file.write(data)
|
||||
|
||||
|
||||
class XmlItemExporter(BaseItemExporter):
|
||||
|
|
@ -255,6 +256,9 @@ class CsvItemExporter(BaseItemExporter):
|
|||
values = list(self._build_row(x for _, x in fields))
|
||||
self.csv_writer.writerow(values)
|
||||
|
||||
def finish_exporting(self):
|
||||
self.stream.detach() # Avoid closing the wrapped file.
|
||||
|
||||
def _build_row(self, values):
|
||||
for s in values:
|
||||
try:
|
||||
|
|
@ -324,13 +328,7 @@ class PythonItemExporter(BaseItemExporter):
|
|||
"""
|
||||
|
||||
def _configure(self, options, dont_fail=False):
|
||||
self.binary = options.pop("binary", True)
|
||||
super()._configure(options, dont_fail)
|
||||
if self.binary:
|
||||
warnings.warn(
|
||||
"PythonItemExporter will drop support for binary export in the future",
|
||||
ScrapyDeprecationWarning,
|
||||
)
|
||||
if not self.encoding:
|
||||
self.encoding = "utf-8"
|
||||
|
||||
|
|
@ -345,18 +343,14 @@ class PythonItemExporter(BaseItemExporter):
|
|||
return dict(self._serialize_item(value))
|
||||
if is_listlike(value):
|
||||
return [self._serialize_value(v) for v in value]
|
||||
encode_func = to_bytes if self.binary else to_unicode
|
||||
if isinstance(value, (str, bytes)):
|
||||
return encode_func(value, encoding=self.encoding)
|
||||
return to_unicode(value, encoding=self.encoding)
|
||||
return value
|
||||
|
||||
def _serialize_item(self, item):
|
||||
for key, value in ItemAdapter(item).items():
|
||||
key = to_bytes(key) if self.binary else key
|
||||
yield key, self._serialize_value(value)
|
||||
|
||||
def export_item(self, item):
|
||||
result = dict(self._get_serialized_fields(item))
|
||||
if self.binary:
|
||||
result = dict(self._serialize_item(result))
|
||||
return result
|
||||
|
|
|
|||
|
|
@ -4,11 +4,14 @@ conditions are met.
|
|||
See documentation in docs/topics/extensions.rst
|
||||
"""
|
||||
|
||||
import logging
|
||||
from collections import defaultdict
|
||||
|
||||
from scrapy import signals
|
||||
from scrapy.exceptions import NotConfigured
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class CloseSpider:
|
||||
def __init__(self, crawler):
|
||||
|
|
@ -19,6 +22,7 @@ class CloseSpider:
|
|||
"itemcount": crawler.settings.getint("CLOSESPIDER_ITEMCOUNT"),
|
||||
"pagecount": crawler.settings.getint("CLOSESPIDER_PAGECOUNT"),
|
||||
"errorcount": crawler.settings.getint("CLOSESPIDER_ERRORCOUNT"),
|
||||
"timeout_no_item": crawler.settings.getint("CLOSESPIDER_TIMEOUT_NO_ITEM"),
|
||||
}
|
||||
|
||||
if not any(self.close_on.values()):
|
||||
|
|
@ -34,6 +38,15 @@ class CloseSpider:
|
|||
crawler.signals.connect(self.spider_opened, signal=signals.spider_opened)
|
||||
if self.close_on.get("itemcount"):
|
||||
crawler.signals.connect(self.item_scraped, signal=signals.item_scraped)
|
||||
if self.close_on.get("timeout_no_item"):
|
||||
self.timeout_no_item = self.close_on["timeout_no_item"]
|
||||
self.items_in_period = 0
|
||||
crawler.signals.connect(
|
||||
self.spider_opened_no_item, signal=signals.spider_opened
|
||||
)
|
||||
crawler.signals.connect(
|
||||
self.item_scraped_no_item, signal=signals.item_scraped
|
||||
)
|
||||
crawler.signals.connect(self.spider_closed, signal=signals.spider_closed)
|
||||
|
||||
@classmethod
|
||||
|
|
@ -69,3 +82,31 @@ class CloseSpider:
|
|||
task = getattr(self, "task", False)
|
||||
if task and task.active():
|
||||
task.cancel()
|
||||
|
||||
task_no_item = getattr(self, "task_no_item", False)
|
||||
if task_no_item and task_no_item.running:
|
||||
task_no_item.stop()
|
||||
|
||||
def spider_opened_no_item(self, spider):
|
||||
from twisted.internet import task
|
||||
|
||||
self.task_no_item = task.LoopingCall(self._count_items_produced, spider)
|
||||
self.task_no_item.start(self.timeout_no_item, now=False)
|
||||
|
||||
logger.info(
|
||||
f"Spider will stop when no items are produced after "
|
||||
f"{self.timeout_no_item} seconds."
|
||||
)
|
||||
|
||||
def item_scraped_no_item(self, item, spider):
|
||||
self.items_in_period += 1
|
||||
|
||||
def _count_items_produced(self, spider):
|
||||
if self.items_in_period >= 1:
|
||||
self.items_in_period = 0
|
||||
else:
|
||||
logger.info(
|
||||
f"Closing spider since no items were produced in the last "
|
||||
f"{self.timeout_no_item} seconds."
|
||||
)
|
||||
self.crawler.engine.close_spider(spider, "closespider_timeout_no_item")
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
"""
|
||||
Extension for collecting core stats like items scraped and start/finish times
|
||||
"""
|
||||
from datetime import datetime
|
||||
from datetime import datetime, timezone
|
||||
|
||||
from scrapy import signals
|
||||
|
||||
|
|
@ -22,11 +22,11 @@ class CoreStats:
|
|||
return o
|
||||
|
||||
def spider_opened(self, spider):
|
||||
self.start_time = datetime.utcnow()
|
||||
self.start_time = datetime.now(tz=timezone.utc)
|
||||
self.stats.set_value("start_time", self.start_time, spider=spider)
|
||||
|
||||
def spider_closed(self, spider, reason):
|
||||
finish_time = datetime.utcnow()
|
||||
finish_time = datetime.now(tz=timezone.utc)
|
||||
elapsed_time = finish_time - self.start_time
|
||||
elapsed_time_seconds = elapsed_time.total_seconds()
|
||||
self.stats.set_value(
|
||||
|
|
|
|||
|
|
@ -8,10 +8,10 @@ import logging
|
|||
import re
|
||||
import sys
|
||||
import warnings
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path, PureWindowsPath
|
||||
from tempfile import NamedTemporaryFile
|
||||
from typing import IO, Any, Callable, List, Optional, Tuple, Union
|
||||
from typing import IO, Any, Callable, Dict, List, Optional, Tuple, Union
|
||||
from urllib.parse import unquote, urlparse
|
||||
|
||||
from twisted.internet import defer, threads
|
||||
|
|
@ -33,6 +33,13 @@ from scrapy.utils.python import get_func_args, without_none_values
|
|||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
try:
|
||||
import boto3 # noqa: F401
|
||||
|
||||
IS_BOTO3_AVAILABLE = True
|
||||
except ImportError:
|
||||
IS_BOTO3_AVAILABLE = False
|
||||
|
||||
|
||||
def build_storage(builder, uri, *args, feed_options=None, preargs=(), **kwargs):
|
||||
argument_names = get_func_args(builder)
|
||||
|
|
@ -165,6 +172,7 @@ class S3FeedStorage(BlockingFeedStorage):
|
|||
*,
|
||||
feed_options=None,
|
||||
session_token=None,
|
||||
region_name=None,
|
||||
):
|
||||
if not is_botocore_available():
|
||||
raise NotConfigured("missing botocore library")
|
||||
|
|
@ -176,16 +184,41 @@ class S3FeedStorage(BlockingFeedStorage):
|
|||
self.keyname = u.path[1:] # remove first "/"
|
||||
self.acl = acl
|
||||
self.endpoint_url = endpoint_url
|
||||
import botocore.session
|
||||
self.region_name = region_name
|
||||
|
||||
if IS_BOTO3_AVAILABLE:
|
||||
import boto3.session
|
||||
|
||||
session = boto3.session.Session()
|
||||
|
||||
self.s3_client = session.client(
|
||||
"s3",
|
||||
aws_access_key_id=self.access_key,
|
||||
aws_secret_access_key=self.secret_key,
|
||||
aws_session_token=self.session_token,
|
||||
endpoint_url=self.endpoint_url,
|
||||
region_name=self.region_name,
|
||||
)
|
||||
else:
|
||||
warnings.warn(
|
||||
"`botocore` usage has been deprecated for S3 feed "
|
||||
"export, please use `boto3` to avoid problems",
|
||||
category=ScrapyDeprecationWarning,
|
||||
)
|
||||
|
||||
import botocore.session
|
||||
|
||||
session = botocore.session.get_session()
|
||||
|
||||
self.s3_client = session.create_client(
|
||||
"s3",
|
||||
aws_access_key_id=self.access_key,
|
||||
aws_secret_access_key=self.secret_key,
|
||||
aws_session_token=self.session_token,
|
||||
endpoint_url=self.endpoint_url,
|
||||
region_name=self.region_name,
|
||||
)
|
||||
|
||||
session = botocore.session.get_session()
|
||||
self.s3_client = session.create_client(
|
||||
"s3",
|
||||
aws_access_key_id=self.access_key,
|
||||
aws_secret_access_key=self.secret_key,
|
||||
aws_session_token=self.session_token,
|
||||
endpoint_url=self.endpoint_url,
|
||||
)
|
||||
if feed_options and feed_options.get("overwrite", True) is False:
|
||||
logger.warning(
|
||||
"S3 does not support appending to files. To "
|
||||
|
|
@ -203,15 +236,22 @@ class S3FeedStorage(BlockingFeedStorage):
|
|||
session_token=crawler.settings["AWS_SESSION_TOKEN"],
|
||||
acl=crawler.settings["FEED_STORAGE_S3_ACL"] or None,
|
||||
endpoint_url=crawler.settings["AWS_ENDPOINT_URL"] or None,
|
||||
region_name=crawler.settings["AWS_REGION_NAME"] or None,
|
||||
feed_options=feed_options,
|
||||
)
|
||||
|
||||
def _store_in_thread(self, file):
|
||||
file.seek(0)
|
||||
kwargs = {"ACL": self.acl} if self.acl else {}
|
||||
self.s3_client.put_object(
|
||||
Bucket=self.bucketname, Key=self.keyname, Body=file, **kwargs
|
||||
)
|
||||
if IS_BOTO3_AVAILABLE:
|
||||
kwargs = {"ExtraArgs": {"ACL": self.acl}} if self.acl else {}
|
||||
self.s3_client.upload_fileobj(
|
||||
Bucket=self.bucketname, Key=self.keyname, Fileobj=file, **kwargs
|
||||
)
|
||||
else:
|
||||
kwargs = {"ACL": self.acl} if self.acl else {}
|
||||
self.s3_client.put_object(
|
||||
Bucket=self.bucketname, Key=self.keyname, Body=file, **kwargs
|
||||
)
|
||||
file.close()
|
||||
|
||||
|
||||
|
|
@ -242,15 +282,23 @@ class GCSFeedStorage(BlockingFeedStorage):
|
|||
|
||||
|
||||
class FTPFeedStorage(BlockingFeedStorage):
|
||||
def __init__(self, uri, use_active_mode=False, *, feed_options=None):
|
||||
def __init__(
|
||||
self,
|
||||
uri: str,
|
||||
use_active_mode: bool = False,
|
||||
*,
|
||||
feed_options: Optional[Dict[str, Any]] = None,
|
||||
):
|
||||
u = urlparse(uri)
|
||||
self.host = u.hostname
|
||||
self.port = int(u.port or "21")
|
||||
self.username = u.username
|
||||
self.password = unquote(u.password or "")
|
||||
self.path = u.path
|
||||
self.use_active_mode = use_active_mode
|
||||
self.overwrite = not feed_options or feed_options.get("overwrite", True)
|
||||
if not u.hostname:
|
||||
raise ValueError(f"Got a storage URI without a hostname: {uri}")
|
||||
self.host: str = u.hostname
|
||||
self.port: int = int(u.port or "21")
|
||||
self.username: str = u.username or ""
|
||||
self.password: str = unquote(u.password or "")
|
||||
self.path: str = u.path
|
||||
self.use_active_mode: bool = use_active_mode
|
||||
self.overwrite: bool = not feed_options or feed_options.get("overwrite", True)
|
||||
|
||||
@classmethod
|
||||
def from_crawler(cls, crawler, uri, *, feed_options=None):
|
||||
|
|
@ -277,8 +325,6 @@ class FTPFeedStorage(BlockingFeedStorage):
|
|||
class FeedSlot:
|
||||
def __init__(
|
||||
self,
|
||||
file,
|
||||
exporter,
|
||||
storage,
|
||||
uri,
|
||||
format,
|
||||
|
|
@ -286,9 +332,14 @@ class FeedSlot:
|
|||
batch_id,
|
||||
uri_template,
|
||||
filter,
|
||||
feed_options,
|
||||
spider,
|
||||
exporters,
|
||||
settings,
|
||||
crawler,
|
||||
):
|
||||
self.file = file
|
||||
self.exporter = exporter
|
||||
self.file = None
|
||||
self.exporter = None
|
||||
self.storage = storage
|
||||
# feed params
|
||||
self.batch_id = batch_id
|
||||
|
|
@ -297,15 +348,44 @@ class FeedSlot:
|
|||
self.uri_template = uri_template
|
||||
self.uri = uri
|
||||
self.filter = filter
|
||||
# exporter params
|
||||
self.feed_options = feed_options
|
||||
self.spider = spider
|
||||
self.exporters = exporters
|
||||
self.settings = settings
|
||||
self.crawler = crawler
|
||||
# flags
|
||||
self.itemcount = 0
|
||||
self._exporting = False
|
||||
self._fileloaded = False
|
||||
|
||||
def start_exporting(self):
|
||||
if not self._fileloaded:
|
||||
self.file = self.storage.open(self.spider)
|
||||
if "postprocessing" in self.feed_options:
|
||||
self.file = PostProcessingManager(
|
||||
self.feed_options["postprocessing"], self.file, self.feed_options
|
||||
)
|
||||
self.exporter = self._get_exporter(
|
||||
file=self.file,
|
||||
format=self.feed_options["format"],
|
||||
fields_to_export=self.feed_options["fields"],
|
||||
encoding=self.feed_options["encoding"],
|
||||
indent=self.feed_options["indent"],
|
||||
**self.feed_options["item_export_kwargs"],
|
||||
)
|
||||
self._fileloaded = True
|
||||
|
||||
if not self._exporting:
|
||||
self.exporter.start_exporting()
|
||||
self._exporting = True
|
||||
|
||||
def _get_instance(self, objcls, *args, **kwargs):
|
||||
return create_instance(objcls, self.settings, self.crawler, *args, **kwargs)
|
||||
|
||||
def _get_exporter(self, file, format, *args, **kwargs):
|
||||
return self._get_instance(self.exporters[format], file, *args, **kwargs)
|
||||
|
||||
def finish_exporting(self):
|
||||
if self._exporting:
|
||||
self.exporter.finish_exporting()
|
||||
|
|
@ -347,7 +427,9 @@ class FeedExporter:
|
|||
category=ScrapyDeprecationWarning,
|
||||
stacklevel=2,
|
||||
)
|
||||
uri = str(self.settings["FEED_URI"]) # handle pathlib.Path objects
|
||||
uri = self.settings["FEED_URI"]
|
||||
# handle pathlib.Path objects
|
||||
uri = str(uri) if not isinstance(uri, Path) else uri.absolute().as_uri()
|
||||
feed_options = {"format": self.settings.get("FEED_FORMAT", "jsonlines")}
|
||||
self.feeds[uri] = feed_complete_default_values_from_settings(
|
||||
feed_options, self.settings
|
||||
|
|
@ -357,7 +439,8 @@ class FeedExporter:
|
|||
|
||||
# 'FEEDS' setting takes precedence over 'FEED_URI'
|
||||
for uri, feed_options in self.settings.getdict("FEEDS").items():
|
||||
uri = str(uri) # handle pathlib.Path objects
|
||||
# handle pathlib.Path objects
|
||||
uri = str(uri) if not isinstance(uri, Path) else uri.absolute().as_uri()
|
||||
self.feeds[uri] = feed_complete_default_values_from_settings(
|
||||
feed_options, self.settings
|
||||
)
|
||||
|
|
@ -406,11 +489,16 @@ class FeedExporter:
|
|||
return slot_.file.file
|
||||
return slot_.file
|
||||
|
||||
slot.finish_exporting()
|
||||
if not slot.itemcount and not slot.store_empty:
|
||||
# We need to call slot.storage.store nonetheless to get the file
|
||||
# properly closed.
|
||||
return defer.maybeDeferred(slot.storage.store, get_file(slot))
|
||||
if slot.itemcount:
|
||||
# Normal case
|
||||
slot.finish_exporting()
|
||||
elif slot.store_empty and slot.batch_id == 1:
|
||||
# Need to store the empty file
|
||||
slot.start_exporting()
|
||||
slot.finish_exporting()
|
||||
else:
|
||||
# In this case, the file is not stored, so no processing is required.
|
||||
return None
|
||||
|
||||
logmsg = f"{slot.format} feed ({slot.itemcount} items) in: {slot.uri}"
|
||||
d = defer.maybeDeferred(slot.storage.store, get_file(slot))
|
||||
|
|
@ -455,23 +543,7 @@ class FeedExporter:
|
|||
:param uri_template: template of uri which contains %(batch_time)s or %(batch_id)d to create new uri
|
||||
"""
|
||||
storage = self._get_storage(uri, feed_options)
|
||||
file = storage.open(spider)
|
||||
if "postprocessing" in feed_options:
|
||||
file = PostProcessingManager(
|
||||
feed_options["postprocessing"], file, feed_options
|
||||
)
|
||||
|
||||
exporter = self._get_exporter(
|
||||
file=file,
|
||||
format=feed_options["format"],
|
||||
fields_to_export=feed_options["fields"],
|
||||
encoding=feed_options["encoding"],
|
||||
indent=feed_options["indent"],
|
||||
**feed_options["item_export_kwargs"],
|
||||
)
|
||||
slot = FeedSlot(
|
||||
file=file,
|
||||
exporter=exporter,
|
||||
storage=storage,
|
||||
uri=uri,
|
||||
format=feed_options["format"],
|
||||
|
|
@ -479,9 +551,12 @@ class FeedExporter:
|
|||
batch_id=batch_id,
|
||||
uri_template=uri_template,
|
||||
filter=self.filters[uri_template],
|
||||
feed_options=feed_options,
|
||||
spider=spider,
|
||||
exporters=self.exporters,
|
||||
settings=self.settings,
|
||||
crawler=getattr(self, "crawler", None),
|
||||
)
|
||||
if slot.store_empty:
|
||||
slot.start_exporting()
|
||||
return slot
|
||||
|
||||
def item_scraped(self, item, spider):
|
||||
|
|
@ -553,7 +628,7 @@ class FeedExporter:
|
|||
|
||||
def _storage_supported(self, uri, feed_options):
|
||||
scheme = urlparse(uri).scheme
|
||||
if scheme in self.storages:
|
||||
if scheme in self.storages or PureWindowsPath(uri).drive:
|
||||
try:
|
||||
self._get_storage(uri, feed_options)
|
||||
return True
|
||||
|
|
@ -565,21 +640,13 @@ class FeedExporter:
|
|||
else:
|
||||
logger.error("Unknown feed storage scheme: %(scheme)s", {"scheme": scheme})
|
||||
|
||||
def _get_instance(self, objcls, *args, **kwargs):
|
||||
return create_instance(
|
||||
objcls, self.settings, getattr(self, "crawler", None), *args, **kwargs
|
||||
)
|
||||
|
||||
def _get_exporter(self, file, format, *args, **kwargs):
|
||||
return self._get_instance(self.exporters[format], file, *args, **kwargs)
|
||||
|
||||
def _get_storage(self, uri, feed_options):
|
||||
"""Fork of create_instance specific to feed storage classes
|
||||
|
||||
It supports not passing the *feed_options* parameters to classes that
|
||||
do not support it, and issuing a deprecation warning instead.
|
||||
"""
|
||||
feedcls = self.storages[urlparse(uri).scheme]
|
||||
feedcls = self.storages.get(urlparse(uri).scheme, self.storages["file"])
|
||||
crawler = getattr(self, "crawler", None)
|
||||
|
||||
def build_instance(builder, *preargs):
|
||||
|
|
@ -609,25 +676,16 @@ class FeedExporter:
|
|||
params = {}
|
||||
for k in dir(spider):
|
||||
params[k] = getattr(spider, k)
|
||||
utc_now = datetime.utcnow()
|
||||
utc_now = datetime.now(tz=timezone.utc)
|
||||
params["time"] = utc_now.replace(microsecond=0).isoformat().replace(":", "-")
|
||||
params["batch_time"] = utc_now.isoformat().replace(":", "-")
|
||||
params["batch_id"] = slot.batch_id + 1 if slot is not None else 1
|
||||
original_params = params.copy()
|
||||
uripar_function = (
|
||||
load_object(uri_params_function)
|
||||
if uri_params_function
|
||||
else lambda params, _: params
|
||||
)
|
||||
new_params = uripar_function(params, spider)
|
||||
if new_params is None or original_params != params:
|
||||
warnings.warn(
|
||||
"Modifying the params dictionary in-place in the function defined in "
|
||||
"the FEED_URI_PARAMS setting or in the uri_params key of the FEEDS "
|
||||
"setting is deprecated. The function must return a new dictionary "
|
||||
"instead.",
|
||||
category=ScrapyDeprecationWarning,
|
||||
)
|
||||
return new_params if new_params is not None else params
|
||||
|
||||
def _load_filter(self, feed_options):
|
||||
|
|
|
|||
|
|
@ -2,7 +2,7 @@ from collections.abc import Mapping
|
|||
|
||||
from w3lib.http import headers_dict_to_raw
|
||||
|
||||
from scrapy.utils.datatypes import CaselessDict
|
||||
from scrapy.utils.datatypes import CaseInsensitiveDict, CaselessDict
|
||||
from scrapy.utils.python import to_unicode
|
||||
|
||||
|
||||
|
|
@ -88,7 +88,7 @@ class Headers(CaselessDict):
|
|||
"""Return headers as a CaselessDict with unicode keys
|
||||
and unicode values. Multiple values are joined with ','.
|
||||
"""
|
||||
return CaselessDict(
|
||||
return CaseInsensitiveDict(
|
||||
(
|
||||
to_unicode(key, encoding=self.encoding),
|
||||
to_unicode(b",".join(value), encoding=self.encoding),
|
||||
|
|
|
|||
|
|
@ -4,10 +4,11 @@ discovering (through HTTP headers) to base Response class.
|
|||
|
||||
See documentation in docs/topics/request-response.rst
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from contextlib import suppress
|
||||
from typing import Generator, Tuple
|
||||
from typing import TYPE_CHECKING, Any, Generator, Optional, Tuple
|
||||
from urllib.parse import urljoin
|
||||
|
||||
import parsel
|
||||
|
|
@ -25,6 +26,9 @@ from scrapy.http.response import Response
|
|||
from scrapy.utils.python import memoizemethod_noargs, to_unicode
|
||||
from scrapy.utils.response import get_base_url
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from scrapy.selector import Selector
|
||||
|
||||
_NONE = object()
|
||||
|
||||
|
||||
|
|
@ -34,11 +38,11 @@ class TextResponse(Response):
|
|||
|
||||
attributes: Tuple[str, ...] = Response.attributes + ("encoding",)
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
def __init__(self, *args: Any, **kwargs: Any):
|
||||
self._encoding = kwargs.pop("encoding", None)
|
||||
self._cached_benc = None
|
||||
self._cached_ubody = None
|
||||
self._cached_selector = None
|
||||
self._cached_benc: Optional[str] = None
|
||||
self._cached_ubody: Optional[str] = None
|
||||
self._cached_selector: Optional[Selector] = None
|
||||
super().__init__(*args, **kwargs)
|
||||
|
||||
def _set_url(self, url):
|
||||
|
|
@ -82,7 +86,7 @@ class TextResponse(Response):
|
|||
return self._cached_decoded_json
|
||||
|
||||
@property
|
||||
def text(self):
|
||||
def text(self) -> str:
|
||||
"""Body as unicode"""
|
||||
# access self.encoding before _cached_ubody to make sure
|
||||
# _body_inferred_encoding is called
|
||||
|
|
|
|||
|
|
@ -4,6 +4,7 @@ This module defines the Link object used in Link extractors.
|
|||
For actual link extractors implementation see scrapy.linkextractors, or
|
||||
its documentation in: docs/topics/link-extractors.rst
|
||||
"""
|
||||
from typing import Any
|
||||
|
||||
|
||||
class Link:
|
||||
|
|
@ -26,16 +27,20 @@ class Link:
|
|||
|
||||
__slots__ = ["url", "text", "fragment", "nofollow"]
|
||||
|
||||
def __init__(self, url, text="", fragment="", nofollow=False):
|
||||
def __init__(
|
||||
self, url: str, text: str = "", fragment: str = "", nofollow: bool = False
|
||||
):
|
||||
if not isinstance(url, str):
|
||||
got = url.__class__.__name__
|
||||
raise TypeError(f"Link urls must be str objects, got {got}")
|
||||
self.url = url
|
||||
self.text = text
|
||||
self.fragment = fragment
|
||||
self.nofollow = nofollow
|
||||
self.url: str = url
|
||||
self.text: str = text
|
||||
self.fragment: str = fragment
|
||||
self.nofollow: bool = nofollow
|
||||
|
||||
def __eq__(self, other):
|
||||
def __eq__(self, other: Any) -> bool:
|
||||
if not isinstance(other, Link):
|
||||
raise NotImplementedError
|
||||
return (
|
||||
self.url == other.url
|
||||
and self.text == other.text
|
||||
|
|
@ -43,12 +48,12 @@ class Link:
|
|||
and self.nofollow == other.nofollow
|
||||
)
|
||||
|
||||
def __hash__(self):
|
||||
def __hash__(self) -> int:
|
||||
return (
|
||||
hash(self.url) ^ hash(self.text) ^ hash(self.fragment) ^ hash(self.nofollow)
|
||||
)
|
||||
|
||||
def __repr__(self):
|
||||
def __repr__(self) -> str:
|
||||
return (
|
||||
f"Link(url={self.url!r}, text={self.text!r}, "
|
||||
f"fragment={self.fragment!r}, nofollow={self.nofollow!r})"
|
||||
|
|
|
|||
|
|
@ -1,6 +1,8 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
from typing import Any, Dict, Optional, Union
|
||||
from typing import TYPE_CHECKING, Any, Dict, Optional, Union
|
||||
|
||||
from twisted.python.failure import Failure
|
||||
|
||||
|
|
@ -8,6 +10,13 @@ from scrapy import Request, Spider
|
|||
from scrapy.http import Response
|
||||
from scrapy.utils.request import referer_str
|
||||
|
||||
if TYPE_CHECKING:
|
||||
# typing.Self requires Python 3.11
|
||||
from typing_extensions import Self
|
||||
|
||||
from scrapy.crawler import Crawler
|
||||
|
||||
|
||||
SCRAPEDMSG = "Scraped from %(src)s" + os.linesep + "%(item)s"
|
||||
DROPPEDMSG = "Dropped: %(exception)s" + os.linesep + "%(item)s"
|
||||
CRAWLEDMSG = "Crawled (%(status)s) %(request)s%(request_flags)s (referer: %(referer)s)%(response_flags)s"
|
||||
|
|
@ -105,7 +114,7 @@ class LogFormatter:
|
|||
}
|
||||
|
||||
def item_error(
|
||||
self, item: Any, exception, response: Response, spider: Spider
|
||||
self, item: Any, exception: BaseException, response: Response, spider: Spider
|
||||
) -> dict:
|
||||
"""Logs a message when an item causes an error while it is passing
|
||||
through the item pipeline.
|
||||
|
|
@ -161,5 +170,5 @@ class LogFormatter:
|
|||
}
|
||||
|
||||
@classmethod
|
||||
def from_crawler(cls, crawler):
|
||||
def from_crawler(cls, crawler: Crawler) -> Self:
|
||||
return cls()
|
||||
|
|
|
|||
|
|
@ -1,7 +1,21 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import pprint
|
||||
from collections import defaultdict, deque
|
||||
from typing import Any, Callable, Deque, Dict, Iterable, List, Tuple, Union, cast
|
||||
from typing import (
|
||||
TYPE_CHECKING,
|
||||
Any,
|
||||
Callable,
|
||||
Deque,
|
||||
Dict,
|
||||
Iterable,
|
||||
List,
|
||||
Optional,
|
||||
Tuple,
|
||||
Union,
|
||||
cast,
|
||||
)
|
||||
|
||||
from twisted.internet.defer import Deferred
|
||||
|
||||
|
|
@ -11,6 +25,13 @@ from scrapy.settings import Settings
|
|||
from scrapy.utils.defer import process_chain, process_parallel
|
||||
from scrapy.utils.misc import create_instance, load_object
|
||||
|
||||
if TYPE_CHECKING:
|
||||
# typing.Self requires Python 3.11
|
||||
from typing_extensions import Self
|
||||
|
||||
from scrapy.crawler import Crawler
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
|
|
@ -34,7 +55,9 @@ class MiddlewareManager:
|
|||
raise NotImplementedError
|
||||
|
||||
@classmethod
|
||||
def from_settings(cls, settings: Settings, crawler=None):
|
||||
def from_settings(
|
||||
cls, settings: Settings, crawler: Optional[Crawler] = None
|
||||
) -> Self:
|
||||
mwlist = cls._get_mwlist_from_settings(settings)
|
||||
middlewares = []
|
||||
enabled = []
|
||||
|
|
@ -46,10 +69,9 @@ class MiddlewareManager:
|
|||
enabled.append(clspath)
|
||||
except NotConfigured as e:
|
||||
if e.args:
|
||||
clsname = clspath.split(".")[-1]
|
||||
logger.warning(
|
||||
"Disabled %(clsname)s: %(eargs)s",
|
||||
{"clsname": clsname, "eargs": e.args[0]},
|
||||
"Disabled %(clspath)s: %(eargs)s",
|
||||
{"clspath": clspath, "eargs": e.args[0]},
|
||||
extra={"crawler": crawler},
|
||||
)
|
||||
|
||||
|
|
@ -64,7 +86,7 @@ class MiddlewareManager:
|
|||
return cls(*middlewares)
|
||||
|
||||
@classmethod
|
||||
def from_crawler(cls, crawler):
|
||||
def from_crawler(cls, crawler: Crawler) -> Self:
|
||||
return cls.from_settings(crawler.settings, crawler)
|
||||
|
||||
def _add_middleware(self, mw: Any) -> None:
|
||||
|
|
|
|||
|
|
@ -28,7 +28,7 @@ from scrapy.http.request import NO_CALLBACK
|
|||
from scrapy.pipelines.media import MediaPipeline
|
||||
from scrapy.settings import Settings
|
||||
from scrapy.utils.boto import is_botocore_available
|
||||
from scrapy.utils.datatypes import CaselessDict
|
||||
from scrapy.utils.datatypes import CaseInsensitiveDict
|
||||
from scrapy.utils.ftp import ftp_store_file
|
||||
from scrapy.utils.log import failure_to_exc_info
|
||||
from scrapy.utils.misc import md5sum
|
||||
|
|
@ -155,7 +155,7 @@ class S3FilesStore:
|
|||
def _headers_to_botocore_kwargs(self, headers):
|
||||
"""Convert headers to botocore keyword arguments."""
|
||||
# This is required while we need to support both boto and botocore.
|
||||
mapping = CaselessDict(
|
||||
mapping = CaseInsensitiveDict(
|
||||
{
|
||||
"Content-Type": "ContentType",
|
||||
"Cache-Control": "CacheControl",
|
||||
|
|
|
|||
|
|
@ -1,3 +1,5 @@
|
|||
from typing import Any
|
||||
|
||||
from twisted.internet import defer
|
||||
from twisted.internet.base import ThreadedResolver
|
||||
from twisted.internet.interfaces import (
|
||||
|
|
@ -11,7 +13,7 @@ from zope.interface.declarations import implementer, provider
|
|||
from scrapy.utils.datatypes import LocalCache
|
||||
|
||||
# TODO: cache misses
|
||||
dnscache = LocalCache(10000)
|
||||
dnscache: LocalCache[str, Any] = LocalCache(10000)
|
||||
|
||||
|
||||
@implementer(IResolverSimple)
|
||||
|
|
@ -36,7 +38,7 @@ class CachingThreadedResolver(ThreadedResolver):
|
|||
def install_on_reactor(self):
|
||||
self.reactor.installResolver(self)
|
||||
|
||||
def getHostByName(self, name, timeout=None):
|
||||
def getHostByName(self, name: str, timeout=None):
|
||||
if name in dnscache:
|
||||
return defer.succeed(dnscache[name])
|
||||
# in Twisted<=16.6, getHostByName() is always called with
|
||||
|
|
@ -110,7 +112,7 @@ class CachingHostnameResolver:
|
|||
def resolveHostName(
|
||||
self,
|
||||
resolutionReceiver,
|
||||
hostName,
|
||||
hostName: str,
|
||||
portNumber=0,
|
||||
addressTypes=None,
|
||||
transportSemantics="TCP",
|
||||
|
|
|
|||
|
|
@ -5,6 +5,7 @@ based on different criteria.
|
|||
from io import StringIO
|
||||
from mimetypes import MimeTypes
|
||||
from pkgutil import get_data
|
||||
from typing import Dict, Mapping, Optional, Type, Union
|
||||
|
||||
from scrapy.http import Response
|
||||
from scrapy.utils.misc import load_object
|
||||
|
|
@ -29,15 +30,19 @@ class ResponseTypes:
|
|||
"text/*": "scrapy.http.TextResponse",
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.classes = {}
|
||||
self.mimetypes = MimeTypes()
|
||||
mimedata = get_data("scrapy", "mime.types").decode("utf8")
|
||||
self.mimetypes.readfp(StringIO(mimedata))
|
||||
def __init__(self) -> None:
|
||||
self.classes: Dict[str, Type[Response]] = {}
|
||||
self.mimetypes: MimeTypes = MimeTypes()
|
||||
mimedata = get_data("scrapy", "mime.types")
|
||||
if not mimedata:
|
||||
raise ValueError(
|
||||
"The mime.types file is not found in the Scrapy installation"
|
||||
)
|
||||
self.mimetypes.readfp(StringIO(mimedata.decode("utf8")))
|
||||
for mimetype, cls in self.CLASSES.items():
|
||||
self.classes[mimetype] = load_object(cls)
|
||||
|
||||
def from_mimetype(self, mimetype):
|
||||
def from_mimetype(self, mimetype: str) -> Type[Response]:
|
||||
"""Return the most appropriate Response class for the given mimetype"""
|
||||
if mimetype is None:
|
||||
return Response
|
||||
|
|
@ -46,7 +51,9 @@ class ResponseTypes:
|
|||
basetype = f"{mimetype.split('/')[0]}/*"
|
||||
return self.classes.get(basetype, Response)
|
||||
|
||||
def from_content_type(self, content_type, content_encoding=None):
|
||||
def from_content_type(
|
||||
self, content_type: Union[str, bytes], content_encoding: Optional[bytes] = None
|
||||
) -> Type[Response]:
|
||||
"""Return the most appropriate Response class from an HTTP Content-Type
|
||||
header"""
|
||||
if content_encoding:
|
||||
|
|
@ -56,7 +63,9 @@ class ResponseTypes:
|
|||
)
|
||||
return self.from_mimetype(mimetype)
|
||||
|
||||
def from_content_disposition(self, content_disposition):
|
||||
def from_content_disposition(
|
||||
self, content_disposition: Union[str, bytes]
|
||||
) -> Type[Response]:
|
||||
try:
|
||||
filename = (
|
||||
to_unicode(content_disposition, encoding="latin-1", errors="replace")
|
||||
|
|
@ -68,7 +77,7 @@ class ResponseTypes:
|
|||
except IndexError:
|
||||
return Response
|
||||
|
||||
def from_headers(self, headers):
|
||||
def from_headers(self, headers: Mapping[bytes, bytes]) -> Type[Response]:
|
||||
"""Return the most appropriate Response class by looking at the HTTP
|
||||
headers"""
|
||||
cls = Response
|
||||
|
|
@ -81,14 +90,14 @@ class ResponseTypes:
|
|||
cls = self.from_content_disposition(headers[b"Content-Disposition"])
|
||||
return cls
|
||||
|
||||
def from_filename(self, filename):
|
||||
def from_filename(self, filename: str) -> Type[Response]:
|
||||
"""Return the most appropriate Response class from a file name"""
|
||||
mimetype, encoding = self.mimetypes.guess_type(filename)
|
||||
if mimetype and not encoding:
|
||||
return self.from_mimetype(mimetype)
|
||||
return Response
|
||||
|
||||
def from_body(self, body):
|
||||
def from_body(self, body: bytes) -> Type[Response]:
|
||||
"""Try to guess the appropriate response based on the body content.
|
||||
This method is a bit magic and could be improved in the future, but
|
||||
it's not meant to be used except for special cases where response types
|
||||
|
|
@ -106,7 +115,13 @@ class ResponseTypes:
|
|||
return self.from_mimetype("text/html")
|
||||
return self.from_mimetype("text")
|
||||
|
||||
def from_args(self, headers=None, url=None, filename=None, body=None):
|
||||
def from_args(
|
||||
self,
|
||||
headers: Optional[Mapping[bytes, bytes]] = None,
|
||||
url: Optional[str] = None,
|
||||
filename: Optional[str] = None,
|
||||
body: Optional[bytes] = None,
|
||||
) -> Type[Response]:
|
||||
"""Guess the most appropriate Response class based on
|
||||
the given arguments."""
|
||||
cls = Response
|
||||
|
|
|
|||
|
|
@ -1,10 +1,11 @@
|
|||
"""
|
||||
XPath selectors based on lxml
|
||||
"""
|
||||
from typing import Any, Optional, Type, Union
|
||||
|
||||
from parsel import Selector as _ParselSelector
|
||||
|
||||
from scrapy.http import HtmlResponse, XmlResponse
|
||||
from scrapy.http import HtmlResponse, TextResponse, XmlResponse
|
||||
from scrapy.utils.python import to_bytes
|
||||
from scrapy.utils.trackref import object_ref
|
||||
|
||||
|
|
@ -13,14 +14,14 @@ __all__ = ["Selector", "SelectorList"]
|
|||
_NOT_SET = object()
|
||||
|
||||
|
||||
def _st(response, st):
|
||||
def _st(response: Optional[TextResponse], st: Optional[str]) -> str:
|
||||
if st is None:
|
||||
return "xml" if isinstance(response, XmlResponse) else "html"
|
||||
return st
|
||||
|
||||
|
||||
def _response_from_text(text, st):
|
||||
rt = XmlResponse if st == "xml" else HtmlResponse
|
||||
def _response_from_text(text: Union[str, bytes], st: Optional[str]) -> TextResponse:
|
||||
rt: Type[TextResponse] = XmlResponse if st == "xml" else HtmlResponse
|
||||
return rt(url="about:blank", encoding="utf-8", body=to_bytes(text, "utf-8"))
|
||||
|
||||
|
||||
|
|
@ -65,7 +66,14 @@ class Selector(_ParselSelector, object_ref):
|
|||
__slots__ = ["response"]
|
||||
selectorlist_cls = SelectorList
|
||||
|
||||
def __init__(self, response=None, text=None, type=None, root=_NOT_SET, **kwargs):
|
||||
def __init__(
|
||||
self,
|
||||
response: Optional[TextResponse] = None,
|
||||
text: Optional[str] = None,
|
||||
type: Optional[str] = None,
|
||||
root: Optional[Any] = _NOT_SET,
|
||||
**kwargs: Any,
|
||||
):
|
||||
if response is not None and text is not None:
|
||||
raise ValueError(
|
||||
f"{self.__class__.__name__}.__init__() received "
|
||||
|
|
|
|||
|
|
@ -1,21 +1,52 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import copy
|
||||
import json
|
||||
from collections.abc import MutableMapping
|
||||
from importlib import import_module
|
||||
from pprint import pformat
|
||||
from types import ModuleType
|
||||
from typing import (
|
||||
TYPE_CHECKING,
|
||||
Any,
|
||||
Dict,
|
||||
Iterable,
|
||||
Iterator,
|
||||
List,
|
||||
Mapping,
|
||||
MutableMapping,
|
||||
Optional,
|
||||
Tuple,
|
||||
Union,
|
||||
cast,
|
||||
)
|
||||
|
||||
from scrapy.settings import default_settings
|
||||
|
||||
SETTINGS_PRIORITIES = {
|
||||
# The key types are restricted in BaseSettings._get_key() to ones supported by JSON,
|
||||
# see https://github.com/scrapy/scrapy/issues/5383.
|
||||
_SettingsKeyT = Union[bool, float, int, str, None]
|
||||
|
||||
if TYPE_CHECKING:
|
||||
# https://github.com/python/typing/issues/445#issuecomment-1131458824
|
||||
from _typeshed import SupportsItems
|
||||
|
||||
# typing.Self requires Python 3.11
|
||||
from typing_extensions import Self
|
||||
|
||||
_SettingsInputT = Union[SupportsItems[_SettingsKeyT, Any], str, None]
|
||||
|
||||
|
||||
SETTINGS_PRIORITIES: Dict[str, int] = {
|
||||
"default": 0,
|
||||
"command": 10,
|
||||
"addon": 15,
|
||||
"project": 20,
|
||||
"spider": 30,
|
||||
"cmdline": 40,
|
||||
}
|
||||
|
||||
|
||||
def get_settings_priority(priority):
|
||||
def get_settings_priority(priority: Union[int, str]) -> int:
|
||||
"""
|
||||
Small helper function that looks up a given string priority in the
|
||||
:attr:`~scrapy.settings.SETTINGS_PRIORITIES` dictionary and returns its
|
||||
|
|
@ -34,14 +65,15 @@ class SettingsAttribute:
|
|||
for settings configuration, not this one.
|
||||
"""
|
||||
|
||||
def __init__(self, value, priority):
|
||||
self.value = value
|
||||
def __init__(self, value: Any, priority: int):
|
||||
self.value: Any = value
|
||||
self.priority: int
|
||||
if isinstance(self.value, BaseSettings):
|
||||
self.priority = max(self.value.maxpriority(), priority)
|
||||
else:
|
||||
self.priority = priority
|
||||
|
||||
def set(self, value, priority):
|
||||
def set(self, value: Any, priority: int) -> None:
|
||||
"""Sets value if priority is higher or equal than current priority."""
|
||||
if priority >= self.priority:
|
||||
if isinstance(self.value, BaseSettings):
|
||||
|
|
@ -49,11 +81,11 @@ class SettingsAttribute:
|
|||
self.value = value
|
||||
self.priority = priority
|
||||
|
||||
def __repr__(self):
|
||||
def __repr__(self) -> str:
|
||||
return f"<SettingsAttribute value={self.value!r} priority={self.priority}>"
|
||||
|
||||
|
||||
class BaseSettings(MutableMapping):
|
||||
class BaseSettings(MutableMapping[_SettingsKeyT, Any]):
|
||||
"""
|
||||
Instances of this class behave like dictionaries, but store priorities
|
||||
along with their ``(key, value)`` pairs, and can be frozen (i.e. marked
|
||||
|
|
@ -75,21 +107,25 @@ class BaseSettings(MutableMapping):
|
|||
highest priority will be retrieved.
|
||||
"""
|
||||
|
||||
def __init__(self, values=None, priority="project"):
|
||||
self.frozen = False
|
||||
self.attributes = {}
|
||||
__default = object()
|
||||
|
||||
def __init__(
|
||||
self, values: _SettingsInputT = None, priority: Union[int, str] = "project"
|
||||
):
|
||||
self.frozen: bool = False
|
||||
self.attributes: dict[_SettingsKeyT, SettingsAttribute] = {}
|
||||
if values:
|
||||
self.update(values, priority)
|
||||
|
||||
def __getitem__(self, opt_name):
|
||||
def __getitem__(self, opt_name: _SettingsKeyT) -> Any:
|
||||
if opt_name not in self:
|
||||
return None
|
||||
return self.attributes[opt_name].value
|
||||
|
||||
def __contains__(self, name):
|
||||
def __contains__(self, name: Any) -> bool:
|
||||
return name in self.attributes
|
||||
|
||||
def get(self, name, default=None):
|
||||
def get(self, name: _SettingsKeyT, default: Any = None) -> Any:
|
||||
"""
|
||||
Get a setting value without affecting its original type.
|
||||
|
||||
|
|
@ -101,7 +137,7 @@ class BaseSettings(MutableMapping):
|
|||
"""
|
||||
return self[name] if self[name] is not None else default
|
||||
|
||||
def getbool(self, name, default=False):
|
||||
def getbool(self, name: _SettingsKeyT, default: bool = False) -> bool:
|
||||
"""
|
||||
Get a setting value as a boolean.
|
||||
|
||||
|
|
@ -131,7 +167,7 @@ class BaseSettings(MutableMapping):
|
|||
"'True'/'False' and 'true'/'false'"
|
||||
)
|
||||
|
||||
def getint(self, name, default=0):
|
||||
def getint(self, name: _SettingsKeyT, default: int = 0) -> int:
|
||||
"""
|
||||
Get a setting value as an int.
|
||||
|
||||
|
|
@ -143,7 +179,7 @@ class BaseSettings(MutableMapping):
|
|||
"""
|
||||
return int(self.get(name, default))
|
||||
|
||||
def getfloat(self, name, default=0.0):
|
||||
def getfloat(self, name: _SettingsKeyT, default: float = 0.0) -> float:
|
||||
"""
|
||||
Get a setting value as a float.
|
||||
|
||||
|
|
@ -155,7 +191,9 @@ class BaseSettings(MutableMapping):
|
|||
"""
|
||||
return float(self.get(name, default))
|
||||
|
||||
def getlist(self, name, default=None):
|
||||
def getlist(
|
||||
self, name: _SettingsKeyT, default: Optional[List[Any]] = None
|
||||
) -> List[Any]:
|
||||
"""
|
||||
Get a setting value as a list. If the setting original type is a list, a
|
||||
copy of it will be returned. If it's a string it will be split by ",".
|
||||
|
|
@ -174,7 +212,9 @@ class BaseSettings(MutableMapping):
|
|||
value = value.split(",")
|
||||
return list(value)
|
||||
|
||||
def getdict(self, name, default=None):
|
||||
def getdict(
|
||||
self, name: _SettingsKeyT, default: Optional[Dict[Any, Any]] = None
|
||||
) -> Dict[Any, Any]:
|
||||
"""
|
||||
Get a setting value as a dictionary. If the setting original type is a
|
||||
dictionary, a copy of it will be returned. If it is a string it will be
|
||||
|
|
@ -195,7 +235,11 @@ class BaseSettings(MutableMapping):
|
|||
value = json.loads(value)
|
||||
return dict(value)
|
||||
|
||||
def getdictorlist(self, name, default=None):
|
||||
def getdictorlist(
|
||||
self,
|
||||
name: _SettingsKeyT,
|
||||
default: Union[Dict[Any, Any], List[Any], Tuple[Any], None] = None,
|
||||
) -> Union[Dict[Any, Any], List[Any]]:
|
||||
"""Get a setting value as either a :class:`dict` or a :class:`list`.
|
||||
|
||||
If the setting is already a dict or a list, a copy of it will be
|
||||
|
|
@ -222,24 +266,31 @@ class BaseSettings(MutableMapping):
|
|||
return {}
|
||||
if isinstance(value, str):
|
||||
try:
|
||||
return json.loads(value)
|
||||
value_loaded = json.loads(value)
|
||||
assert isinstance(value_loaded, (dict, list))
|
||||
return value_loaded
|
||||
except ValueError:
|
||||
return value.split(",")
|
||||
if isinstance(value, tuple):
|
||||
return list(value)
|
||||
assert isinstance(value, (dict, list))
|
||||
return copy.deepcopy(value)
|
||||
|
||||
def getwithbase(self, name):
|
||||
def getwithbase(self, name: _SettingsKeyT) -> "BaseSettings":
|
||||
"""Get a composition of a dictionary-like setting and its `_BASE`
|
||||
counterpart.
|
||||
|
||||
:param name: name of the dictionary-like setting
|
||||
:type name: str
|
||||
"""
|
||||
if not isinstance(name, str):
|
||||
raise ValueError(f"Base setting key must be a string, got {name}")
|
||||
compbs = BaseSettings()
|
||||
compbs.update(self[name + "_BASE"])
|
||||
compbs.update(self[name])
|
||||
return compbs
|
||||
|
||||
def getpriority(self, name):
|
||||
def getpriority(self, name: _SettingsKeyT) -> Optional[int]:
|
||||
"""
|
||||
Return the current numerical priority value of a setting, or ``None`` if
|
||||
the given ``name`` does not exist.
|
||||
|
|
@ -251,7 +302,7 @@ class BaseSettings(MutableMapping):
|
|||
return None
|
||||
return self.attributes[name].priority
|
||||
|
||||
def maxpriority(self):
|
||||
def maxpriority(self) -> int:
|
||||
"""
|
||||
Return the numerical value of the highest priority present throughout
|
||||
all settings, or the numerical value for ``default`` from
|
||||
|
|
@ -259,13 +310,15 @@ class BaseSettings(MutableMapping):
|
|||
stored.
|
||||
"""
|
||||
if len(self) > 0:
|
||||
return max(self.getpriority(name) for name in self)
|
||||
return max(cast(int, self.getpriority(name)) for name in self)
|
||||
return get_settings_priority("default")
|
||||
|
||||
def __setitem__(self, name, value):
|
||||
def __setitem__(self, name: _SettingsKeyT, value: Any) -> None:
|
||||
self.set(name, value)
|
||||
|
||||
def set(self, name, value, priority="project"):
|
||||
def set(
|
||||
self, name: _SettingsKeyT, value: Any, priority: Union[int, str] = "project"
|
||||
) -> None:
|
||||
"""
|
||||
Store a key/value attribute with a given priority.
|
||||
|
||||
|
|
@ -293,17 +346,26 @@ class BaseSettings(MutableMapping):
|
|||
else:
|
||||
self.attributes[name].set(value, priority)
|
||||
|
||||
def setdefault(self, name, default=None, priority="project"):
|
||||
def setdefault(
|
||||
self,
|
||||
name: _SettingsKeyT,
|
||||
default: Any = None,
|
||||
priority: Union[int, str] = "project",
|
||||
) -> Any:
|
||||
if name not in self:
|
||||
self.set(name, default, priority)
|
||||
return default
|
||||
|
||||
return self.attributes[name].value
|
||||
|
||||
def setdict(self, values, priority="project"):
|
||||
def setdict(
|
||||
self, values: _SettingsInputT, priority: Union[int, str] = "project"
|
||||
) -> None:
|
||||
self.update(values, priority)
|
||||
|
||||
def setmodule(self, module, priority="project"):
|
||||
def setmodule(
|
||||
self, module: Union[ModuleType, str], priority: Union[int, str] = "project"
|
||||
) -> None:
|
||||
"""
|
||||
Store settings from a module with a given priority.
|
||||
|
||||
|
|
@ -325,7 +387,8 @@ class BaseSettings(MutableMapping):
|
|||
if key.isupper():
|
||||
self.set(key, getattr(module, key), priority)
|
||||
|
||||
def update(self, values, priority="project"):
|
||||
# BaseSettings.update() doesn't support all inputs that MutableMapping.update() supports
|
||||
def update(self, values: _SettingsInputT, priority: Union[int, str] = "project") -> None: # type: ignore[override]
|
||||
"""
|
||||
Store key/value pairs with a given priority.
|
||||
|
||||
|
|
@ -349,30 +412,34 @@ class BaseSettings(MutableMapping):
|
|||
"""
|
||||
self._assert_mutability()
|
||||
if isinstance(values, str):
|
||||
values = json.loads(values)
|
||||
values = cast(dict, json.loads(values))
|
||||
if values is not None:
|
||||
if isinstance(values, BaseSettings):
|
||||
for name, value in values.items():
|
||||
self.set(name, value, values.getpriority(name))
|
||||
self.set(name, value, cast(int, values.getpriority(name)))
|
||||
else:
|
||||
for name, value in values.items():
|
||||
self.set(name, value, priority)
|
||||
|
||||
def delete(self, name, priority="project"):
|
||||
def delete(
|
||||
self, name: _SettingsKeyT, priority: Union[int, str] = "project"
|
||||
) -> None:
|
||||
if name not in self:
|
||||
raise KeyError(name)
|
||||
self._assert_mutability()
|
||||
priority = get_settings_priority(priority)
|
||||
if priority >= self.getpriority(name):
|
||||
if priority >= cast(int, self.getpriority(name)):
|
||||
del self.attributes[name]
|
||||
|
||||
def __delitem__(self, name):
|
||||
def __delitem__(self, name: _SettingsKeyT) -> None:
|
||||
self._assert_mutability()
|
||||
del self.attributes[name]
|
||||
|
||||
def _assert_mutability(self):
|
||||
def _assert_mutability(self) -> None:
|
||||
if self.frozen:
|
||||
raise TypeError("Trying to modify an immutable Settings object")
|
||||
|
||||
def copy(self):
|
||||
def copy(self) -> "Self":
|
||||
"""
|
||||
Make a deep copy of current settings.
|
||||
|
||||
|
|
@ -384,7 +451,7 @@ class BaseSettings(MutableMapping):
|
|||
"""
|
||||
return copy.deepcopy(self)
|
||||
|
||||
def freeze(self):
|
||||
def freeze(self) -> None:
|
||||
"""
|
||||
Disable further changes to the current settings.
|
||||
|
||||
|
|
@ -394,7 +461,7 @@ class BaseSettings(MutableMapping):
|
|||
"""
|
||||
self.frozen = True
|
||||
|
||||
def frozencopy(self):
|
||||
def frozencopy(self) -> "Self":
|
||||
"""
|
||||
Return an immutable copy of the current settings.
|
||||
|
||||
|
|
@ -404,26 +471,26 @@ class BaseSettings(MutableMapping):
|
|||
copy.freeze()
|
||||
return copy
|
||||
|
||||
def __iter__(self):
|
||||
def __iter__(self) -> Iterator[_SettingsKeyT]:
|
||||
return iter(self.attributes)
|
||||
|
||||
def __len__(self):
|
||||
def __len__(self) -> int:
|
||||
return len(self.attributes)
|
||||
|
||||
def _to_dict(self):
|
||||
def _to_dict(self) -> Dict[_SettingsKeyT, Any]:
|
||||
return {
|
||||
self._get_key(k): (v._to_dict() if isinstance(v, BaseSettings) else v)
|
||||
for k, v in self.items()
|
||||
}
|
||||
|
||||
def _get_key(self, key_value):
|
||||
def _get_key(self, key_value: Any) -> _SettingsKeyT:
|
||||
return (
|
||||
key_value
|
||||
if isinstance(key_value, (bool, float, int, str, type(None)))
|
||||
else str(key_value)
|
||||
)
|
||||
|
||||
def copy_to_dict(self):
|
||||
def copy_to_dict(self) -> Dict[_SettingsKeyT, Any]:
|
||||
"""
|
||||
Make a copy of current settings and convert to a dict.
|
||||
|
||||
|
|
@ -439,12 +506,25 @@ class BaseSettings(MutableMapping):
|
|||
settings = self.copy()
|
||||
return settings._to_dict()
|
||||
|
||||
def _repr_pretty_(self, p, cycle):
|
||||
# https://ipython.readthedocs.io/en/stable/config/integrating.html#pretty-printing
|
||||
def _repr_pretty_(self, p: Any, cycle: bool) -> None:
|
||||
if cycle:
|
||||
p.text(repr(self))
|
||||
else:
|
||||
p.text(pformat(self.copy_to_dict()))
|
||||
|
||||
def pop(self, name: _SettingsKeyT, default: Any = __default) -> Any:
|
||||
try:
|
||||
value = self.attributes[name].value
|
||||
except KeyError:
|
||||
if default is self.__default:
|
||||
raise
|
||||
|
||||
return default
|
||||
else:
|
||||
self.__delitem__(name)
|
||||
return value
|
||||
|
||||
|
||||
class Settings(BaseSettings):
|
||||
"""
|
||||
|
|
@ -457,7 +537,9 @@ class Settings(BaseSettings):
|
|||
described on :ref:`topics-settings-ref` already populated.
|
||||
"""
|
||||
|
||||
def __init__(self, values=None, priority="project"):
|
||||
def __init__(
|
||||
self, values: _SettingsInputT = None, priority: Union[int, str] = "project"
|
||||
):
|
||||
# Do not pass kwarg values here. We don't want to promote user-defined
|
||||
# dicts, and we want to update, not replace, default dicts with the
|
||||
# values given by the user
|
||||
|
|
@ -471,14 +553,16 @@ class Settings(BaseSettings):
|
|||
self.update(values, priority)
|
||||
|
||||
|
||||
def iter_default_settings():
|
||||
def iter_default_settings() -> Iterable[Tuple[str, Any]]:
|
||||
"""Return the default settings as an iterator of (name, value) tuples"""
|
||||
for name in dir(default_settings):
|
||||
if name.isupper():
|
||||
yield name, getattr(default_settings, name)
|
||||
|
||||
|
||||
def overridden_settings(settings):
|
||||
def overridden_settings(
|
||||
settings: Mapping[_SettingsKeyT, Any]
|
||||
) -> Iterable[Tuple[str, Any]]:
|
||||
"""Return a dict of the settings that have been overridden"""
|
||||
for name, defvalue in iter_default_settings():
|
||||
value = settings[name]
|
||||
|
|
|
|||
|
|
@ -17,6 +17,8 @@ import sys
|
|||
from importlib import import_module
|
||||
from pathlib import Path
|
||||
|
||||
ADDONS = {}
|
||||
|
||||
AJAXCRAWL_ENABLED = False
|
||||
|
||||
ASYNCIO_EVENT_LOOP = None
|
||||
|
|
@ -141,7 +143,7 @@ EXTENSIONS_BASE = {
|
|||
FEED_TEMPDIR = None
|
||||
FEEDS = {}
|
||||
FEED_URI_PARAMS = None # a function to extend uri arguments
|
||||
FEED_STORE_EMPTY = False
|
||||
FEED_STORE_EMPTY = True
|
||||
FEED_EXPORT_ENCODING = None
|
||||
FEED_EXPORT_FIELDS = None
|
||||
FEED_STORAGES = {}
|
||||
|
|
@ -262,6 +264,21 @@ RETRY_ENABLED = True
|
|||
RETRY_TIMES = 2 # initial response + 2 retries = 3 requests
|
||||
RETRY_HTTP_CODES = [500, 502, 503, 504, 522, 524, 408, 429]
|
||||
RETRY_PRIORITY_ADJUST = -1
|
||||
RETRY_EXCEPTIONS = [
|
||||
"twisted.internet.defer.TimeoutError",
|
||||
"twisted.internet.error.TimeoutError",
|
||||
"twisted.internet.error.DNSLookupError",
|
||||
"twisted.internet.error.ConnectionRefusedError",
|
||||
"twisted.internet.error.ConnectionDone",
|
||||
"twisted.internet.error.ConnectError",
|
||||
"twisted.internet.error.ConnectionLost",
|
||||
"twisted.internet.error.TCPTimedOutError",
|
||||
"twisted.web.client.ResponseFailed",
|
||||
# OSError is raised by the HttpCompression middleware when trying to
|
||||
# decompress an empty response
|
||||
OSError,
|
||||
"scrapy.core.downloader.handlers.http11.TunnelError",
|
||||
]
|
||||
|
||||
ROBOTSTXT_OBEY = False
|
||||
ROBOTSTXT_PARSER = "scrapy.robotstxt.ProtegoRobotParser"
|
||||
|
|
|
|||
|
|
@ -1,13 +1,23 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import traceback
|
||||
import warnings
|
||||
from collections import defaultdict
|
||||
from types import ModuleType
|
||||
from typing import TYPE_CHECKING, DefaultDict, Dict, List, Tuple, Type
|
||||
|
||||
from zope.interface import implementer
|
||||
|
||||
from scrapy import Request, Spider
|
||||
from scrapy.interfaces import ISpiderLoader
|
||||
from scrapy.settings import BaseSettings
|
||||
from scrapy.utils.misc import walk_modules
|
||||
from scrapy.utils.spider import iter_spider_classes
|
||||
|
||||
if TYPE_CHECKING:
|
||||
# typing.Self requires Python 3.11
|
||||
from typing_extensions import Self
|
||||
|
||||
|
||||
@implementer(ISpiderLoader)
|
||||
class SpiderLoader:
|
||||
|
|
@ -16,14 +26,14 @@ class SpiderLoader:
|
|||
in a Scrapy project.
|
||||
"""
|
||||
|
||||
def __init__(self, settings):
|
||||
self.spider_modules = settings.getlist("SPIDER_MODULES")
|
||||
self.warn_only = settings.getbool("SPIDER_LOADER_WARN_ONLY")
|
||||
self._spiders = {}
|
||||
self._found = defaultdict(list)
|
||||
def __init__(self, settings: BaseSettings):
|
||||
self.spider_modules: List[str] = settings.getlist("SPIDER_MODULES")
|
||||
self.warn_only: bool = settings.getbool("SPIDER_LOADER_WARN_ONLY")
|
||||
self._spiders: Dict[str, Type[Spider]] = {}
|
||||
self._found: DefaultDict[str, List[Tuple[str, str]]] = defaultdict(list)
|
||||
self._load_all_spiders()
|
||||
|
||||
def _check_name_duplicates(self):
|
||||
def _check_name_duplicates(self) -> None:
|
||||
dupes = []
|
||||
for name, locations in self._found.items():
|
||||
dupes.extend(
|
||||
|
|
@ -42,12 +52,12 @@ class SpiderLoader:
|
|||
category=UserWarning,
|
||||
)
|
||||
|
||||
def _load_spiders(self, module):
|
||||
def _load_spiders(self, module: ModuleType) -> None:
|
||||
for spcls in iter_spider_classes(module):
|
||||
self._found[spcls.name].append((module.__name__, spcls.__name__))
|
||||
self._spiders[spcls.name] = spcls
|
||||
|
||||
def _load_all_spiders(self):
|
||||
def _load_all_spiders(self) -> None:
|
||||
for name in self.spider_modules:
|
||||
try:
|
||||
for module in walk_modules(name):
|
||||
|
|
@ -65,10 +75,10 @@ class SpiderLoader:
|
|||
self._check_name_duplicates()
|
||||
|
||||
@classmethod
|
||||
def from_settings(cls, settings):
|
||||
def from_settings(cls, settings: BaseSettings) -> Self:
|
||||
return cls(settings)
|
||||
|
||||
def load(self, spider_name):
|
||||
def load(self, spider_name: str) -> Type[Spider]:
|
||||
"""
|
||||
Return the Spider class for the given spider name. If the spider
|
||||
name is not found, raise a KeyError.
|
||||
|
|
@ -78,7 +88,7 @@ class SpiderLoader:
|
|||
except KeyError:
|
||||
raise KeyError(f"Spider not found: {spider_name}")
|
||||
|
||||
def find_by_request(self, request):
|
||||
def find_by_request(self, request: Request) -> List[str]:
|
||||
"""
|
||||
Return the list of spider names that can handle the given request.
|
||||
"""
|
||||
|
|
@ -86,7 +96,7 @@ class SpiderLoader:
|
|||
name for name, cls in self._spiders.items() if cls.handles_request(request)
|
||||
]
|
||||
|
||||
def list(self):
|
||||
def list(self) -> List[str]:
|
||||
"""
|
||||
Return a list with the names of all spiders available in the project.
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -6,15 +6,21 @@ See documentation in docs/topics/spiders.rst
|
|||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import TYPE_CHECKING, Optional
|
||||
from typing import TYPE_CHECKING, Any, Iterable, List, Optional, Union, cast
|
||||
|
||||
from twisted.internet.defer import Deferred
|
||||
|
||||
from scrapy import signals
|
||||
from scrapy.http import Request
|
||||
from scrapy.http import Request, Response
|
||||
from scrapy.utils.trackref import object_ref
|
||||
from scrapy.utils.url import url_is_from_spider
|
||||
|
||||
if TYPE_CHECKING:
|
||||
# typing.Self requires Python 3.11
|
||||
from typing_extensions import Self
|
||||
|
||||
from scrapy.crawler import Crawler
|
||||
from scrapy.settings import BaseSettings
|
||||
|
||||
|
||||
class Spider(object_ref):
|
||||
|
|
@ -25,21 +31,21 @@ class Spider(object_ref):
|
|||
name: str
|
||||
custom_settings: Optional[dict] = None
|
||||
|
||||
def __init__(self, name=None, **kwargs):
|
||||
def __init__(self, name: Optional[str] = None, **kwargs: Any):
|
||||
if name is not None:
|
||||
self.name = name
|
||||
elif not getattr(self, "name", None):
|
||||
raise ValueError(f"{type(self).__name__} must have a name")
|
||||
self.__dict__.update(kwargs)
|
||||
if not hasattr(self, "start_urls"):
|
||||
self.start_urls = []
|
||||
self.start_urls: List[str] = []
|
||||
|
||||
@property
|
||||
def logger(self):
|
||||
def logger(self) -> logging.LoggerAdapter:
|
||||
logger = logging.getLogger(self.name)
|
||||
return logging.LoggerAdapter(logger, {"spider": self})
|
||||
|
||||
def log(self, message, level=logging.DEBUG, **kw):
|
||||
def log(self, message: Any, level: int = logging.DEBUG, **kw: Any) -> None:
|
||||
"""Log the given message at the given log level
|
||||
|
||||
This helper wraps a log call to the logger within the spider, but you
|
||||
|
|
@ -49,17 +55,17 @@ class Spider(object_ref):
|
|||
self.logger.log(level, message, **kw)
|
||||
|
||||
@classmethod
|
||||
def from_crawler(cls, crawler, *args, **kwargs):
|
||||
def from_crawler(cls, crawler: Crawler, *args: Any, **kwargs: Any) -> Self:
|
||||
spider = cls(*args, **kwargs)
|
||||
spider._set_crawler(crawler)
|
||||
return spider
|
||||
|
||||
def _set_crawler(self, crawler: Crawler):
|
||||
def _set_crawler(self, crawler: Crawler) -> None:
|
||||
self.crawler = crawler
|
||||
self.settings = crawler.settings
|
||||
crawler.signals.connect(self.close, signals.spider_closed)
|
||||
|
||||
def start_requests(self):
|
||||
def start_requests(self) -> Iterable[Request]:
|
||||
if not self.start_urls and hasattr(self, "start_url"):
|
||||
raise AttributeError(
|
||||
"Crawling could not start: 'start_urls' not found "
|
||||
|
|
@ -69,29 +75,30 @@ class Spider(object_ref):
|
|||
for url in self.start_urls:
|
||||
yield Request(url, dont_filter=True)
|
||||
|
||||
def _parse(self, response, **kwargs):
|
||||
def _parse(self, response: Response, **kwargs: Any) -> Any:
|
||||
return self.parse(response, **kwargs)
|
||||
|
||||
def parse(self, response, **kwargs):
|
||||
def parse(self, response: Response, **kwargs: Any) -> Any:
|
||||
raise NotImplementedError(
|
||||
f"{self.__class__.__name__}.parse callback is not defined"
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def update_settings(cls, settings):
|
||||
def update_settings(cls, settings: BaseSettings) -> None:
|
||||
settings.setdict(cls.custom_settings or {}, priority="spider")
|
||||
|
||||
@classmethod
|
||||
def handles_request(cls, request):
|
||||
def handles_request(cls, request: Request) -> bool:
|
||||
return url_is_from_spider(request.url, cls)
|
||||
|
||||
@staticmethod
|
||||
def close(spider, reason):
|
||||
def close(spider: Spider, reason: str) -> Union[Deferred, None]:
|
||||
closed = getattr(spider, "closed", None)
|
||||
if callable(closed):
|
||||
return closed(reason)
|
||||
return cast(Union[Deferred, None], closed(reason))
|
||||
return None
|
||||
|
||||
def __repr__(self):
|
||||
def __repr__(self) -> str:
|
||||
return f"<{type(self).__name__} {self.name!r} at 0x{id(self):0x}>"
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -10,7 +10,6 @@ from typing import Union
|
|||
|
||||
from queuelib import queue
|
||||
|
||||
from scrapy.utils.deprecate import create_deprecated_class
|
||||
from scrapy.utils.request import request_from_dict
|
||||
|
||||
|
||||
|
|
@ -143,32 +142,3 @@ MarshalFifoDiskQueue = _scrapy_serialization_queue(_MarshalFifoSerializationDisk
|
|||
MarshalLifoDiskQueue = _scrapy_serialization_queue(_MarshalLifoSerializationDiskQueue)
|
||||
FifoMemoryQueue = _scrapy_non_serialization_queue(queue.FifoMemoryQueue)
|
||||
LifoMemoryQueue = _scrapy_non_serialization_queue(queue.LifoMemoryQueue)
|
||||
|
||||
|
||||
# deprecated queue classes
|
||||
_subclass_warn_message = "{cls} inherits from deprecated class {old}"
|
||||
_instance_warn_message = "{cls} is deprecated"
|
||||
PickleFifoDiskQueueNonRequest = create_deprecated_class(
|
||||
name="PickleFifoDiskQueueNonRequest",
|
||||
new_class=_PickleFifoSerializationDiskQueue,
|
||||
subclass_warn_message=_subclass_warn_message,
|
||||
instance_warn_message=_instance_warn_message,
|
||||
)
|
||||
PickleLifoDiskQueueNonRequest = create_deprecated_class(
|
||||
name="PickleLifoDiskQueueNonRequest",
|
||||
new_class=_PickleLifoSerializationDiskQueue,
|
||||
subclass_warn_message=_subclass_warn_message,
|
||||
instance_warn_message=_instance_warn_message,
|
||||
)
|
||||
MarshalFifoDiskQueueNonRequest = create_deprecated_class(
|
||||
name="MarshalFifoDiskQueueNonRequest",
|
||||
new_class=_MarshalFifoSerializationDiskQueue,
|
||||
subclass_warn_message=_subclass_warn_message,
|
||||
instance_warn_message=_instance_warn_message,
|
||||
)
|
||||
MarshalLifoDiskQueueNonRequest = create_deprecated_class(
|
||||
name="MarshalLifoDiskQueueNonRequest",
|
||||
new_class=_MarshalLifoSerializationDiskQueue,
|
||||
subclass_warn_message=_subclass_warn_message,
|
||||
instance_warn_message=_instance_warn_message,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
"""Boto/botocore helpers"""
|
||||
|
||||
|
||||
def is_botocore_available():
|
||||
def is_botocore_available() -> bool:
|
||||
try:
|
||||
import botocore # noqa: F401
|
||||
|
||||
|
|
|
|||
|
|
@ -5,7 +5,18 @@ import warnings
|
|||
from configparser import ConfigParser
|
||||
from operator import itemgetter
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional, Union
|
||||
from typing import (
|
||||
Any,
|
||||
Callable,
|
||||
Collection,
|
||||
Dict,
|
||||
Iterable,
|
||||
List,
|
||||
Mapping,
|
||||
MutableMapping,
|
||||
Optional,
|
||||
Union,
|
||||
)
|
||||
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning, UsageError
|
||||
from scrapy.settings import BaseSettings
|
||||
|
|
@ -13,21 +24,26 @@ from scrapy.utils.deprecate import update_classpath
|
|||
from scrapy.utils.python import without_none_values
|
||||
|
||||
|
||||
def build_component_list(compdict, custom=None, convert=update_classpath):
|
||||
def build_component_list(
|
||||
compdict: MutableMapping[Any, Any],
|
||||
custom: Any = None,
|
||||
convert: Callable[[Any], Any] = update_classpath,
|
||||
) -> List[Any]:
|
||||
"""Compose a component list from a { class: order } dictionary."""
|
||||
|
||||
def _check_components(complist):
|
||||
def _check_components(complist: Collection[Any]) -> None:
|
||||
if len({convert(c) for c in complist}) != len(complist):
|
||||
raise ValueError(
|
||||
f"Some paths in {complist!r} convert to the same object, "
|
||||
"please update your settings"
|
||||
)
|
||||
|
||||
def _map_keys(compdict):
|
||||
def _map_keys(compdict: Mapping[Any, Any]) -> Union[BaseSettings, Dict[Any, Any]]:
|
||||
if isinstance(compdict, BaseSettings):
|
||||
compbs = BaseSettings()
|
||||
for k, v in compdict.items():
|
||||
prio = compdict.getpriority(k)
|
||||
assert prio is not None
|
||||
if compbs.getpriority(convert(k)) == prio:
|
||||
raise ValueError(
|
||||
f"Some paths in {list(compdict.keys())!r} "
|
||||
|
|
@ -40,7 +56,7 @@ def build_component_list(compdict, custom=None, convert=update_classpath):
|
|||
_check_components(compdict)
|
||||
return {convert(k): v for k, v in compdict.items()}
|
||||
|
||||
def _validate_values(compdict):
|
||||
def _validate_values(compdict: Mapping[Any, Any]) -> None:
|
||||
"""Fail if a value in the components dict is not a real number or None."""
|
||||
for name, value in compdict.items():
|
||||
if value is not None and not isinstance(value, numbers.Real):
|
||||
|
|
@ -49,11 +65,17 @@ def build_component_list(compdict, custom=None, convert=update_classpath):
|
|||
"please provide a real number or None instead"
|
||||
)
|
||||
|
||||
if isinstance(custom, (list, tuple)):
|
||||
_check_components(custom)
|
||||
return type(custom)(convert(c) for c in custom)
|
||||
|
||||
if custom is not None:
|
||||
warnings.warn(
|
||||
"The 'custom' attribute of build_component_list() is deprecated. "
|
||||
"Please merge its value into 'compdict' manually or change your "
|
||||
"code to use Settings.getwithbase().",
|
||||
category=ScrapyDeprecationWarning,
|
||||
stacklevel=2,
|
||||
)
|
||||
if isinstance(custom, (list, tuple)):
|
||||
_check_components(custom)
|
||||
return type(custom)(convert(c) for c in custom) # type: ignore[return-value]
|
||||
compdict.update(custom)
|
||||
|
||||
_validate_values(compdict)
|
||||
|
|
@ -61,7 +83,7 @@ def build_component_list(compdict, custom=None, convert=update_classpath):
|
|||
return [k for k, v in sorted(compdict.items(), key=itemgetter(1))]
|
||||
|
||||
|
||||
def arglist_to_dict(arglist):
|
||||
def arglist_to_dict(arglist: List[str]) -> Dict[str, str]:
|
||||
"""Convert a list of arguments like ['arg1=val1', 'arg2=val2', ...] to a
|
||||
dict
|
||||
"""
|
||||
|
|
@ -84,7 +106,7 @@ def closest_scrapy_cfg(
|
|||
return closest_scrapy_cfg(path.parent, path)
|
||||
|
||||
|
||||
def init_env(project="default", set_syspath=True):
|
||||
def init_env(project: str = "default", set_syspath: bool = True) -> None:
|
||||
"""Initialize environment to use command-line tool from inside a project
|
||||
dir. This sets the Scrapy settings module and modifies the Python path to
|
||||
be able to locate the project module.
|
||||
|
|
@ -99,7 +121,7 @@ def init_env(project="default", set_syspath=True):
|
|||
sys.path.append(projdir)
|
||||
|
||||
|
||||
def get_config(use_closest=True):
|
||||
def get_config(use_closest: bool = True) -> ConfigParser:
|
||||
"""Get Scrapy config file as a ConfigParser"""
|
||||
sources = get_sources(use_closest)
|
||||
cfg = ConfigParser()
|
||||
|
|
@ -107,7 +129,7 @@ def get_config(use_closest=True):
|
|||
return cfg
|
||||
|
||||
|
||||
def get_sources(use_closest=True) -> List[str]:
|
||||
def get_sources(use_closest: bool = True) -> List[str]:
|
||||
xdg_config_home = (
|
||||
os.environ.get("XDG_CONFIG_HOME") or Path("~/.config").expanduser()
|
||||
)
|
||||
|
|
@ -122,7 +144,9 @@ def get_sources(use_closest=True) -> List[str]:
|
|||
return sources
|
||||
|
||||
|
||||
def feed_complete_default_values_from_settings(feed, settings):
|
||||
def feed_complete_default_values_from_settings(
|
||||
feed: Dict[str, Any], settings: BaseSettings
|
||||
) -> Dict[str, Any]:
|
||||
out = feed.copy()
|
||||
out.setdefault("batch_item_count", settings.getint("FEED_EXPORT_BATCH_ITEM_COUNT"))
|
||||
out.setdefault("encoding", settings["FEED_EXPORT_ENCODING"])
|
||||
|
|
@ -138,21 +162,21 @@ def feed_complete_default_values_from_settings(feed, settings):
|
|||
|
||||
|
||||
def feed_process_params_from_cli(
|
||||
settings,
|
||||
settings: BaseSettings,
|
||||
output: List[str],
|
||||
output_format=None,
|
||||
output_format: Optional[str] = None,
|
||||
overwrite_output: Optional[List[str]] = None,
|
||||
):
|
||||
) -> Dict[str, Dict[str, Any]]:
|
||||
"""
|
||||
Receives feed export params (from the 'crawl' or 'runspider' commands),
|
||||
checks for inconsistencies in their quantities and returns a dictionary
|
||||
suitable to be used as the FEEDS setting.
|
||||
"""
|
||||
valid_output_formats = without_none_values(
|
||||
valid_output_formats: Iterable[str] = without_none_values(
|
||||
settings.getwithbase("FEED_EXPORTERS")
|
||||
).keys()
|
||||
|
||||
def check_valid_format(output_format):
|
||||
def check_valid_format(output_format: str) -> None:
|
||||
if output_format not in valid_output_formats:
|
||||
raise UsageError(
|
||||
f"Unrecognized output format '{output_format}'. "
|
||||
|
|
@ -202,7 +226,8 @@ def feed_process_params_from_cli(
|
|||
for element in output:
|
||||
try:
|
||||
feed_uri, feed_format = element.rsplit(":", 1)
|
||||
except ValueError:
|
||||
check_valid_format(feed_format)
|
||||
except (ValueError, UsageError):
|
||||
feed_uri = element
|
||||
feed_format = Path(element).suffix.replace(".", "")
|
||||
else:
|
||||
|
|
|
|||
|
|
@ -6,13 +6,32 @@ This module must not depend on any module outside the Standard Library.
|
|||
"""
|
||||
|
||||
import collections
|
||||
import warnings
|
||||
import weakref
|
||||
from collections.abc import Mapping
|
||||
from typing import Any, AnyStr, Optional, OrderedDict, Sequence, TypeVar
|
||||
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
|
||||
_KT = TypeVar("_KT")
|
||||
_VT = TypeVar("_VT")
|
||||
|
||||
|
||||
class CaselessDict(dict):
|
||||
__slots__ = ()
|
||||
|
||||
def __new__(cls, *args, **kwargs):
|
||||
from scrapy.http.headers import Headers
|
||||
|
||||
if issubclass(cls, CaselessDict) and not issubclass(cls, Headers):
|
||||
warnings.warn(
|
||||
"scrapy.utils.datatypes.CaselessDict is deprecated,"
|
||||
" please use scrapy.utils.datatypes.CaseInsensitiveDict instead",
|
||||
category=ScrapyDeprecationWarning,
|
||||
stacklevel=2,
|
||||
)
|
||||
return super().__new__(cls, *args, **kwargs)
|
||||
|
||||
def __init__(self, seq=None):
|
||||
super().__init__()
|
||||
if seq:
|
||||
|
|
@ -64,17 +83,59 @@ class CaselessDict(dict):
|
|||
return dict.pop(self, self.normkey(key), *args)
|
||||
|
||||
|
||||
class LocalCache(collections.OrderedDict):
|
||||
class CaseInsensitiveDict(collections.UserDict):
|
||||
"""A dict-like structure that accepts strings or bytes
|
||||
as keys and allows case-insensitive lookups.
|
||||
"""
|
||||
|
||||
def __init__(self, *args, **kwargs) -> None:
|
||||
self._keys: dict = {}
|
||||
super().__init__(*args, **kwargs)
|
||||
|
||||
def __getitem__(self, key: AnyStr) -> Any:
|
||||
normalized_key = self._normkey(key)
|
||||
return super().__getitem__(self._keys[normalized_key.lower()])
|
||||
|
||||
def __setitem__(self, key: AnyStr, value: Any) -> None:
|
||||
normalized_key = self._normkey(key)
|
||||
try:
|
||||
lower_key = self._keys[normalized_key.lower()]
|
||||
del self[lower_key]
|
||||
except KeyError:
|
||||
pass
|
||||
super().__setitem__(normalized_key, self._normvalue(value))
|
||||
self._keys[normalized_key.lower()] = normalized_key
|
||||
|
||||
def __delitem__(self, key: AnyStr) -> None:
|
||||
normalized_key = self._normkey(key)
|
||||
stored_key = self._keys.pop(normalized_key.lower())
|
||||
super().__delitem__(stored_key)
|
||||
|
||||
def __contains__(self, key: AnyStr) -> bool: # type: ignore[override]
|
||||
normalized_key = self._normkey(key)
|
||||
return normalized_key.lower() in self._keys
|
||||
|
||||
def __repr__(self) -> str:
|
||||
return f"<{self.__class__.__name__}: {super().__repr__()}>"
|
||||
|
||||
def _normkey(self, key: AnyStr) -> AnyStr:
|
||||
return key
|
||||
|
||||
def _normvalue(self, value: Any) -> Any:
|
||||
return value
|
||||
|
||||
|
||||
class LocalCache(OrderedDict[_KT, _VT]):
|
||||
"""Dictionary with a finite number of keys.
|
||||
|
||||
Older items expires first.
|
||||
"""
|
||||
|
||||
def __init__(self, limit=None):
|
||||
def __init__(self, limit: Optional[int] = None):
|
||||
super().__init__()
|
||||
self.limit = limit
|
||||
self.limit: Optional[int] = limit
|
||||
|
||||
def __setitem__(self, key, value):
|
||||
def __setitem__(self, key: _KT, value: _VT) -> None:
|
||||
if self.limit:
|
||||
while len(self) >= self.limit:
|
||||
self.popitem(last=False)
|
||||
|
|
@ -93,17 +154,17 @@ class LocalWeakReferencedCache(weakref.WeakKeyDictionary):
|
|||
it cannot be instantiated with an initial dictionary.
|
||||
"""
|
||||
|
||||
def __init__(self, limit=None):
|
||||
def __init__(self, limit: Optional[int] = None):
|
||||
super().__init__()
|
||||
self.data = LocalCache(limit=limit)
|
||||
self.data: LocalCache = LocalCache(limit=limit)
|
||||
|
||||
def __setitem__(self, key, value):
|
||||
def __setitem__(self, key: _KT, value: _VT) -> None:
|
||||
try:
|
||||
super().__setitem__(key, value)
|
||||
except TypeError:
|
||||
pass # key is not weak-referenceable, skip caching
|
||||
|
||||
def __getitem__(self, key):
|
||||
def __getitem__(self, key: _KT) -> Optional[_VT]: # type: ignore[override]
|
||||
try:
|
||||
return super().__getitem__(key)
|
||||
except (TypeError, KeyError):
|
||||
|
|
@ -113,8 +174,8 @@ class LocalWeakReferencedCache(weakref.WeakKeyDictionary):
|
|||
class SequenceExclude:
|
||||
"""Object to test if an item is NOT within some sequence."""
|
||||
|
||||
def __init__(self, seq):
|
||||
self.seq = seq
|
||||
def __init__(self, seq: Sequence):
|
||||
self.seq: Sequence = seq
|
||||
|
||||
def __contains__(self, item):
|
||||
def __contains__(self, item: Any) -> bool:
|
||||
return item not in self.seq
|
||||
|
|
|
|||
|
|
@ -1,19 +1,21 @@
|
|||
import warnings
|
||||
from functools import wraps
|
||||
from typing import Any, Callable
|
||||
|
||||
from twisted.internet import defer, threads
|
||||
from twisted.internet.defer import Deferred
|
||||
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
|
||||
|
||||
def deprecated(use_instead=None):
|
||||
def deprecated(use_instead: Any = None) -> Callable:
|
||||
"""This is a decorator which can be used to mark functions
|
||||
as deprecated. It will result in a warning being emitted
|
||||
when the function is used."""
|
||||
|
||||
def deco(func):
|
||||
def deco(func: Callable) -> Callable:
|
||||
@wraps(func)
|
||||
def wrapped(*args, **kwargs):
|
||||
def wrapped(*args: Any, **kwargs: Any) -> Any:
|
||||
message = f"Call to deprecated function {func.__name__}."
|
||||
if use_instead:
|
||||
message += f" Use {use_instead} instead."
|
||||
|
|
@ -28,23 +30,23 @@ def deprecated(use_instead=None):
|
|||
return deco
|
||||
|
||||
|
||||
def defers(func):
|
||||
def defers(func: Callable) -> Callable[..., Deferred]:
|
||||
"""Decorator to make sure a function always returns a deferred"""
|
||||
|
||||
@wraps(func)
|
||||
def wrapped(*a, **kw):
|
||||
def wrapped(*a: Any, **kw: Any) -> Deferred:
|
||||
return defer.maybeDeferred(func, *a, **kw)
|
||||
|
||||
return wrapped
|
||||
|
||||
|
||||
def inthread(func):
|
||||
def inthread(func: Callable) -> Callable[..., Deferred]:
|
||||
"""Decorator to call a function in a thread and return a deferred with the
|
||||
result
|
||||
"""
|
||||
|
||||
@wraps(func)
|
||||
def wrapped(*a, **kw):
|
||||
def wrapped(*a: Any, **kw: Any) -> Deferred:
|
||||
return threads.deferToThread(func, *a, **kw)
|
||||
|
||||
return wrapped
|
||||
|
|
|
|||
|
|
@ -5,19 +5,26 @@ import asyncio
|
|||
import inspect
|
||||
from asyncio import Future
|
||||
from functools import wraps
|
||||
from types import CoroutineType
|
||||
from typing import (
|
||||
Any,
|
||||
AsyncGenerator,
|
||||
AsyncIterable,
|
||||
AsyncIterator,
|
||||
Awaitable,
|
||||
Callable,
|
||||
Coroutine,
|
||||
Dict,
|
||||
Generator,
|
||||
Iterable,
|
||||
Iterator,
|
||||
List,
|
||||
Optional,
|
||||
Tuple,
|
||||
TypeVar,
|
||||
Union,
|
||||
cast,
|
||||
overload,
|
||||
)
|
||||
|
||||
from twisted.internet import defer
|
||||
|
|
@ -44,7 +51,7 @@ def defer_fail(_failure: Failure) -> Deferred:
|
|||
return d
|
||||
|
||||
|
||||
def defer_succeed(result) -> Deferred:
|
||||
def defer_succeed(result: Any) -> Deferred:
|
||||
"""Same as twisted.internet.defer.succeed but delay calling callback until
|
||||
next reactor loop
|
||||
|
||||
|
|
@ -58,7 +65,7 @@ def defer_succeed(result) -> Deferred:
|
|||
return d
|
||||
|
||||
|
||||
def defer_result(result) -> Deferred:
|
||||
def defer_result(result: Any) -> Deferred:
|
||||
if isinstance(result, Deferred):
|
||||
return result
|
||||
if isinstance(result, failure.Failure):
|
||||
|
|
@ -66,7 +73,7 @@ def defer_result(result) -> Deferred:
|
|||
return defer_succeed(result)
|
||||
|
||||
|
||||
def mustbe_deferred(f: Callable, *args, **kw) -> Deferred:
|
||||
def mustbe_deferred(f: Callable, *args: Any, **kw: Any) -> Deferred:
|
||||
"""Same as twisted.internet.defer.maybeDeferred, but delay calling
|
||||
callback/errback to next reactor loop
|
||||
"""
|
||||
|
|
@ -84,7 +91,7 @@ def mustbe_deferred(f: Callable, *args, **kw) -> Deferred:
|
|||
|
||||
|
||||
def parallel(
|
||||
iterable: Iterable, count: int, callable: Callable, *args, **named
|
||||
iterable: Iterable, count: int, callable: Callable, *args: Any, **named: Any
|
||||
) -> Deferred:
|
||||
"""Execute a callable over the objects in the given iterable, in parallel,
|
||||
using no more than ``count`` concurrent calls.
|
||||
|
|
@ -146,14 +153,14 @@ class _AsyncCooperatorAdapter(Iterator):
|
|||
self,
|
||||
aiterable: AsyncIterable,
|
||||
callable: Callable,
|
||||
*callable_args,
|
||||
**callable_kwargs
|
||||
*callable_args: Any,
|
||||
**callable_kwargs: Any,
|
||||
):
|
||||
self.aiterator = aiterable.__aiter__()
|
||||
self.callable = callable
|
||||
self.callable_args = callable_args
|
||||
self.callable_kwargs = callable_kwargs
|
||||
self.finished = False
|
||||
self.aiterator: AsyncIterator = aiterable.__aiter__()
|
||||
self.callable: Callable = callable
|
||||
self.callable_args: Tuple[Any, ...] = callable_args
|
||||
self.callable_kwargs: Dict[str, Any] = callable_kwargs
|
||||
self.finished: bool = False
|
||||
self.waiting_deferreds: List[Deferred] = []
|
||||
self.anext_deferred: Optional[Deferred] = None
|
||||
|
||||
|
|
@ -183,9 +190,7 @@ class _AsyncCooperatorAdapter(Iterator):
|
|||
def _call_anext(self) -> None:
|
||||
# This starts waiting for the next result from aiterator.
|
||||
# If aiterator is exhausted, _errback will be called.
|
||||
self.anext_deferred = cast(
|
||||
Deferred, deferred_from_coro(self.aiterator.__anext__())
|
||||
)
|
||||
self.anext_deferred = deferred_from_coro(self.aiterator.__anext__())
|
||||
self.anext_deferred.addCallbacks(self._callback, self._errback)
|
||||
|
||||
def __next__(self) -> Deferred:
|
||||
|
|
@ -201,7 +206,11 @@ class _AsyncCooperatorAdapter(Iterator):
|
|||
|
||||
|
||||
def parallel_async(
|
||||
async_iterable: AsyncIterable, count: int, callable: Callable, *args, **named
|
||||
async_iterable: AsyncIterable,
|
||||
count: int,
|
||||
callable: Callable,
|
||||
*args: Any,
|
||||
**named: Any,
|
||||
) -> Deferred:
|
||||
"""Like parallel but for async iterators"""
|
||||
coop = Cooperator()
|
||||
|
|
@ -210,7 +219,9 @@ def parallel_async(
|
|||
return dl
|
||||
|
||||
|
||||
def process_chain(callbacks: Iterable[Callable], input, *a, **kw) -> Deferred:
|
||||
def process_chain(
|
||||
callbacks: Iterable[Callable], input: Any, *a: Any, **kw: Any
|
||||
) -> Deferred:
|
||||
"""Return a Deferred built by chaining the given callbacks"""
|
||||
d: Deferred = Deferred()
|
||||
for x in callbacks:
|
||||
|
|
@ -220,7 +231,11 @@ def process_chain(callbacks: Iterable[Callable], input, *a, **kw) -> Deferred:
|
|||
|
||||
|
||||
def process_chain_both(
|
||||
callbacks: Iterable[Callable], errbacks: Iterable[Callable], input, *a, **kw
|
||||
callbacks: Iterable[Callable],
|
||||
errbacks: Iterable[Callable],
|
||||
input: Any,
|
||||
*a: Any,
|
||||
**kw: Any,
|
||||
) -> Deferred:
|
||||
"""Return a Deferred built by chaining the given callbacks and errbacks"""
|
||||
d: Deferred = Deferred()
|
||||
|
|
@ -240,7 +255,9 @@ def process_chain_both(
|
|||
return d
|
||||
|
||||
|
||||
def process_parallel(callbacks: Iterable[Callable], input, *a, **kw) -> Deferred:
|
||||
def process_parallel(
|
||||
callbacks: Iterable[Callable], input: Any, *a: Any, **kw: Any
|
||||
) -> Deferred:
|
||||
"""Return a Deferred with the output of all successful calls to the given
|
||||
callbacks
|
||||
"""
|
||||
|
|
@ -250,7 +267,9 @@ def process_parallel(callbacks: Iterable[Callable], input, *a, **kw) -> Deferred
|
|||
return d
|
||||
|
||||
|
||||
def iter_errback(iterable: Iterable, errback: Callable, *a, **kw) -> Generator:
|
||||
def iter_errback(
|
||||
iterable: Iterable, errback: Callable, *a: Any, **kw: Any
|
||||
) -> Generator:
|
||||
"""Wraps an iterable calling an errback if an error is caught while
|
||||
iterating it.
|
||||
"""
|
||||
|
|
@ -265,7 +284,7 @@ def iter_errback(iterable: Iterable, errback: Callable, *a, **kw) -> Generator:
|
|||
|
||||
|
||||
async def aiter_errback(
|
||||
aiterable: AsyncIterable, errback: Callable, *a, **kw
|
||||
aiterable: AsyncIterable, errback: Callable, *a: Any, **kw: Any
|
||||
) -> AsyncGenerator:
|
||||
"""Wraps an async iterable calling an errback if an error is caught while
|
||||
iterating it. Similar to scrapy.utils.defer.iter_errback()
|
||||
|
|
@ -280,7 +299,21 @@ async def aiter_errback(
|
|||
errback(failure.Failure(), *a, **kw)
|
||||
|
||||
|
||||
def deferred_from_coro(o) -> Any:
|
||||
_CT = TypeVar("_CT", bound=Union[Awaitable, CoroutineType, Future])
|
||||
_T = TypeVar("_T")
|
||||
|
||||
|
||||
@overload
|
||||
def deferred_from_coro(o: _CT) -> Deferred:
|
||||
...
|
||||
|
||||
|
||||
@overload
|
||||
def deferred_from_coro(o: _T) -> _T:
|
||||
...
|
||||
|
||||
|
||||
def deferred_from_coro(o: _T) -> Union[Deferred, _T]:
|
||||
"""Converts a coroutine into a Deferred, or returns the object as is if it isn't a coroutine"""
|
||||
if isinstance(o, Deferred):
|
||||
return o
|
||||
|
|
@ -303,13 +336,13 @@ def deferred_f_from_coro_f(coro_f: Callable[..., Coroutine]) -> Callable:
|
|||
"""
|
||||
|
||||
@wraps(coro_f)
|
||||
def f(*coro_args, **coro_kwargs):
|
||||
def f(*coro_args: Any, **coro_kwargs: Any) -> Any:
|
||||
return deferred_from_coro(coro_f(*coro_args, **coro_kwargs))
|
||||
|
||||
return f
|
||||
|
||||
|
||||
def maybeDeferred_coro(f: Callable, *args, **kw) -> Deferred:
|
||||
def maybeDeferred_coro(f: Callable, *args: Any, **kw: Any) -> Deferred:
|
||||
"""Copy of defer.maybeDeferred that also converts coroutines to Deferreds."""
|
||||
try:
|
||||
result = f(*args, **kw)
|
||||
|
|
@ -340,8 +373,9 @@ def deferred_to_future(d: Deferred) -> Future:
|
|||
class MySpider(Spider):
|
||||
...
|
||||
async def parse(self, response):
|
||||
d = treq.get('https://example.com/additional')
|
||||
additional_response = await deferred_to_future(d)
|
||||
additional_request = scrapy.Request('https://example.org/price')
|
||||
deferred = self.crawler.engine.download(additional_request)
|
||||
additional_response = await deferred_to_future(deferred)
|
||||
"""
|
||||
return d.asFuture(_get_asyncio_event_loop())
|
||||
|
||||
|
|
@ -368,8 +402,9 @@ def maybe_deferred_to_future(d: Deferred) -> Union[Deferred, Future]:
|
|||
class MySpider(Spider):
|
||||
...
|
||||
async def parse(self, response):
|
||||
d = treq.get('https://example.com/additional')
|
||||
extra_response = await maybe_deferred_to_future(d)
|
||||
additional_request = scrapy.Request('https://example.org/price')
|
||||
deferred = self.crawler.engine.download(additional_request)
|
||||
additional_response = await maybe_deferred_to_future(deferred)
|
||||
"""
|
||||
if not is_asyncio_reactor_installed():
|
||||
return d
|
||||
|
|
|
|||
|
|
@ -2,12 +2,12 @@
|
|||
|
||||
import inspect
|
||||
import warnings
|
||||
from typing import List, Tuple
|
||||
from typing import Any, Dict, List, Optional, Tuple, Type, overload
|
||||
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
|
||||
|
||||
def attribute(obj, oldattr, newattr, version="0.12"):
|
||||
def attribute(obj: Any, oldattr: str, newattr: str, version: str = "0.12") -> None:
|
||||
cname = obj.__class__.__name__
|
||||
warnings.warn(
|
||||
f"{cname}.{oldattr} attribute is deprecated and will be no longer supported "
|
||||
|
|
@ -18,16 +18,16 @@ def attribute(obj, oldattr, newattr, version="0.12"):
|
|||
|
||||
|
||||
def create_deprecated_class(
|
||||
name,
|
||||
new_class,
|
||||
clsdict=None,
|
||||
warn_category=ScrapyDeprecationWarning,
|
||||
warn_once=True,
|
||||
old_class_path=None,
|
||||
new_class_path=None,
|
||||
subclass_warn_message="{cls} inherits from deprecated class {old}, please inherit from {new}.",
|
||||
instance_warn_message="{cls} is deprecated, instantiate {new} instead.",
|
||||
):
|
||||
name: str,
|
||||
new_class: type,
|
||||
clsdict: Optional[Dict[str, Any]] = None,
|
||||
warn_category: Type[Warning] = ScrapyDeprecationWarning,
|
||||
warn_once: bool = True,
|
||||
old_class_path: Optional[str] = None,
|
||||
new_class_path: Optional[str] = None,
|
||||
subclass_warn_message: str = "{cls} inherits from deprecated class {old}, please inherit from {new}.",
|
||||
instance_warn_message: str = "{cls} is deprecated, instantiate {new} instead.",
|
||||
) -> type:
|
||||
"""
|
||||
Return a "deprecated" class that causes its subclasses to issue a warning.
|
||||
Subclasses of ``new_class`` are considered subclasses of this class.
|
||||
|
|
@ -53,17 +53,20 @@ def create_deprecated_class(
|
|||
OldName.
|
||||
"""
|
||||
|
||||
class DeprecatedClass(new_class.__class__):
|
||||
deprecated_class = None
|
||||
warned_on_subclass = False
|
||||
# https://github.com/python/mypy/issues/4177
|
||||
class DeprecatedClass(new_class.__class__): # type: ignore[misc, name-defined]
|
||||
deprecated_class: Optional[type] = None
|
||||
warned_on_subclass: bool = False
|
||||
|
||||
def __new__(metacls, name, bases, clsdict_):
|
||||
def __new__(
|
||||
metacls, name: str, bases: Tuple[type, ...], clsdict_: Dict[str, Any]
|
||||
) -> type:
|
||||
cls = super().__new__(metacls, name, bases, clsdict_)
|
||||
if metacls.deprecated_class is None:
|
||||
metacls.deprecated_class = cls
|
||||
return cls
|
||||
|
||||
def __init__(cls, name, bases, clsdict_):
|
||||
def __init__(cls, name: str, bases: Tuple[type, ...], clsdict_: Dict[str, Any]):
|
||||
meta = cls.__class__
|
||||
old = meta.deprecated_class
|
||||
if old in bases and not (warn_once and meta.warned_on_subclass):
|
||||
|
|
@ -81,10 +84,10 @@ def create_deprecated_class(
|
|||
# see https://www.python.org/dev/peps/pep-3119/#overloading-isinstance-and-issubclass
|
||||
# and https://docs.python.org/reference/datamodel.html#customizing-instance-and-subclass-checks
|
||||
# for implementation details
|
||||
def __instancecheck__(cls, inst):
|
||||
def __instancecheck__(cls, inst: Any) -> bool:
|
||||
return any(cls.__subclasscheck__(c) for c in (type(inst), inst.__class__))
|
||||
|
||||
def __subclasscheck__(cls, sub):
|
||||
def __subclasscheck__(cls, sub: type) -> bool:
|
||||
if cls is not DeprecatedClass.deprecated_class:
|
||||
# we should do the magic only if second `issubclass` argument
|
||||
# is the deprecated class itself - subclasses of the
|
||||
|
|
@ -98,7 +101,7 @@ def create_deprecated_class(
|
|||
mro = getattr(sub, "__mro__", ())
|
||||
return any(c in {cls, new_class} for c in mro)
|
||||
|
||||
def __call__(cls, *args, **kwargs):
|
||||
def __call__(cls, *args: Any, **kwargs: Any) -> Any:
|
||||
old = DeprecatedClass.deprecated_class
|
||||
if cls is old:
|
||||
msg = instance_warn_message.format(
|
||||
|
|
@ -125,7 +128,7 @@ def create_deprecated_class(
|
|||
return deprecated_cls
|
||||
|
||||
|
||||
def _clspath(cls, forced=None):
|
||||
def _clspath(cls: type, forced: Optional[str] = None) -> str:
|
||||
if forced is not None:
|
||||
return forced
|
||||
return f"{cls.__module__}.{cls.__name__}"
|
||||
|
|
@ -134,7 +137,17 @@ def _clspath(cls, forced=None):
|
|||
DEPRECATION_RULES: List[Tuple[str, str]] = []
|
||||
|
||||
|
||||
def update_classpath(path):
|
||||
@overload
|
||||
def update_classpath(path: str) -> str:
|
||||
...
|
||||
|
||||
|
||||
@overload
|
||||
def update_classpath(path: Any) -> Any:
|
||||
...
|
||||
|
||||
|
||||
def update_classpath(path: Any) -> Any:
|
||||
"""Update a deprecated path from an object with its new location"""
|
||||
for prefix, replacement in DEPRECATION_RULES:
|
||||
if isinstance(path, str) and path.startswith(prefix):
|
||||
|
|
@ -147,7 +160,7 @@ def update_classpath(path):
|
|||
return path
|
||||
|
||||
|
||||
def method_is_overridden(subclass, base_class, method_name):
|
||||
def method_is_overridden(subclass: type, base_class: type, method_name: str) -> bool:
|
||||
"""
|
||||
Return True if a method named ``method_name`` of a ``base_class``
|
||||
is overridden in a ``subclass``.
|
||||
|
|
|
|||
|
|
@ -6,17 +6,18 @@ import ctypes
|
|||
import platform
|
||||
import sys
|
||||
from pprint import pformat as pformat_
|
||||
from typing import Any
|
||||
|
||||
from packaging.version import Version as parse_version
|
||||
|
||||
|
||||
def _enable_windows_terminal_processing():
|
||||
def _enable_windows_terminal_processing() -> bool:
|
||||
# https://stackoverflow.com/a/36760881
|
||||
kernel32 = ctypes.windll.kernel32
|
||||
kernel32 = ctypes.windll.kernel32 # type: ignore[attr-defined]
|
||||
return bool(kernel32.SetConsoleMode(kernel32.GetStdHandle(-11), 7))
|
||||
|
||||
|
||||
def _tty_supports_color():
|
||||
def _tty_supports_color() -> bool:
|
||||
if sys.platform != "win32":
|
||||
return True
|
||||
|
||||
|
|
@ -28,7 +29,7 @@ def _tty_supports_color():
|
|||
return _enable_windows_terminal_processing()
|
||||
|
||||
|
||||
def _colorize(text, colorize=True):
|
||||
def _colorize(text: str, colorize: bool = True) -> str:
|
||||
if not colorize or not sys.stdout.isatty() or not _tty_supports_color():
|
||||
return text
|
||||
try:
|
||||
|
|
@ -42,9 +43,9 @@ def _colorize(text, colorize=True):
|
|||
return highlight(text, PythonLexer(), TerminalFormatter())
|
||||
|
||||
|
||||
def pformat(obj, *args, **kwargs):
|
||||
def pformat(obj: Any, *args: Any, **kwargs: Any) -> str:
|
||||
return _colorize(pformat_(obj), kwargs.pop("colorize", True))
|
||||
|
||||
|
||||
def pprint(obj, *args, **kwargs):
|
||||
def pprint(obj: Any, *args: Any, **kwargs: Any) -> None:
|
||||
print(pformat(obj, *args, **kwargs))
|
||||
|
|
|
|||
|
|
@ -2,9 +2,13 @@
|
|||
|
||||
# used in global tests code
|
||||
from time import time # noqa: F401
|
||||
from typing import TYPE_CHECKING, Any, List, Tuple
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from scrapy.core.engine import ExecutionEngine
|
||||
|
||||
|
||||
def get_engine_status(engine):
|
||||
def get_engine_status(engine: "ExecutionEngine") -> List[Tuple[str, Any]]:
|
||||
"""Return a report of the current engine status"""
|
||||
tests = [
|
||||
"time()-engine.start_time",
|
||||
|
|
@ -23,7 +27,7 @@ def get_engine_status(engine):
|
|||
"engine.scraper.slot.needs_backout()",
|
||||
]
|
||||
|
||||
checks = []
|
||||
checks: List[Tuple[str, Any]] = []
|
||||
for test in tests:
|
||||
try:
|
||||
checks += [(test, eval(test))]
|
||||
|
|
@ -33,7 +37,7 @@ def get_engine_status(engine):
|
|||
return checks
|
||||
|
||||
|
||||
def format_engine_status(engine=None):
|
||||
def format_engine_status(engine: "ExecutionEngine") -> str:
|
||||
checks = get_engine_status(engine)
|
||||
s = "Execution engine status\n\n"
|
||||
for test, result in checks:
|
||||
|
|
@ -43,5 +47,5 @@ def format_engine_status(engine=None):
|
|||
return s
|
||||
|
||||
|
||||
def print_engine_status(engine):
|
||||
def print_engine_status(engine: "ExecutionEngine") -> None:
|
||||
print(format_engine_status(engine))
|
||||
|
|
|
|||
|
|
@ -1,9 +1,10 @@
|
|||
import posixpath
|
||||
from ftplib import FTP, error_perm
|
||||
from posixpath import dirname
|
||||
from typing import IO
|
||||
|
||||
|
||||
def ftp_makedirs_cwd(ftp, path, first_call=True):
|
||||
def ftp_makedirs_cwd(ftp: FTP, path: str, first_call: bool = True) -> None:
|
||||
"""Set the current directory of the FTP connection given in the ``ftp``
|
||||
argument (as a ftplib.FTP object), creating all parent directories if they
|
||||
don't exist. The ftplib.FTP object must be already connected and logged in.
|
||||
|
|
@ -18,8 +19,16 @@ def ftp_makedirs_cwd(ftp, path, first_call=True):
|
|||
|
||||
|
||||
def ftp_store_file(
|
||||
*, path, file, host, port, username, password, use_active_mode=False, overwrite=True
|
||||
):
|
||||
*,
|
||||
path: str,
|
||||
file: IO,
|
||||
host: str,
|
||||
port: int,
|
||||
username: str,
|
||||
password: str,
|
||||
use_active_mode: bool = False,
|
||||
overwrite: bool = True,
|
||||
) -> None:
|
||||
"""Opens a FTP connection with passed credentials,sets current directory
|
||||
to the directory extracted from given path, then uploads the file to server
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -1,34 +1,32 @@
|
|||
import struct
|
||||
from gzip import GzipFile
|
||||
from io import BytesIO
|
||||
from typing import List
|
||||
|
||||
from scrapy.http import Response
|
||||
|
||||
|
||||
def gunzip(data):
|
||||
def gunzip(data: bytes) -> bytes:
|
||||
"""Gunzip the given data and return as much data as possible.
|
||||
|
||||
This is resilient to CRC checksum errors.
|
||||
"""
|
||||
f = GzipFile(fileobj=BytesIO(data))
|
||||
output_list = []
|
||||
output_list: List[bytes] = []
|
||||
chunk = b"."
|
||||
while chunk:
|
||||
try:
|
||||
chunk = f.read1(8196)
|
||||
output_list.append(chunk)
|
||||
except (IOError, EOFError, struct.error):
|
||||
except (OSError, EOFError, struct.error):
|
||||
# complete only if there is some data, otherwise re-raise
|
||||
# see issue 87 about catching struct.error
|
||||
# some pages are quite small so output_list is empty and f.extrabuf
|
||||
# contains the whole page content
|
||||
if output_list or getattr(f, "extrabuf", None):
|
||||
try:
|
||||
output_list.append(f.extrabuf[-f.extrasize :])
|
||||
finally:
|
||||
break
|
||||
else:
|
||||
raise
|
||||
# some pages are quite small so output_list is empty
|
||||
if output_list:
|
||||
break
|
||||
raise
|
||||
return b"".join(output_list)
|
||||
|
||||
|
||||
def gzip_magic_number(response):
|
||||
def gzip_magic_number(response: Response) -> bool:
|
||||
return response.body[:3] == b"\x1f\x8b\x08"
|
||||
|
|
|
|||
|
|
@ -2,15 +2,34 @@ import csv
|
|||
import logging
|
||||
import re
|
||||
from io import StringIO
|
||||
from typing import (
|
||||
TYPE_CHECKING,
|
||||
Any,
|
||||
Callable,
|
||||
Dict,
|
||||
Generator,
|
||||
Iterable,
|
||||
List,
|
||||
Literal,
|
||||
Optional,
|
||||
Union,
|
||||
cast,
|
||||
overload,
|
||||
)
|
||||
|
||||
from scrapy.http import Response, TextResponse
|
||||
from scrapy.selector import Selector
|
||||
from scrapy.utils.python import re_rsearch, to_unicode
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from lxml._types import SupportsReadClose
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def xmliter(obj, nodename):
|
||||
def xmliter(
|
||||
obj: Union[Response, str, bytes], nodename: str
|
||||
) -> Generator[Selector, Any, None]:
|
||||
"""Return a iterator of Selector's over all nodes of a XML document,
|
||||
given the name of the node to iterate. Useful for parsing XML feeds.
|
||||
|
||||
|
|
@ -27,20 +46,22 @@ def xmliter(obj, nodename):
|
|||
NAMESPACE_RE = re.compile(r"((xmlns[:A-Za-z]*)=[^>\s]+)", re.S)
|
||||
text = _body_or_str(obj)
|
||||
|
||||
document_header = re.search(DOCUMENT_HEADER_RE, text)
|
||||
document_header = document_header.group().strip() if document_header else ""
|
||||
document_header_match = re.search(DOCUMENT_HEADER_RE, text)
|
||||
document_header = (
|
||||
document_header_match.group().strip() if document_header_match else ""
|
||||
)
|
||||
header_end_idx = re_rsearch(HEADER_END_RE, text)
|
||||
header_end = text[header_end_idx[1] :].strip() if header_end_idx else ""
|
||||
namespaces = {}
|
||||
namespaces: Dict[str, str] = {}
|
||||
if header_end:
|
||||
for tagname in reversed(re.findall(END_TAG_RE, header_end)):
|
||||
assert header_end_idx
|
||||
tag = re.search(
|
||||
rf"<\s*{tagname}.*?xmlns[:=][^>]*>", text[: header_end_idx[1]], re.S
|
||||
)
|
||||
if tag:
|
||||
namespaces.update(
|
||||
reversed(x) for x in re.findall(NAMESPACE_RE, tag.group())
|
||||
)
|
||||
for x in re.findall(NAMESPACE_RE, tag.group()):
|
||||
namespaces[x[1]] = x[0]
|
||||
|
||||
r = re.compile(rf"<{nodename_patt}[\s>].*?</{nodename_patt}>", re.DOTALL)
|
||||
for match in r.finditer(text):
|
||||
|
|
@ -54,12 +75,19 @@ def xmliter(obj, nodename):
|
|||
yield Selector(text=nodetext, type="xml")
|
||||
|
||||
|
||||
def xmliter_lxml(obj, nodename, namespace=None, prefix="x"):
|
||||
def xmliter_lxml(
|
||||
obj: Union[Response, str, bytes],
|
||||
nodename: str,
|
||||
namespace: Optional[str] = None,
|
||||
prefix: str = "x",
|
||||
) -> Generator[Selector, Any, None]:
|
||||
from lxml import etree
|
||||
|
||||
reader = _StreamReader(obj)
|
||||
tag = f"{{{namespace}}}{nodename}" if namespace else nodename
|
||||
iterable = etree.iterparse(reader, tag=tag, encoding=reader.encoding)
|
||||
iterable = etree.iterparse(
|
||||
cast("SupportsReadClose[bytes]", reader), tag=tag, encoding=reader.encoding
|
||||
)
|
||||
selxpath = "//" + (f"{prefix}:{nodename}" if namespace else nodename)
|
||||
for _, node in iterable:
|
||||
nodetext = etree.tostring(node, encoding="unicode")
|
||||
|
|
@ -71,30 +99,46 @@ def xmliter_lxml(obj, nodename, namespace=None, prefix="x"):
|
|||
|
||||
|
||||
class _StreamReader:
|
||||
def __init__(self, obj):
|
||||
self._ptr = 0
|
||||
if isinstance(obj, Response):
|
||||
def __init__(self, obj: Union[Response, str, bytes]):
|
||||
self._ptr: int = 0
|
||||
self._text: Union[str, bytes]
|
||||
if isinstance(obj, TextResponse):
|
||||
self._text, self.encoding = obj.body, obj.encoding
|
||||
elif isinstance(obj, Response):
|
||||
self._text, self.encoding = obj.body, "utf-8"
|
||||
else:
|
||||
self._text, self.encoding = obj, "utf-8"
|
||||
self._is_unicode = isinstance(self._text, str)
|
||||
self._is_unicode: bool = isinstance(self._text, str)
|
||||
self._is_first_read: bool = True
|
||||
|
||||
def read(self, n=65535):
|
||||
self.read = self._read_unicode if self._is_unicode else self._read_string
|
||||
return self.read(n).lstrip()
|
||||
def read(self, n: int = 65535) -> bytes:
|
||||
method: Callable[[int], bytes] = (
|
||||
self._read_unicode if self._is_unicode else self._read_string
|
||||
)
|
||||
result = method(n)
|
||||
if self._is_first_read:
|
||||
self._is_first_read = False
|
||||
result = result.lstrip()
|
||||
return result
|
||||
|
||||
def _read_string(self, n=65535):
|
||||
def _read_string(self, n: int = 65535) -> bytes:
|
||||
s, e = self._ptr, self._ptr + n
|
||||
self._ptr = e
|
||||
return self._text[s:e]
|
||||
return cast(bytes, self._text)[s:e]
|
||||
|
||||
def _read_unicode(self, n=65535):
|
||||
def _read_unicode(self, n: int = 65535) -> bytes:
|
||||
s, e = self._ptr, self._ptr + n
|
||||
self._ptr = e
|
||||
return self._text[s:e].encode("utf-8")
|
||||
return cast(str, self._text)[s:e].encode("utf-8")
|
||||
|
||||
|
||||
def csviter(obj, delimiter=None, headers=None, encoding=None, quotechar=None):
|
||||
def csviter(
|
||||
obj: Union[Response, str, bytes],
|
||||
delimiter: Optional[str] = None,
|
||||
headers: Optional[List[str]] = None,
|
||||
encoding: Optional[str] = None,
|
||||
quotechar: Optional[str] = None,
|
||||
) -> Generator[Dict[str, str], Any, None]:
|
||||
"""Returns an iterator of dictionaries from the given csv object
|
||||
|
||||
obj can be:
|
||||
|
|
@ -112,12 +156,12 @@ def csviter(obj, delimiter=None, headers=None, encoding=None, quotechar=None):
|
|||
|
||||
encoding = obj.encoding if isinstance(obj, TextResponse) else encoding or "utf-8"
|
||||
|
||||
def row_to_unicode(row_):
|
||||
def row_to_unicode(row_: Iterable) -> List[str]:
|
||||
return [to_unicode(field, encoding) for field in row_]
|
||||
|
||||
lines = StringIO(_body_or_str(obj, unicode=True))
|
||||
|
||||
kwargs = {}
|
||||
kwargs: Dict[str, Any] = {}
|
||||
if delimiter:
|
||||
kwargs["delimiter"] = delimiter
|
||||
if quotechar:
|
||||
|
|
@ -147,7 +191,24 @@ def csviter(obj, delimiter=None, headers=None, encoding=None, quotechar=None):
|
|||
yield dict(zip(headers, row))
|
||||
|
||||
|
||||
def _body_or_str(obj, unicode=True):
|
||||
@overload
|
||||
def _body_or_str(obj: Union[Response, str, bytes]) -> str:
|
||||
...
|
||||
|
||||
|
||||
@overload
|
||||
def _body_or_str(obj: Union[Response, str, bytes], unicode: Literal[True]) -> str:
|
||||
...
|
||||
|
||||
|
||||
@overload
|
||||
def _body_or_str(obj: Union[Response, str, bytes], unicode: Literal[False]) -> bytes:
|
||||
...
|
||||
|
||||
|
||||
def _body_or_str(
|
||||
obj: Union[Response, str, bytes], unicode: bool = True
|
||||
) -> Union[str, bytes]:
|
||||
expected_types = (Response, str, bytes)
|
||||
if not isinstance(obj, expected_types):
|
||||
expected_types_str = " or ".join(t.__name__ for t in expected_types)
|
||||
|
|
@ -156,10 +217,10 @@ def _body_or_str(obj, unicode=True):
|
|||
)
|
||||
if isinstance(obj, Response):
|
||||
if not unicode:
|
||||
return obj.body
|
||||
return cast(bytes, obj.body)
|
||||
if isinstance(obj, TextResponse):
|
||||
return obj.text
|
||||
return obj.body.decode("utf-8")
|
||||
return cast(bytes, obj.body).decode("utf-8")
|
||||
if isinstance(obj, str):
|
||||
return obj if unicode else obj.encode("utf-8")
|
||||
return obj.decode("utf-8") if unicode else obj
|
||||
|
|
|
|||
|
|
@ -5,7 +5,7 @@ from scrapy.settings import BaseSettings
|
|||
|
||||
|
||||
def job_dir(settings: BaseSettings) -> Optional[str]:
|
||||
path = settings["JOBDIR"]
|
||||
path: str = settings["JOBDIR"]
|
||||
if path and not Path(path).exists():
|
||||
Path(path).mkdir(parents=True)
|
||||
return path
|
||||
|
|
|
|||
|
|
@ -1,8 +1,11 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import sys
|
||||
import warnings
|
||||
from logging.config import dictConfig
|
||||
from typing import Tuple
|
||||
from types import TracebackType
|
||||
from typing import TYPE_CHECKING, Any, List, Optional, Tuple, Type, Union, cast
|
||||
|
||||
from twisted.python import log as twisted_log
|
||||
from twisted.python.failure import Failure
|
||||
|
|
@ -12,13 +15,25 @@ from scrapy.exceptions import ScrapyDeprecationWarning
|
|||
from scrapy.settings import Settings
|
||||
from scrapy.utils.versions import scrapy_components_versions
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from scrapy.crawler import Crawler
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def failure_to_exc_info(failure: Failure):
|
||||
def failure_to_exc_info(
|
||||
failure: Failure,
|
||||
) -> Optional[Tuple[Type[BaseException], BaseException, Optional[TracebackType]]]:
|
||||
"""Extract exc_info from Failure instances"""
|
||||
if isinstance(failure, Failure):
|
||||
return (failure.type, failure.value, failure.getTracebackObject())
|
||||
assert failure.type
|
||||
assert failure.value
|
||||
return (
|
||||
failure.type,
|
||||
failure.value,
|
||||
cast(Optional[TracebackType], failure.getTracebackObject()),
|
||||
)
|
||||
return None
|
||||
|
||||
|
||||
class TopLevelFormatter(logging.Filter):
|
||||
|
|
@ -33,10 +48,10 @@ class TopLevelFormatter(logging.Filter):
|
|||
``loggers`` list where it should act.
|
||||
"""
|
||||
|
||||
def __init__(self, loggers=None):
|
||||
self.loggers = loggers or []
|
||||
def __init__(self, loggers: Optional[List[str]] = None):
|
||||
self.loggers: List[str] = loggers or []
|
||||
|
||||
def filter(self, record):
|
||||
def filter(self, record: logging.LogRecord) -> bool:
|
||||
if any(record.name.startswith(logger + ".") for logger in self.loggers):
|
||||
record.name = record.name.split(".", 1)[0]
|
||||
return True
|
||||
|
|
@ -62,7 +77,9 @@ DEFAULT_LOGGING = {
|
|||
}
|
||||
|
||||
|
||||
def configure_logging(settings=None, install_root_handler=True):
|
||||
def configure_logging(
|
||||
settings: Union[Settings, dict, None] = None, install_root_handler: bool = True
|
||||
) -> None:
|
||||
"""
|
||||
Initialize logging defaults for Scrapy.
|
||||
|
||||
|
|
@ -99,13 +116,16 @@ def configure_logging(settings=None, install_root_handler=True):
|
|||
settings = Settings(settings)
|
||||
|
||||
if settings.getbool("LOG_STDOUT"):
|
||||
sys.stdout = StreamLogger(logging.getLogger("stdout"))
|
||||
sys.stdout = StreamLogger(logging.getLogger("stdout")) # type: ignore[assignment]
|
||||
|
||||
if install_root_handler:
|
||||
install_scrapy_root_handler(settings)
|
||||
|
||||
|
||||
def install_scrapy_root_handler(settings):
|
||||
_scrapy_root_handler: Optional[logging.Handler] = None
|
||||
|
||||
|
||||
def install_scrapy_root_handler(settings: Settings) -> None:
|
||||
global _scrapy_root_handler
|
||||
|
||||
if (
|
||||
|
|
@ -118,16 +138,14 @@ def install_scrapy_root_handler(settings):
|
|||
logging.root.addHandler(_scrapy_root_handler)
|
||||
|
||||
|
||||
def get_scrapy_root_handler():
|
||||
def get_scrapy_root_handler() -> Optional[logging.Handler]:
|
||||
return _scrapy_root_handler
|
||||
|
||||
|
||||
_scrapy_root_handler = None
|
||||
|
||||
|
||||
def _get_handler(settings):
|
||||
def _get_handler(settings: Settings) -> logging.Handler:
|
||||
"""Return a log handler object according to settings"""
|
||||
filename = settings.get("LOG_FILE")
|
||||
handler: logging.Handler
|
||||
if filename:
|
||||
mode = "a" if settings.getbool("LOG_FILE_APPEND") else "w"
|
||||
encoding = settings.get("LOG_ENCODING")
|
||||
|
|
@ -181,16 +199,16 @@ class StreamLogger:
|
|||
https://www.electricmonk.nl/log/2011/08/14/redirect-stdout-and-stderr-to-a-logger-in-python/
|
||||
"""
|
||||
|
||||
def __init__(self, logger, log_level=logging.INFO):
|
||||
self.logger = logger
|
||||
self.log_level = log_level
|
||||
self.linebuf = ""
|
||||
def __init__(self, logger: logging.Logger, log_level: int = logging.INFO):
|
||||
self.logger: logging.Logger = logger
|
||||
self.log_level: int = log_level
|
||||
self.linebuf: str = ""
|
||||
|
||||
def write(self, buf):
|
||||
def write(self, buf: str) -> None:
|
||||
for line in buf.rstrip().splitlines():
|
||||
self.logger.log(self.log_level, line.rstrip())
|
||||
|
||||
def flush(self):
|
||||
def flush(self) -> None:
|
||||
for h in self.logger.handlers:
|
||||
h.flush()
|
||||
|
||||
|
|
@ -198,11 +216,11 @@ class StreamLogger:
|
|||
class LogCounterHandler(logging.Handler):
|
||||
"""Record log levels count into a crawler stats"""
|
||||
|
||||
def __init__(self, crawler, *args, **kwargs):
|
||||
def __init__(self, crawler: Crawler, *args: Any, **kwargs: Any):
|
||||
super().__init__(*args, **kwargs)
|
||||
self.crawler = crawler
|
||||
self.crawler: Crawler = crawler
|
||||
|
||||
def emit(self, record):
|
||||
def emit(self, record: logging.LogRecord) -> None:
|
||||
sname = f"log_count/{record.levelname}"
|
||||
self.crawler.stats.inc_value(sname)
|
||||
|
||||
|
|
|
|||
|
|
@ -10,7 +10,21 @@ from contextlib import contextmanager
|
|||
from functools import partial
|
||||
from importlib import import_module
|
||||
from pkgutil import iter_modules
|
||||
from typing import TYPE_CHECKING, Any, Callable, Union
|
||||
from types import ModuleType
|
||||
from typing import (
|
||||
IO,
|
||||
TYPE_CHECKING,
|
||||
Any,
|
||||
Callable,
|
||||
Deque,
|
||||
Generator,
|
||||
Iterable,
|
||||
List,
|
||||
Optional,
|
||||
Pattern,
|
||||
Union,
|
||||
cast,
|
||||
)
|
||||
|
||||
from w3lib.html import replace_entities
|
||||
|
||||
|
|
@ -26,7 +40,7 @@ if TYPE_CHECKING:
|
|||
_ITERABLE_SINGLE_VALUES = dict, Item, str, bytes
|
||||
|
||||
|
||||
def arg_to_iter(arg):
|
||||
def arg_to_iter(arg: Any) -> Iterable[Any]:
|
||||
"""Convert an argument to an iterable. The argument can be a None, single
|
||||
value, or an iterable.
|
||||
|
||||
|
|
@ -35,7 +49,7 @@ def arg_to_iter(arg):
|
|||
if arg is None:
|
||||
return []
|
||||
if not isinstance(arg, _ITERABLE_SINGLE_VALUES) and hasattr(arg, "__iter__"):
|
||||
return arg
|
||||
return cast(Iterable[Any], arg)
|
||||
return [arg]
|
||||
|
||||
|
||||
|
|
@ -53,7 +67,7 @@ def load_object(path: Union[str, Callable]) -> Any:
|
|||
if callable(path):
|
||||
return path
|
||||
raise TypeError(
|
||||
"Unexpected argument type, expected string " f"or object, got: {type(path)}"
|
||||
f"Unexpected argument type, expected string or object, got: {type(path)}"
|
||||
)
|
||||
|
||||
try:
|
||||
|
|
@ -72,7 +86,7 @@ def load_object(path: Union[str, Callable]) -> Any:
|
|||
return obj
|
||||
|
||||
|
||||
def walk_modules(path):
|
||||
def walk_modules(path: str) -> List[ModuleType]:
|
||||
"""Loads a module and all its submodules from the given module path and
|
||||
returns them. If *any* module throws an exception while importing, that
|
||||
exception is thrown back.
|
||||
|
|
@ -80,7 +94,7 @@ def walk_modules(path):
|
|||
For example: walk_modules('scrapy.utils')
|
||||
"""
|
||||
|
||||
mods = []
|
||||
mods: List[ModuleType] = []
|
||||
mod = import_module(path)
|
||||
mods.append(mod)
|
||||
if hasattr(mod, "__path__"):
|
||||
|
|
@ -94,7 +108,9 @@ def walk_modules(path):
|
|||
return mods
|
||||
|
||||
|
||||
def extract_regex(regex, text, encoding="utf-8"):
|
||||
def extract_regex(
|
||||
regex: Union[str, Pattern], text: str, encoding: str = "utf-8"
|
||||
) -> List[str]:
|
||||
"""Extract a list of unicode strings from the given text/encoding using the following policies:
|
||||
|
||||
* if the regex contains a named group called "extract" that will be returned
|
||||
|
|
@ -111,9 +127,11 @@ def extract_regex(regex, text, encoding="utf-8"):
|
|||
regex = re.compile(regex, re.UNICODE)
|
||||
|
||||
try:
|
||||
strings = [regex.search(text).group("extract")] # named group
|
||||
# named group
|
||||
strings = [regex.search(text).group("extract")] # type: ignore[union-attr]
|
||||
except Exception:
|
||||
strings = regex.findall(text) # full regex or numbered groups
|
||||
# full regex or numbered groups
|
||||
strings = regex.findall(text)
|
||||
strings = flatten(strings)
|
||||
|
||||
if isinstance(text, str):
|
||||
|
|
@ -123,7 +141,7 @@ def extract_regex(regex, text, encoding="utf-8"):
|
|||
]
|
||||
|
||||
|
||||
def md5sum(file):
|
||||
def md5sum(file: IO) -> str:
|
||||
"""Calculate the md5 checksum of a file-like object without reading its
|
||||
whole content in memory.
|
||||
|
||||
|
|
@ -140,7 +158,7 @@ def md5sum(file):
|
|||
return m.hexdigest()
|
||||
|
||||
|
||||
def rel_has_nofollow(rel):
|
||||
def rel_has_nofollow(rel: Optional[str]) -> bool:
|
||||
"""Return True if link rel attribute has nofollow type"""
|
||||
return rel is not None and "nofollow" in rel.replace(",", " ").split()
|
||||
|
||||
|
|
@ -181,7 +199,7 @@ def create_instance(objcls, settings, crawler, *args, **kwargs):
|
|||
|
||||
|
||||
@contextmanager
|
||||
def set_environ(**kwargs):
|
||||
def set_environ(**kwargs: str) -> Generator[None, Any, None]:
|
||||
"""Temporarily set environment variables inside the context manager and
|
||||
fully restore previous environment afterwards
|
||||
"""
|
||||
|
|
@ -198,11 +216,11 @@ def set_environ(**kwargs):
|
|||
os.environ[k] = v
|
||||
|
||||
|
||||
def walk_callable(node):
|
||||
def walk_callable(node: ast.AST) -> Generator[ast.AST, Any, None]:
|
||||
"""Similar to ``ast.walk``, but walks only function body and skips nested
|
||||
functions defined within the node.
|
||||
"""
|
||||
todo = deque([node])
|
||||
todo: Deque[ast.AST] = deque([node])
|
||||
walked_func_def = False
|
||||
while todo:
|
||||
node = todo.popleft()
|
||||
|
|
@ -217,15 +235,15 @@ def walk_callable(node):
|
|||
_generator_callbacks_cache = LocalWeakReferencedCache(limit=128)
|
||||
|
||||
|
||||
def is_generator_with_return_value(callable):
|
||||
def is_generator_with_return_value(callable: Callable) -> bool:
|
||||
"""
|
||||
Returns True if a callable is a generator function which includes a
|
||||
'return' statement with a value different than None, False otherwise
|
||||
"""
|
||||
if callable in _generator_callbacks_cache:
|
||||
return _generator_callbacks_cache[callable]
|
||||
return bool(_generator_callbacks_cache[callable])
|
||||
|
||||
def returns_none(return_node):
|
||||
def returns_none(return_node: ast.Return) -> bool:
|
||||
value = return_node.value
|
||||
return (
|
||||
value is None or isinstance(value, ast.NameConstant) and value.value is None
|
||||
|
|
@ -248,10 +266,10 @@ def is_generator_with_return_value(callable):
|
|||
for node in walk_callable(tree):
|
||||
if isinstance(node, ast.Return) and not returns_none(node):
|
||||
_generator_callbacks_cache[callable] = True
|
||||
return _generator_callbacks_cache[callable]
|
||||
return bool(_generator_callbacks_cache[callable])
|
||||
|
||||
_generator_callbacks_cache[callable] = False
|
||||
return _generator_callbacks_cache[callable]
|
||||
return bool(_generator_callbacks_cache[callable])
|
||||
|
||||
|
||||
def warn_on_generator_with_return_value(spider: "Spider", callable: Callable) -> None:
|
||||
|
|
|
|||
|
|
@ -1,6 +1,13 @@
|
|||
import signal
|
||||
from types import FrameType
|
||||
from typing import Any, Callable, Dict, Optional, Union
|
||||
|
||||
signal_names = {}
|
||||
# copy of _HANDLER from typeshed/stdlib/signal.pyi
|
||||
SignalHandlerT = Union[
|
||||
Callable[[int, Optional[FrameType]], Any], int, signal.Handlers, None
|
||||
]
|
||||
|
||||
signal_names: Dict[int, str] = {}
|
||||
for signame in dir(signal):
|
||||
if signame.startswith("SIG") and not signame.startswith("SIG_"):
|
||||
signum = getattr(signal, signame)
|
||||
|
|
@ -8,7 +15,9 @@ for signame in dir(signal):
|
|||
signal_names[signum] = signame
|
||||
|
||||
|
||||
def install_shutdown_handlers(function, override_sigint=True):
|
||||
def install_shutdown_handlers(
|
||||
function: SignalHandlerT, override_sigint: bool = True
|
||||
) -> None:
|
||||
"""Install the given function as a signal handler for all common shutdown
|
||||
signals (such as SIGINT, SIGTERM, etc). If override_sigint is ``False`` the
|
||||
SIGINT handler won't be install if there is already a handler in place
|
||||
|
|
|
|||
|
|
@ -11,9 +11,9 @@ ENVVAR = "SCRAPY_SETTINGS_MODULE"
|
|||
DATADIR_CFG_SECTION = "datadir"
|
||||
|
||||
|
||||
def inside_project():
|
||||
scrapy_module = os.environ.get("SCRAPY_SETTINGS_MODULE")
|
||||
if scrapy_module is not None:
|
||||
def inside_project() -> bool:
|
||||
scrapy_module = os.environ.get(ENVVAR)
|
||||
if scrapy_module:
|
||||
try:
|
||||
import_module(scrapy_module)
|
||||
except ImportError as exc:
|
||||
|
|
@ -25,7 +25,7 @@ def inside_project():
|
|||
return bool(closest_scrapy_cfg())
|
||||
|
||||
|
||||
def project_data_dir(project="default") -> str:
|
||||
def project_data_dir(project: str = "default") -> str:
|
||||
"""Return the current project data dir, creating it if it doesn't exist"""
|
||||
if not inside_project():
|
||||
raise NotConfigured("Not inside a project")
|
||||
|
|
@ -44,7 +44,7 @@ def project_data_dir(project="default") -> str:
|
|||
return str(d)
|
||||
|
||||
|
||||
def data_path(path: str, createdir=False) -> str:
|
||||
def data_path(path: str, createdir: bool = False) -> str:
|
||||
"""
|
||||
Return the given path joined with the .scrapy data directory.
|
||||
If given an absolute path, return it unmodified.
|
||||
|
|
@ -60,7 +60,7 @@ def data_path(path: str, createdir=False) -> str:
|
|||
return str(path_obj)
|
||||
|
||||
|
||||
def get_project_settings():
|
||||
def get_project_settings() -> Settings:
|
||||
if ENVVAR not in os.environ:
|
||||
project = os.environ.get("SCRAPY_PROJECT", "default")
|
||||
init_env(project)
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
"""
|
||||
This module contains essential stuff that should've come with Python itself ;)
|
||||
"""
|
||||
import collections.abc
|
||||
import gc
|
||||
import inspect
|
||||
import re
|
||||
|
|
@ -12,9 +13,17 @@ from typing import (
|
|||
Any,
|
||||
AsyncGenerator,
|
||||
AsyncIterable,
|
||||
AsyncIterator,
|
||||
Callable,
|
||||
Dict,
|
||||
Generator,
|
||||
Iterable,
|
||||
Iterator,
|
||||
List,
|
||||
Mapping,
|
||||
Optional,
|
||||
Pattern,
|
||||
Tuple,
|
||||
Union,
|
||||
overload,
|
||||
)
|
||||
|
|
@ -22,7 +31,7 @@ from typing import (
|
|||
from scrapy.utils.asyncgen import as_async_generator
|
||||
|
||||
|
||||
def flatten(x):
|
||||
def flatten(x: Iterable) -> list:
|
||||
"""flatten(sequence) -> list
|
||||
|
||||
Returns a single, flat list which contains all elements retrieved
|
||||
|
|
@ -42,7 +51,7 @@ def flatten(x):
|
|||
return list(iflatten(x))
|
||||
|
||||
|
||||
def iflatten(x):
|
||||
def iflatten(x: Iterable) -> Iterable:
|
||||
"""iflatten(sequence) -> iterator
|
||||
|
||||
Similar to ``.flatten()``, but returns iterator instead"""
|
||||
|
|
@ -78,7 +87,7 @@ def is_listlike(x: Any) -> bool:
|
|||
return hasattr(x, "__iter__") and not isinstance(x, (str, bytes))
|
||||
|
||||
|
||||
def unique(list_, key=lambda x: x):
|
||||
def unique(list_: Iterable, key: Callable[[Any], Any] = lambda x: x) -> list:
|
||||
"""efficient function to uniquify a list preserving item order"""
|
||||
seen = set()
|
||||
result = []
|
||||
|
|
@ -124,7 +133,9 @@ def to_bytes(
|
|||
return text.encode(encoding, errors)
|
||||
|
||||
|
||||
def re_rsearch(pattern, text, chunk_size=1024):
|
||||
def re_rsearch(
|
||||
pattern: Union[str, Pattern], text: str, chunk_size: int = 1024
|
||||
) -> Optional[Tuple[int, int]]:
|
||||
"""
|
||||
This function does a reverse search in a text using a regular expression
|
||||
given in the attribute 'pattern'.
|
||||
|
|
@ -138,7 +149,7 @@ def re_rsearch(pattern, text, chunk_size=1024):
|
|||
the start position of the match, and the ending (regarding the entire text).
|
||||
"""
|
||||
|
||||
def _chunk_iter():
|
||||
def _chunk_iter() -> Generator[Tuple[str, int], Any, None]:
|
||||
offset = len(text)
|
||||
while True:
|
||||
offset -= chunk_size * 1024
|
||||
|
|
@ -158,14 +169,14 @@ def re_rsearch(pattern, text, chunk_size=1024):
|
|||
return None
|
||||
|
||||
|
||||
def memoizemethod_noargs(method):
|
||||
def memoizemethod_noargs(method: Callable) -> Callable:
|
||||
"""Decorator to cache the result of a method (without arguments) using a
|
||||
weak reference to its object
|
||||
"""
|
||||
cache = weakref.WeakKeyDictionary()
|
||||
cache: weakref.WeakKeyDictionary[Any, Any] = weakref.WeakKeyDictionary()
|
||||
|
||||
@wraps(method)
|
||||
def new_method(self, *args, **kwargs):
|
||||
def new_method(self: Any, *args: Any, **kwargs: Any) -> Any:
|
||||
if self not in cache:
|
||||
cache[self] = method(self, *args, **kwargs)
|
||||
return cache[self]
|
||||
|
|
@ -187,12 +198,12 @@ def binary_is_text(data: bytes) -> bool:
|
|||
return all(c not in _BINARYCHARS for c in data)
|
||||
|
||||
|
||||
def get_func_args(func, stripself=False):
|
||||
def get_func_args(func: Callable, stripself: bool = False) -> List[str]:
|
||||
"""Return the argument name list of a callable object"""
|
||||
if not callable(func):
|
||||
raise TypeError(f"func must be callable, got '{type(func).__name__}'")
|
||||
|
||||
args = []
|
||||
args: List[str] = []
|
||||
try:
|
||||
sig = inspect.signature(func)
|
||||
except ValueError:
|
||||
|
|
@ -217,7 +228,7 @@ def get_func_args(func, stripself=False):
|
|||
return args
|
||||
|
||||
|
||||
def get_spec(func):
|
||||
def get_spec(func: Callable) -> Tuple[List[str], Dict[str, Any]]:
|
||||
"""Returns (args, kwargs) tuple for a function
|
||||
>>> import re
|
||||
>>> get_spec(re.match)
|
||||
|
|
@ -246,7 +257,7 @@ def get_spec(func):
|
|||
else:
|
||||
raise TypeError(f"{type(func)} is not callable")
|
||||
|
||||
defaults = spec.defaults or []
|
||||
defaults: Tuple[Any, ...] = spec.defaults or ()
|
||||
|
||||
firstdefault = len(spec.args) - len(defaults)
|
||||
args = spec.args[:firstdefault]
|
||||
|
|
@ -254,7 +265,9 @@ def get_spec(func):
|
|||
return args, kwargs
|
||||
|
||||
|
||||
def equal_attributes(obj1, obj2, attributes):
|
||||
def equal_attributes(
|
||||
obj1: Any, obj2: Any, attributes: Optional[List[Union[str, Callable]]]
|
||||
) -> bool:
|
||||
"""Compare two objects attributes"""
|
||||
# not attributes given return False by default
|
||||
if not attributes:
|
||||
|
|
@ -282,19 +295,20 @@ def without_none_values(iterable: Iterable) -> Iterable:
|
|||
...
|
||||
|
||||
|
||||
def without_none_values(iterable):
|
||||
def without_none_values(iterable: Union[Mapping, Iterable]) -> Union[dict, Iterable]:
|
||||
"""Return a copy of ``iterable`` with all ``None`` entries removed.
|
||||
|
||||
If ``iterable`` is a mapping, return a dictionary where all pairs that have
|
||||
value ``None`` have been removed.
|
||||
"""
|
||||
try:
|
||||
if isinstance(iterable, collections.abc.Mapping):
|
||||
return {k: v for k, v in iterable.items() if v is not None}
|
||||
except AttributeError:
|
||||
return type(iterable)((v for v in iterable if v is not None))
|
||||
else:
|
||||
# the iterable __init__ must take another iterable
|
||||
return type(iterable)(v for v in iterable if v is not None) # type: ignore[call-arg]
|
||||
|
||||
|
||||
def global_object_name(obj):
|
||||
def global_object_name(obj: Any) -> str:
|
||||
"""
|
||||
Return full name of a global object.
|
||||
|
||||
|
|
@ -307,14 +321,14 @@ def global_object_name(obj):
|
|||
|
||||
if hasattr(sys, "pypy_version_info"):
|
||||
|
||||
def garbage_collect():
|
||||
def garbage_collect() -> None:
|
||||
# Collecting weakreferences can take two collections on PyPy.
|
||||
gc.collect()
|
||||
gc.collect()
|
||||
|
||||
else:
|
||||
|
||||
def garbage_collect():
|
||||
def garbage_collect() -> None:
|
||||
gc.collect()
|
||||
|
||||
|
||||
|
|
@ -329,10 +343,10 @@ class MutableChain(Iterable):
|
|||
def extend(self, *iterables: Iterable) -> None:
|
||||
self.data = chain(self.data, chain.from_iterable(iterables))
|
||||
|
||||
def __iter__(self):
|
||||
def __iter__(self) -> Iterator:
|
||||
return self
|
||||
|
||||
def __next__(self):
|
||||
def __next__(self) -> Any:
|
||||
return next(self.data)
|
||||
|
||||
|
||||
|
|
@ -353,8 +367,8 @@ class MutableAsyncChain(AsyncIterable):
|
|||
def extend(self, *iterables: Union[Iterable, AsyncIterable]) -> None:
|
||||
self.data = _async_chain(self.data, _async_chain(*iterables))
|
||||
|
||||
def __aiter__(self):
|
||||
def __aiter__(self) -> AsyncIterator:
|
||||
return self
|
||||
|
||||
async def __anext__(self):
|
||||
async def __anext__(self) -> Any:
|
||||
return await self.data.__anext__()
|
||||
|
|
|
|||
|
|
@ -1,7 +1,8 @@
|
|||
import asyncio
|
||||
import sys
|
||||
from asyncio import AbstractEventLoop, AbstractEventLoopPolicy
|
||||
from contextlib import suppress
|
||||
from typing import Any, Callable, Dict, Optional, Sequence
|
||||
from typing import Any, Callable, Dict, Optional, Sequence, Type
|
||||
from warnings import catch_warnings, filterwarnings, warn
|
||||
|
||||
from twisted.internet import asyncioreactor, error
|
||||
|
|
@ -57,7 +58,7 @@ class CallLaterOnce:
|
|||
return self._func(*self._a, **self._kw)
|
||||
|
||||
|
||||
def set_asyncio_event_loop_policy():
|
||||
def set_asyncio_event_loop_policy() -> None:
|
||||
"""The policy functions from asyncio often behave unexpectedly,
|
||||
so we restrict their use to the absolutely essential case.
|
||||
This should only be used to install the reactor.
|
||||
|
|
@ -65,7 +66,7 @@ def set_asyncio_event_loop_policy():
|
|||
_get_asyncio_event_loop_policy()
|
||||
|
||||
|
||||
def get_asyncio_event_loop_policy():
|
||||
def get_asyncio_event_loop_policy() -> AbstractEventLoopPolicy:
|
||||
warn(
|
||||
"Call to deprecated function "
|
||||
"scrapy.utils.reactor.get_asyncio_event_loop_policy().\n"
|
||||
|
|
@ -81,7 +82,7 @@ def get_asyncio_event_loop_policy():
|
|||
return _get_asyncio_event_loop_policy()
|
||||
|
||||
|
||||
def _get_asyncio_event_loop_policy():
|
||||
def _get_asyncio_event_loop_policy() -> AbstractEventLoopPolicy:
|
||||
policy = asyncio.get_event_loop_policy()
|
||||
if (
|
||||
sys.version_info >= (3, 8)
|
||||
|
|
@ -93,7 +94,7 @@ def _get_asyncio_event_loop_policy():
|
|||
return policy
|
||||
|
||||
|
||||
def install_reactor(reactor_path, event_loop_path=None):
|
||||
def install_reactor(reactor_path: str, event_loop_path: Optional[str] = None) -> None:
|
||||
"""Installs the :mod:`~twisted.internet.reactor` with the specified
|
||||
import path. Also installs the asyncio event loop with the specified import
|
||||
path if the asyncio reactor is enabled"""
|
||||
|
|
@ -111,14 +112,14 @@ def install_reactor(reactor_path, event_loop_path=None):
|
|||
installer()
|
||||
|
||||
|
||||
def _get_asyncio_event_loop():
|
||||
def _get_asyncio_event_loop() -> AbstractEventLoop:
|
||||
return set_asyncio_event_loop(None)
|
||||
|
||||
|
||||
def set_asyncio_event_loop(event_loop_path):
|
||||
def set_asyncio_event_loop(event_loop_path: Optional[str]) -> AbstractEventLoop:
|
||||
"""Sets and returns the event loop with specified import path."""
|
||||
if event_loop_path is not None:
|
||||
event_loop_class = load_object(event_loop_path)
|
||||
event_loop_class: Type[AbstractEventLoop] = load_object(event_loop_path)
|
||||
event_loop = event_loop_class()
|
||||
asyncio.set_event_loop(event_loop)
|
||||
else:
|
||||
|
|
@ -146,7 +147,7 @@ def set_asyncio_event_loop(event_loop_path):
|
|||
return event_loop
|
||||
|
||||
|
||||
def verify_installed_reactor(reactor_path):
|
||||
def verify_installed_reactor(reactor_path: str) -> None:
|
||||
"""Raises :exc:`Exception` if the installed
|
||||
:mod:`~twisted.internet.reactor` does not match the specified import
|
||||
path."""
|
||||
|
|
@ -162,7 +163,7 @@ def verify_installed_reactor(reactor_path):
|
|||
raise Exception(msg)
|
||||
|
||||
|
||||
def verify_installed_asyncio_event_loop(loop_path):
|
||||
def verify_installed_asyncio_event_loop(loop_path: str) -> None:
|
||||
from twisted.internet import reactor
|
||||
|
||||
loop_class = load_object(loop_path)
|
||||
|
|
@ -181,7 +182,7 @@ def verify_installed_asyncio_event_loop(loop_path):
|
|||
)
|
||||
|
||||
|
||||
def is_asyncio_reactor_installed():
|
||||
def is_asyncio_reactor_installed() -> bool:
|
||||
from twisted.internet import reactor
|
||||
|
||||
return isinstance(reactor, asyncioreactor.AsyncioSelectorReactor)
|
||||
|
|
|
|||
|
|
@ -1,27 +0,0 @@
|
|||
import warnings
|
||||
from typing import Optional
|
||||
|
||||
import scrapy
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
from scrapy.utils.request import request_from_dict as _from_dict
|
||||
|
||||
warnings.warn(
|
||||
(
|
||||
"Module scrapy.utils.reqser is deprecated, please use request.to_dict method"
|
||||
" and/or scrapy.utils.request.request_from_dict instead"
|
||||
),
|
||||
category=ScrapyDeprecationWarning,
|
||||
stacklevel=2,
|
||||
)
|
||||
|
||||
|
||||
def request_to_dict(
|
||||
request: "scrapy.Request", spider: Optional["scrapy.Spider"] = None
|
||||
) -> dict:
|
||||
return request.to_dict(spider=spider)
|
||||
|
||||
|
||||
def request_from_dict(
|
||||
d: dict, spider: Optional["scrapy.Spider"] = None
|
||||
) -> "scrapy.Request":
|
||||
return _from_dict(d, spider=spider)
|
||||
|
|
@ -6,7 +6,19 @@ scrapy.http.Request objects
|
|||
import hashlib
|
||||
import json
|
||||
import warnings
|
||||
from typing import Dict, Iterable, List, Optional, Tuple, Union
|
||||
from typing import (
|
||||
TYPE_CHECKING,
|
||||
Any,
|
||||
Dict,
|
||||
Generator,
|
||||
Iterable,
|
||||
List,
|
||||
Optional,
|
||||
Protocol,
|
||||
Tuple,
|
||||
Type,
|
||||
Union,
|
||||
)
|
||||
from urllib.parse import urlunparse
|
||||
from weakref import WeakKeyDictionary
|
||||
|
||||
|
|
@ -19,11 +31,16 @@ from scrapy.utils.httpobj import urlparse_cached
|
|||
from scrapy.utils.misc import load_object
|
||||
from scrapy.utils.python import to_bytes, to_unicode
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from scrapy.crawler import Crawler
|
||||
|
||||
_deprecated_fingerprint_cache: "WeakKeyDictionary[Request, Dict[Tuple[Optional[Tuple[bytes, ...]], bool], str]]"
|
||||
_deprecated_fingerprint_cache = WeakKeyDictionary()
|
||||
|
||||
|
||||
def _serialize_headers(headers, request):
|
||||
def _serialize_headers(
|
||||
headers: Iterable[bytes], request: Request
|
||||
) -> Generator[bytes, Any, None]:
|
||||
for header in headers:
|
||||
if header in request.headers:
|
||||
yield header
|
||||
|
|
@ -139,7 +156,7 @@ def request_fingerprint(
|
|||
return cache[cache_key]
|
||||
|
||||
|
||||
def _request_fingerprint_as_bytes(*args, **kwargs):
|
||||
def _request_fingerprint_as_bytes(*args: Any, **kwargs: Any) -> bytes:
|
||||
with warnings.catch_warnings():
|
||||
warnings.simplefilter("ignore")
|
||||
return bytes.fromhex(request_fingerprint(*args, **kwargs))
|
||||
|
|
@ -214,6 +231,11 @@ def fingerprint(
|
|||
return cache[cache_key]
|
||||
|
||||
|
||||
class RequestFingerprinterProtocol(Protocol):
|
||||
def fingerprint(self, request: Request) -> bytes:
|
||||
...
|
||||
|
||||
|
||||
class RequestFingerprinter:
|
||||
"""Default fingerprinter.
|
||||
|
||||
|
|
@ -231,7 +253,7 @@ class RequestFingerprinter:
|
|||
def from_crawler(cls, crawler):
|
||||
return cls(crawler)
|
||||
|
||||
def __init__(self, crawler=None):
|
||||
def __init__(self, crawler: Optional["Crawler"] = None):
|
||||
if crawler:
|
||||
implementation = crawler.settings.get(
|
||||
"REQUEST_FINGERPRINTER_IMPLEMENTATION"
|
||||
|
|
@ -265,7 +287,7 @@ class RequestFingerprinter:
|
|||
f"and '2.7'."
|
||||
)
|
||||
|
||||
def fingerprint(self, request: Request):
|
||||
def fingerprint(self, request: Request) -> bytes:
|
||||
return self._fingerprint(request)
|
||||
|
||||
|
||||
|
|
@ -311,7 +333,7 @@ def request_from_dict(d: dict, *, spider: Optional[Spider] = None) -> Request:
|
|||
If a spider is given, it will try to resolve the callbacks looking at the
|
||||
spider for methods with the same name.
|
||||
"""
|
||||
request_cls = load_object(d["_class"]) if "_class" in d else Request
|
||||
request_cls: Type[Request] = load_object(d["_class"]) if "_class" in d else Request
|
||||
kwargs = {key: value for key, value in d.items() if key in request_cls.attributes}
|
||||
if d.get("callback") and spider:
|
||||
kwargs["callback"] = _get_method(spider, d["callback"])
|
||||
|
|
@ -320,7 +342,7 @@ def request_from_dict(d: dict, *, spider: Optional[Spider] = None) -> Request:
|
|||
return request_cls(**kwargs)
|
||||
|
||||
|
||||
def _get_method(obj, name):
|
||||
def _get_method(obj: Any, name: Any) -> Any:
|
||||
"""Helper function for request_from_dict"""
|
||||
name = str(name)
|
||||
try:
|
||||
|
|
|
|||
|
|
@ -42,8 +42,7 @@ def get_meta_refresh(
|
|||
"""Parse the http-equiv refresh parameter from the given response"""
|
||||
if response not in _metaref_cache:
|
||||
text = response.text[0:4096]
|
||||
# a w3lib typing bug here, fixed in https://github.com/scrapy/w3lib/pull/211
|
||||
_metaref_cache[response] = html.get_meta_refresh( # type: ignore[assignment]
|
||||
_metaref_cache[response] = html.get_meta_refresh(
|
||||
text, response.url, response.encoding, ignore_tags=ignore_tags
|
||||
)
|
||||
return _metaref_cache[response]
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
import datetime
|
||||
import decimal
|
||||
import json
|
||||
from typing import Any
|
||||
|
||||
from itemadapter import ItemAdapter, is_item
|
||||
from twisted.internet import defer
|
||||
|
|
@ -12,7 +13,7 @@ class ScrapyJSONEncoder(json.JSONEncoder):
|
|||
DATE_FORMAT = "%Y-%m-%d"
|
||||
TIME_FORMAT = "%H:%M:%S"
|
||||
|
||||
def default(self, o):
|
||||
def default(self, o: Any) -> Any:
|
||||
if isinstance(o, set):
|
||||
return list(o)
|
||||
if isinstance(o, datetime.datetime):
|
||||
|
|
|
|||
|
|
@ -23,7 +23,10 @@ logger = logging.getLogger(__name__)
|
|||
|
||||
|
||||
def send_catch_log(
|
||||
signal=Any, sender=Anonymous, *arguments, **named
|
||||
signal: TypingAny = Any,
|
||||
sender: TypingAny = Anonymous,
|
||||
*arguments: TypingAny,
|
||||
**named: TypingAny
|
||||
) -> List[Tuple[TypingAny, TypingAny]]:
|
||||
"""Like pydispatcher.robust.sendRobust but it also logs errors and returns
|
||||
Failures instead of exceptions.
|
||||
|
|
@ -65,13 +68,18 @@ def send_catch_log(
|
|||
return responses
|
||||
|
||||
|
||||
def send_catch_log_deferred(signal=Any, sender=Anonymous, *arguments, **named):
|
||||
def send_catch_log_deferred(
|
||||
signal: TypingAny = Any,
|
||||
sender: TypingAny = Anonymous,
|
||||
*arguments: TypingAny,
|
||||
**named: TypingAny
|
||||
) -> Deferred:
|
||||
"""Like send_catch_log but supports returning deferreds on signal handlers.
|
||||
Returns a deferred that gets fired once all signal handlers deferreds were
|
||||
fired.
|
||||
"""
|
||||
|
||||
def logerror(failure, recv):
|
||||
def logerror(failure: Failure, recv: Any) -> Failure:
|
||||
if dont_log is None or not isinstance(failure.value, dont_log):
|
||||
logger.error(
|
||||
"Error caught on signal handler: %(receiver)s",
|
||||
|
|
@ -96,7 +104,7 @@ def send_catch_log_deferred(signal=Any, sender=Anonymous, *arguments, **named):
|
|||
return d
|
||||
|
||||
|
||||
def disconnect_all(signal=Any, sender=Any):
|
||||
def disconnect_all(signal: TypingAny = Any, sender: TypingAny = Any) -> None:
|
||||
"""Disconnect all signal handlers. Useful for cleaning up after running
|
||||
tests
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -4,7 +4,7 @@ Module for processing Sitemaps.
|
|||
Note: The main purpose of this module is to provide support for the
|
||||
SitemapSpider, its API is subject to change without notice.
|
||||
"""
|
||||
|
||||
from typing import Any, Dict, Generator, Iterator, Optional
|
||||
from urllib.parse import urljoin
|
||||
|
||||
import lxml.etree
|
||||
|
|
@ -14,7 +14,7 @@ class Sitemap:
|
|||
"""Class to parse Sitemap (type=urlset) and Sitemap Index
|
||||
(type=sitemapindex) files"""
|
||||
|
||||
def __init__(self, xmltext):
|
||||
def __init__(self, xmltext: str):
|
||||
xmlp = lxml.etree.XMLParser(
|
||||
recover=True, remove_comments=True, resolve_entities=False
|
||||
)
|
||||
|
|
@ -22,9 +22,9 @@ class Sitemap:
|
|||
rt = self._root.tag
|
||||
self.type = self._root.tag.split("}", 1)[1] if "}" in rt else rt
|
||||
|
||||
def __iter__(self):
|
||||
def __iter__(self) -> Iterator[Dict[str, Any]]:
|
||||
for elem in self._root.getchildren():
|
||||
d = {}
|
||||
d: Dict[str, Any] = {}
|
||||
for el in elem.getchildren():
|
||||
tag = el.tag
|
||||
name = tag.split("}", 1)[1] if "}" in tag else tag
|
||||
|
|
@ -39,11 +39,13 @@ class Sitemap:
|
|||
yield d
|
||||
|
||||
|
||||
def sitemap_urls_from_robots(robots_text, base_url=None):
|
||||
def sitemap_urls_from_robots(
|
||||
robots_text: str, base_url: Optional[str] = None
|
||||
) -> Generator[str, Any, None]:
|
||||
"""Return an iterator over all sitemap urls contained in the given
|
||||
robots.txt file
|
||||
"""
|
||||
for line in robots_text.splitlines():
|
||||
if line.lstrip().lower().startswith("sitemap:"):
|
||||
url = line.split(":", 1)[1].strip()
|
||||
yield urljoin(base_url, url)
|
||||
yield urljoin(base_url or "", url)
|
||||
|
|
|
|||
|
|
@ -1,14 +1,54 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import inspect
|
||||
import logging
|
||||
from types import CoroutineType, ModuleType
|
||||
from typing import (
|
||||
TYPE_CHECKING,
|
||||
Any,
|
||||
AsyncGenerator,
|
||||
Generator,
|
||||
Iterable,
|
||||
Literal,
|
||||
Optional,
|
||||
Type,
|
||||
TypeVar,
|
||||
Union,
|
||||
overload,
|
||||
)
|
||||
|
||||
from twisted.internet.defer import Deferred
|
||||
|
||||
from scrapy import Request
|
||||
from scrapy.spiders import Spider
|
||||
from scrapy.utils.defer import deferred_from_coro
|
||||
from scrapy.utils.misc import arg_to_iter
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from scrapy.spiderloader import SpiderLoader
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_T = TypeVar("_T")
|
||||
|
||||
def iterate_spider_output(result):
|
||||
|
||||
# https://stackoverflow.com/questions/60222982
|
||||
@overload
|
||||
def iterate_spider_output(result: AsyncGenerator) -> AsyncGenerator: # type: ignore[misc]
|
||||
...
|
||||
|
||||
|
||||
@overload
|
||||
def iterate_spider_output(result: CoroutineType) -> Deferred:
|
||||
...
|
||||
|
||||
|
||||
@overload
|
||||
def iterate_spider_output(result: _T) -> Iterable:
|
||||
...
|
||||
|
||||
|
||||
def iterate_spider_output(result: Any) -> Union[Iterable, AsyncGenerator, Deferred]:
|
||||
if inspect.isasyncgen(result):
|
||||
return result
|
||||
if inspect.iscoroutine(result):
|
||||
|
|
@ -18,7 +58,7 @@ def iterate_spider_output(result):
|
|||
return arg_to_iter(deferred_from_coro(result))
|
||||
|
||||
|
||||
def iter_spider_classes(module):
|
||||
def iter_spider_classes(module: ModuleType) -> Generator[Type[Spider], Any, None]:
|
||||
"""Return an iterator over all spider classes defined in the given module
|
||||
that can be instantiated (i.e. which have name)
|
||||
"""
|
||||
|
|
@ -36,9 +76,46 @@ def iter_spider_classes(module):
|
|||
yield obj
|
||||
|
||||
|
||||
@overload
|
||||
def spidercls_for_request(
|
||||
spider_loader, request, default_spidercls=None, log_none=False, log_multiple=False
|
||||
):
|
||||
spider_loader: SpiderLoader,
|
||||
request: Request,
|
||||
default_spidercls: Type[Spider],
|
||||
log_none: bool = ...,
|
||||
log_multiple: bool = ...,
|
||||
) -> Type[Spider]:
|
||||
...
|
||||
|
||||
|
||||
@overload
|
||||
def spidercls_for_request(
|
||||
spider_loader: SpiderLoader,
|
||||
request: Request,
|
||||
default_spidercls: Literal[None],
|
||||
log_none: bool = ...,
|
||||
log_multiple: bool = ...,
|
||||
) -> Optional[Type[Spider]]:
|
||||
...
|
||||
|
||||
|
||||
@overload
|
||||
def spidercls_for_request(
|
||||
spider_loader: SpiderLoader,
|
||||
request: Request,
|
||||
*,
|
||||
log_none: bool = ...,
|
||||
log_multiple: bool = ...,
|
||||
) -> Optional[Type[Spider]]:
|
||||
...
|
||||
|
||||
|
||||
def spidercls_for_request(
|
||||
spider_loader: SpiderLoader,
|
||||
request: Request,
|
||||
default_spidercls: Optional[Type[Spider]] = None,
|
||||
log_none: bool = False,
|
||||
log_multiple: bool = False,
|
||||
) -> Optional[Type[Spider]]:
|
||||
"""Return a spider class that handles the given Request.
|
||||
|
||||
This will look for the spiders that can handle the given request (using
|
||||
|
|
|
|||
|
|
@ -1,4 +1,4 @@
|
|||
from typing import Any, Optional, cast
|
||||
from typing import Any, Optional
|
||||
|
||||
import OpenSSL._util as pyOpenSSLutil
|
||||
import OpenSSL.SSL
|
||||
|
|
@ -58,9 +58,6 @@ def get_temp_key_info(ssl_object: Any) -> Optional[str]:
|
|||
|
||||
|
||||
def get_openssl_version() -> str:
|
||||
# https://github.com/python/typeshed/issues/10024
|
||||
system_openssl_bytes = cast(
|
||||
bytes, OpenSSL.SSL.SSLeay_version(OpenSSL.SSL.SSLEAY_VERSION)
|
||||
)
|
||||
system_openssl_bytes = OpenSSL.SSL.SSLeay_version(OpenSSL.SSL.SSLEAY_VERSION)
|
||||
system_openssl = system_openssl_bytes.decode("ascii", errors="replace")
|
||||
return f"{OpenSSL.version.__version__} ({system_openssl})"
|
||||
|
|
|
|||
|
|
@ -4,10 +4,10 @@ import re
|
|||
import string
|
||||
from os import PathLike
|
||||
from pathlib import Path
|
||||
from typing import Union
|
||||
from typing import Any, Union
|
||||
|
||||
|
||||
def render_templatefile(path: Union[str, PathLike], **kwargs):
|
||||
def render_templatefile(path: Union[str, PathLike], **kwargs: Any) -> None:
|
||||
path_obj = Path(path)
|
||||
raw = path_obj.read_text("utf8")
|
||||
|
||||
|
|
@ -24,7 +24,7 @@ def render_templatefile(path: Union[str, PathLike], **kwargs):
|
|||
CAMELCASE_INVALID_CHARS = re.compile(r"[^a-zA-Z\d]")
|
||||
|
||||
|
||||
def string_camelcase(string):
|
||||
def string_camelcase(string: str) -> str:
|
||||
"""Convert a word to its CamelCase version and remove invalid chars
|
||||
|
||||
>>> string_camelcase('lost-pound')
|
||||
|
|
|
|||
|
|
@ -7,24 +7,30 @@ import os
|
|||
from importlib import import_module
|
||||
from pathlib import Path
|
||||
from posixpath import split
|
||||
from unittest import mock
|
||||
from typing import Any, Coroutine, Dict, List, Optional, Tuple, Type
|
||||
from unittest import TestCase, mock
|
||||
|
||||
from twisted.internet.defer import Deferred
|
||||
from twisted.trial.unittest import SkipTest
|
||||
|
||||
from scrapy import Spider
|
||||
from scrapy.crawler import Crawler
|
||||
from scrapy.utils.boto import is_botocore_available
|
||||
|
||||
|
||||
def assert_gcs_environ():
|
||||
def assert_gcs_environ() -> None:
|
||||
if "GCS_PROJECT_ID" not in os.environ:
|
||||
raise SkipTest("GCS_PROJECT_ID not found")
|
||||
|
||||
|
||||
def skip_if_no_boto():
|
||||
def skip_if_no_boto() -> None:
|
||||
if not is_botocore_available():
|
||||
raise SkipTest("missing botocore library")
|
||||
|
||||
|
||||
def get_gcs_content_and_delete(bucket, path):
|
||||
def get_gcs_content_and_delete(
|
||||
bucket: Any, path: str
|
||||
) -> Tuple[bytes, List[Dict[str, str]], Any]:
|
||||
from google.cloud import storage
|
||||
|
||||
client = storage.Client(project=os.environ.get("GCS_PROJECT_ID"))
|
||||
|
|
@ -37,8 +43,13 @@ def get_gcs_content_and_delete(bucket, path):
|
|||
|
||||
|
||||
def get_ftp_content_and_delete(
|
||||
path, host, port, username, password, use_active_mode=False
|
||||
):
|
||||
path: str,
|
||||
host: str,
|
||||
port: int,
|
||||
username: str,
|
||||
password: str,
|
||||
use_active_mode: bool = False,
|
||||
) -> bytes:
|
||||
from ftplib import FTP
|
||||
|
||||
ftp = FTP()
|
||||
|
|
@ -46,19 +57,23 @@ def get_ftp_content_and_delete(
|
|||
ftp.login(username, password)
|
||||
if use_active_mode:
|
||||
ftp.set_pasv(False)
|
||||
ftp_data = []
|
||||
ftp_data: List[bytes] = []
|
||||
|
||||
def buffer_data(data):
|
||||
def buffer_data(data: bytes) -> None:
|
||||
ftp_data.append(data)
|
||||
|
||||
ftp.retrbinary(f"RETR {path}", buffer_data)
|
||||
dirname, filename = split(path)
|
||||
ftp.cwd(dirname)
|
||||
ftp.delete(filename)
|
||||
return "".join(ftp_data)
|
||||
return b"".join(ftp_data)
|
||||
|
||||
|
||||
def get_crawler(spidercls=None, settings_dict=None, prevent_warnings=True):
|
||||
def get_crawler(
|
||||
spidercls: Optional[Type[Spider]] = None,
|
||||
settings_dict: Optional[Dict[str, Any]] = None,
|
||||
prevent_warnings: bool = True,
|
||||
) -> Crawler:
|
||||
"""Return an unconfigured Crawler object. If settings_dict is given, it
|
||||
will be used to populate the crawler settings with a project level
|
||||
priority.
|
||||
|
|
@ -82,7 +97,7 @@ def get_pythonpath() -> str:
|
|||
return str(Path(scrapy_path).parent) + os.pathsep + os.environ.get("PYTHONPATH", "")
|
||||
|
||||
|
||||
def get_testenv():
|
||||
def get_testenv() -> Dict[str, str]:
|
||||
"""Return a OS environment dict suitable to fork processes that need to import
|
||||
this installation of Scrapy, instead of a system installed one.
|
||||
"""
|
||||
|
|
@ -91,21 +106,23 @@ def get_testenv():
|
|||
return env
|
||||
|
||||
|
||||
def assert_samelines(testcase, text1, text2, msg=None):
|
||||
def assert_samelines(
|
||||
testcase: TestCase, text1: str, text2: str, msg: Optional[str] = None
|
||||
) -> None:
|
||||
"""Asserts text1 and text2 have the same lines, ignoring differences in
|
||||
line endings between platforms
|
||||
"""
|
||||
testcase.assertEqual(text1.splitlines(), text2.splitlines(), msg)
|
||||
|
||||
|
||||
def get_from_asyncio_queue(value):
|
||||
q = asyncio.Queue()
|
||||
def get_from_asyncio_queue(value: Any) -> Coroutine:
|
||||
q: asyncio.Queue = asyncio.Queue()
|
||||
getter = q.get()
|
||||
q.put_nowait(value)
|
||||
return getter
|
||||
|
||||
|
||||
def mock_google_cloud_storage():
|
||||
def mock_google_cloud_storage() -> Tuple[Any, Any, Any]:
|
||||
"""Creates autospec mocks for google-cloud-storage Client, Bucket and Blob
|
||||
classes and set their proper return values.
|
||||
"""
|
||||
|
|
@ -122,7 +139,7 @@ def mock_google_cloud_storage():
|
|||
return (client_mock, bucket_mock, blob_mock)
|
||||
|
||||
|
||||
def get_web_client_agent_req(url):
|
||||
def get_web_client_agent_req(url: str) -> Deferred:
|
||||
from twisted.internet import reactor
|
||||
from twisted.web.client import Agent # imports twisted.internet.reactor
|
||||
|
||||
|
|
|
|||
|
|
@ -1,7 +1,13 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
from typing import Iterable, Optional, Tuple, cast
|
||||
|
||||
from twisted.internet import defer, protocol
|
||||
from twisted.internet.defer import Deferred
|
||||
from twisted.internet.error import ProcessTerminated
|
||||
from twisted.internet.protocol import ProcessProtocol
|
||||
from twisted.python.failure import Failure
|
||||
|
||||
|
||||
class ProcessTest:
|
||||
|
|
@ -9,7 +15,12 @@ class ProcessTest:
|
|||
prefix = [sys.executable, "-m", "scrapy.cmdline"]
|
||||
cwd = os.getcwd() # trial chdirs to temp dir
|
||||
|
||||
def execute(self, args, check_code=True, settings=None):
|
||||
def execute(
|
||||
self,
|
||||
args: Iterable[str],
|
||||
check_code: bool = True,
|
||||
settings: Optional[str] = None,
|
||||
) -> Deferred:
|
||||
from twisted.internet import reactor
|
||||
|
||||
env = os.environ.copy()
|
||||
|
|
@ -21,29 +32,31 @@ class ProcessTest:
|
|||
reactor.spawnProcess(pp, cmd[0], cmd, env=env, path=self.cwd)
|
||||
return pp.deferred
|
||||
|
||||
def _process_finished(self, pp, cmd, check_code):
|
||||
def _process_finished(
|
||||
self, pp: TestProcessProtocol, cmd: str, check_code: bool
|
||||
) -> Tuple[int, bytes, bytes]:
|
||||
if pp.exitcode and check_code:
|
||||
msg = f"process {cmd} exit with code {pp.exitcode}"
|
||||
msg += f"\n>>> stdout <<<\n{pp.out}"
|
||||
msg += f"\n>>> stdout <<<\n{pp.out.decode()}"
|
||||
msg += "\n"
|
||||
msg += f"\n>>> stderr <<<\n{pp.err}"
|
||||
msg += f"\n>>> stderr <<<\n{pp.err.decode()}"
|
||||
raise RuntimeError(msg)
|
||||
return pp.exitcode, pp.out, pp.err
|
||||
return cast(int, pp.exitcode), pp.out, pp.err
|
||||
|
||||
|
||||
class TestProcessProtocol(protocol.ProcessProtocol):
|
||||
def __init__(self):
|
||||
self.deferred = defer.Deferred()
|
||||
self.out = b""
|
||||
self.err = b""
|
||||
self.exitcode = None
|
||||
class TestProcessProtocol(ProcessProtocol):
|
||||
def __init__(self) -> None:
|
||||
self.deferred: Deferred = Deferred()
|
||||
self.out: bytes = b""
|
||||
self.err: bytes = b""
|
||||
self.exitcode: Optional[int] = None
|
||||
|
||||
def outReceived(self, data):
|
||||
def outReceived(self, data: bytes) -> None:
|
||||
self.out += data
|
||||
|
||||
def errReceived(self, data):
|
||||
def errReceived(self, data: bytes) -> None:
|
||||
self.err += data
|
||||
|
||||
def processEnded(self, status):
|
||||
self.exitcode = status.value.exitCode
|
||||
def processEnded(self, status: Failure) -> None:
|
||||
self.exitcode = cast(ProcessTerminated, status.value).exitCode
|
||||
self.deferred.callback(self)
|
||||
|
|
|
|||
|
|
@ -12,9 +12,14 @@ alias to object in that case).
|
|||
from collections import defaultdict
|
||||
from operator import itemgetter
|
||||
from time import time
|
||||
from typing import DefaultDict
|
||||
from typing import TYPE_CHECKING, Any, DefaultDict, Iterable
|
||||
from weakref import WeakKeyDictionary
|
||||
|
||||
if TYPE_CHECKING:
|
||||
# typing.Self requires Python 3.11
|
||||
from typing_extensions import Self
|
||||
|
||||
|
||||
NoneType = type(None)
|
||||
live_refs: DefaultDict[type, WeakKeyDictionary] = defaultdict(WeakKeyDictionary)
|
||||
|
||||
|
|
@ -24,13 +29,14 @@ class object_ref:
|
|||
|
||||
__slots__ = ()
|
||||
|
||||
def __new__(cls, *args, **kwargs):
|
||||
def __new__(cls, *args: Any, **kwargs: Any) -> "Self":
|
||||
obj = object.__new__(cls)
|
||||
live_refs[cls][obj] = time()
|
||||
return obj
|
||||
|
||||
|
||||
def format_live_refs(ignore=NoneType):
|
||||
# using Any as it's hard to type type(None)
|
||||
def format_live_refs(ignore: Any = NoneType) -> str:
|
||||
"""Return a tabular representation of tracked objects"""
|
||||
s = "Live References\n\n"
|
||||
now = time()
|
||||
|
|
@ -44,12 +50,12 @@ def format_live_refs(ignore=NoneType):
|
|||
return s
|
||||
|
||||
|
||||
def print_live_refs(*a, **kw):
|
||||
def print_live_refs(*a: Any, **kw: Any) -> None:
|
||||
"""Print tracked objects"""
|
||||
print(format_live_refs(*a, **kw))
|
||||
|
||||
|
||||
def get_oldest(class_name):
|
||||
def get_oldest(class_name: str) -> Any:
|
||||
"""Get the oldest object for a specific class name"""
|
||||
for cls, wdict in live_refs.items():
|
||||
if cls.__name__ == class_name:
|
||||
|
|
@ -58,8 +64,9 @@ def get_oldest(class_name):
|
|||
return min(wdict.items(), key=itemgetter(1))[0]
|
||||
|
||||
|
||||
def iter_all(class_name):
|
||||
def iter_all(class_name: str) -> Iterable[Any]:
|
||||
"""Iterate over all objects of the same class by its class name"""
|
||||
for cls, wdict in live_refs.items():
|
||||
if cls.__name__ == class_name:
|
||||
return wdict.keys()
|
||||
return []
|
||||
|
|
|
|||
|
|
@ -6,6 +6,7 @@ Some of the functions that used to be imported from this module have been moved
|
|||
to the w3lib.url module. Always import those from there instead.
|
||||
"""
|
||||
import re
|
||||
from typing import TYPE_CHECKING, Iterable, Optional, Type, Union, cast
|
||||
from urllib.parse import ParseResult, urldefrag, urlparse, urlunparse
|
||||
|
||||
# scrapy.utils.url was moved to w3lib.url and import * ensures this
|
||||
|
|
@ -15,8 +16,14 @@ from w3lib.url import _safe_chars, _unquotepath # noqa: F401
|
|||
|
||||
from scrapy.utils.python import to_unicode
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from scrapy import Spider
|
||||
|
||||
def url_is_from_any_domain(url, domains):
|
||||
|
||||
UrlT = Union[str, bytes, ParseResult]
|
||||
|
||||
|
||||
def url_is_from_any_domain(url: UrlT, domains: Iterable[str]) -> bool:
|
||||
"""Return True if the url belongs to any of the given domains"""
|
||||
host = parse_url(url).netloc.lower()
|
||||
if not host:
|
||||
|
|
@ -25,29 +32,29 @@ def url_is_from_any_domain(url, domains):
|
|||
return any((host == d) or (host.endswith(f".{d}")) for d in domains)
|
||||
|
||||
|
||||
def url_is_from_spider(url, spider):
|
||||
def url_is_from_spider(url: UrlT, spider: Type["Spider"]) -> bool:
|
||||
"""Return True if the url belongs to the given spider"""
|
||||
return url_is_from_any_domain(
|
||||
url, [spider.name] + list(getattr(spider, "allowed_domains", []))
|
||||
)
|
||||
|
||||
|
||||
def url_has_any_extension(url, extensions):
|
||||
def url_has_any_extension(url: UrlT, extensions: Iterable[str]) -> bool:
|
||||
"""Return True if the url ends with one of the extensions provided"""
|
||||
lowercase_path = parse_url(url).path.lower()
|
||||
return any(lowercase_path.endswith(ext) for ext in extensions)
|
||||
|
||||
|
||||
def parse_url(url, encoding=None):
|
||||
def parse_url(url: UrlT, encoding: Optional[str] = None) -> ParseResult:
|
||||
"""Return urlparsed url from the given argument (which could be an already
|
||||
parsed url)
|
||||
"""
|
||||
if isinstance(url, ParseResult):
|
||||
return url
|
||||
return urlparse(to_unicode(url, encoding))
|
||||
return cast(ParseResult, urlparse(to_unicode(url, encoding)))
|
||||
|
||||
|
||||
def escape_ajax(url):
|
||||
def escape_ajax(url: str) -> str:
|
||||
"""
|
||||
Return the crawlable url according to:
|
||||
https://developers.google.com/webmasters/ajax-crawling/docs/getting-started
|
||||
|
|
@ -76,7 +83,7 @@ def escape_ajax(url):
|
|||
return add_or_replace_parameter(defrag, "_escaped_fragment_", frag[1:])
|
||||
|
||||
|
||||
def add_http_if_no_scheme(url):
|
||||
def add_http_if_no_scheme(url: str) -> str:
|
||||
"""Add http as the default scheme if it is missing from the url."""
|
||||
match = re.match(r"^\w+://", url, flags=re.I)
|
||||
if not match:
|
||||
|
|
@ -87,7 +94,7 @@ def add_http_if_no_scheme(url):
|
|||
return url
|
||||
|
||||
|
||||
def _is_posix_path(string):
|
||||
def _is_posix_path(string: str) -> bool:
|
||||
return bool(
|
||||
re.match(
|
||||
r"""
|
||||
|
|
@ -109,7 +116,7 @@ def _is_posix_path(string):
|
|||
)
|
||||
|
||||
|
||||
def _is_windows_path(string):
|
||||
def _is_windows_path(string: str) -> bool:
|
||||
return bool(
|
||||
re.match(
|
||||
r"""
|
||||
|
|
@ -125,11 +132,11 @@ def _is_windows_path(string):
|
|||
)
|
||||
|
||||
|
||||
def _is_filesystem_path(string):
|
||||
def _is_filesystem_path(string: str) -> bool:
|
||||
return _is_posix_path(string) or _is_windows_path(string)
|
||||
|
||||
|
||||
def guess_scheme(url):
|
||||
def guess_scheme(url: str) -> str:
|
||||
"""Add an URL scheme if missing: file:// for filepath-like input or
|
||||
http:// otherwise."""
|
||||
if _is_filesystem_path(url):
|
||||
|
|
@ -138,12 +145,12 @@ def guess_scheme(url):
|
|||
|
||||
|
||||
def strip_url(
|
||||
url,
|
||||
strip_credentials=True,
|
||||
strip_default_port=True,
|
||||
origin_only=False,
|
||||
strip_fragment=True,
|
||||
):
|
||||
url: str,
|
||||
strip_credentials: bool = True,
|
||||
strip_default_port: bool = True,
|
||||
origin_only: bool = False,
|
||||
strip_fragment: bool = True,
|
||||
) -> str:
|
||||
"""Strip URL string from some of its components:
|
||||
|
||||
- ``strip_credentials`` removes "user:password@"
|
||||
|
|
|
|||
|
|
@ -1,5 +1,6 @@
|
|||
import platform
|
||||
import sys
|
||||
from typing import List, Tuple
|
||||
|
||||
import cryptography
|
||||
import cssselect
|
||||
|
|
@ -12,7 +13,7 @@ import scrapy
|
|||
from scrapy.utils.ssl import get_openssl_version
|
||||
|
||||
|
||||
def scrapy_components_versions():
|
||||
def scrapy_components_versions() -> List[Tuple[str, str]]:
|
||||
lxml_version = ".".join(map(str, lxml.etree.LXML_VERSION))
|
||||
libxml2_version = ".".join(map(str, lxml.etree.LIBXML_VERSION))
|
||||
|
||||
|
|
|
|||
113
sep/sep-021.rst
113
sep/sep-021.rst
|
|
@ -1,113 +0,0 @@
|
|||
======= ===================
|
||||
SEP 21
|
||||
Title Add-ons
|
||||
Author Pablo Hoffman
|
||||
Created 2014-02-14
|
||||
Status Draft
|
||||
======= ===================
|
||||
|
||||
================
|
||||
SEP-021: Add-ons
|
||||
================
|
||||
|
||||
This proposal introduces add-ons, a unified way to manage Scrapy extensions,
|
||||
middlewares and pipelines.
|
||||
|
||||
Scrapy currently supports many hooks and mechanisms for extending its
|
||||
functionality, but no single entry point for enabling and configuring them.
|
||||
Instead, the hooks are spread over:
|
||||
|
||||
* Spider middlewares (SPIDER_MIDDLEWARES)
|
||||
* Downloader middlewares (DOWNLOADER_MIDDLEWARES)
|
||||
* Downloader handlers (DOWNLOADER_HANDLERS)
|
||||
* Item pipelines (ITEM_PIPELINES)
|
||||
* Feed exporters and storages (FEED_EXPORTERS, FEED_STORAGES)
|
||||
* Overridable components (DUPEFILTER_CLASS, STATS_CLASS, SCHEDULER, SPIDER_MANAGER_CLASS, ITEM_PROCESSOR, etc)
|
||||
* Generic extensions (EXTENSIONS)
|
||||
* CLI commands (COMMANDS_MODULE)
|
||||
|
||||
One problem of this approach is that enabling an extension often requires
|
||||
modifying many settings, often in a coordinated way, which is complex and error
|
||||
prone. Add-ons are meant to fix this by providing a simple mechanism for
|
||||
enabling extensions.
|
||||
|
||||
Design goals and non-goals
|
||||
==========================
|
||||
|
||||
Goals:
|
||||
|
||||
* simple to manage: adding or removing extensions should be just a matter of
|
||||
adding or removing lines in a ``scrapy.cfg`` file
|
||||
* backward compatibility with enabling extension the "old way" (i.e. modifying
|
||||
settings directly)
|
||||
|
||||
Non-goals:
|
||||
|
||||
* a way to publish, distribute or discover extensions (use pypi for that)
|
||||
|
||||
|
||||
Managing add-ons
|
||||
================
|
||||
|
||||
Add-ons are defined in the ``scrapy.cfg`` file, inside the ``[addons]``
|
||||
section.
|
||||
|
||||
To enable the "httpcache" addon, either shipped with Scrapy or in the Python
|
||||
search path, create an entry for it in your ``scrapy.cfg``, like this::
|
||||
|
||||
[addons]
|
||||
httpcache =
|
||||
|
||||
You may also specify the full path to an add-on (which may be either a .py file
|
||||
or a folder containing __init__.py)::
|
||||
|
||||
[addons]
|
||||
mongodb_pipeline = /path/to/mongodb_pipeline.py
|
||||
|
||||
|
||||
Writing add-ons
|
||||
===============
|
||||
|
||||
Add-ons are Python modules that implement the following callbacks.
|
||||
|
||||
addon_configure
|
||||
---------------
|
||||
|
||||
Receives the Settings object and modifies it to enable the required components.
|
||||
If it raises an exception, Scrapy will print it and exit.
|
||||
|
||||
Examples:
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
def addon_configure(settings):
|
||||
settings.overrides["DOWNLOADER_MIDDLEWARES"].update(
|
||||
{
|
||||
"scrapy.contrib.downloadermiddleware.httpcache.HttpCacheMiddleware": 900,
|
||||
}
|
||||
)
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
def addon_configure(settings):
|
||||
try:
|
||||
import boto
|
||||
except ImportError:
|
||||
raise RuntimeError("boto library is required")
|
||||
|
||||
|
||||
crawler_ready
|
||||
-------------
|
||||
|
||||
``crawler_ready`` receives a Crawler object after it has been initialized and
|
||||
is meant to be used to perform post-initialization checks like making sure the
|
||||
extension and its dependencies were configured properly. If it raises an
|
||||
exception, Scrapy will print and exit.
|
||||
|
||||
Examples:
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
def crawler_ready(crawler):
|
||||
if "some.other.addon" not in crawler.extensions.enabled:
|
||||
raise RuntimeError("Some other addon is required to use this addon")
|
||||
38
setup.py
38
setup.py
|
|
@ -1,26 +1,13 @@
|
|||
from pathlib import Path
|
||||
|
||||
from pkg_resources import parse_version
|
||||
from setuptools import __version__ as setuptools_version
|
||||
from setuptools import find_packages, setup
|
||||
|
||||
version = (Path(__file__).parent / "scrapy/VERSION").read_text("ascii").strip()
|
||||
|
||||
|
||||
def has_environment_marker_platform_impl_support():
|
||||
"""Code extracted from 'pytest/setup.py'
|
||||
https://github.com/pytest-dev/pytest/blob/7538680c/setup.py#L31
|
||||
|
||||
The first known release to support environment marker with range operators
|
||||
it is 18.5, see:
|
||||
https://setuptools.readthedocs.io/en/latest/history.html#id235
|
||||
"""
|
||||
return parse_version(setuptools_version) >= parse_version("18.5")
|
||||
|
||||
|
||||
install_requires = [
|
||||
"Twisted>=18.9.0",
|
||||
"cryptography>=3.4.6",
|
||||
"cryptography>=36.0.0",
|
||||
"cssselect>=0.9.1",
|
||||
"itemloaders>=1.0.1",
|
||||
"parsel>=1.5.0",
|
||||
|
|
@ -34,21 +21,12 @@ install_requires = [
|
|||
"setuptools",
|
||||
"packaging",
|
||||
"tldextract",
|
||||
"lxml>=4.3.0",
|
||||
"lxml>=4.4.1",
|
||||
]
|
||||
extras_require = {}
|
||||
cpython_dependencies = [
|
||||
"PyDispatcher>=2.0.5",
|
||||
]
|
||||
if has_environment_marker_platform_impl_support():
|
||||
extras_require[
|
||||
':platform_python_implementation == "CPython"'
|
||||
] = cpython_dependencies
|
||||
extras_require[':platform_python_implementation == "PyPy"'] = [
|
||||
"PyPyDispatcher>=2.1.0",
|
||||
]
|
||||
else:
|
||||
install_requires.extend(cpython_dependencies)
|
||||
extras_require = {
|
||||
':platform_python_implementation == "CPython"': ["PyDispatcher>=2.0.5"],
|
||||
':platform_python_implementation == "PyPy"': ["PyPyDispatcher>=2.1.0"],
|
||||
}
|
||||
|
||||
|
||||
setup(
|
||||
|
|
@ -80,18 +58,18 @@ setup(
|
|||
"Operating System :: OS Independent",
|
||||
"Programming Language :: Python",
|
||||
"Programming Language :: Python :: 3",
|
||||
"Programming Language :: Python :: 3.7",
|
||||
"Programming Language :: Python :: 3.8",
|
||||
"Programming Language :: Python :: 3.9",
|
||||
"Programming Language :: Python :: 3.10",
|
||||
"Programming Language :: Python :: 3.11",
|
||||
"Programming Language :: Python :: 3.12",
|
||||
"Programming Language :: Python :: Implementation :: CPython",
|
||||
"Programming Language :: Python :: Implementation :: PyPy",
|
||||
"Topic :: Internet :: WWW/HTTP",
|
||||
"Topic :: Software Development :: Libraries :: Application Frameworks",
|
||||
"Topic :: Software Development :: Libraries :: Python Modules",
|
||||
],
|
||||
python_requires=">=3.7",
|
||||
python_requires=">=3.8",
|
||||
install_requires=install_requires,
|
||||
extras_require=extras_require,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -1,4 +1,4 @@
|
|||
from datetime import datetime, timedelta
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from pathlib import Path
|
||||
|
||||
from cryptography.hazmat.backends import default_backend
|
||||
|
|
@ -50,8 +50,8 @@ def generate_keys():
|
|||
.issuer_name(issuer)
|
||||
.public_key(key.public_key())
|
||||
.serial_number(random_serial_number())
|
||||
.not_valid_before(datetime.utcnow())
|
||||
.not_valid_after(datetime.utcnow() + timedelta(days=10))
|
||||
.not_valid_before(datetime.now(tz=timezone.utc))
|
||||
.not_valid_after(datetime.now(tz=timezone.utc) + timedelta(days=10))
|
||||
.add_extension(
|
||||
SubjectAlternativeName([DNSName("localhost")]),
|
||||
critical=False,
|
||||
|
|
|
|||
|
|
@ -1,11 +1,13 @@
|
|||
import argparse
|
||||
import json
|
||||
import os
|
||||
import random
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from shutil import rmtree
|
||||
from subprocess import PIPE, Popen
|
||||
from tempfile import mkdtemp
|
||||
from typing import Dict
|
||||
from urllib.parse import urlencode
|
||||
|
||||
from OpenSSL import SSL
|
||||
|
|
@ -20,7 +22,6 @@ from twisted.web.static import File
|
|||
from twisted.web.util import redirectTo
|
||||
|
||||
from scrapy.utils.python import to_bytes, to_unicode
|
||||
from scrapy.utils.test import get_testenv
|
||||
|
||||
|
||||
def getarg(request, name, default=None, type=None):
|
||||
|
|
@ -32,6 +33,16 @@ def getarg(request, name, default=None, type=None):
|
|||
return default
|
||||
|
||||
|
||||
def get_mockserver_env() -> Dict[str, str]:
|
||||
"""Return a OS environment dict suitable to run mockserver processes."""
|
||||
|
||||
tests_path = Path(__file__).parent.parent
|
||||
pythonpath = str(tests_path) + os.pathsep + os.environ.get("PYTHONPATH", "")
|
||||
env = os.environ.copy()
|
||||
env["PYTHONPATH"] = pythonpath
|
||||
return env
|
||||
|
||||
|
||||
# most of the following resources are copied from twisted.web.test.test_webclient
|
||||
class ForeverTakingResource(resource.Resource):
|
||||
"""
|
||||
|
|
@ -264,7 +275,7 @@ class MockServer:
|
|||
self.proc = Popen(
|
||||
[sys.executable, "-u", "-m", "tests.mockserver", "-t", "http"],
|
||||
stdout=PIPE,
|
||||
env=get_testenv(),
|
||||
env=get_mockserver_env(),
|
||||
)
|
||||
http_address = self.proc.stdout.readline().strip().decode("ascii")
|
||||
https_address = self.proc.stdout.readline().strip().decode("ascii")
|
||||
|
|
@ -308,7 +319,7 @@ class MockDNSServer:
|
|||
self.proc = Popen(
|
||||
[sys.executable, "-u", "-m", "tests.mockserver", "-t", "dns"],
|
||||
stdout=PIPE,
|
||||
env=get_testenv(),
|
||||
env=get_mockserver_env(),
|
||||
)
|
||||
self.host = "127.0.0.1"
|
||||
self.port = int(
|
||||
|
|
@ -331,7 +342,7 @@ class MockFTPServer:
|
|||
self.proc = Popen(
|
||||
[sys.executable, "-u", "-m", "tests.ftpserver", "-d", str(self.path)],
|
||||
stderr=PIPE,
|
||||
env=get_testenv(),
|
||||
env=get_mockserver_env(),
|
||||
)
|
||||
for line in self.proc.stderr:
|
||||
if b"starting FTP server" in line:
|
||||
|
|
|
|||
|
|
@ -1,15 +1,17 @@
|
|||
# Tests requirements
|
||||
attrs
|
||||
pyftpdlib
|
||||
# https://github.com/giampaolo/pyftpdlib/issues/560
|
||||
pyftpdlib; python_version < "3.12"
|
||||
pytest
|
||||
pytest-cov==4.0.0
|
||||
pytest-xdist
|
||||
sybil >= 1.3.0 # https://github.com/cjw296/sybil/issues/20#issuecomment-605433422
|
||||
testfixtures
|
||||
uvloop; platform_system != "Windows"
|
||||
# uvloop currently doesn't build on 3.12
|
||||
uvloop; platform_system != "Windows" and python_version < "3.12"
|
||||
|
||||
# optional for shell wrapper tests
|
||||
bpython
|
||||
# bpython requires greenlet which currently doesn't build on 3.12
|
||||
bpython; python_version < "3.12" # optional for shell wrapper tests
|
||||
brotli # optional for HTTP compress downloader middleware tests
|
||||
zstandard; implementation_name != 'pypy' # optional for HTTP compress downloader middleware tests
|
||||
ipython
|
||||
|
|
|
|||
|
|
@ -77,6 +77,22 @@ class DelaySpider(MetaSpider):
|
|||
self.t2_err = time.time()
|
||||
|
||||
|
||||
class SlowSpider(DelaySpider):
|
||||
name = "slow"
|
||||
|
||||
def start_requests(self):
|
||||
# 1st response is fast
|
||||
url = self.mockserver.url("/delay?n=0&b=0")
|
||||
yield Request(url, callback=self.parse, errback=self.errback)
|
||||
|
||||
# 2nd response is slow
|
||||
url = self.mockserver.url(f"/delay?n={self.n}&b={self.b}")
|
||||
yield Request(url, callback=self.parse, errback=self.errback)
|
||||
|
||||
def parse(self, response):
|
||||
yield Item()
|
||||
|
||||
|
||||
class SimpleSpider(MetaSpider):
|
||||
name = "simple"
|
||||
|
||||
|
|
|
|||
|
|
@ -0,0 +1,158 @@
|
|||
import itertools
|
||||
import unittest
|
||||
from typing import Any, Dict
|
||||
|
||||
from scrapy import Spider
|
||||
from scrapy.crawler import Crawler, CrawlerRunner
|
||||
from scrapy.exceptions import NotConfigured
|
||||
from scrapy.settings import BaseSettings, Settings
|
||||
from scrapy.utils.test import get_crawler
|
||||
|
||||
|
||||
class SimpleAddon:
|
||||
def update_settings(self, settings):
|
||||
pass
|
||||
|
||||
|
||||
def get_addon_cls(config: Dict[str, Any]) -> type:
|
||||
class AddonWithConfig:
|
||||
def update_settings(self, settings: BaseSettings):
|
||||
settings.update(config, priority="addon")
|
||||
|
||||
return AddonWithConfig
|
||||
|
||||
|
||||
class CreateInstanceAddon:
|
||||
def __init__(self, crawler: Crawler) -> None:
|
||||
super().__init__()
|
||||
self.crawler = crawler
|
||||
self.config = crawler.settings.getdict("MYADDON")
|
||||
|
||||
@classmethod
|
||||
def from_crawler(cls, crawler: Crawler):
|
||||
return cls(crawler)
|
||||
|
||||
def update_settings(self, settings):
|
||||
settings.update(self.config, "addon")
|
||||
|
||||
|
||||
class AddonTest(unittest.TestCase):
|
||||
def test_update_settings(self):
|
||||
settings = BaseSettings()
|
||||
settings.set("KEY1", "default", priority="default")
|
||||
settings.set("KEY2", "project", priority="project")
|
||||
addon_config = {"KEY1": "addon", "KEY2": "addon", "KEY3": "addon"}
|
||||
testaddon = get_addon_cls(addon_config)()
|
||||
testaddon.update_settings(settings)
|
||||
self.assertEqual(settings["KEY1"], "addon")
|
||||
self.assertEqual(settings["KEY2"], "project")
|
||||
self.assertEqual(settings["KEY3"], "addon")
|
||||
|
||||
|
||||
class AddonManagerTest(unittest.TestCase):
|
||||
def test_load_settings(self):
|
||||
settings_dict = {
|
||||
"ADDONS": {"tests.test_addons.SimpleAddon": 0},
|
||||
}
|
||||
crawler = get_crawler(settings_dict=settings_dict)
|
||||
manager = crawler.addons
|
||||
self.assertIsInstance(manager.addons[0], SimpleAddon)
|
||||
|
||||
def test_notconfigured(self):
|
||||
class NotConfiguredAddon:
|
||||
def update_settings(self, settings):
|
||||
raise NotConfigured()
|
||||
|
||||
settings_dict = {
|
||||
"ADDONS": {NotConfiguredAddon: 0},
|
||||
}
|
||||
crawler = get_crawler(settings_dict=settings_dict)
|
||||
manager = crawler.addons
|
||||
self.assertFalse(manager.addons)
|
||||
|
||||
def test_load_settings_order(self):
|
||||
# Get three addons with different settings
|
||||
addonlist = []
|
||||
for i in range(3):
|
||||
addon = get_addon_cls({"KEY1": i})
|
||||
addon.number = i
|
||||
addonlist.append(addon)
|
||||
# Test for every possible ordering
|
||||
for ordered_addons in itertools.permutations(addonlist):
|
||||
expected_order = [a.number for a in ordered_addons]
|
||||
settings = {"ADDONS": {a: i for i, a in enumerate(ordered_addons)}}
|
||||
crawler = get_crawler(settings_dict=settings)
|
||||
manager = crawler.addons
|
||||
self.assertEqual([a.number for a in manager.addons], expected_order)
|
||||
self.assertEqual(crawler.settings.getint("KEY1"), expected_order[-1])
|
||||
|
||||
def test_create_instance(self):
|
||||
settings_dict = {
|
||||
"ADDONS": {"tests.test_addons.CreateInstanceAddon": 0},
|
||||
"MYADDON": {"MYADDON_KEY": "val"},
|
||||
}
|
||||
crawler = get_crawler(settings_dict=settings_dict)
|
||||
manager = crawler.addons
|
||||
self.assertIsInstance(manager.addons[0], CreateInstanceAddon)
|
||||
self.assertEqual(crawler.settings.get("MYADDON_KEY"), "val")
|
||||
|
||||
def test_settings_priority(self):
|
||||
config = {
|
||||
"KEY": 15, # priority=addon
|
||||
}
|
||||
settings_dict = {
|
||||
"ADDONS": {get_addon_cls(config): 1},
|
||||
}
|
||||
crawler = get_crawler(settings_dict=settings_dict)
|
||||
self.assertEqual(crawler.settings.getint("KEY"), 15)
|
||||
|
||||
settings = Settings(settings_dict)
|
||||
settings.set("KEY", 0, priority="default")
|
||||
runner = CrawlerRunner(settings)
|
||||
crawler = runner.create_crawler(Spider)
|
||||
self.assertEqual(crawler.settings.getint("KEY"), 15)
|
||||
|
||||
settings_dict = {
|
||||
"KEY": 20, # priority=project
|
||||
"ADDONS": {get_addon_cls(config): 1},
|
||||
}
|
||||
settings = Settings(settings_dict)
|
||||
settings.set("KEY", 0, priority="default")
|
||||
runner = CrawlerRunner(settings)
|
||||
crawler = runner.create_crawler(Spider)
|
||||
self.assertEqual(crawler.settings.getint("KEY"), 20)
|
||||
|
||||
def test_fallback_workflow(self):
|
||||
FALLBACK_SETTING = "MY_FALLBACK_DOWNLOAD_HANDLER"
|
||||
|
||||
class AddonWithFallback:
|
||||
def update_settings(self, settings):
|
||||
if not settings.get(FALLBACK_SETTING):
|
||||
settings.set(
|
||||
FALLBACK_SETTING,
|
||||
settings.getwithbase("DOWNLOAD_HANDLERS")["https"],
|
||||
"addon",
|
||||
)
|
||||
settings["DOWNLOAD_HANDLERS"]["https"] = "AddonHandler"
|
||||
|
||||
settings_dict = {
|
||||
"ADDONS": {AddonWithFallback: 1},
|
||||
}
|
||||
crawler = get_crawler(settings_dict=settings_dict)
|
||||
self.assertEqual(
|
||||
crawler.settings.getwithbase("DOWNLOAD_HANDLERS")["https"], "AddonHandler"
|
||||
)
|
||||
self.assertEqual(
|
||||
crawler.settings.get(FALLBACK_SETTING),
|
||||
"scrapy.core.downloader.handlers.http.HTTPDownloadHandler",
|
||||
)
|
||||
|
||||
settings_dict = {
|
||||
"ADDONS": {AddonWithFallback: 1},
|
||||
"DOWNLOAD_HANDLERS": {"https": "UserHandler"},
|
||||
}
|
||||
crawler = get_crawler(settings_dict=settings_dict)
|
||||
self.assertEqual(
|
||||
crawler.settings.getwithbase("DOWNLOAD_HANDLERS")["https"], "AddonHandler"
|
||||
)
|
||||
self.assertEqual(crawler.settings.get(FALLBACK_SETTING), "UserHandler")
|
||||
|
|
@ -3,7 +3,7 @@ from twisted.trial.unittest import TestCase
|
|||
|
||||
from scrapy.utils.test import get_crawler
|
||||
from tests.mockserver import MockServer
|
||||
from tests.spiders import ErrorSpider, FollowAllSpider, ItemSpider
|
||||
from tests.spiders import ErrorSpider, FollowAllSpider, ItemSpider, SlowSpider
|
||||
|
||||
|
||||
class TestCloseSpider(TestCase):
|
||||
|
|
@ -54,3 +54,13 @@ class TestCloseSpider(TestCase):
|
|||
self.assertEqual(reason, "closespider_timeout")
|
||||
total_seconds = crawler.stats.get_value("elapsed_time_seconds")
|
||||
self.assertTrue(total_seconds >= close_on)
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_closespider_timeout_no_item(self):
|
||||
timeout = 1
|
||||
crawler = get_crawler(SlowSpider, {"CLOSESPIDER_TIMEOUT_NO_ITEM": timeout})
|
||||
yield crawler.crawl(n=3, mockserver=self.mockserver)
|
||||
reason = crawler.spider.meta["close_reason"]
|
||||
self.assertEqual(reason, "closespider_timeout_no_item")
|
||||
total_seconds = crawler.stats.get_value("elapsed_time_seconds")
|
||||
self.assertTrue(total_seconds >= timeout)
|
||||
|
|
|
|||
Some files were not shown because too many files have changed in this diff Show More
Loading…
Reference in New Issue