Merge branch 'master' into pathlib

This commit is contained in:
Andrey Rahmatullin 2022-11-25 18:20:25 +05:00 committed by GitHub
commit d19a216e10
No known key found for this signature in database
GPG Key ID: 4AEE18F83AFDEB23
83 changed files with 1166 additions and 605 deletions

View File

@ -1,5 +1,5 @@
[bumpversion]
current_version = 2.6.2
current_version = 2.7.1
commit = True
tag = True
tag_name = {new_version}

17
.flake8
View File

@ -6,16 +6,17 @@ ignore = W503
exclude =
docs/conf.py
per-file-ignores =
# Exclude files that are meant to provide top-level imports
# E402: Module level import not at top of file
# F401: Module imported but unused
scrapy/__init__.py E402
scrapy/core/downloader/handlers/http.py F401
scrapy/http/__init__.py F401
scrapy/linkextractors/__init__.py E402 F401
scrapy/selector/__init__.py F401
scrapy/spiders/__init__.py E402 F401
scrapy/__init__.py:E402
scrapy/core/downloader/handlers/http.py:F401
scrapy/http/__init__.py:F401
scrapy/linkextractors/__init__.py:E402,F401
scrapy/selector/__init__.py:F401
scrapy/spiders/__init__.py:E402,F401
# Issues pending a review:
scrapy/utils/url.py F403 F405
tests/test_loader.py E741
scrapy/utils/url.py:F403,F405
tests/test_loader.py:E741

View File

@ -3,15 +3,15 @@ on: [push, pull_request]
jobs:
checks:
runs-on: ubuntu-18.04
runs-on: ubuntu-latest
strategy:
fail-fast: false
matrix:
include:
- python-version: "3.10"
- python-version: "3.11"
env:
TOXENV: security
- python-version: "3.10"
- python-version: "3.11"
env:
TOXENV: flake8
# Pylint requires installing reppy, which does not support Python 3.9
@ -22,10 +22,10 @@ jobs:
- python-version: 3.7
env:
TOXENV: typing
- python-version: "3.10" # Keep in sync with .readthedocs.yml
- python-version: "3.11" # Keep in sync with .readthedocs.yml
env:
TOXENV: docs
- python-version: "3.10"
- python-version: "3.11"
env:
TOXENV: twinecheck

View File

@ -3,7 +3,7 @@ on: [push]
jobs:
publish:
runs-on: ubuntu-18.04
runs-on: ubuntu-latest
if: startsWith(github.event.ref, 'refs/tags/')
steps:
@ -12,7 +12,7 @@ jobs:
- name: Set up Python
uses: actions/setup-python@v4
with:
python-version: "3.10"
python-version: "3.11"
- name: Check Tag
id: check-release-tag

View File

@ -3,11 +3,11 @@ on: [push, pull_request]
jobs:
tests:
runs-on: macos-10.15
runs-on: macos-11
strategy:
fail-fast: false
matrix:
python-version: ["3.7", "3.8", "3.9", "3.10"]
python-version: ["3.7", "3.8", "3.9", "3.10", "3.11"]
steps:
- uses: actions/checkout@v3

View File

@ -3,7 +3,7 @@ on: [push, pull_request]
jobs:
tests:
runs-on: ubuntu-18.04
runs-on: ubuntu-latest
strategy:
fail-fast: false
matrix:
@ -17,13 +17,10 @@ jobs:
- python-version: "3.10"
env:
TOXENV: py
- python-version: "3.10"
env:
TOXENV: asyncio
- python-version: "3.11.0-rc.2"
- python-version: "3.11"
env:
TOXENV: py
- python-version: "3.11.0-rc.2"
- python-version: "3.11"
env:
TOXENV: asyncio
- python-version: pypy3.9
@ -57,7 +54,7 @@ jobs:
python-version: ${{ matrix.python-version }}
- name: Install system libraries
if: matrix.python-version == 'pypy3.9' || contains(matrix.env.TOXENV, 'pinned') || matrix.python-version == '3.11.0-rc.2'
if: matrix.python-version == 'pypy3.9' || contains(matrix.env.TOXENV, 'pinned')
run: |
sudo apt-get update
sudo apt-get install libxml2-dev libxslt-dev

View File

@ -23,6 +23,13 @@ jobs:
- python-version: "3.10"
env:
TOXENV: asyncio
# no binary package for lxml for 3.11 yet
# - python-version: "3.11"
# env:
# TOXENV: py
# - python-version: "3.11"
# env:
# TOXENV: asyncio
steps:
- uses: actions/checkout@v3

View File

@ -9,7 +9,7 @@ build:
tools:
# For available versions, see:
# https://docs.readthedocs.io/en/stable/config-file/v2.html#build-tools-python
python: "3.10" # Keep in sync with .github/workflows/checks.yml
python: "3.11" # Keep in sync with .github/workflows/checks.yml
python:
install:

View File

@ -1,77 +1,133 @@
# Contributor Covenant Code of Conduct
## Our Pledge
In the interest of fostering an open and welcoming environment, we as
contributors and maintainers pledge to make participation in our project and
our community a harassment-free experience for everyone, regardless of age, body
size, disability, ethnicity, gender identity and expression, level of experience,
nationality, personal appearance, race, religion, or sexual identity and
orientation.
We as members, contributors, and leaders pledge to make participation in our
community a harassment-free experience for everyone, regardless of age, body
size, visible or invisible disability, ethnicity, sex characteristics, gender
identity and expression, level of experience, education, socio-economic status,
nationality, personal appearance, race, caste, color, religion, or sexual
identity and orientation.
We pledge to act and interact in ways that contribute to an open, welcoming,
diverse, inclusive, and healthy community.
## Our Standards
Examples of behavior that contributes to creating a positive environment
include:
Examples of behavior that contributes to a positive environment for our
community include:
* Using welcoming and inclusive language
* Being respectful of differing viewpoints and experiences
* Gracefully accepting constructive criticism
* Focusing on what is best for the community
* Showing empathy towards other community members
* Demonstrating empathy and kindness toward other people
* Being respectful of differing opinions, viewpoints, and experiences
* Giving and gracefully accepting constructive feedback
* Accepting responsibility and apologizing to those affected by our mistakes,
and learning from the experience
* Focusing on what is best not just for us as individuals, but for the overall
community
Examples of unacceptable behavior by participants include:
Examples of unacceptable behavior include:
* The use of sexualized language or imagery and unwelcome sexual attention or
advances
* Trolling, insulting/derogatory comments, and personal or political attacks
* The use of sexualized language or imagery, and sexual attention or advances of
any kind
* Trolling, insulting or derogatory comments, and personal or political attacks
* Public or private harassment
* Publishing others' private information, such as a physical or electronic
address, without explicit permission
* Publishing others' private information, such as a physical or email address,
without their explicit permission
* Other conduct which could reasonably be considered inappropriate in a
professional setting
## Our Responsibilities
## Enforcement Responsibilities
Project maintainers are responsible for clarifying the standards of acceptable
behavior and are expected to take appropriate and fair corrective action in
response to any instances of unacceptable behavior.
Community leaders are responsible for clarifying and enforcing our standards of
acceptable behavior and will take appropriate and fair corrective action in
response to any behavior that they deem inappropriate, threatening, offensive,
or harmful.
Project maintainers have the right and responsibility to remove, edit, or
reject comments, commits, code, wiki edits, issues, and other contributions
that are not aligned to this Code of Conduct, or to ban temporarily or
permanently any contributor for other behaviors that they deem inappropriate,
threatening, offensive, or harmful.
Community leaders have the right and responsibility to remove, edit, or reject
comments, commits, code, wiki edits, issues, and other contributions that are
not aligned to this Code of Conduct, and will communicate reasons for moderation
decisions when appropriate.
## Scope
This Code of Conduct applies both within project spaces and in public spaces
when an individual is representing the project or its community. Examples of
representing a project or community include using an official project e-mail
address, posting via an official social media account, or acting as an appointed
representative at an online or offline event. Representation of a project may be
further defined and clarified by project maintainers.
This Code of Conduct applies within all community spaces, and also applies when
an individual is officially representing the community in public spaces.
Examples of representing our community include using an official e-mail address,
posting via an official social media account, or acting as an appointed
representative at an online or offline event.
## Enforcement
Instances of abusive, harassing, or otherwise unacceptable behavior may be
reported by contacting the project team at opensource@zyte.com. All
complaints will be reviewed and investigated and will result in a response that
is deemed necessary and appropriate to the circumstances. The project team is
obligated to maintain confidentiality with regard to the reporter of an incident.
Further details of specific enforcement policies may be posted separately.
reported to the community leaders responsible for enforcement at
opensource@zyte.com.
All complaints will be reviewed and investigated promptly and fairly.
Project maintainers who do not follow or enforce the Code of Conduct in good
faith may face temporary or permanent repercussions as determined by other
members of the project's leadership.
All community leaders are obligated to respect the privacy and security of the
reporter of any incident.
## Enforcement Guidelines
Community leaders will follow these Community Impact Guidelines in determining
the consequences for any action they deem in violation of this Code of Conduct:
### 1. Correction
**Community Impact**: Use of inappropriate language or other behavior deemed
unprofessional or unwelcome in the community.
**Consequence**: A private, written warning from community leaders, providing
clarity around the nature of the violation and an explanation of why the
behavior was inappropriate. A public apology may be requested.
### 2. Warning
**Community Impact**: A violation through a single incident or series of
actions.
**Consequence**: A warning with consequences for continued behavior. No
interaction with the people involved, including unsolicited interaction with
those enforcing the Code of Conduct, for a specified period of time. This
includes avoiding interactions in community spaces as well as external channels
like social media. Violating these terms may lead to a temporary or permanent
ban.
### 3. Temporary Ban
**Community Impact**: A serious violation of community standards, including
sustained inappropriate behavior.
**Consequence**: A temporary ban from any sort of interaction or public
communication with the community for a specified period of time. No public or
private interaction with the people involved, including unsolicited interaction
with those enforcing the Code of Conduct, is allowed during this period.
Violating these terms may lead to a permanent ban.
### 4. Permanent Ban
**Community Impact**: Demonstrating a pattern of violation of community
standards, including sustained inappropriate behavior, harassment of an
individual, or aggression toward or disparagement of classes of individuals.
**Consequence**: A permanent ban from any sort of public interaction within the
community.
## Attribution
This Code of Conduct is adapted from the [Contributor Covenant][homepage], version 1.4,
available at [http://contributor-covenant.org/version/1/4][version].
This Code of Conduct is adapted from the [Contributor Covenant][homepage],
version 2.1, available at
[https://www.contributor-covenant.org/version/2/1/code_of_conduct.html][v2.1].
[homepage]: http://contributor-covenant.org
[version]: http://contributor-covenant.org/version/1/4/
Community Impact Guidelines were inspired by
[Mozilla's code of conduct enforcement ladder][Mozilla CoC].
For answers to common questions about this code of conduct, see
https://www.contributor-covenant.org/faq
For answers to common questions about this code of conduct, see the FAQ at
[https://www.contributor-covenant.org/faq][FAQ]. Translations are available at
[https://www.contributor-covenant.org/translations][translations].
[homepage]: https://www.contributor-covenant.org
[v2.1]: https://www.contributor-covenant.org/version/2/1/code_of_conduct.html
[Mozilla CoC]: https://github.com/mozilla/diversity
[FAQ]: https://www.contributor-covenant.org/faq
[translations]: https://www.contributor-covenant.org/translations

View File

@ -1,4 +0,0 @@
For information about installing Scrapy see:
* docs/intro/install.rst (local file)
* https://docs.scrapy.org/en/latest/intro/install.html (online version)

4
INSTALL.md Normal file
View File

@ -0,0 +1,4 @@
For information about installing Scrapy see:
* [Local docs](docs/intro/install.rst)
* [Online docs](https://docs.scrapy.org/en/latest/intro/install.html)

View File

@ -64,7 +64,9 @@ Requirements
Install
=======
The quick way::
The quick way:
.. code:: bash
pip install scrapy
@ -95,8 +97,7 @@ See https://docs.scrapy.org/en/master/contributing.html for details.
Code of Conduct
---------------
Please note that this project is released with a Contributor Code of Conduct
(see https://github.com/scrapy/scrapy/blob/master/CODE_OF_CONDUCT.md).
Please note that this project is released with a Contributor `Code of Conduct <https://github.com/scrapy/scrapy/blob/master/CODE_OF_CONDUCT.md>`_.
By participating in this project you agree to abide by its terms.
Please report unacceptable behavior to opensource@zyte.com.

View File

@ -15,7 +15,7 @@ class SettingsListDirective(Directive):
def is_setting_index(node):
if node.tagname == 'index':
if node.tagname == 'index' and node['entries']:
# index entries for setting directives look like:
# [('pair', 'SETTING_NAME; setting', 'std:setting-SETTING_NAME', '')]
entry_type, info, refid = node['entries'][0][:3]

View File

@ -10,7 +10,7 @@ Supported Python versions
=========================
Scrapy requires Python 3.7+, either the CPython implementation (default) or
the PyPy 7.3.5+ implementation (see :ref:`python:implementations`).
the PyPy implementation (see :ref:`python:implementations`).
.. _intro-install-scrapy:
@ -52,7 +52,7 @@ Scrapy is written in pure Python and depends on a few key Python packages (among
* `twisted`_, an asynchronous networking framework
* `cryptography`_ and `pyOpenSSL`_, to deal with various network-level security needs
Some of these packages themselves depends on non-Python packages
Some of these packages themselves depend on non-Python packages
that might require additional installation steps depending on your platform.
Please check :ref:`platform-specific guides below <intro-install-platform-notes>`.
@ -187,7 +187,7 @@ solutions:
* Install `homebrew`_ following the instructions in https://brew.sh/
* Update your ``PATH`` variable to state that homebrew packages should be
used before system packages (Change ``.bashrc`` to ``.zshrc`` accordantly
used before system packages (Change ``.bashrc`` to ``.zshrc`` accordingly
if you're using `zsh`_ as default shell)::
echo "export PATH=/usr/local/bin:/usr/local/sbin:$PATH" >> ~/.bashrc
@ -219,7 +219,7 @@ After any of these workarounds you should be able to install Scrapy::
PyPy
----
We recommend using the latest PyPy version. The version tested is 5.9.0.
We recommend using the latest PyPy version.
For PyPy3, only Linux installation was tested.
Most Scrapy dependencies now have binary wheels for CPython, but not for PyPy.

View File

@ -3,6 +3,247 @@
Release notes
=============
.. _release-2.7.1:
Scrapy 2.7.1 (2022-11-02)
-------------------------
New features
~~~~~~~~~~~~
- Relaxed the restriction introduced in 2.6.2 so that the
``Proxy-Authentication`` header can again be set explicitly, as long as the
proxy URL in the :reqmeta:`proxy` metadata has no other credentials, and
for as long as that proxy URL remains the same; this restores compatibility
with scrapy-zyte-smartproxy 2.1.0 and older (:issue:`5626`).
Bug fixes
~~~~~~~~~
- Using ``-O``/``--overwrite-output`` and ``-t``/``--output-format`` options
together now produces an error instead of ignoring the former option
(:issue:`5516`, :issue:`5605`).
- Replaced deprecated :mod:`asyncio` APIs that implicitly use the current
event loop with code that explicitly requests a loop from the event loop
policy (:issue:`5685`, :issue:`5689`).
- Fixed uses of deprecated Scrapy APIs in Scrapy itself (:issue:`5588`,
:issue:`5589`).
- Fixed uses of a deprecated Pillow API (:issue:`5684`, :issue:`5692`).
- Improved code that checks if generators return values, so that it no longer
fails on decorated methods and partial methods (:issue:`5323`,
:issue:`5592`, :issue:`5599`, :issue:`5691`).
Documentation
~~~~~~~~~~~~~
- Upgraded the Code of Conduct to Contributor Covenant v2.1 (:issue:`5698`).
- Fixed typos (:issue:`5681`, :issue:`5694`).
Quality assurance
~~~~~~~~~~~~~~~~~
- Re-enabled some erroneously disabled flake8 checks (:issue:`5688`).
- Ignored harmless deprecation warnings from :mod:`typing` in tests
(:issue:`5686`, :issue:`5697`).
- Modernized our CI configuration (:issue:`5695`, :issue:`5696`).
.. _release-2.7.0:
Scrapy 2.7.0 (2022-10-17)
-----------------------------
Highlights:
- Added Python 3.11 support, dropped Python 3.6 support
- Improved support for :ref:`asynchronous callbacks <topics-coroutines>`
- :ref:`Asyncio support <using-asyncio>` is enabled by default on new
projects
- Output names of item fields can now be arbitrary strings
- Centralized :ref:`request fingerprinting <request-fingerprints>`
configuration is now possible
Modified requirements
~~~~~~~~~~~~~~~~~~~~~
Python 3.7 or greater is now required; support for Python 3.6 has been dropped.
Support for the upcoming Python 3.11 has been added.
The minimum required version of some dependencies has changed as well:
- lxml_: 3.5.0 → 4.3.0
- Pillow_ (:ref:`images pipeline <images-pipeline>`): 4.0.0 → 7.1.0
- zope.interface_: 5.0.0 → 5.1.0
(:issue:`5512`, :issue:`5514`, :issue:`5524`, :issue:`5563`, :issue:`5664`,
:issue:`5670`, :issue:`5678`)
Deprecations
~~~~~~~~~~~~
- :meth:`ImagesPipeline.thumb_path
<scrapy.pipelines.images.ImagesPipeline.thumb_path>` must now accept an
``item`` parameter (:issue:`5504`, :issue:`5508`).
- The ``scrapy.downloadermiddlewares.decompression`` module is now
deprecated (:issue:`5546`, :issue:`5547`).
New features
~~~~~~~~~~~~
- The
:meth:`~scrapy.spidermiddlewares.SpiderMiddleware.process_spider_output`
method of :ref:`spider middlewares <topics-spider-middleware>` can now be
defined as an :term:`asynchronous generator` (:issue:`4978`).
- The output of :class:`~scrapy.Request` callbacks defined as
:ref:`coroutines <topics-coroutines>` is now processed asynchronously
(:issue:`4978`).
- :class:`~scrapy.spiders.crawl.CrawlSpider` now supports :ref:`asynchronous
callbacks <topics-coroutines>` (:issue:`5657`).
- New projects created with the :command:`startproject` command have
:ref:`asyncio support <using-asyncio>` enabled by default (:issue:`5590`,
:issue:`5679`).
- The :setting:`FEED_EXPORT_FIELDS` setting can now be defined as a
dictionary to customize the output name of item fields, lifting the
restriction that required output names to be valid Python identifiers, e.g.
preventing them to have whitespace (:issue:`1008`, :issue:`3266`,
:issue:`3696`).
- You can now customize :ref:`request fingerprinting <request-fingerprints>`
through the new :setting:`REQUEST_FINGERPRINTER_CLASS` setting, instead of
having to change it on every Scrapy component that relies on request
fingerprinting (:issue:`900`, :issue:`3420`, :issue:`4113`, :issue:`4762`,
:issue:`4524`).
- ``jsonl`` is now supported and encouraged as a file extension for `JSON
Lines`_ files (:issue:`4848`).
.. _JSON Lines: https://jsonlines.org/
- :meth:`ImagesPipeline.thumb_path
<scrapy.pipelines.images.ImagesPipeline.thumb_path>` now receives the
source :ref:`item <topics-items>` (:issue:`5504`, :issue:`5508`).
Bug fixes
~~~~~~~~~
- When using Google Cloud Storage with a :ref:`media pipeline
<topics-media-pipeline>`, :setting:`FILES_EXPIRES` now also works when
:setting:`FILES_STORE` does not point at the root of your Google Cloud
Storage bucket (:issue:`5317`, :issue:`5318`).
- The :command:`parse` command now supports :ref:`asynchronous callbacks
<topics-coroutines>` (:issue:`5424`, :issue:`5577`).
- When using the :command:`parse` command with a URL for which there is no
available spider, an exception is no longer raised (:issue:`3264`,
:issue:`3265`, :issue:`5375`, :issue:`5376`, :issue:`5497`).
- :class:`~scrapy.http.TextResponse` now gives higher priority to the `byte
order mark`_ when determining the text encoding of the response body,
following the `HTML living standard`_ (:issue:`5601`, :issue:`5611`).
.. _byte order mark: https://en.wikipedia.org/wiki/Byte_order_mark
.. _HTML living standard: https://html.spec.whatwg.org/multipage/parsing.html#determining-the-character-encoding
- MIME sniffing takes the response body into account in FTP and HTTP/1.0
requests, as well as in cached requests (:issue:`4873`).
- MIME sniffing now detects valid HTML 5 documents even if the ``html`` tag
is missing (:issue:`4873`).
- An exception is now raised if :setting:`ASYNCIO_EVENT_LOOP` has a value
that does not match the asyncio event loop actually installed
(:issue:`5529`).
- Fixed :meth:`Headers.getlist <scrapy.http.headers.Headers.getlist>`
returning only the last header (:issue:`5515`, :issue:`5526`).
- Fixed :class:`LinkExtractor
<scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor>` not ignoring the
``tar.gz`` file extension by default (:issue:`1837`, :issue:`2067`,
:issue:`4066`)
Documentation
~~~~~~~~~~~~~
- Clarified the return type of :meth:`Spider.parse <scrapy.Spider.parse>`
(:issue:`5602`, :issue:`5608`).
- To enable
:class:`~scrapy.downloadermiddlewares.httpcompression.HttpCompressionMiddleware`
to do `brotli compression`_, installing brotli_ is now recommended instead
of installing brotlipy_, as the former provides a more recent version of
brotli.
.. _brotli: https://github.com/google/brotli
.. _brotli compression: https://www.ietf.org/rfc/rfc7932.txt
- :ref:`Signal documentation <topics-signals>` now mentions :ref:`coroutine
support <topics-coroutines>` and uses it in code examples (:issue:`4852`,
:issue:`5358`).
- :ref:`bans` now recommends `Common Crawl`_ instead of `Google cache`_
(:issue:`3582`, :issue:`5432`).
.. _Common Crawl: https://commoncrawl.org/
.. _Google cache: http://www.googleguide.com/cached_pages.html
- The new :ref:`topics-components` topic covers enforcing requirements on
Scrapy components, like :ref:`downloader middlewares
<topics-downloader-middleware>`, :ref:`extensions <topics-extensions>`,
:ref:`item pipelines <topics-item-pipeline>`, :ref:`spider middlewares
<topics-spider-middleware>`, and more; :ref:`enforce-asyncio-requirement`
has also been added (:issue:`4978`).
- :ref:`topics-settings` now indicates that setting values must be
:ref:`picklable <pickle-picklable>` (:issue:`5607`, :issue:`5629`).
- Removed outdated documentation (:issue:`5446`, :issue:`5373`,
:issue:`5369`, :issue:`5370`, :issue:`5554`).
- Fixed typos (:issue:`5442`, :issue:`5455`, :issue:`5457`, :issue:`5461`,
:issue:`5538`, :issue:`5553`, :issue:`5558`, :issue:`5624`, :issue:`5631`).
- Fixed other issues (:issue:`5283`, :issue:`5284`, :issue:`5559`,
:issue:`5567`, :issue:`5648`, :issue:`5659`, :issue:`5665`).
Quality assurance
~~~~~~~~~~~~~~~~~
- Added a continuous integration job to run `twine check`_ (:issue:`5655`,
:issue:`5656`).
.. _twine check: https://twine.readthedocs.io/en/stable/#twine-check
- Addressed test issues and warnings (:issue:`5560`, :issue:`5561`,
:issue:`5612`, :issue:`5617`, :issue:`5639`, :issue:`5645`, :issue:`5662`,
:issue:`5671`, :issue:`5675`).
- Cleaned up code (:issue:`4991`, :issue:`4995`, :issue:`5451`,
:issue:`5487`, :issue:`5542`, :issue:`5667`, :issue:`5668`, :issue:`5672`).
- Applied minor code improvements (:issue:`5661`).
.. _release-2.6.3:
Scrapy 2.6.3 (2022-09-27)
@ -3139,7 +3380,7 @@ New Features
~~~~~~~~~~~~
- Accept proxy credentials in :reqmeta:`proxy` request meta key (:issue:`2526`)
- Support `brotli`_-compressed content; requires optional `brotlipy`_
- Support `brotli-compressed`_ content; requires optional `brotlipy`_
(:issue:`2535`)
- New :ref:`response.follow <response-follow-example>` shortcut
for creating requests (:issue:`1940`)
@ -3176,7 +3417,7 @@ New Features
- ``python -m scrapy`` as a more explicit alternative to ``scrapy`` command
(:issue:`2740`)
.. _brotli: https://github.com/google/brotli
.. _brotli-compressed: https://www.ietf.org/rfc/rfc7932.txt
.. _brotlipy: https://github.com/python-hyper/brotlipy/
Bug fixes

View File

@ -68,7 +68,7 @@ IP (:setting:`CONCURRENT_REQUESTS_PER_IP`).
The default global concurrency limit in Scrapy is not suitable for crawling
many different domains in parallel, so you will want to increase it. How much
to increase it will depend on how much CPU and memory you crawler will have
to increase it will depend on how much CPU and memory your crawler will have
available.
A good starting point is ``100``::

View File

@ -271,11 +271,31 @@ crawl
Start crawling using a spider.
Supported options:
* ``-h, --help``: show a help message and exit
* ``-a NAME=VALUE``: set a spider argument (may be repeated)
* ``--output FILE`` or ``-o FILE``: append scraped items to the end of FILE (use - for stdout), to define format set a colon at the end of the output URI (i.e. ``-o FILE:FORMAT``)
* ``--overwrite-output FILE`` or ``-O FILE``: dump scraped items into FILE, overwriting any existing file, to define format set a colon at the end of the output URI (i.e. ``-O FILE:FORMAT``)
* ``--output-format FORMAT`` or ``-t FORMAT``: deprecated way to define format to use for dumping items, does not work in combination with ``-O``
Usage examples::
$ scrapy crawl myspider
[ ... myspider starts crawling ... ]
$ scrapy -o myfile:csv myspider
[ ... myspider starts crawling and appends the result to the file myfile in csv format ... ]
$ scrapy -O myfile:json myspider
[ ... myspider starts crawling and saves the result in myfile in json format overwriting the original content... ]
$ scrapy -o myfile -t csv myspider
[ ... myspider starts crawling and appends the result to the file myfile in csv format ... ]
.. command:: check

View File

@ -75,9 +75,9 @@ If your requirement is a minimum Scrapy version, you may use
class MyComponent:
def __init__(self):
if parse_version(scrapy.__version__) < parse_version('VERSION'):
if parse_version(scrapy.__version__) < parse_version('2.7'):
raise RuntimeError(
f"{MyComponent.__qualname__} requires Scrapy VERSION or "
f"{MyComponent.__qualname__} requires Scrapy 2.7 or "
f"later, which allow defining the process_spider_output "
f"method of spider middlewares as an asynchronous "
f"generator."

View File

@ -102,7 +102,7 @@ override three methods:
.. method:: Contract.post_process(output)
This allows processing the output of the callback. Iterators are
converted listified before being passed to this hook.
converted to lists before being passed to this hook.
Raise :class:`~scrapy.exceptions.ContractFail` from
:class:`~scrapy.contracts.Contract.pre_process` or

View File

@ -22,7 +22,7 @@ hence use coroutine syntax (e.g. ``await``, ``async for``, ``async with``):
If you are using any custom or third-party :ref:`spider middleware
<topics-spider-middleware>`, see :ref:`sync-async-spider-middleware`.
.. versionchanged:: VERSION
.. versionchanged:: 2.7
Output of async callbacks is now processed asynchronously instead of
collecting all of it first.
@ -49,7 +49,7 @@ hence use coroutine syntax (e.g. ``await``, ``async for``, ``async with``):
See also :ref:`sync-async-spider-middleware` and
:ref:`universal-spider-middleware`.
.. versionadded:: VERSION
.. versionadded:: 2.7
General usage
=============
@ -129,7 +129,7 @@ Common use cases for asynchronous code include:
Mixing synchronous and asynchronous spider middlewares
======================================================
.. versionadded:: VERSION
.. versionadded:: 2.7
The output of a :class:`~scrapy.Request` callback is passed as the ``result``
parameter to the
@ -182,10 +182,10 @@ process_spider_output_async method <universal-spider-middleware>`.
Universal spider middlewares
============================
.. versionadded:: VERSION
.. versionadded:: 2.7
To allow writing a spider middleware that supports asynchronous execution of
its ``process_spider_output`` method in Scrapy VERSION and later (avoiding
its ``process_spider_output`` method in Scrapy 2.7 and later (avoiding
:ref:`asynchronous-to-synchronous conversions <sync-async-spider-middleware>`)
while maintaining support for older Scrapy versions, you may define
``process_spider_output`` as a synchronous method and define an
@ -206,7 +206,7 @@ For example::
yield r
.. note:: This is an interim measure to allow, for a time, to write code that
works in Scrapy VERSION and later without requiring
works in Scrapy 2.7 and later without requiring
asynchronous-to-synchronous conversions, and works in earlier Scrapy
versions as well.

View File

@ -150,3 +150,31 @@ available in all future runs should they be necessary again::
For more information, check the :ref:`topics-logging` section.
.. _base tag: https://www.w3schools.com/tags/tag_base.asp
Visual Studio Code
==================
.. highlight:: json
To debug spiders with Visual Studio Code you can use the following ``launch.json``::
{
"version": "0.1.0",
"configurations": [
{
"name": "Python: Launch Scrapy Spider",
"type": "python",
"request": "launch",
"module": "scrapy",
"args": [
"runspider",
"${file}"
],
"console": "integratedTerminal"
}
]
}
Also, make sure you enable "User Uncaught Exceptions", to catch exceptions in
your Scrapy spider.

View File

@ -17,7 +17,7 @@ settings, just like any other Scrapy code.
It is customary for extensions to prefix their settings with their own name, to
avoid collision with existing (and future) extensions. For example, a
hypothetic extension to handle `Google Sitemaps`_ would use settings like
hypothetical extension to handle `Google Sitemaps`_ would use settings like
``GOOGLESITEMAP_ENABLED``, ``GOOGLESITEMAP_DEPTH``, and so on.
.. _Google Sitemaps: https://en.wikipedia.org/wiki/Sitemaps

View File

@ -154,7 +154,7 @@ Too many spiders?
If your project has too many spiders executed in parallel,
the output of :func:`prefs()` can be difficult to read.
For this reason, that function has a ``ignore`` argument which can be used to
ignore a particular class (and all its subclases). For
ignore a particular class (and all its subclasses). For
example, this won't show any live references to spiders:
>>> from scrapy.spiders import Spider

View File

@ -394,7 +394,7 @@ To change how request fingerprints are built for your requests, use the
REQUEST_FINGERPRINTER_CLASS
~~~~~~~~~~~~~~~~~~~~~~~~~~~
.. versionadded:: VERSION
.. versionadded:: 2.7
Default: :class:`scrapy.utils.request.RequestFingerprinter`
@ -409,54 +409,54 @@ import path.
REQUEST_FINGERPRINTER_IMPLEMENTATION
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
.. versionadded:: VERSION
.. versionadded:: 2.7
Default: ``'PREVIOUS_VERSION'``
Default: ``'2.6'``
Determines which request fingerprinting algorithm is used by the default
request fingerprinter class (see :setting:`REQUEST_FINGERPRINTER_CLASS`).
Possible values are:
- ``'PREVIOUS_VERSION'`` (default)
- ``'2.6'`` (default)
This implementation uses the same request fingerprinting algorithm as
Scrapy PREVIOUS_VERSION and earlier versions.
Scrapy 2.6 and earlier versions.
Even though this is the default value for backward compatibility reasons,
it is a deprecated value.
- ``'VERSION'``
- ``'2.7'``
This implementation was introduced in Scrapy VERSION to fix an issue of the
This implementation was introduced in Scrapy 2.7 to fix an issue of the
previous implementation.
New projects should use this value. The :command:`startproject` command
sets this value in the generated ``settings.py`` file.
If you are using the default value (``'PREVIOUS_VERSION'``) for this setting, and you are
If you are using the default value (``'2.6'``) for this setting, and you are
using Scrapy components where changing the request fingerprinting algorithm
would cause undesired results, you need to carefully decide when to change the
value of this setting, or switch the :setting:`REQUEST_FINGERPRINTER_CLASS`
setting to a custom request fingerprinter class that implements the PREVIOUS_VERSION request
setting to a custom request fingerprinter class that implements the 2.6 request
fingerprinting algorithm and does not log this warning (
:ref:`PREVIOUS_VERSION-request-fingerprinter` includes an example implementation of such a
:ref:`2.6-request-fingerprinter` includes an example implementation of such a
class).
Scenarios where changing the request fingerprinting algorithm may cause
undesired results include, for example, using the HTTP cache middleware (see
:class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`).
Changing the request fingerprinting algorithm would invalidade the current
Changing the request fingerprinting algorithm would invalidate the current
cache, requiring you to redownload all requests again.
Otherwise, set :setting:`REQUEST_FINGERPRINTER_IMPLEMENTATION` to ``'VERSION'`` in
Otherwise, set :setting:`REQUEST_FINGERPRINTER_IMPLEMENTATION` to ``'2.7'`` in
your settings to switch already to the request fingerprinting implementation
that will be the only request fingerprinting implementation available in a
future version of Scrapy, and remove the deprecation warning triggered by using
the default value (``'PREVIOUS_VERSION'``).
the default value (``'2.6'``).
.. _PREVIOUS_VERSION-request-fingerprinter:
.. _2.6-request-fingerprinter:
.. _custom-request-fingerprinter:
Writing your own request fingerprinter
@ -464,6 +464,8 @@ Writing your own request fingerprinter
A request fingerprinter is a class that must implement the following method:
.. currentmodule:: None
.. method:: fingerprint(self, request)
Return a :class:`bytes` object that uniquely identifies *request*.
@ -476,6 +478,7 @@ A request fingerprinter is a class that must implement the following method:
Additionally, it may also implement the following methods:
.. classmethod:: from_crawler(cls, crawler)
:noindex:
If present, this class method is called to create a request fingerprinter
instance from a :class:`~scrapy.crawler.Crawler` object. It must return a
@ -495,11 +498,13 @@ Additionally, it may also implement the following methods:
:class:`~scrapy.settings.Settings` object. It must return a new instance of
the request fingerprinter.
The ``fingerprint`` method of the default request fingerprinter,
.. currentmodule:: scrapy.http
The :meth:`fingerprint` method of the default request fingerprinter,
:class:`scrapy.utils.request.RequestFingerprinter`, uses
:func:`scrapy.utils.request.fingerprint` with its default parameters. For some
common use cases you can use :func:`~scrapy.utils.request.fingerprint` as well
in your ``fingerprint`` method implementation:
common use cases you can use :func:`scrapy.utils.request.fingerprint` as well
in your :meth:`fingerprint` method implementation:
.. autofunction:: scrapy.utils.request.fingerprint
@ -519,7 +524,7 @@ account::
You can also write your own fingerprinting logic from scratch.
However, if you do not use :func:`~scrapy.utils.request.fingerprint`, make sure
However, if you do not use :func:`scrapy.utils.request.fingerprint`, make sure
you use :class:`~weakref.WeakKeyDictionary` to cache request fingerprints:
- Caching saves CPU by ensuring that fingerprints are calculated only once
@ -553,7 +558,7 @@ If you need to be able to override the request fingerprinting for arbitrary
requests from your spider callbacks, you may implement a request fingerprinter
that reads fingerprints from :attr:`request.meta <scrapy.http.Request.meta>`
when available, and then falls back to
:func:`~scrapy.utils.request.fingerprint`. For example::
:func:`scrapy.utils.request.fingerprint`. For example::
from scrapy.utils.request import fingerprint
@ -564,8 +569,8 @@ when available, and then falls back to
return request.meta['fingerprint']
return fingerprint(request)
If you need to reproduce the same fingerprinting algorithm as Scrapy PREVIOUS_VERSION
without using the deprecated ``'PREVIOUS_VERSION'`` value of the
If you need to reproduce the same fingerprinting algorithm as Scrapy 2.6
without using the deprecated ``'2.6'`` value of the
:setting:`REQUEST_FINGERPRINTER_IMPLEMENTATION` setting, use the following
request fingerprinter::

View File

@ -1642,6 +1642,11 @@ install the default reactor defined by Twisted for the current platform. This
is to maintain backward compatibility and avoid possible problems caused by
using a non-default reactor.
.. versionchanged:: 2.7
The :command:`startproject` command now sets this setting to
``twisted.internet.asyncioreactor.AsyncioSelectorReactor`` in the generated
``settings.py`` file.
For additional information, see :doc:`core/howto/choosing-reactor`.
@ -1656,14 +1661,14 @@ Scope: ``spidermiddlewares.urllength``
The maximum URL length to allow for crawled URLs.
This setting can act as a stopping condition in case of URLs of ever-increasing
length, which may be caused for example by a programming error either in the
target server or in your code. See also :setting:`REDIRECT_MAX_TIMES` and
This setting can act as a stopping condition in case of URLs of ever-increasing
length, which may be caused for example by a programming error either in the
target server or in your code. See also :setting:`REDIRECT_MAX_TIMES` and
:setting:`DEPTH_LIMIT`.
Use ``0`` to allow URLs of any length.
The default value is copied from the `Microsoft Internet Explorer maximum URL
The default value is copied from the `Microsoft Internet Explorer maximum URL
length`_, even though this setting exists for different reasons.
.. _Microsoft Internet Explorer maximum URL length: https://support.microsoft.com/en-us/topic/maximum-url-length-is-2-083-characters-in-internet-explorer-174e7c8a-6666-f4e0-6fd6-908b53c12246

View File

@ -105,17 +105,17 @@ object gives you access, for example, to the :ref:`settings <topics-settings>`.
:class:`~scrapy.Request` objects and :ref:`item objects
<topics-items>`.
.. versionchanged:: VERSION
.. versionchanged:: 2.7
This method may be defined as an :term:`asynchronous generator`, in
which case ``result`` is an :term:`asynchronous iterable`.
Consider defining this method as an :term:`asynchronous generator`,
which will be a requirement in a future version of Scrapy. However, if
you plan on sharing your spider middleware with other people, consider
either :ref:`enforcing Scrapy VERSION <enforce-component-requirements>`
either :ref:`enforcing Scrapy 2.7 <enforce-component-requirements>`
as a minimum requirement of your spider middleware, or :ref:`making
your spider middleware universal <universal-spider-middleware>` so that
it works with Scrapy versions earlier than Scrapy VERSION.
it works with Scrapy versions earlier than Scrapy 2.7.
:param response: the response which generated this output from the
spider
@ -130,7 +130,7 @@ object gives you access, for example, to the :ref:`settings <topics-settings>`.
.. method:: process_spider_output_async(response, result, spider)
.. versionadded:: VERSION
.. versionadded:: 2.7
If defined, this method must be an :term:`asynchronous generator`,
which will be called instead of :meth:`process_spider_output` if

View File

@ -24,3 +24,5 @@ markers =
filterwarnings =
ignore:scrapy.downloadermiddlewares.decompression is deprecated
ignore:Module scrapy.utils.reqser is deprecated
ignore:typing.re is deprecated
ignore:typing.io is deprecated

View File

@ -1 +1 @@
2.6.2
2.7.1

View File

@ -7,7 +7,7 @@ import pkg_resources
import scrapy
from scrapy.crawler import CrawlerProcess
from scrapy.commands import ScrapyCommand, ScrapyHelpFormatter
from scrapy.commands import ScrapyCommand, ScrapyHelpFormatter, BaseRunSpiderCommand
from scrapy.exceptions import UsageError
from scrapy.utils.misc import walk_modules
from scrapy.utils.project import inside_project, get_project_settings
@ -32,7 +32,7 @@ def _iter_command_classes(module_name):
inspect.isclass(obj)
and issubclass(obj, ScrapyCommand)
and obj.__module__ == module.__name__
and not obj == ScrapyCommand
and obj not in (ScrapyCommand, BaseRunSpiderCommand)
):
yield obj
@ -78,7 +78,8 @@ def _pop_command_name(argv):
def _print_header(settings, inproject):
version = scrapy.__version__
if inproject:
print(f"Scrapy {version} - project: {settings['BOT_NAME']}\n")
print(f"Scrapy {version} - active project: {settings['BOT_NAME']}\n")
else:
print(f"Scrapy {version} - no active project\n")

View File

@ -116,9 +116,11 @@ class BaseRunSpiderCommand(ScrapyCommand):
parser.add_argument("-a", dest="spargs", action="append", default=[], metavar="NAME=VALUE",
help="set spider argument (may be repeated)")
parser.add_argument("-o", "--output", metavar="FILE", action="append",
help="append scraped items to the end of FILE (use - for stdout)")
help="append scraped items to the end of FILE (use - for stdout),"
" to define format set a colon at the end of the output URI (i.e. -o FILE:FORMAT)")
parser.add_argument("-O", "--overwrite-output", metavar="FILE", action="append",
help="dump scraped items into FILE, overwriting any existing file")
help="dump scraped items into FILE, overwriting any existing file,"
" to define format set a colon at the end of the output URI (i.e. -O FILE:FORMAT)")
parser.add_argument("-t", "--output-format", metavar="FORMAT",
help="format to use for dumping items")

View File

@ -5,6 +5,8 @@ from typing import Dict
from itemadapter import is_item, ItemAdapter
from w3lib.url import is_url
from twisted.internet.defer import maybeDeferred
from scrapy.commands import BaseRunSpiderCommand
from scrapy.http import Request
from scrapy.utils import display
@ -110,16 +112,19 @@ class Command(BaseRunSpiderCommand):
if not opts.nolinks:
self.print_requests(colour=colour)
def run_callback(self, response, callback, cb_kwargs=None):
cb_kwargs = cb_kwargs or {}
def _get_items_and_requests(self, spider_output, opts, depth, spider, callback):
items, requests = [], []
for x in iterate_spider_output(callback(response, **cb_kwargs)):
for x in spider_output:
if is_item(x):
items.append(x)
elif isinstance(x, Request):
requests.append(x)
return items, requests
return items, requests, opts, depth, spider, callback
def run_callback(self, response, callback, cb_kwargs=None):
cb_kwargs = cb_kwargs or {}
d = maybeDeferred(iterate_spider_output, callback(response, **cb_kwargs))
return d
def get_callback_from_rules(self, spider, response):
if getattr(spider, 'rules', None):
@ -158,6 +163,25 @@ class Command(BaseRunSpiderCommand):
logger.error('No response downloaded for: %(url)s',
{'url': url})
def scraped_data(self, args):
items, requests, opts, depth, spider, callback = args
if opts.pipelines:
itemproc = self.pcrawler.engine.scraper.itemproc
for item in items:
itemproc.process_item(item, spider)
self.add_items(depth, items)
self.add_requests(depth, requests)
scraped_data = items if opts.output else []
if depth < opts.depth:
for req in requests:
req.meta['_depth'] = depth + 1
req.meta['_callback'] = req.callback
req.callback = callback
scraped_data += requests
return scraped_data
def prepare_request(self, spider, request, opts):
def callback(response, **cb_kwargs):
# memorize first request
@ -191,23 +215,10 @@ class Command(BaseRunSpiderCommand):
# parse items and requests
depth = response.meta['_depth']
items, requests = self.run_callback(response, cb, cb_kwargs)
if opts.pipelines:
itemproc = self.pcrawler.engine.scraper.itemproc
for item in items:
itemproc.process_item(item, spider)
self.add_items(depth, items)
self.add_requests(depth, requests)
scraped_data = items if opts.output else []
if depth < opts.depth:
for req in requests:
req.meta['_depth'] = depth + 1
req.meta['_callback'] = req.callback
req.callback = callback
scraped_data += requests
return scraped_data
d = self.run_callback(response, cb, cb_kwargs)
d.addCallback(self._get_items_and_requests, opts, depth, spider, callback)
d.addCallback(self.scraped_data)
return d
# update request meta if any extra meta was passed through the --meta/-m opts.
if opts.meta:

View File

@ -3,7 +3,6 @@
import ipaddress
import logging
import re
import warnings
from contextlib import suppress
from io import BytesIO
from time import time
@ -22,7 +21,7 @@ from zope.interface import implementer
from scrapy import signals
from scrapy.core.downloader.contextfactory import load_context_factory_from_settings
from scrapy.core.downloader.webclient import _parse
from scrapy.exceptions import ScrapyDeprecationWarning, StopDownload
from scrapy.exceptions import StopDownload
from scrapy.http import Headers
from scrapy.responsetypes import responsetypes
from scrapy.utils.python import to_bytes, to_unicode
@ -279,17 +278,7 @@ class ScrapyAgent:
proxyScheme, proxyNetloc, proxyHost, proxyPort, proxyParams = _parse(proxy)
scheme = _parse(request.url)[0]
proxyHost = to_unicode(proxyHost)
omitConnectTunnel = b'noconnect' in proxyParams
if omitConnectTunnel:
warnings.warn(
"Using HTTPS proxies in the noconnect mode is deprecated. "
"If you use Zyte Smart Proxy Manager, it doesn't require "
"this mode anymore, so you should update scrapy-crawlera "
"to scrapy-zyte-smartproxy and remove '?noconnect' "
"from the Zyte Smart Proxy Manager URL.",
ScrapyDeprecationWarning,
)
if scheme == b'https' and not omitConnectTunnel:
if scheme == b'https':
proxyAuth = request.headers.get(b'Proxy-Authorization', None)
proxyConf = (proxyHost, proxyPort, proxyAuth)
return self._TunnelingAgent(
@ -302,8 +291,6 @@ class ScrapyAgent:
)
else:
proxyScheme = proxyScheme or b'http'
proxyHost = to_bytes(proxyHost, encoding='ascii')
proxyPort = to_bytes(str(proxyPort), encoding='ascii')
proxyURI = urlunparse((proxyScheme, proxyNetloc, proxyParams, '', '', ''))
return self._ProxyAgent(
reactor=reactor,
@ -384,8 +371,7 @@ class ScrapyAgent:
logger.debug("Download stopped for %(request)s from signal handler %(handler)s",
{"request": request, "handler": handler.__qualname__})
txresponse._transport.stopProducing()
with suppress(AttributeError):
txresponse._transport._producer.loseConnection()
txresponse._transport.loseConnection()
return {
"txresponse": txresponse,
"body": b"",
@ -417,7 +403,7 @@ class ScrapyAgent:
logger.warning(warning_msg, warning_args)
txresponse._transport._producer.loseConnection()
txresponse._transport.loseConnection()
raise defer.CancelledError(warning_msg % warning_args)
if warnsize and expected_size > warnsize:
@ -543,7 +529,7 @@ class _ResponseReader(protocol.Protocol):
logger.debug("Download stopped for %(request)s from signal handler %(handler)s",
{"request": self._request, "handler": handler.__qualname__})
self.transport.stopProducing()
self.transport._producer.loseConnection()
self.transport.loseConnection()
failure = result if result.value.fail else None
self._finish_response(flags=["download_stopped"], failure=failure)

View File

@ -1,4 +1,3 @@
import warnings
from time import time
from typing import Optional, Type, TypeVar
from urllib.parse import urldefrag
@ -69,19 +68,8 @@ class ScrapyH2Agent:
if proxy:
_, _, proxy_host, proxy_port, proxy_params = _parse(proxy)
scheme = _parse(request.url)[0]
proxy_host = proxy_host.decode()
omit_connect_tunnel = b'noconnect' in proxy_params
if omit_connect_tunnel:
warnings.warn(
"Using HTTPS proxies in the noconnect mode is not "
"supported by the downloader handler. If you use Zyte "
"Smart Proxy Manager, it doesn't require this mode "
"anymore, so you should update scrapy-crawlera to "
"scrapy-zyte-smartproxy and remove '?noconnect' from the "
"Zyte Smart Proxy Manager URL."
)
if scheme == b'https' and not omit_connect_tunnel:
if scheme == b'https':
# ToDo
raise NotImplementedError('Tunneling via CONNECT method using HTTP/2.0 is not yet supported')
return self._ProxyAgent(

View File

@ -257,9 +257,7 @@ class ExecutionEngine:
def download(self, request: Request, spider: Optional[Spider] = None) -> Deferred:
"""Return a Deferred which fires with a Response as result, only downloader middlewares are applied"""
if spider is None:
spider = self.spider
else:
if spider is not None:
warnings.warn(
"Passing a 'spider' argument to ExecutionEngine.download is deprecated",
category=ScrapyDeprecationWarning,
@ -267,7 +265,7 @@ class ExecutionEngine:
)
if spider is not self.spider:
logger.warning("The spider '%s' does not match the open spider", spider.name)
if spider is None:
if self.spider is None:
raise RuntimeError(f"No open spider to crawl: {request}")
return self._download(request, spider).addBoth(self._downloaded, request, spider)
@ -278,11 +276,14 @@ class ExecutionEngine:
self.slot.remove_request(request)
return self.download(result, spider) if isinstance(result, Request) else result
def _download(self, request: Request, spider: Spider) -> Deferred:
def _download(self, request: Request, spider: Optional[Spider]) -> Deferred:
assert self.slot is not None # typing
self.slot.add_request(request)
if spider is None:
spider = self.spider
def _on_success(result: Union[Response, Request]) -> Union[Response, Request]:
if not isinstance(result, (Response, Request)):
raise TypeError(f"Incorrect type: expected Response or Request, got {type(result)}: {result!r}")

View File

@ -151,11 +151,9 @@ class Stream:
self._deferred_response = Deferred(_cancel)
def __str__(self) -> str:
def __repr__(self) -> str:
return f'Stream(id={self.stream_id!r})'
__repr__ = __str__
@property
def _log_warnsize(self) -> bool:
"""Checks if we have received data which exceeds the download warnsize

View File

@ -34,7 +34,12 @@ from scrapy.utils.log import (
)
from scrapy.utils.misc import create_instance, load_object
from scrapy.utils.ossignal import install_shutdown_handlers, signal_names
from scrapy.utils.reactor import install_reactor, verify_installed_reactor
from scrapy.utils.reactor import (
install_reactor,
is_asyncio_reactor_installed,
verify_installed_asyncio_event_loop,
verify_installed_reactor,
)
if TYPE_CHECKING:
from scrapy.utils.request import RequestFingerprinter
@ -84,17 +89,20 @@ class Crawler:
crawler=self,
)
reactor_class = self.settings.get("TWISTED_REACTOR")
reactor_class = self.settings["TWISTED_REACTOR"]
event_loop = self.settings["ASYNCIO_EVENT_LOOP"]
if init_reactor:
# this needs to be done after the spider settings are merged,
# but before something imports twisted.internet.reactor
if reactor_class:
install_reactor(reactor_class, self.settings["ASYNCIO_EVENT_LOOP"])
install_reactor(reactor_class, event_loop)
else:
from twisted.internet import reactor # noqa: F401
log_reactor_info()
if reactor_class:
verify_installed_reactor(reactor_class)
if is_asyncio_reactor_installed() and event_loop:
verify_installed_asyncio_event_loop(event_loop)
self.extensions = ExtensionManager.from_crawler(self)

View File

@ -78,4 +78,7 @@ class HttpProxyMiddleware:
del request.headers[b'Proxy-Authorization']
del request.meta['_auth_proxy']
elif b'Proxy-Authorization' in request.headers:
del request.headers[b'Proxy-Authorization']
if proxy_url:
request.meta['_auth_proxy'] = proxy_url
else:
del request.headers[b'Proxy-Authorization']

View File

@ -75,15 +75,16 @@ class MemoryUsage:
self.crawler.stats.max_value('memusage/max', self.get_virtual_size())
def _check_limit(self):
if self.get_virtual_size() > self.limit:
peak_mem_usage = self.get_virtual_size()
if peak_mem_usage > self.limit:
self.crawler.stats.set_value('memusage/limit_reached', 1)
mem = self.limit / 1024 / 1024
logger.error("Memory usage exceeded %(memusage)dM. Shutting down Scrapy...",
logger.error("Memory usage exceeded %(memusage)dMiB. Shutting down Scrapy...",
{'memusage': mem}, extra={'crawler': self.crawler})
if self.notify_mails:
subj = (
f"{self.crawler.settings['BOT_NAME']} terminated: "
f"memory usage exceeded {mem}M at {socket.gethostname()}"
f"memory usage exceeded {mem}MiB at {socket.gethostname()}"
)
self._send_report(self.notify_mails, subj)
self.crawler.stats.set_value('memusage/limit_notified', 1)
@ -92,6 +93,8 @@ class MemoryUsage:
self.crawler.engine.close_spider(self.crawler.engine.spider, 'memusage_exceeded')
else:
self.crawler.stop()
else:
logger.info("Peak memory usage is %(virtualsize)dMiB", {'virtualsize': peak_mem_usage / 1024 / 1024})
def _check_warning(self):
if self.warned: # warn only once
@ -99,12 +102,12 @@ class MemoryUsage:
if self.get_virtual_size() > self.warning:
self.crawler.stats.set_value('memusage/warning_reached', 1)
mem = self.warning / 1024 / 1024
logger.warning("Memory usage reached %(memusage)dM",
logger.warning("Memory usage reached %(memusage)dMiB",
{'memusage': mem}, extra={'crawler': self.crawler})
if self.notify_mails:
subj = (
f"{self.crawler.settings['BOT_NAME']} warning: "
f"memory usage reached {mem}M at {socket.gethostname()}"
f"memory usage reached {mem}MiB at {socket.gethostname()}"
)
self._send_report(self.notify_mails, subj)
self.crawler.stats.set_value('memusage/warning_notified', 1)

View File

@ -121,11 +121,9 @@ class Request(object_ref):
def encoding(self) -> str:
return self._encoding
def __str__(self) -> str:
def __repr__(self) -> str:
return f"<{self.method} {self.url}>"
__repr__ = __str__
def copy(self) -> "Request":
return self.replace()

View File

@ -100,11 +100,9 @@ class Response(object_ref):
body = property(_get_body, obsolete_setter(_set_body, 'body'))
def __str__(self):
def __repr__(self):
return f"<{self.status} {self.url}>"
__repr__ = __str__
def copy(self):
"""Return a copy of this Response"""
return self.replace()

View File

@ -6,18 +6,6 @@ This package contains a collection of Link Extractors.
For more info see docs/topics/link-extractors.rst
"""
import re
from urllib.parse import urlparse
from warnings import warn
from parsel.csstranslator import HTMLTranslator
from w3lib.url import canonicalize_url
from scrapy.utils.deprecate import ScrapyDeprecationWarning
from scrapy.utils.misc import arg_to_iter
from scrapy.utils.url import (
url_is_from_any_domain, url_has_any_extension,
)
# common file extensions that are not followed if they occur in links
IGNORED_EXTENSIONS = [
@ -55,82 +43,5 @@ def _is_valid_url(url):
return url.split('://', 1)[0] in {'http', 'https', 'file', 'ftp'}
class FilteringLinkExtractor:
_csstranslator = HTMLTranslator()
def __new__(cls, *args, **kwargs):
from scrapy.linkextractors.lxmlhtml import LxmlLinkExtractor
if issubclass(cls, FilteringLinkExtractor) and not issubclass(cls, LxmlLinkExtractor):
warn('scrapy.linkextractors.FilteringLinkExtractor is deprecated, '
'please use scrapy.linkextractors.LinkExtractor instead',
ScrapyDeprecationWarning, stacklevel=2)
return super().__new__(cls)
def __init__(self, link_extractor, allow, deny, allow_domains, deny_domains,
restrict_xpaths, canonicalize, deny_extensions, restrict_css, restrict_text):
self.link_extractor = link_extractor
self.allow_res = [x if isinstance(x, _re_type) else re.compile(x)
for x in arg_to_iter(allow)]
self.deny_res = [x if isinstance(x, _re_type) else re.compile(x)
for x in arg_to_iter(deny)]
self.allow_domains = set(arg_to_iter(allow_domains))
self.deny_domains = set(arg_to_iter(deny_domains))
self.restrict_xpaths = tuple(arg_to_iter(restrict_xpaths))
self.restrict_xpaths += tuple(map(self._csstranslator.css_to_xpath,
arg_to_iter(restrict_css)))
self.canonicalize = canonicalize
if deny_extensions is None:
deny_extensions = IGNORED_EXTENSIONS
self.deny_extensions = {'.' + e for e in arg_to_iter(deny_extensions)}
self.restrict_text = [x if isinstance(x, _re_type) else re.compile(x)
for x in arg_to_iter(restrict_text)]
def _link_allowed(self, link):
if not _is_valid_url(link.url):
return False
if self.allow_res and not _matches(link.url, self.allow_res):
return False
if self.deny_res and _matches(link.url, self.deny_res):
return False
parsed_url = urlparse(link.url)
if self.allow_domains and not url_is_from_any_domain(parsed_url, self.allow_domains):
return False
if self.deny_domains and url_is_from_any_domain(parsed_url, self.deny_domains):
return False
if self.deny_extensions and url_has_any_extension(parsed_url, self.deny_extensions):
return False
if self.restrict_text and not _matches(link.text, self.restrict_text):
return False
return True
def matches(self, url):
if self.allow_domains and not url_is_from_any_domain(url, self.allow_domains):
return False
if self.deny_domains and url_is_from_any_domain(url, self.deny_domains):
return False
allowed = (regex.search(url) for regex in self.allow_res) if self.allow_res else [True]
denied = (regex.search(url) for regex in self.deny_res) if self.deny_res else []
return any(allowed) and not any(denied)
def _process_links(self, links):
links = [x for x in links if self._link_allowed(x)]
if self.canonicalize:
for link in links:
link.url = canonicalize_url(link.url)
links = self.link_extractor._process_links(links)
return links
def _extract_links(self, *args, **kwargs):
return self.link_extractor._extract_links(*args, **kwargs)
# Top-level imports
from scrapy.linkextractors.lxmlhtml import LxmlLinkExtractor as LinkExtractor

View File

@ -3,18 +3,20 @@ Link extractor based on lxml.html
"""
import operator
from functools import partial
from urllib.parse import urljoin
from urllib.parse import urljoin, urlparse
import lxml.etree as etree
from parsel.csstranslator import HTMLTranslator
from w3lib.html import strip_html5_whitespace
from w3lib.url import canonicalize_url, safe_url_string
from scrapy.link import Link
from scrapy.linkextractors import FilteringLinkExtractor
from scrapy.linkextractors import (IGNORED_EXTENSIONS, _is_valid_url, _matches,
_re_type, re)
from scrapy.utils.misc import arg_to_iter, rel_has_nofollow
from scrapy.utils.python import unique as unique_list
from scrapy.utils.response import get_base_url
from scrapy.utils.url import url_has_any_extension, url_is_from_any_domain
# from lxml/src/lxml/html/__init__.py
XHTML_NAMESPACE = "http://www.w3.org/1999/xhtml"
@ -98,7 +100,8 @@ class LxmlParserLinkExtractor:
return links
class LxmlLinkExtractor(FilteringLinkExtractor):
class LxmlLinkExtractor:
_csstranslator = HTMLTranslator()
def __init__(
self,
@ -118,7 +121,7 @@ class LxmlLinkExtractor(FilteringLinkExtractor):
restrict_text=None,
):
tags, attrs = set(arg_to_iter(tags)), set(arg_to_iter(attrs))
lx = LxmlParserLinkExtractor(
self.link_extractor = LxmlParserLinkExtractor(
tag=partial(operator.contains, tags),
attr=partial(operator.contains, attrs),
unique=unique,
@ -126,18 +129,64 @@ class LxmlLinkExtractor(FilteringLinkExtractor):
strip=strip,
canonicalized=canonicalize
)
super().__init__(
link_extractor=lx,
allow=allow,
deny=deny,
allow_domains=allow_domains,
deny_domains=deny_domains,
restrict_xpaths=restrict_xpaths,
restrict_css=restrict_css,
canonicalize=canonicalize,
deny_extensions=deny_extensions,
restrict_text=restrict_text,
)
self.allow_res = [x if isinstance(x, _re_type) else re.compile(x)
for x in arg_to_iter(allow)]
self.deny_res = [x if isinstance(x, _re_type) else re.compile(x)
for x in arg_to_iter(deny)]
self.allow_domains = set(arg_to_iter(allow_domains))
self.deny_domains = set(arg_to_iter(deny_domains))
self.restrict_xpaths = tuple(arg_to_iter(restrict_xpaths))
self.restrict_xpaths += tuple(map(self._csstranslator.css_to_xpath,
arg_to_iter(restrict_css)))
if deny_extensions is None:
deny_extensions = IGNORED_EXTENSIONS
self.canonicalize = canonicalize
self.deny_extensions = {'.' + e for e in arg_to_iter(deny_extensions)}
self.restrict_text = [x if isinstance(x, _re_type) else re.compile(x)
for x in arg_to_iter(restrict_text)]
def _link_allowed(self, link):
if not _is_valid_url(link.url):
return False
if self.allow_res and not _matches(link.url, self.allow_res):
return False
if self.deny_res and _matches(link.url, self.deny_res):
return False
parsed_url = urlparse(link.url)
if self.allow_domains and not url_is_from_any_domain(parsed_url, self.allow_domains):
return False
if self.deny_domains and url_is_from_any_domain(parsed_url, self.deny_domains):
return False
if self.deny_extensions and url_has_any_extension(parsed_url, self.deny_extensions):
return False
if self.restrict_text and not _matches(link.text, self.restrict_text):
return False
return True
def matches(self, url):
if self.allow_domains and not url_is_from_any_domain(url, self.allow_domains):
return False
if self.deny_domains and url_is_from_any_domain(url, self.deny_domains):
return False
allowed = (regex.search(url) for regex in self.allow_res) if self.allow_res else [True]
denied = (regex.search(url) for regex in self.deny_res) if self.deny_res else []
return any(allowed) and not any(denied)
def _process_links(self, links):
links = [x for x in links if self._link_allowed(x)]
if self.canonicalize:
for link in links:
link.url = canonicalize_url(link.url)
links = self.link_extractor._process_links(links)
return links
def _extract_links(self, *args, **kwargs):
return self.link_extractor._extract_links(*args, **kwargs)
def extract_links(self, response):
"""Returns a list of :class:`~scrapy.link.Link` objects from the

View File

@ -5,23 +5,28 @@ See documentation in topics/media-pipeline.rst
"""
import functools
import hashlib
import warnings
from contextlib import suppress
from io import BytesIO
from itemadapter import ItemAdapter
from scrapy.exceptions import DropItem, NotConfigured
from scrapy.exceptions import DropItem, NotConfigured, ScrapyDeprecationWarning
from scrapy.http import Request
from scrapy.pipelines.files import FileException, FilesPipeline
# TODO: from scrapy.pipelines.media import MediaPipeline
from scrapy.settings import Settings
from scrapy.utils.misc import md5sum
from scrapy.utils.python import to_bytes
from scrapy.utils.python import get_func_args, to_bytes
class NoimagesDrop(DropItem):
"""Product with no images exception"""
def __init__(self, *args, **kwargs):
warnings.warn("The NoimagesDrop class is deprecated", category=ScrapyDeprecationWarning, stacklevel=2)
super().__init__(*args, **kwargs)
class ImageException(FileException):
"""General image error exception"""
@ -87,6 +92,8 @@ class ImagesPipeline(FilesPipeline):
resolve('IMAGES_THUMBS'), self.THUMBS
)
self._deprecated_convert_image = None
@classmethod
def from_settings(cls, settings):
s3store = cls.STORE_SCHEMES['s3']
@ -137,15 +144,33 @@ class ImagesPipeline(FilesPipeline):
f"({width}x{height} < "
f"{self.min_width}x{self.min_height})")
image, buf = self.convert_image(orig_image)
if self._deprecated_convert_image is None:
self._deprecated_convert_image = 'response_body' not in get_func_args(self.convert_image)
if self._deprecated_convert_image:
warnings.warn(f'{self.__class__.__name__}.convert_image() method overriden in a deprecated way, '
'overriden method does not accept response_body argument.',
category=ScrapyDeprecationWarning)
if self._deprecated_convert_image:
image, buf = self.convert_image(orig_image)
else:
image, buf = self.convert_image(orig_image, response_body=BytesIO(response.body))
yield path, image, buf
for thumb_id, size in self.thumbs.items():
thumb_path = self.thumb_path(request, thumb_id, response=response, info=info, item=item)
thumb_image, thumb_buf = self.convert_image(image, size)
if self._deprecated_convert_image:
thumb_image, thumb_buf = self.convert_image(image, size)
else:
thumb_image, thumb_buf = self.convert_image(image, size, buf)
yield thumb_path, thumb_image, thumb_buf
def convert_image(self, image, size=None):
def convert_image(self, image, size=None, response_body=None):
if response_body is None:
warnings.warn(f'{self.__class__.__name__}.convert_image() method called in a deprecated way, '
'method called without response_body argument.',
category=ScrapyDeprecationWarning, stacklevel=2)
if image.format == 'PNG' and image.mode == 'RGBA':
background = self._Image.new('RGBA', image.size, (255, 255, 255))
background.paste(image, image)
@ -160,7 +185,16 @@ class ImagesPipeline(FilesPipeline):
if size:
image = image.copy()
image.thumbnail(size, self._Image.ANTIALIAS)
try:
# Image.Resampling.LANCZOS was added in Pillow 9.1.0
# remove this try except block,
# when updating the minimum requirements for Pillow.
resampling_filter = self._Image.Resampling.LANCZOS
except AttributeError:
resampling_filter = self._Image.ANTIALIAS
image.thumbnail(size, resampling_filter)
elif response_body is not None and image.format == 'JPEG':
return image, response_body
buf = BytesIO()
image.save(buf, 'JPEG')

View File

@ -51,11 +51,9 @@ class SettingsAttribute:
self.value = value
self.priority = priority
def __str__(self):
def __repr__(self):
return f"<SettingsAttribute value={self.value!r} priority={self.priority}>"
__repr__ = __str__
class BaseSettings(MutableMapping):
"""
@ -437,30 +435,6 @@ class BaseSettings(MutableMapping):
p.text(pformat(self.copy_to_dict()))
class _DictProxy(MutableMapping):
def __init__(self, settings, priority):
self.o = {}
self.settings = settings
self.priority = priority
def __len__(self):
return len(self.o)
def __getitem__(self, k):
return self.o[k]
def __setitem__(self, k, v):
self.settings.set(k, v, priority=self.priority)
self.o[k] = v
def __delitem__(self, k):
del self.o[k]
def __iter__(self, k, v):
return iter(self.o)
class Settings(BaseSettings):
"""
This object stores Scrapy settings for the configuration of internal

View File

@ -248,7 +248,7 @@ REFERER_ENABLED = True
REFERRER_POLICY = 'scrapy.spidermiddlewares.referer.DefaultReferrerPolicy'
REQUEST_FINGERPRINTER_CLASS = 'scrapy.utils.request.RequestFingerprinter'
REQUEST_FINGERPRINTER_IMPLEMENTATION = 'PREVIOUS_VERSION'
REQUEST_FINGERPRINTER_IMPLEMENTATION = '2.6'
RETRY_ENABLED = True
RETRY_TIMES = 2 # initial response + 2 retries = 3 requests

View File

@ -88,11 +88,9 @@ class Spider(object_ref):
if callable(closed):
return closed(reason)
def __str__(self):
def __repr__(self):
return f"<{type(self).__name__} {self.name!r} at 0x{id(self):0x}>"
__repr__ = __str__
# Top-level imports
from scrapy.spiders.crawl import CrawlSpider, Rule

View File

@ -102,9 +102,9 @@ class CrawlSpider(Spider):
request = self._build_request(rule_index, link)
yield rule.process_request(request, response)
def _callback(self, response):
def _callback(self, response, **cb_kwargs):
rule = self._rules[response.meta['rule']]
return self._parse_response(response, rule.callback, rule.cb_kwargs, rule.follow)
return self._parse_response(response, rule.callback, {**rule.cb_kwargs, **cb_kwargs}, rule.follow)
def _errback(self, failure):
rule = self._rules[failure.request.meta['rule']]

View File

@ -88,4 +88,5 @@ ROBOTSTXT_OBEY = True
#HTTPCACHE_STORAGE = 'scrapy.extensions.httpcache.FilesystemCacheStorage'
# Set settings whose default value is deprecated to a future-proof value
REQUEST_FINGERPRINTER_IMPLEMENTATION = 'VERSION'
REQUEST_FINGERPRINTER_IMPLEMENTATION = '2.7'
TWISTED_REACTOR = 'twisted.internet.asyncioreactor.AsyncioSelectorReactor'

View File

@ -1,27 +1,4 @@
"""Boto/botocore helpers"""
import warnings
from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning
def is_botocore():
""" Returns True if botocore is available, otherwise raises NotConfigured. Never returns False.
Previously, when boto was supported in addition to botocore, this returned False if boto was available
but botocore wasn't.
"""
message = (
'is_botocore() is deprecated and always returns True or raises an Exception, '
'so it cannot be used for checking if boto is available instead of botocore. '
'You can use scrapy.utils.boto.is_botocore_available() to check if botocore '
'is available.'
)
warnings.warn(message, ScrapyDeprecationWarning, stacklevel=2)
try:
import botocore # noqa: F401
return True
except ImportError:
raise NotConfigured('missing botocore library')
def is_botocore_available():

View File

@ -46,14 +46,12 @@ def build_component_list(compdict, custom=None, convert=update_classpath):
raise ValueError(f'Invalid value {value} for component {name}, '
'please provide a real number or None instead')
# BEGIN Backward compatibility for old (base, custom) call signature
if isinstance(custom, (list, tuple)):
_check_components(custom)
return type(custom)(convert(c) for c in custom)
if custom is not None:
compdict.update(custom)
# END Backward compatibility
_validate_values(compdict)
compdict = without_none_values(_map_keys(compdict))
@ -157,6 +155,15 @@ def feed_process_params_from_cli(settings, output: List[str], output_format=None
raise UsageError(
"Please use only one of -o/--output and -O/--overwrite-output"
)
if output_format:
raise UsageError(
"-t/--output-format is a deprecated command line option"
" and does not work in combination with -O/--overwrite-output."
" To specify a format please specify it after a colon at the end of the"
" output URI (i.e. -O <URI>:<FORMAT>)."
" Example working in the tutorial: "
"scrapy crawl quotes -O quotes.json:json"
)
output = overwrite_output
overwrite = True
@ -164,9 +171,13 @@ def feed_process_params_from_cli(settings, output: List[str], output_format=None
if len(output) == 1:
check_valid_format(output_format)
message = (
'The -t command line option is deprecated in favor of '
'specifying the output format within the output URI. See the '
'documentation of the -o and -O options for more information.'
"The -t/--output-format command line option is deprecated in favor of "
"specifying the output format within the output URI using the -o/--output or the"
" -O/--overwrite-output option (i.e. -o/-O <URI>:<FORMAT>). See the documentation"
" of the -o or -O option or the following examples for more information. "
"Examples working in the tutorial: "
"scrapy crawl quotes -o quotes.csv:csv or "
"scrapy crawl quotes -O quotes.json:json"
)
warnings.warn(message, ScrapyDeprecationWarning, stacklevel=2)
return {output[0]: {'format': output_format}}

View File

@ -26,7 +26,7 @@ from twisted.python import failure
from twisted.python.failure import Failure
from scrapy.exceptions import IgnoreRequest
from scrapy.utils.reactor import is_asyncio_reactor_installed
from scrapy.utils.reactor import is_asyncio_reactor_installed, get_asyncio_event_loop_policy
def defer_fail(_failure: Failure) -> Deferred:
@ -269,7 +269,8 @@ def deferred_from_coro(o) -> Any:
return ensureDeferred(o)
else:
# wrapping the coroutine into a Future and then into a Deferred, this requires AsyncioSelectorReactor
return Deferred.fromFuture(asyncio.ensure_future(o))
event_loop = get_asyncio_event_loop_policy().get_event_loop()
return Deferred.fromFuture(asyncio.ensure_future(o, loop=event_loop))
return o
@ -320,7 +321,8 @@ def deferred_to_future(d: Deferred) -> Future:
d = treq.get('https://example.com/additional')
additional_response = await deferred_to_future(d)
"""
return d.asFuture(asyncio.get_event_loop())
policy = get_asyncio_event_loop_policy()
return d.asFuture(policy.get_event_loop())
def maybe_deferred_to_future(d: Deferred) -> Union[Deferred, Future]:

View File

@ -2,6 +2,7 @@
import warnings
import inspect
from typing import List, Tuple
from scrapy.exceptions import ScrapyDeprecationWarning
@ -126,9 +127,7 @@ def _clspath(cls, forced=None):
return f'{cls.__module__}.{cls.__name__}'
DEPRECATION_RULES = [
('scrapy.telnet.', 'scrapy.extensions.telnet.'),
]
DEPRECATION_RULES: List[Tuple[str, str]] = []
def update_classpath(path):

View File

@ -2,17 +2,6 @@ import struct
from gzip import GzipFile
from io import BytesIO
from scrapy.utils.decorators import deprecated
# - GzipFile's read() has issues returning leftover uncompressed data when
# input is corrupted
# - read1(), which fetches data before raising EOFError on next call
# works here
@deprecated('GzipFile.read1')
def read1(gzf, size=-1):
return gzf.read1(size)
def gunzip(data):
"""Gunzip the given data and return as much data as possible.

View File

@ -9,6 +9,7 @@ from collections import deque
from contextlib import contextmanager
from importlib import import_module
from pkgutil import iter_modules
from functools import partial
from w3lib.html import replace_entities
@ -226,7 +227,18 @@ def is_generator_with_return_value(callable):
return value is None or isinstance(value, ast.NameConstant) and value.value is None
if inspect.isgeneratorfunction(callable):
code = re.sub(r"^[\t ]+", "", inspect.getsource(callable))
func = callable
while isinstance(func, partial):
func = func.func
src = inspect.getsource(func)
pattern = re.compile(r"(^[\t ]+)")
code = pattern.sub("", src)
match = pattern.match(src) # finds indentation
if match:
code = re.sub(f"\n{match.group(0)}", "\n", code) # remove indentation
tree = ast.parse(code)
for node in walk_callable(tree):
if isinstance(node, ast.Return) and not returns_none(node):

View File

@ -6,7 +6,7 @@ from pathlib import Path
from scrapy.utils.conf import closest_scrapy_cfg, get_config, init_env
from scrapy.settings import Settings
from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning
from scrapy.exceptions import NotConfigured
ENVVAR = 'SCRAPY_SETTINGS_MODULE'
@ -68,23 +68,16 @@ def get_project_settings():
if settings_module_path:
settings.setmodule(settings_module_path, priority='project')
scrapy_envvars = {k[7:]: v for k, v in os.environ.items() if
k.startswith('SCRAPY_')}
valid_envvars = {
'CHECK',
'PROJECT',
'PYTHON_SHELL',
'SETTINGS_MODULE',
}
setting_envvars = {k for k in scrapy_envvars if k not in valid_envvars}
if setting_envvars:
setting_envvar_list = ', '.join(sorted(setting_envvars))
warnings.warn(
'Use of environment variables prefixed with SCRAPY_ to override '
'settings is deprecated. The following environment variables are '
f'currently defined: {setting_envvar_list}',
ScrapyDeprecationWarning
)
scrapy_envvars = {k[7:]: v for k, v in os.environ.items() if
k.startswith('SCRAPY_') and k.replace('SCRAPY_', '') in valid_envvars}
settings.setdict(scrapy_envvars, priority='project')
return settings

View File

@ -1,20 +1,16 @@
"""
This module contains essential stuff that should've come with Python itself ;)
"""
import errno
import gc
import inspect
import re
import sys
import warnings
import weakref
from functools import partial, wraps
from itertools import chain
from typing import AsyncGenerator, AsyncIterable, Iterable, Union
from scrapy.exceptions import ScrapyDeprecationWarning
from scrapy.utils.asyncgen import as_async_generator
from scrapy.utils.decorators import deprecated
def flatten(x):
@ -112,12 +108,6 @@ def to_bytes(text, encoding=None, errors='strict'):
return text.encode(encoding, errors)
@deprecated('to_unicode')
def to_native_str(text, encoding=None, errors='strict'):
""" Return str representation of ``text``. """
return to_unicode(text, encoding, errors)
def re_rsearch(pattern, text, chunk_size=1024):
"""
This function does a reverse search in a text using a regular expression
@ -263,30 +253,6 @@ def equal_attributes(obj1, obj2, attributes):
return True
class WeakKeyCache:
def __init__(self, default_factory):
warnings.warn("The WeakKeyCache class is deprecated", category=ScrapyDeprecationWarning, stacklevel=2)
self.default_factory = default_factory
self._weakdict = weakref.WeakKeyDictionary()
def __getitem__(self, key):
if key not in self._weakdict:
self._weakdict[key] = self.default_factory(key)
return self._weakdict[key]
@deprecated
def retry_on_eintr(function, *args, **kw):
"""Run a function and retry it while getting EINTR errors"""
while True:
try:
return function(*args, **kw)
except IOError as e:
if e.errno != errno.EINTR:
raise
def without_none_values(iterable):
"""Return a copy of ``iterable`` with all ``None`` entries removed.
@ -337,10 +303,6 @@ class MutableChain(Iterable):
def __next__(self):
return next(self.data)
@deprecated("scrapy.utils.python.MutableChain.__next__")
def next(self):
return self.__next__()
async def _async_chain(*iterables: Union[Iterable, AsyncIterable]) -> AsyncGenerator:
for it in iterables:

View File

@ -51,6 +51,19 @@ class CallLaterOnce:
return self._func(*self._a, **self._kw)
def get_asyncio_event_loop_policy():
policy = asyncio.get_event_loop_policy()
if (
sys.version_info >= (3, 8)
and sys.platform == "win32"
and not isinstance(policy, asyncio.WindowsSelectorEventLoopPolicy)
):
policy = asyncio.WindowsSelectorEventLoopPolicy()
asyncio.set_event_loop_policy(policy)
return policy
def install_reactor(reactor_path, event_loop_path=None):
"""Installs the :mod:`~twisted.internet.reactor` with the specified
import path. Also installs the asyncio event loop with the specified import
@ -58,16 +71,14 @@ def install_reactor(reactor_path, event_loop_path=None):
reactor_class = load_object(reactor_path)
if reactor_class is asyncioreactor.AsyncioSelectorReactor:
with suppress(error.ReactorAlreadyInstalledError):
if sys.version_info >= (3, 8) and sys.platform == "win32":
policy = asyncio.get_event_loop_policy()
if not isinstance(policy, asyncio.WindowsSelectorEventLoopPolicy):
asyncio.set_event_loop_policy(asyncio.WindowsSelectorEventLoopPolicy())
policy = get_asyncio_event_loop_policy()
if event_loop_path is not None:
event_loop_class = load_object(event_loop_path)
event_loop = event_loop_class()
asyncio.set_event_loop(event_loop)
else:
event_loop = asyncio.get_event_loop()
event_loop = policy.get_event_loop()
asyncioreactor.install(eventloop=event_loop)
else:
*module, _ = reactor_path.split(".")
@ -90,6 +101,24 @@ def verify_installed_reactor(reactor_path):
raise Exception(msg)
def verify_installed_asyncio_event_loop(loop_path):
from twisted.internet import reactor
loop_class = load_object(loop_path)
if isinstance(reactor._asyncioEventloop, loop_class):
return
installed = (
f"{reactor._asyncioEventloop.__class__.__module__}"
f".{reactor._asyncioEventloop.__class__.__qualname__}"
)
specified = f"{loop_class.__module__}.{loop_class.__qualname__}"
raise Exception(
"Scrapy found an asyncio Twisted reactor already "
f"installed, and its event loop class ({installed}) does "
"not match the one specified in the ASYNCIO_EVENT_LOOP "
f"setting ({specified})"
)
def is_asyncio_reactor_installed():
from twisted.internet import reactor
return isinstance(reactor, asyncioreactor.AsyncioSelectorReactor)

View File

@ -236,10 +236,10 @@ class RequestFingerprinter:
'REQUEST_FINGERPRINTER_IMPLEMENTATION'
)
else:
implementation = 'PREVIOUS_VERSION'
if implementation == 'PREVIOUS_VERSION':
implementation = '2.6'
if implementation == '2.6':
message = (
'\'PREVIOUS_VERSION\' is a deprecated value for the '
'\'2.6\' is a deprecated value for the '
'\'REQUEST_FINGERPRINTER_IMPLEMENTATION\' setting.\n'
'\n'
'It is also the default value. In other words, it is normal '
@ -254,14 +254,14 @@ class RequestFingerprinter:
)
warnings.warn(message, category=ScrapyDeprecationWarning, stacklevel=2)
self._fingerprint = _request_fingerprint_as_bytes
elif implementation == 'VERSION':
elif implementation == '2.7':
self._fingerprint = fingerprint
else:
raise ValueError(
f'Got an invalid value on setting '
f'\'REQUEST_FINGERPRINTER_IMPLEMENTATION\': '
f'{implementation!r}. Valid values are \'PREVIOUS_VERSION\' (deprecated) '
f'and \'VERSION\'.'
f'{implementation!r}. Valid values are \'2.6\' (deprecated) '
f'and \'2.7\'.'
)
def fingerprint(self, request: Request):

View File

@ -66,7 +66,7 @@ def get_crawler(spidercls=None, settings_dict=None, prevent_warnings=True):
# Set by default settings that prevent deprecation warnings.
settings = {}
if prevent_warnings:
settings['REQUEST_FINGERPRINTER_IMPLEMENTATION'] = 'VERSION'
settings['REQUEST_FINGERPRINTER_IMPLEMENTATION'] = '2.7'
settings.update(settings_dict or {})
runner = CrawlerRunner(settings)
return runner.create_crawler(spidercls or Spider)

View File

@ -34,6 +34,7 @@ def url_has_any_extension(url, extensions):
lowercase_path = parse_url(url).path.lower()
return any(lowercase_path.endswith(ext) for ext in extensions)
def parse_url(url, encoding=None):
"""Return urlparsed url from the given argument (which could be an already
parsed url)

View File

@ -0,0 +1,25 @@
import asyncio
import sys
from twisted.internet import asyncioreactor
if sys.version_info >= (3, 8) and sys.platform == "win32":
asyncio.set_event_loop_policy(asyncio.WindowsSelectorEventLoopPolicy())
asyncioreactor.install(asyncio.get_event_loop())
import scrapy # noqa: E402
from scrapy.crawler import CrawlerProcess # noqa: E402
class NoRequestsSpider(scrapy.Spider):
name = 'no_request'
def start_requests(self):
return []
process = CrawlerProcess(settings={
"TWISTED_REACTOR": "twisted.internet.asyncioreactor.AsyncioSelectorReactor",
"ASYNCIO_EVENT_LOOP": "uvloop.Loop",
})
process.crawl(NoRequestsSpider)
process.start()

View File

@ -0,0 +1,28 @@
import asyncio
import sys
from uvloop import Loop
from twisted.internet import asyncioreactor
if sys.version_info >= (3, 8) and sys.platform == "win32":
asyncio.set_event_loop_policy(asyncio.WindowsSelectorEventLoopPolicy())
asyncio.set_event_loop(Loop())
asyncioreactor.install(asyncio.get_event_loop())
import scrapy # noqa: E402
from scrapy.crawler import CrawlerProcess # noqa: E402
class NoRequestsSpider(scrapy.Spider):
name = 'no_request'
def start_requests(self):
return []
process = CrawlerProcess(settings={
"TWISTED_REACTOR": "twisted.internet.asyncioreactor.AsyncioSelectorReactor",
"ASYNCIO_EVENT_LOOP": "uvloop.Loop",
})
process.crawl(NoRequestsSpider)
process.start()

View File

@ -419,6 +419,17 @@ class CrawlSpiderWithErrback(CrawlSpiderWithParseMethod):
self.logger.info('[errback] status %i', failure.value.response.status)
class CrawlSpiderWithProcessRequestCallbackKeywordArguments(CrawlSpiderWithParseMethod):
name = 'crawl_spider_with_process_request_cb_kwargs'
rules = (
Rule(LinkExtractor(), callback='parse', follow=True, process_request="process_request"),
)
def process_request(self, request, response):
request.cb_kwargs["foo"] = "process_request"
return request
class BytesReceivedCallbackSpider(MetaSpider):
full_response_length = 2**18

View File

@ -31,10 +31,6 @@ class CmdlineTest(unittest.TestCase):
self.assertEqual(self._execute('settings', '--get', 'TEST1', '-s',
'TEST1=override'), 'override')
def test_override_settings_using_envvar(self):
self.env['SCRAPY_TEST1'] = 'override'
self.assertEqual(self._execute('settings', '--get', 'TEST1'), 'override')
def test_profiling(self):
path = Path(tempfile.mkdtemp())
filename = path / 'res.prof'

View File

@ -27,7 +27,15 @@ class ParseCommandTest(ProcessTest, SiteTest, CommandTest):
import scrapy
from scrapy.linkextractors import LinkExtractor
from scrapy.spiders import CrawlSpider, Rule
from scrapy.utils.test import get_from_asyncio_queue
class AsyncDefAsyncioSpider(scrapy.Spider):
name = 'asyncdef{self.spider_name}'
async def parse(self, response):
status = await get_from_asyncio_queue(response.status)
return [scrapy.Item(), dict(foo='bar')]
class MySpider(scrapy.Spider):
name = '{self.spider_name}'
@ -155,6 +163,13 @@ ITEM_PIPELINES = {{'{self.project_name}.pipelines.MyPipeline': 1}}
self.url('/html')])
self.assertIn("INFO: It Works!", _textmode(stderr))
@defer.inlineCallbacks
def test_asyncio_parse_items(self):
status, out, stderr = yield self.execute(
['--spider', 'asyncdef' + self.spider_name, '-c', 'parse', self.url('/html')]
)
self.assertIn("""[{}, {'foo': 'bar'}]""", _textmode(out))
@defer.inlineCallbacks
def test_parse_items(self):
status, out, stderr = yield self.execute(

View File

@ -689,8 +689,15 @@ class MySpider(scrapy.Spider):
])
self.assertIn("Using reactor: twisted.internet.asyncioreactor.AsyncioSelectorReactor", log)
def test_asyncio_enabled_false(self):
def test_asyncio_enabled_default(self):
log = self.get_log(self.debug_log_spider, args=[])
self.assertIn("Using reactor: twisted.internet.asyncioreactor.AsyncioSelectorReactor", log)
def test_asyncio_enabled_false(self):
log = self.get_log(self.debug_log_spider, args=[
'-s', 'TWISTED_REACTOR=twisted.internet.selectreactor.SelectReactor'
])
self.assertIn("Using reactor: twisted.internet.selectreactor.SelectReactor", log)
self.assertNotIn("Using reactor: twisted.internet.asyncioreactor.AsyncioSelectorReactor", log)
@mark.skipif(sys.implementation.name == 'pypy', reason='uvloop does not support pypy properly')

View File

@ -41,6 +41,7 @@ from tests.spiders import (
CrawlSpiderWithAsyncGeneratorCallback,
CrawlSpiderWithErrback,
CrawlSpiderWithParseMethod,
CrawlSpiderWithProcessRequestCallbackKeywordArguments,
DelaySpider,
DuplicateStartRequestsSpider,
FollowAllSpider,
@ -350,7 +351,7 @@ with multiples lines
@defer.inlineCallbacks
def test_crawl_multiple(self):
runner = CrawlerRunner({'REQUEST_FINGERPRINTER_IMPLEMENTATION': 'VERSION'})
runner = CrawlerRunner({'REQUEST_FINGERPRINTER_IMPLEMENTATION': '2.7'})
runner.crawl(SimpleSpider, self.mockserver.url("/status?n=200"), mockserver=self.mockserver)
runner.crawl(SimpleSpider, self.mockserver.url("/status?n=503"), mockserver=self.mockserver)
@ -426,6 +427,16 @@ class CrawlSpiderTestCase(TestCase):
self.assertIn("[errback] status 500", str(log))
self.assertIn("[errback] status 501", str(log))
@defer.inlineCallbacks
def test_crawlspider_process_request_cb_kwargs(self):
crawler = get_crawler(CrawlSpiderWithProcessRequestCallbackKeywordArguments)
with LogCapture() as log:
yield crawler.crawl(mockserver=self.mockserver)
self.assertIn("[parse] status 200 (foo: process_request)", str(log))
self.assertIn("[parse] status 201 (foo: process_request)", str(log))
self.assertIn("[parse] status 202 (foo: bar)", str(log))
@defer.inlineCallbacks
def test_async_def_parse(self):
crawler = get_crawler(AsyncDefSpider)

View File

@ -109,7 +109,7 @@ class CrawlerLoggingTestCase(unittest.TestCase):
'LOG_LEVEL': 'INFO',
'LOG_FILE': str(log_file),
# settings to avoid extra warnings
'REQUEST_FINGERPRINTER_IMPLEMENTATION': 'VERSION',
'REQUEST_FINGERPRINTER_IMPLEMENTATION': '2.7',
'TELNETCONSOLE_ENABLED': telnet.TWISTED_CONCH_AVAILABLE,
}
@ -231,7 +231,7 @@ class NoRequestsSpider(scrapy.Spider):
class CrawlerRunnerHasSpider(unittest.TestCase):
def _runner(self):
return CrawlerRunner({'REQUEST_FINGERPRINTER_IMPLEMENTATION': 'VERSION'})
return CrawlerRunner({'REQUEST_FINGERPRINTER_IMPLEMENTATION': '2.7'})
@defer.inlineCallbacks
def test_crawler_runner_bootstrap_successful(self):
@ -279,14 +279,14 @@ class CrawlerRunnerHasSpider(unittest.TestCase):
if self.reactor_pytest == 'asyncio':
CrawlerRunner(settings={
"TWISTED_REACTOR": "twisted.internet.asyncioreactor.AsyncioSelectorReactor",
"REQUEST_FINGERPRINTER_IMPLEMENTATION": "VERSION",
"REQUEST_FINGERPRINTER_IMPLEMENTATION": "2.7",
})
else:
msg = r"The installed reactor \(.*?\) does not match the requested one \(.*?\)"
with self.assertRaisesRegex(Exception, msg):
runner = CrawlerRunner(settings={
"TWISTED_REACTOR": "twisted.internet.asyncioreactor.AsyncioSelectorReactor",
"REQUEST_FINGERPRINTER_IMPLEMENTATION": "VERSION",
"REQUEST_FINGERPRINTER_IMPLEMENTATION": "2.7",
})
yield runner.crawl(NoRequestsSpider)
@ -452,6 +452,29 @@ class CrawlerProcessSubprocess(ScriptRunnerMixin, unittest.TestCase):
self.assertIn("Using asyncio event loop: uvloop.Loop", log)
self.assertIn("async pipeline opened!", log)
@mark.skipif(sys.implementation.name == 'pypy', reason='uvloop does not support pypy properly')
@mark.skipif(platform.system() == 'Windows', reason='uvloop does not support Windows')
@mark.skipif(twisted_version == Version('twisted', 21, 2, 0), reason='https://twistedmatrix.com/trac/ticket/10106')
def test_asyncio_enabled_reactor_same_loop(self):
log = self.run_script("asyncio_enabled_reactor_same_loop.py")
self.assertIn("Spider closed (finished)", log)
self.assertIn("Using reactor: twisted.internet.asyncioreactor.AsyncioSelectorReactor", log)
self.assertIn("Using asyncio event loop: uvloop.Loop", log)
@mark.skipif(sys.implementation.name == 'pypy', reason='uvloop does not support pypy properly')
@mark.skipif(platform.system() == 'Windows', reason='uvloop does not support Windows')
@mark.skipif(twisted_version == Version('twisted', 21, 2, 0), reason='https://twistedmatrix.com/trac/ticket/10106')
def test_asyncio_enabled_reactor_different_loop(self):
log = self.run_script("asyncio_enabled_reactor_different_loop.py")
self.assertNotIn("Spider closed (finished)", log)
self.assertIn(
(
"does not match the one specified in the ASYNCIO_EVENT_LOOP "
"setting (uvloop.Loop)"
),
log,
)
def test_default_loop_asyncio_deferred_signal(self):
log = self.run_script("asyncio_deferred_signal.py")
self.assertIn("Spider closed (finished)", log)

View File

@ -24,7 +24,7 @@ from scrapy.core.downloader.handlers.http import HTTPDownloadHandler
from scrapy.core.downloader.handlers.http10 import HTTP10DownloadHandler
from scrapy.core.downloader.handlers.http11 import HTTP11DownloadHandler
from scrapy.core.downloader.handlers.s3 import S3DownloadHandler
from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning
from scrapy.exceptions import NotConfigured
from scrapy.http import Headers, HtmlResponse, Request
from scrapy.http.response.text import TextResponse
from scrapy.responsetypes import responsetypes
@ -108,6 +108,7 @@ class LoadTestCase(unittest.TestCase):
class FileTestCase(unittest.TestCase):
def setUp(self):
# add a special char to check that they are handled correctly
self.tmpname = Path(self.mktemp() + '^')
Path(self.tmpname).write_text('0123456789')
handler = create_instance(FileDownloadHandler, None, get_crawler())
@ -756,18 +757,6 @@ class HttpProxyTestCase(unittest.TestCase):
request = Request('http://example.com', meta={'proxy': http_proxy})
return self.download_request(request, Spider('foo')).addCallback(_test)
def test_download_with_proxy_https_noconnect(self):
def _test(response):
self.assertEqual(response.status, 200)
self.assertEqual(response.url, request.url)
self.assertEqual(response.body, b'https://example.com')
http_proxy = f'{self.getURL("")}?noconnect'
request = Request('https://example.com', meta={'proxy': http_proxy})
with self.assertWarnsRegex(ScrapyDeprecationWarning,
r'Using HTTPS proxies in the noconnect mode is deprecated'):
return self.download_request(request, Spider('foo')).addCallback(_test)
def test_download_without_proxy(self):
def _test(response):
self.assertEqual(response.status, 200)

View File

@ -242,21 +242,6 @@ class Https2ProxyTestCase(Http11ProxyTestCase):
def getURL(self, path):
return f"{self.scheme}://{self.host}:{self.portno}/{path}"
def test_download_with_proxy_https_noconnect(self):
def _test(response):
self.assertEqual(response.status, 200)
self.assertEqual(response.url, request.url)
self.assertEqual(response.body, b'/')
http_proxy = f"{self.getURL('')}?noconnect"
request = Request('https://example.com', meta={'proxy': http_proxy})
with self.assertWarnsRegex(
Warning,
r'Using HTTPS proxies in the noconnect mode is not supported by the '
r'downloader handler.'
):
return self.download_request(request, Spider('foo')).addCallback(_test)
@defer.inlineCallbacks
def test_download_with_proxy_https_timeout(self):
with self.assertRaises(NotImplementedError):

View File

@ -400,6 +400,9 @@ class TestHttpProxyMiddleware(TestCase):
self.assertNotIn(b'Proxy-Authorization', request.headers)
def test_proxy_authentication_header_proxy_without_credentials(self):
"""As long as the proxy URL in request metadata remains the same, the
Proxy-Authorization header is used and kept, and may even be
changed."""
middleware = HttpProxyMiddleware()
request = Request(
'https://example.com',
@ -408,7 +411,16 @@ class TestHttpProxyMiddleware(TestCase):
)
assert middleware.process_request(request, spider) is None
self.assertEqual(request.meta['proxy'], 'https://example.com')
self.assertNotIn(b'Proxy-Authorization', request.headers)
self.assertEqual(request.headers['Proxy-Authorization'], b'Basic foo')
assert middleware.process_request(request, spider) is None
self.assertEqual(request.meta['proxy'], 'https://example.com')
self.assertEqual(request.headers['Proxy-Authorization'], b'Basic foo')
request.headers['Proxy-Authorization'] = b'Basic bar'
assert middleware.process_request(request, spider) is None
self.assertEqual(request.meta['proxy'], 'https://example.com')
self.assertEqual(request.headers['Proxy-Authorization'], b'Basic bar')
def test_proxy_authentication_header_proxy_with_same_credentials(self):
middleware = HttpProxyMiddleware()

View File

@ -51,7 +51,7 @@ class RFPDupeFilterTest(unittest.TestCase):
def test_df_from_crawler_scheduler(self):
settings = {'DUPEFILTER_DEBUG': True,
'DUPEFILTER_CLASS': FromCrawlerRFPDupeFilter,
'REQUEST_FINGERPRINTER_IMPLEMENTATION': 'VERSION'}
'REQUEST_FINGERPRINTER_IMPLEMENTATION': '2.7'}
crawler = get_crawler(settings_dict=settings)
scheduler = Scheduler.from_crawler(crawler)
self.assertTrue(scheduler.df.debug)
@ -60,7 +60,7 @@ class RFPDupeFilterTest(unittest.TestCase):
def test_df_from_settings_scheduler(self):
settings = {'DUPEFILTER_DEBUG': True,
'DUPEFILTER_CLASS': FromSettingsRFPDupeFilter,
'REQUEST_FINGERPRINTER_IMPLEMENTATION': 'VERSION'}
'REQUEST_FINGERPRINTER_IMPLEMENTATION': '2.7'}
crawler = get_crawler(settings_dict=settings)
scheduler = Scheduler.from_crawler(crawler)
self.assertTrue(scheduler.df.debug)
@ -68,7 +68,7 @@ class RFPDupeFilterTest(unittest.TestCase):
def test_df_direct_scheduler(self):
settings = {'DUPEFILTER_CLASS': DirectDupeFilter,
'REQUEST_FINGERPRINTER_IMPLEMENTATION': 'VERSION'}
'REQUEST_FINGERPRINTER_IMPLEMENTATION': '2.7'}
crawler = get_crawler(settings_dict=settings)
scheduler = Scheduler.from_crawler(crawler)
self.assertEqual(scheduler.df.method, 'n/a')
@ -172,7 +172,7 @@ class RFPDupeFilterTest(unittest.TestCase):
with LogCapture() as log:
settings = {'DUPEFILTER_DEBUG': False,
'DUPEFILTER_CLASS': FromCrawlerRFPDupeFilter,
'REQUEST_FINGERPRINTER_IMPLEMENTATION': 'VERSION'}
'REQUEST_FINGERPRINTER_IMPLEMENTATION': '2.7'}
crawler = get_crawler(SimpleSpider, settings_dict=settings)
spider = SimpleSpider.from_crawler(crawler)
dupefilter = _get_dupefilter(crawler=crawler)
@ -199,7 +199,7 @@ class RFPDupeFilterTest(unittest.TestCase):
with LogCapture() as log:
settings = {'DUPEFILTER_DEBUG': True,
'DUPEFILTER_CLASS': FromCrawlerRFPDupeFilter,
'REQUEST_FINGERPRINTER_IMPLEMENTATION': 'VERSION'}
'REQUEST_FINGERPRINTER_IMPLEMENTATION': '2.7'}
crawler = get_crawler(SimpleSpider, settings_dict=settings)
spider = SimpleSpider.from_crawler(crawler)
dupefilter = _get_dupefilter(crawler=crawler)
@ -233,7 +233,7 @@ class RFPDupeFilterTest(unittest.TestCase):
def test_log_debug_default_dupefilter(self):
with LogCapture() as log:
settings = {'DUPEFILTER_DEBUG': True,
'REQUEST_FINGERPRINTER_IMPLEMENTATION': 'VERSION'}
'REQUEST_FINGERPRINTER_IMPLEMENTATION': '2.7'}
crawler = get_crawler(SimpleSpider, settings_dict=settings)
spider = SimpleSpider.from_crawler(crawler)
dupefilter = _get_dupefilter(crawler=crawler)

View File

@ -1,12 +1,9 @@
import pickle
import re
import unittest
from warnings import catch_warnings
from scrapy.exceptions import ScrapyDeprecationWarning
from scrapy.http import HtmlResponse, XmlResponse
from scrapy.link import Link
from scrapy.linkextractors import FilteringLinkExtractor
from scrapy.linkextractors.lxmlhtml import LxmlLinkExtractor
from tests import get_testdata
@ -517,32 +514,3 @@ class LxmlLinkExtractorTestCase(Base.LinkExtractorTestCase):
def test_restrict_xpaths_with_html_entities(self):
super().test_restrict_xpaths_with_html_entities()
def test_filteringlinkextractor_deprecation_warning(self):
"""Make sure the FilteringLinkExtractor deprecation warning is not
issued for LxmlLinkExtractor"""
with catch_warnings(record=True) as warnings:
LxmlLinkExtractor()
self.assertEqual(len(warnings), 0)
class SubclassedLxmlLinkExtractor(LxmlLinkExtractor):
pass
SubclassedLxmlLinkExtractor()
self.assertEqual(len(warnings), 0)
class FilteringLinkExtractorTest(unittest.TestCase):
def test_deprecation_warning(self):
args = [None] * 10
with catch_warnings(record=True) as warnings:
FilteringLinkExtractor(*args)
self.assertEqual(len(warnings), 1)
self.assertEqual(warnings[0].category, ScrapyDeprecationWarning)
with catch_warnings(record=True) as warnings:
class SubclassedFilteringLinkExtractor(FilteringLinkExtractor):
pass
SubclassedFilteringLinkExtractor(*args)
self.assertEqual(len(warnings), 1)
self.assertEqual(warnings[0].category, ScrapyDeprecationWarning)

View File

@ -295,7 +295,7 @@ class SelectortemLoaderTest(unittest.TestCase):
l.add_css('name', 'div::text')
self.assertEqual(l.get_output_value('name'), ['Marta'])
def test_init_method_with_base_response(self):
"""Selector should be None after initialization"""
response = Response("https://scrapy.org")

View File

@ -64,7 +64,7 @@ class FileDownloadCrawlTestCase(TestCase):
self.tmpmediastore = Path(self.mktemp())
self.tmpmediastore.mkdir()
self.settings = {
'REQUEST_FINGERPRINTER_IMPLEMENTATION': 'VERSION',
'REQUEST_FINGERPRINTER_IMPLEMENTATION': '2.7',
'ITEM_PIPELINES': {self.pipeline_class: 1},
self.store_setting_key: str(self.tmpmediastore),
}

View File

@ -1,17 +1,20 @@
import dataclasses
import hashlib
import io
import random
import warnings
from shutil import rmtree
from tempfile import mkdtemp
import dataclasses
from unittest.mock import patch
import attr
from itemadapter import ItemAdapter
from twisted.trial import unittest
from scrapy.exceptions import ScrapyDeprecationWarning
from scrapy.http import Request, Response
from scrapy.item import Field, Item
from scrapy.pipelines.images import ImagesPipeline
from scrapy.pipelines.images import ImageException, ImagesPipeline, NoimagesDrop
from scrapy.settings import Settings
from scrapy.utils.python import to_bytes
@ -102,32 +105,143 @@ class ImagesPipelineTestCase(unittest.TestCase):
request = Request("http://example.com")
self.assertEqual(thumb_path(request, 'small', item=item), 'thumb/small/path-to-store-file')
def test_convert_image(self):
def test_get_images_exception(self):
self.pipeline.min_width = 100
self.pipeline.min_height = 100
_, buf1 = _create_image('JPEG', 'RGB', (50, 50), (0, 0, 0))
_, buf2 = _create_image('JPEG', 'RGB', (150, 50), (0, 0, 0))
_, buf3 = _create_image('JPEG', 'RGB', (50, 150), (0, 0, 0))
resp1 = Response(url="https://dev.mydeco.com/mydeco.gif", body=buf1.getvalue())
resp2 = Response(url="https://dev.mydeco.com/mydeco.gif", body=buf2.getvalue())
resp3 = Response(url="https://dev.mydeco.com/mydeco.gif", body=buf3.getvalue())
req = Request(url="https://dev.mydeco.com/mydeco.gif")
with self.assertRaises(ImageException):
next(self.pipeline.get_images(response=resp1, request=req, info=object()))
with self.assertRaises(ImageException):
next(self.pipeline.get_images(response=resp2, request=req, info=object()))
with self.assertRaises(ImageException):
next(self.pipeline.get_images(response=resp3, request=req, info=object()))
def test_get_images_new(self):
self.pipeline.min_width = 0
self.pipeline.min_height = 0
self.pipeline.thumbs = {'small': (20, 20)}
orig_im, buf = _create_image('JPEG', 'RGB', (50, 50), (0, 0, 0))
orig_thumb, orig_thumb_buf = _create_image('JPEG', 'RGB', (20, 20), (0, 0, 0))
resp = Response(url="https://dev.mydeco.com/mydeco.gif", body=buf.getvalue())
req = Request(url="https://dev.mydeco.com/mydeco.gif")
get_images_gen = self.pipeline.get_images(response=resp, request=req, info=object())
path, new_im, new_buf = next(get_images_gen)
self.assertEqual(path, 'full/3fd165099d8e71b8a48b2683946e64dbfad8b52d.jpg')
self.assertEqual(orig_im, new_im)
self.assertEqual(buf.getvalue(), new_buf.getvalue())
thumb_path, thumb_img, thumb_buf = next(get_images_gen)
self.assertEqual(thumb_path, 'thumbs/small/3fd165099d8e71b8a48b2683946e64dbfad8b52d.jpg')
self.assertEqual(thumb_img, thumb_img)
self.assertEqual(orig_thumb_buf.getvalue(), thumb_buf.getvalue())
def test_get_images_old(self):
self.pipeline.thumbs = {'small': (20, 20)}
orig_im, buf = _create_image('JPEG', 'RGB', (50, 50), (0, 0, 0))
resp = Response(url="https://dev.mydeco.com/mydeco.gif", body=buf.getvalue())
req = Request(url="https://dev.mydeco.com/mydeco.gif")
def overridden_convert_image(image, size=None):
im, buf = _create_image('JPEG', 'RGB', (50, 50), (0, 0, 0))
return im, buf
with patch.object(self.pipeline, 'convert_image', overridden_convert_image):
with warnings.catch_warnings(record=True) as w:
warnings.simplefilter('always')
get_images_gen = self.pipeline.get_images(response=resp, request=req, info=object())
path, new_im, new_buf = next(get_images_gen)
self.assertEqual(path, 'full/3fd165099d8e71b8a48b2683946e64dbfad8b52d.jpg')
self.assertEqual(orig_im.mode, new_im.mode)
self.assertEqual(orig_im.getcolors(), new_im.getcolors())
self.assertEqual(buf.getvalue(), new_buf.getvalue())
thumb_path, thumb_img, thumb_buf = next(get_images_gen)
self.assertEqual(thumb_path, 'thumbs/small/3fd165099d8e71b8a48b2683946e64dbfad8b52d.jpg')
self.assertEqual(orig_im.mode, thumb_img.mode)
self.assertEqual(orig_im.getcolors(), thumb_img.getcolors())
self.assertEqual(buf.getvalue(), thumb_buf.getvalue())
expected_warning_msg = ('.convert_image() method overriden in a deprecated way, '
'overriden method does not accept response_body argument.')
self.assertEqual(len([warning for warning in w if expected_warning_msg in str(warning.message)]), 1)
def test_convert_image_old(self):
# tests for old API
with warnings.catch_warnings(record=True) as w:
warnings.simplefilter('always')
SIZE = (100, 100)
# straigh forward case: RGB and JPEG
COLOUR = (0, 127, 255)
im, _ = _create_image('JPEG', 'RGB', SIZE, COLOUR)
converted, _ = self.pipeline.convert_image(im)
self.assertEqual(converted.mode, 'RGB')
self.assertEqual(converted.getcolors(), [(10000, COLOUR)])
# check that thumbnail keep image ratio
thumbnail, _ = self.pipeline.convert_image(converted, size=(10, 25))
self.assertEqual(thumbnail.mode, 'RGB')
self.assertEqual(thumbnail.size, (10, 10))
# transparency case: RGBA and PNG
COLOUR = (0, 127, 255, 50)
im, _ = _create_image('PNG', 'RGBA', SIZE, COLOUR)
converted, _ = self.pipeline.convert_image(im)
self.assertEqual(converted.mode, 'RGB')
self.assertEqual(converted.getcolors(), [(10000, (205, 230, 255))])
# transparency case with palette: P and PNG
COLOUR = (0, 127, 255, 50)
im, _ = _create_image('PNG', 'RGBA', SIZE, COLOUR)
im = im.convert('P')
converted, _ = self.pipeline.convert_image(im)
self.assertEqual(converted.mode, 'RGB')
self.assertEqual(converted.getcolors(), [(10000, (205, 230, 255))])
# ensure that we recieved deprecation warnings
expected_warning_msg = '.convert_image() method called in a deprecated way'
self.assertTrue(len([warning for warning in w if expected_warning_msg in str(warning.message)]) == 4)
def test_convert_image_new(self):
# tests for new API
SIZE = (100, 100)
# straigh forward case: RGB and JPEG
COLOUR = (0, 127, 255)
im = _create_image('JPEG', 'RGB', SIZE, COLOUR)
converted, _ = self.pipeline.convert_image(im)
im, buf = _create_image('JPEG', 'RGB', SIZE, COLOUR)
converted, converted_buf = self.pipeline.convert_image(im, response_body=buf)
self.assertEqual(converted.mode, 'RGB')
self.assertEqual(converted.getcolors(), [(10000, COLOUR)])
# check that we don't convert JPEGs again
self.assertEqual(converted_buf, buf)
# check that thumbnail keep image ratio
thumbnail, _ = self.pipeline.convert_image(converted, size=(10, 25))
thumbnail, _ = self.pipeline.convert_image(converted, size=(10, 25), response_body=converted_buf)
self.assertEqual(thumbnail.mode, 'RGB')
self.assertEqual(thumbnail.size, (10, 10))
# transparency case: RGBA and PNG
COLOUR = (0, 127, 255, 50)
im = _create_image('PNG', 'RGBA', SIZE, COLOUR)
converted, _ = self.pipeline.convert_image(im)
im, buf = _create_image('PNG', 'RGBA', SIZE, COLOUR)
converted, _ = self.pipeline.convert_image(im, response_body=buf)
self.assertEqual(converted.mode, 'RGB')
self.assertEqual(converted.getcolors(), [(10000, (205, 230, 255))])
# transparency case with palette: P and PNG
COLOUR = (0, 127, 255, 50)
im = _create_image('PNG', 'RGBA', SIZE, COLOUR)
im, buf = _create_image('PNG', 'RGBA', SIZE, COLOUR)
im = im.convert('P')
converted, _ = self.pipeline.convert_image(im)
converted, _ = self.pipeline.convert_image(im, response_body=buf)
self.assertEqual(converted.mode, 'RGB')
self.assertEqual(converted.getcolors(), [(10000, (205, 230, 255))])
@ -416,11 +530,27 @@ class ImagesPipelineTestCaseCustomSettings(unittest.TestCase):
expected_value)
class NoimagesDropTestCase(unittest.TestCase):
def test_deprecation_warning(self):
arg = str()
with warnings.catch_warnings(record=True) as w:
NoimagesDrop(arg)
self.assertEqual(len(w), 1)
self.assertEqual(w[0].category, ScrapyDeprecationWarning)
with warnings.catch_warnings(record=True) as w:
class SubclassedNoimagesDrop(NoimagesDrop):
pass
SubclassedNoimagesDrop(arg)
self.assertEqual(len(w), 1)
self.assertEqual(w[0].category, ScrapyDeprecationWarning)
def _create_image(format, *a, **kw):
buf = io.BytesIO()
Image.new(*a, **kw).save(buf, format)
buf.seek(0)
return Image.open(buf)
return Image.open(buf), buf
if __name__ == "__main__":

View File

@ -52,7 +52,7 @@ class MockCrawler(Crawler):
SCHEDULER_PRIORITY_QUEUE=priority_queue_cls,
JOBDIR=jobdir,
DUPEFILTER_CLASS='scrapy.dupefilters.BaseDupeFilter',
REQUEST_FINGERPRINTER_IMPLEMENTATION='VERSION',
REQUEST_FINGERPRINTER_IMPLEMENTATION='2.7',
)
super().__init__(Spider, settings)
self.engine = MockEngine(downloader=MockDownloader())

View File

@ -98,7 +98,7 @@ class SpiderLoaderTest(unittest.TestCase):
module = 'tests.test_spiderloader.test_spiders.spider1'
runner = CrawlerRunner({
'SPIDER_MODULES': [module],
'REQUEST_FINGERPRINTER_IMPLEMENTATION': 'VERSION',
'REQUEST_FINGERPRINTER_IMPLEMENTATION': '2.7',
})
self.assertRaisesRegex(KeyError, 'Spider not found',

View File

@ -1,3 +1,4 @@
import warnings
from unittest import TestCase
from pytest import mark
@ -13,5 +14,6 @@ class AsyncioTest(TestCase):
self.assertEqual(is_asyncio_reactor_installed(), self.reactor_pytest == 'asyncio')
def test_install_asyncio_reactor(self):
# this should do nothing
install_reactor("twisted.internet.asyncioreactor.AsyncioSelectorReactor")
with warnings.catch_warnings(record=True) as w:
install_reactor("twisted.internet.asyncioreactor.AsyncioSelectorReactor")
self.assertEqual(len(w), 0)

View File

@ -1,5 +1,6 @@
import unittest
import warnings
from functools import partial
from unittest import mock
from scrapy.utils.misc import is_generator_with_return_value, warn_on_generator_with_return_value
@ -165,9 +166,99 @@ https://example.org
warn_on_generator_with_return_value(None, l2)
self.assertEqual(len(w), 0)
def test_generators_return_none_with_decorator(self):
def decorator(func):
def inner_func():
func()
return inner_func
@decorator
def f3():
yield 1
return None
@decorator
def g3():
yield 1
return
@decorator
def h3():
yield 1
@decorator
def i3():
yield 1
yield from generator_that_returns_stuff()
@decorator
def j3():
yield 1
def helper():
return 0
yield helper()
@decorator
def k3():
"""
docstring
"""
url = """
https://example.org
"""
yield url
return
@decorator
def l3():
return
assert not is_generator_with_return_value(top_level_return_none)
assert not is_generator_with_return_value(f3)
assert not is_generator_with_return_value(g3)
assert not is_generator_with_return_value(h3)
assert not is_generator_with_return_value(i3)
assert not is_generator_with_return_value(j3) # not recursive
assert not is_generator_with_return_value(k3) # not recursive
assert not is_generator_with_return_value(l3)
with warnings.catch_warnings(record=True) as w:
warn_on_generator_with_return_value(None, top_level_return_none)
self.assertEqual(len(w), 0)
with warnings.catch_warnings(record=True) as w:
warn_on_generator_with_return_value(None, f3)
self.assertEqual(len(w), 0)
with warnings.catch_warnings(record=True) as w:
warn_on_generator_with_return_value(None, g3)
self.assertEqual(len(w), 0)
with warnings.catch_warnings(record=True) as w:
warn_on_generator_with_return_value(None, h3)
self.assertEqual(len(w), 0)
with warnings.catch_warnings(record=True) as w:
warn_on_generator_with_return_value(None, i3)
self.assertEqual(len(w), 0)
with warnings.catch_warnings(record=True) as w:
warn_on_generator_with_return_value(None, j3)
self.assertEqual(len(w), 0)
with warnings.catch_warnings(record=True) as w:
warn_on_generator_with_return_value(None, k3)
self.assertEqual(len(w), 0)
with warnings.catch_warnings(record=True) as w:
warn_on_generator_with_return_value(None, l3)
self.assertEqual(len(w), 0)
@mock.patch("scrapy.utils.misc.is_generator_with_return_value", new=_indentation_error)
def test_indentation_error(self):
with warnings.catch_warnings(record=True) as w:
warn_on_generator_with_return_value(None, top_level_return_none)
self.assertEqual(len(w), 1)
self.assertIn('Unable to determine', str(w[0].message))
def test_partial(self):
def cb(arg1, arg2):
yield {}
partial_cb = partial(cb, arg1=42)
assert not is_generator_with_return_value(partial_cb)

View File

@ -6,9 +6,6 @@ import contextlib
import warnings
from pathlib import Path
from pytest import warns
from scrapy.exceptions import ScrapyDeprecationWarning
from scrapy.utils.project import data_path, get_project_settings
@ -79,10 +76,10 @@ class GetProjectSettingsTestCase(unittest.TestCase):
envvars = {
'SCRAPY_FOO': 'bar',
}
with warns(ScrapyDeprecationWarning, match=': FOO') as record:
with set_env(**envvars):
get_project_settings()
assert len(record) == 1
with set_env(**envvars):
settings = get_project_settings()
assert settings.get("SCRAPY_FOO") is None
def test_valid_and_invalid_envvars(self):
value = 'tests.test_cmdline.settings'
@ -90,8 +87,7 @@ class GetProjectSettingsTestCase(unittest.TestCase):
'SCRAPY_FOO': 'bar',
'SCRAPY_SETTINGS_MODULE': value,
}
with warns(ScrapyDeprecationWarning, match=': FOO') as record:
with set_env(**envvars):
settings = get_project_settings()
assert len(record) == 1
with set_env(**envvars):
settings = get_project_settings()
assert settings.get('SETTINGS_MODULE') == value
assert settings.get('SCRAPY_FOO') is None

View File

@ -1,18 +1,14 @@
import functools
import gc
import operator
import platform
from itertools import count
from warnings import catch_warnings, filterwarnings
from twisted.trial import unittest
from scrapy.exceptions import ScrapyDeprecationWarning
from scrapy.utils.asyncgen import as_async_generator, collect_asyncgen
from scrapy.utils.defer import deferred_f_from_coro_f, aiter_errback
from scrapy.utils.python import (
memoizemethod_noargs, binary_is_text, equal_attributes,
WeakKeyCache, get_func_args, to_bytes, to_unicode,
get_func_args, to_bytes, to_unicode,
without_none_values, MutableChain, MutableAsyncChain)
@ -27,12 +23,7 @@ class MutableChainTest(unittest.TestCase):
m.extend([9, 10], (11, 12))
self.assertEqual(next(m), 0)
self.assertEqual(m.__next__(), 1)
with catch_warnings(record=True) as warnings:
self.assertEqual(m.next(), 2)
self.assertEqual(len(warnings), 1)
self.assertIn('scrapy.utils.python.MutableChain.__next__',
str(warnings[0].message))
self.assertEqual(list(m), list(range(3, 13)))
self.assertEqual(list(m), list(range(2, 13)))
class MutableAsyncChainTest(unittest.TestCase):
@ -209,27 +200,6 @@ class UtilsPythonTestCase(unittest.TestCase):
a.meta['z'] = 2
self.assertFalse(equal_attributes(a, b, [compare_z, 'x']))
def test_weakkeycache(self):
class _Weakme:
pass
_values = count()
with catch_warnings():
filterwarnings("ignore", category=ScrapyDeprecationWarning)
wk = WeakKeyCache(lambda k: next(_values))
k = _Weakme()
v = wk[k]
self.assertEqual(v, wk[k])
self.assertNotEqual(v, wk[_Weakme()])
self.assertEqual(v, wk[k])
del k
for _ in range(100):
if wk._weakdict:
gc.collect()
self.assertFalse(len(wk._weakdict))
def test_get_func_args(self):
def f1(a, b, c):
pass

View File

@ -505,7 +505,7 @@ class RequestFingerprinterTestCase(unittest.TestCase):
def test_deprecated_implementation(self):
settings = {
'REQUEST_FINGERPRINTER_IMPLEMENTATION': 'PREVIOUS_VERSION',
'REQUEST_FINGERPRINTER_IMPLEMENTATION': '2.6',
}
with warnings.catch_warnings(record=True) as logged_warnings:
crawler = get_crawler(settings_dict=settings)
@ -518,7 +518,7 @@ class RequestFingerprinterTestCase(unittest.TestCase):
def test_recommended_implementation(self):
settings = {
'REQUEST_FINGERPRINTER_IMPLEMENTATION': 'VERSION',
'REQUEST_FINGERPRINTER_IMPLEMENTATION': '2.7',
}
with warnings.catch_warnings(record=True) as logged_warnings:
crawler = get_crawler(settings_dict=settings)