mirror of https://github.com/scrapy/scrapy.git
Merge branch 'master' into pathlib
This commit is contained in:
commit
d19a216e10
|
|
@ -1,5 +1,5 @@
|
|||
[bumpversion]
|
||||
current_version = 2.6.2
|
||||
current_version = 2.7.1
|
||||
commit = True
|
||||
tag = True
|
||||
tag_name = {new_version}
|
||||
|
|
|
|||
17
.flake8
17
.flake8
|
|
@ -6,16 +6,17 @@ ignore = W503
|
|||
exclude =
|
||||
docs/conf.py
|
||||
|
||||
per-file-ignores =
|
||||
# Exclude files that are meant to provide top-level imports
|
||||
# E402: Module level import not at top of file
|
||||
# F401: Module imported but unused
|
||||
scrapy/__init__.py E402
|
||||
scrapy/core/downloader/handlers/http.py F401
|
||||
scrapy/http/__init__.py F401
|
||||
scrapy/linkextractors/__init__.py E402 F401
|
||||
scrapy/selector/__init__.py F401
|
||||
scrapy/spiders/__init__.py E402 F401
|
||||
scrapy/__init__.py:E402
|
||||
scrapy/core/downloader/handlers/http.py:F401
|
||||
scrapy/http/__init__.py:F401
|
||||
scrapy/linkextractors/__init__.py:E402,F401
|
||||
scrapy/selector/__init__.py:F401
|
||||
scrapy/spiders/__init__.py:E402,F401
|
||||
|
||||
# Issues pending a review:
|
||||
scrapy/utils/url.py F403 F405
|
||||
tests/test_loader.py E741
|
||||
scrapy/utils/url.py:F403,F405
|
||||
tests/test_loader.py:E741
|
||||
|
|
|
|||
|
|
@ -3,15 +3,15 @@ on: [push, pull_request]
|
|||
|
||||
jobs:
|
||||
checks:
|
||||
runs-on: ubuntu-18.04
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- python-version: "3.10"
|
||||
- python-version: "3.11"
|
||||
env:
|
||||
TOXENV: security
|
||||
- python-version: "3.10"
|
||||
- python-version: "3.11"
|
||||
env:
|
||||
TOXENV: flake8
|
||||
# Pylint requires installing reppy, which does not support Python 3.9
|
||||
|
|
@ -22,10 +22,10 @@ jobs:
|
|||
- python-version: 3.7
|
||||
env:
|
||||
TOXENV: typing
|
||||
- python-version: "3.10" # Keep in sync with .readthedocs.yml
|
||||
- python-version: "3.11" # Keep in sync with .readthedocs.yml
|
||||
env:
|
||||
TOXENV: docs
|
||||
- python-version: "3.10"
|
||||
- python-version: "3.11"
|
||||
env:
|
||||
TOXENV: twinecheck
|
||||
|
||||
|
|
|
|||
|
|
@ -3,7 +3,7 @@ on: [push]
|
|||
|
||||
jobs:
|
||||
publish:
|
||||
runs-on: ubuntu-18.04
|
||||
runs-on: ubuntu-latest
|
||||
if: startsWith(github.event.ref, 'refs/tags/')
|
||||
|
||||
steps:
|
||||
|
|
@ -12,7 +12,7 @@ jobs:
|
|||
- name: Set up Python
|
||||
uses: actions/setup-python@v4
|
||||
with:
|
||||
python-version: "3.10"
|
||||
python-version: "3.11"
|
||||
|
||||
- name: Check Tag
|
||||
id: check-release-tag
|
||||
|
|
|
|||
|
|
@ -3,11 +3,11 @@ on: [push, pull_request]
|
|||
|
||||
jobs:
|
||||
tests:
|
||||
runs-on: macos-10.15
|
||||
runs-on: macos-11
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
python-version: ["3.7", "3.8", "3.9", "3.10"]
|
||||
python-version: ["3.7", "3.8", "3.9", "3.10", "3.11"]
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
|
|
|
|||
|
|
@ -3,7 +3,7 @@ on: [push, pull_request]
|
|||
|
||||
jobs:
|
||||
tests:
|
||||
runs-on: ubuntu-18.04
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
|
|
@ -17,13 +17,10 @@ jobs:
|
|||
- python-version: "3.10"
|
||||
env:
|
||||
TOXENV: py
|
||||
- python-version: "3.10"
|
||||
env:
|
||||
TOXENV: asyncio
|
||||
- python-version: "3.11.0-rc.2"
|
||||
- python-version: "3.11"
|
||||
env:
|
||||
TOXENV: py
|
||||
- python-version: "3.11.0-rc.2"
|
||||
- python-version: "3.11"
|
||||
env:
|
||||
TOXENV: asyncio
|
||||
- python-version: pypy3.9
|
||||
|
|
@ -57,7 +54,7 @@ jobs:
|
|||
python-version: ${{ matrix.python-version }}
|
||||
|
||||
- name: Install system libraries
|
||||
if: matrix.python-version == 'pypy3.9' || contains(matrix.env.TOXENV, 'pinned') || matrix.python-version == '3.11.0-rc.2'
|
||||
if: matrix.python-version == 'pypy3.9' || contains(matrix.env.TOXENV, 'pinned')
|
||||
run: |
|
||||
sudo apt-get update
|
||||
sudo apt-get install libxml2-dev libxslt-dev
|
||||
|
|
|
|||
|
|
@ -23,6 +23,13 @@ jobs:
|
|||
- python-version: "3.10"
|
||||
env:
|
||||
TOXENV: asyncio
|
||||
# no binary package for lxml for 3.11 yet
|
||||
# - python-version: "3.11"
|
||||
# env:
|
||||
# TOXENV: py
|
||||
# - python-version: "3.11"
|
||||
# env:
|
||||
# TOXENV: asyncio
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
|
|
|
|||
|
|
@ -9,7 +9,7 @@ build:
|
|||
tools:
|
||||
# For available versions, see:
|
||||
# https://docs.readthedocs.io/en/stable/config-file/v2.html#build-tools-python
|
||||
python: "3.10" # Keep in sync with .github/workflows/checks.yml
|
||||
python: "3.11" # Keep in sync with .github/workflows/checks.yml
|
||||
|
||||
python:
|
||||
install:
|
||||
|
|
|
|||
|
|
@ -1,77 +1,133 @@
|
|||
|
||||
# Contributor Covenant Code of Conduct
|
||||
|
||||
## Our Pledge
|
||||
|
||||
In the interest of fostering an open and welcoming environment, we as
|
||||
contributors and maintainers pledge to make participation in our project and
|
||||
our community a harassment-free experience for everyone, regardless of age, body
|
||||
size, disability, ethnicity, gender identity and expression, level of experience,
|
||||
nationality, personal appearance, race, religion, or sexual identity and
|
||||
orientation.
|
||||
We as members, contributors, and leaders pledge to make participation in our
|
||||
community a harassment-free experience for everyone, regardless of age, body
|
||||
size, visible or invisible disability, ethnicity, sex characteristics, gender
|
||||
identity and expression, level of experience, education, socio-economic status,
|
||||
nationality, personal appearance, race, caste, color, religion, or sexual
|
||||
identity and orientation.
|
||||
|
||||
We pledge to act and interact in ways that contribute to an open, welcoming,
|
||||
diverse, inclusive, and healthy community.
|
||||
|
||||
## Our Standards
|
||||
|
||||
Examples of behavior that contributes to creating a positive environment
|
||||
include:
|
||||
Examples of behavior that contributes to a positive environment for our
|
||||
community include:
|
||||
|
||||
* Using welcoming and inclusive language
|
||||
* Being respectful of differing viewpoints and experiences
|
||||
* Gracefully accepting constructive criticism
|
||||
* Focusing on what is best for the community
|
||||
* Showing empathy towards other community members
|
||||
* Demonstrating empathy and kindness toward other people
|
||||
* Being respectful of differing opinions, viewpoints, and experiences
|
||||
* Giving and gracefully accepting constructive feedback
|
||||
* Accepting responsibility and apologizing to those affected by our mistakes,
|
||||
and learning from the experience
|
||||
* Focusing on what is best not just for us as individuals, but for the overall
|
||||
community
|
||||
|
||||
Examples of unacceptable behavior by participants include:
|
||||
Examples of unacceptable behavior include:
|
||||
|
||||
* The use of sexualized language or imagery and unwelcome sexual attention or
|
||||
advances
|
||||
* Trolling, insulting/derogatory comments, and personal or political attacks
|
||||
* The use of sexualized language or imagery, and sexual attention or advances of
|
||||
any kind
|
||||
* Trolling, insulting or derogatory comments, and personal or political attacks
|
||||
* Public or private harassment
|
||||
* Publishing others' private information, such as a physical or electronic
|
||||
address, without explicit permission
|
||||
* Publishing others' private information, such as a physical or email address,
|
||||
without their explicit permission
|
||||
* Other conduct which could reasonably be considered inappropriate in a
|
||||
professional setting
|
||||
|
||||
## Our Responsibilities
|
||||
## Enforcement Responsibilities
|
||||
|
||||
Project maintainers are responsible for clarifying the standards of acceptable
|
||||
behavior and are expected to take appropriate and fair corrective action in
|
||||
response to any instances of unacceptable behavior.
|
||||
Community leaders are responsible for clarifying and enforcing our standards of
|
||||
acceptable behavior and will take appropriate and fair corrective action in
|
||||
response to any behavior that they deem inappropriate, threatening, offensive,
|
||||
or harmful.
|
||||
|
||||
Project maintainers have the right and responsibility to remove, edit, or
|
||||
reject comments, commits, code, wiki edits, issues, and other contributions
|
||||
that are not aligned to this Code of Conduct, or to ban temporarily or
|
||||
permanently any contributor for other behaviors that they deem inappropriate,
|
||||
threatening, offensive, or harmful.
|
||||
Community leaders have the right and responsibility to remove, edit, or reject
|
||||
comments, commits, code, wiki edits, issues, and other contributions that are
|
||||
not aligned to this Code of Conduct, and will communicate reasons for moderation
|
||||
decisions when appropriate.
|
||||
|
||||
## Scope
|
||||
|
||||
This Code of Conduct applies both within project spaces and in public spaces
|
||||
when an individual is representing the project or its community. Examples of
|
||||
representing a project or community include using an official project e-mail
|
||||
address, posting via an official social media account, or acting as an appointed
|
||||
representative at an online or offline event. Representation of a project may be
|
||||
further defined and clarified by project maintainers.
|
||||
This Code of Conduct applies within all community spaces, and also applies when
|
||||
an individual is officially representing the community in public spaces.
|
||||
Examples of representing our community include using an official e-mail address,
|
||||
posting via an official social media account, or acting as an appointed
|
||||
representative at an online or offline event.
|
||||
|
||||
## Enforcement
|
||||
|
||||
Instances of abusive, harassing, or otherwise unacceptable behavior may be
|
||||
reported by contacting the project team at opensource@zyte.com. All
|
||||
complaints will be reviewed and investigated and will result in a response that
|
||||
is deemed necessary and appropriate to the circumstances. The project team is
|
||||
obligated to maintain confidentiality with regard to the reporter of an incident.
|
||||
Further details of specific enforcement policies may be posted separately.
|
||||
reported to the community leaders responsible for enforcement at
|
||||
opensource@zyte.com.
|
||||
All complaints will be reviewed and investigated promptly and fairly.
|
||||
|
||||
Project maintainers who do not follow or enforce the Code of Conduct in good
|
||||
faith may face temporary or permanent repercussions as determined by other
|
||||
members of the project's leadership.
|
||||
All community leaders are obligated to respect the privacy and security of the
|
||||
reporter of any incident.
|
||||
|
||||
## Enforcement Guidelines
|
||||
|
||||
Community leaders will follow these Community Impact Guidelines in determining
|
||||
the consequences for any action they deem in violation of this Code of Conduct:
|
||||
|
||||
### 1. Correction
|
||||
|
||||
**Community Impact**: Use of inappropriate language or other behavior deemed
|
||||
unprofessional or unwelcome in the community.
|
||||
|
||||
**Consequence**: A private, written warning from community leaders, providing
|
||||
clarity around the nature of the violation and an explanation of why the
|
||||
behavior was inappropriate. A public apology may be requested.
|
||||
|
||||
### 2. Warning
|
||||
|
||||
**Community Impact**: A violation through a single incident or series of
|
||||
actions.
|
||||
|
||||
**Consequence**: A warning with consequences for continued behavior. No
|
||||
interaction with the people involved, including unsolicited interaction with
|
||||
those enforcing the Code of Conduct, for a specified period of time. This
|
||||
includes avoiding interactions in community spaces as well as external channels
|
||||
like social media. Violating these terms may lead to a temporary or permanent
|
||||
ban.
|
||||
|
||||
### 3. Temporary Ban
|
||||
|
||||
**Community Impact**: A serious violation of community standards, including
|
||||
sustained inappropriate behavior.
|
||||
|
||||
**Consequence**: A temporary ban from any sort of interaction or public
|
||||
communication with the community for a specified period of time. No public or
|
||||
private interaction with the people involved, including unsolicited interaction
|
||||
with those enforcing the Code of Conduct, is allowed during this period.
|
||||
Violating these terms may lead to a permanent ban.
|
||||
|
||||
### 4. Permanent Ban
|
||||
|
||||
**Community Impact**: Demonstrating a pattern of violation of community
|
||||
standards, including sustained inappropriate behavior, harassment of an
|
||||
individual, or aggression toward or disparagement of classes of individuals.
|
||||
|
||||
**Consequence**: A permanent ban from any sort of public interaction within the
|
||||
community.
|
||||
|
||||
## Attribution
|
||||
|
||||
This Code of Conduct is adapted from the [Contributor Covenant][homepage], version 1.4,
|
||||
available at [http://contributor-covenant.org/version/1/4][version].
|
||||
This Code of Conduct is adapted from the [Contributor Covenant][homepage],
|
||||
version 2.1, available at
|
||||
[https://www.contributor-covenant.org/version/2/1/code_of_conduct.html][v2.1].
|
||||
|
||||
[homepage]: http://contributor-covenant.org
|
||||
[version]: http://contributor-covenant.org/version/1/4/
|
||||
Community Impact Guidelines were inspired by
|
||||
[Mozilla's code of conduct enforcement ladder][Mozilla CoC].
|
||||
|
||||
For answers to common questions about this code of conduct, see
|
||||
https://www.contributor-covenant.org/faq
|
||||
For answers to common questions about this code of conduct, see the FAQ at
|
||||
[https://www.contributor-covenant.org/faq][FAQ]. Translations are available at
|
||||
[https://www.contributor-covenant.org/translations][translations].
|
||||
|
||||
[homepage]: https://www.contributor-covenant.org
|
||||
[v2.1]: https://www.contributor-covenant.org/version/2/1/code_of_conduct.html
|
||||
[Mozilla CoC]: https://github.com/mozilla/diversity
|
||||
[FAQ]: https://www.contributor-covenant.org/faq
|
||||
[translations]: https://www.contributor-covenant.org/translations
|
||||
|
|
|
|||
4
INSTALL
4
INSTALL
|
|
@ -1,4 +0,0 @@
|
|||
For information about installing Scrapy see:
|
||||
|
||||
* docs/intro/install.rst (local file)
|
||||
* https://docs.scrapy.org/en/latest/intro/install.html (online version)
|
||||
|
|
@ -0,0 +1,4 @@
|
|||
For information about installing Scrapy see:
|
||||
|
||||
* [Local docs](docs/intro/install.rst)
|
||||
* [Online docs](https://docs.scrapy.org/en/latest/intro/install.html)
|
||||
|
|
@ -64,7 +64,9 @@ Requirements
|
|||
Install
|
||||
=======
|
||||
|
||||
The quick way::
|
||||
The quick way:
|
||||
|
||||
.. code:: bash
|
||||
|
||||
pip install scrapy
|
||||
|
||||
|
|
@ -95,8 +97,7 @@ See https://docs.scrapy.org/en/master/contributing.html for details.
|
|||
Code of Conduct
|
||||
---------------
|
||||
|
||||
Please note that this project is released with a Contributor Code of Conduct
|
||||
(see https://github.com/scrapy/scrapy/blob/master/CODE_OF_CONDUCT.md).
|
||||
Please note that this project is released with a Contributor `Code of Conduct <https://github.com/scrapy/scrapy/blob/master/CODE_OF_CONDUCT.md>`_.
|
||||
|
||||
By participating in this project you agree to abide by its terms.
|
||||
Please report unacceptable behavior to opensource@zyte.com.
|
||||
|
|
|
|||
|
|
@ -15,7 +15,7 @@ class SettingsListDirective(Directive):
|
|||
|
||||
|
||||
def is_setting_index(node):
|
||||
if node.tagname == 'index':
|
||||
if node.tagname == 'index' and node['entries']:
|
||||
# index entries for setting directives look like:
|
||||
# [('pair', 'SETTING_NAME; setting', 'std:setting-SETTING_NAME', '')]
|
||||
entry_type, info, refid = node['entries'][0][:3]
|
||||
|
|
|
|||
|
|
@ -10,7 +10,7 @@ Supported Python versions
|
|||
=========================
|
||||
|
||||
Scrapy requires Python 3.7+, either the CPython implementation (default) or
|
||||
the PyPy 7.3.5+ implementation (see :ref:`python:implementations`).
|
||||
the PyPy implementation (see :ref:`python:implementations`).
|
||||
|
||||
.. _intro-install-scrapy:
|
||||
|
||||
|
|
@ -52,7 +52,7 @@ Scrapy is written in pure Python and depends on a few key Python packages (among
|
|||
* `twisted`_, an asynchronous networking framework
|
||||
* `cryptography`_ and `pyOpenSSL`_, to deal with various network-level security needs
|
||||
|
||||
Some of these packages themselves depends on non-Python packages
|
||||
Some of these packages themselves depend on non-Python packages
|
||||
that might require additional installation steps depending on your platform.
|
||||
Please check :ref:`platform-specific guides below <intro-install-platform-notes>`.
|
||||
|
||||
|
|
@ -187,7 +187,7 @@ solutions:
|
|||
* Install `homebrew`_ following the instructions in https://brew.sh/
|
||||
|
||||
* Update your ``PATH`` variable to state that homebrew packages should be
|
||||
used before system packages (Change ``.bashrc`` to ``.zshrc`` accordantly
|
||||
used before system packages (Change ``.bashrc`` to ``.zshrc`` accordingly
|
||||
if you're using `zsh`_ as default shell)::
|
||||
|
||||
echo "export PATH=/usr/local/bin:/usr/local/sbin:$PATH" >> ~/.bashrc
|
||||
|
|
@ -219,7 +219,7 @@ After any of these workarounds you should be able to install Scrapy::
|
|||
PyPy
|
||||
----
|
||||
|
||||
We recommend using the latest PyPy version. The version tested is 5.9.0.
|
||||
We recommend using the latest PyPy version.
|
||||
For PyPy3, only Linux installation was tested.
|
||||
|
||||
Most Scrapy dependencies now have binary wheels for CPython, but not for PyPy.
|
||||
|
|
|
|||
245
docs/news.rst
245
docs/news.rst
|
|
@ -3,6 +3,247 @@
|
|||
Release notes
|
||||
=============
|
||||
|
||||
.. _release-2.7.1:
|
||||
|
||||
Scrapy 2.7.1 (2022-11-02)
|
||||
-------------------------
|
||||
|
||||
New features
|
||||
~~~~~~~~~~~~
|
||||
|
||||
- Relaxed the restriction introduced in 2.6.2 so that the
|
||||
``Proxy-Authentication`` header can again be set explicitly, as long as the
|
||||
proxy URL in the :reqmeta:`proxy` metadata has no other credentials, and
|
||||
for as long as that proxy URL remains the same; this restores compatibility
|
||||
with scrapy-zyte-smartproxy 2.1.0 and older (:issue:`5626`).
|
||||
|
||||
Bug fixes
|
||||
~~~~~~~~~
|
||||
|
||||
- Using ``-O``/``--overwrite-output`` and ``-t``/``--output-format`` options
|
||||
together now produces an error instead of ignoring the former option
|
||||
(:issue:`5516`, :issue:`5605`).
|
||||
|
||||
- Replaced deprecated :mod:`asyncio` APIs that implicitly use the current
|
||||
event loop with code that explicitly requests a loop from the event loop
|
||||
policy (:issue:`5685`, :issue:`5689`).
|
||||
|
||||
- Fixed uses of deprecated Scrapy APIs in Scrapy itself (:issue:`5588`,
|
||||
:issue:`5589`).
|
||||
|
||||
- Fixed uses of a deprecated Pillow API (:issue:`5684`, :issue:`5692`).
|
||||
|
||||
- Improved code that checks if generators return values, so that it no longer
|
||||
fails on decorated methods and partial methods (:issue:`5323`,
|
||||
:issue:`5592`, :issue:`5599`, :issue:`5691`).
|
||||
|
||||
Documentation
|
||||
~~~~~~~~~~~~~
|
||||
|
||||
- Upgraded the Code of Conduct to Contributor Covenant v2.1 (:issue:`5698`).
|
||||
|
||||
- Fixed typos (:issue:`5681`, :issue:`5694`).
|
||||
|
||||
Quality assurance
|
||||
~~~~~~~~~~~~~~~~~
|
||||
|
||||
- Re-enabled some erroneously disabled flake8 checks (:issue:`5688`).
|
||||
|
||||
- Ignored harmless deprecation warnings from :mod:`typing` in tests
|
||||
(:issue:`5686`, :issue:`5697`).
|
||||
|
||||
- Modernized our CI configuration (:issue:`5695`, :issue:`5696`).
|
||||
|
||||
|
||||
.. _release-2.7.0:
|
||||
|
||||
Scrapy 2.7.0 (2022-10-17)
|
||||
-----------------------------
|
||||
|
||||
Highlights:
|
||||
|
||||
- Added Python 3.11 support, dropped Python 3.6 support
|
||||
- Improved support for :ref:`asynchronous callbacks <topics-coroutines>`
|
||||
- :ref:`Asyncio support <using-asyncio>` is enabled by default on new
|
||||
projects
|
||||
- Output names of item fields can now be arbitrary strings
|
||||
- Centralized :ref:`request fingerprinting <request-fingerprints>`
|
||||
configuration is now possible
|
||||
|
||||
Modified requirements
|
||||
~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
Python 3.7 or greater is now required; support for Python 3.6 has been dropped.
|
||||
Support for the upcoming Python 3.11 has been added.
|
||||
|
||||
The minimum required version of some dependencies has changed as well:
|
||||
|
||||
- lxml_: 3.5.0 → 4.3.0
|
||||
|
||||
- Pillow_ (:ref:`images pipeline <images-pipeline>`): 4.0.0 → 7.1.0
|
||||
|
||||
- zope.interface_: 5.0.0 → 5.1.0
|
||||
|
||||
(:issue:`5512`, :issue:`5514`, :issue:`5524`, :issue:`5563`, :issue:`5664`,
|
||||
:issue:`5670`, :issue:`5678`)
|
||||
|
||||
|
||||
Deprecations
|
||||
~~~~~~~~~~~~
|
||||
|
||||
- :meth:`ImagesPipeline.thumb_path
|
||||
<scrapy.pipelines.images.ImagesPipeline.thumb_path>` must now accept an
|
||||
``item`` parameter (:issue:`5504`, :issue:`5508`).
|
||||
|
||||
- The ``scrapy.downloadermiddlewares.decompression`` module is now
|
||||
deprecated (:issue:`5546`, :issue:`5547`).
|
||||
|
||||
|
||||
New features
|
||||
~~~~~~~~~~~~
|
||||
|
||||
- The
|
||||
:meth:`~scrapy.spidermiddlewares.SpiderMiddleware.process_spider_output`
|
||||
method of :ref:`spider middlewares <topics-spider-middleware>` can now be
|
||||
defined as an :term:`asynchronous generator` (:issue:`4978`).
|
||||
|
||||
- The output of :class:`~scrapy.Request` callbacks defined as
|
||||
:ref:`coroutines <topics-coroutines>` is now processed asynchronously
|
||||
(:issue:`4978`).
|
||||
|
||||
- :class:`~scrapy.spiders.crawl.CrawlSpider` now supports :ref:`asynchronous
|
||||
callbacks <topics-coroutines>` (:issue:`5657`).
|
||||
|
||||
- New projects created with the :command:`startproject` command have
|
||||
:ref:`asyncio support <using-asyncio>` enabled by default (:issue:`5590`,
|
||||
:issue:`5679`).
|
||||
|
||||
- The :setting:`FEED_EXPORT_FIELDS` setting can now be defined as a
|
||||
dictionary to customize the output name of item fields, lifting the
|
||||
restriction that required output names to be valid Python identifiers, e.g.
|
||||
preventing them to have whitespace (:issue:`1008`, :issue:`3266`,
|
||||
:issue:`3696`).
|
||||
|
||||
- You can now customize :ref:`request fingerprinting <request-fingerprints>`
|
||||
through the new :setting:`REQUEST_FINGERPRINTER_CLASS` setting, instead of
|
||||
having to change it on every Scrapy component that relies on request
|
||||
fingerprinting (:issue:`900`, :issue:`3420`, :issue:`4113`, :issue:`4762`,
|
||||
:issue:`4524`).
|
||||
|
||||
- ``jsonl`` is now supported and encouraged as a file extension for `JSON
|
||||
Lines`_ files (:issue:`4848`).
|
||||
|
||||
.. _JSON Lines: https://jsonlines.org/
|
||||
|
||||
- :meth:`ImagesPipeline.thumb_path
|
||||
<scrapy.pipelines.images.ImagesPipeline.thumb_path>` now receives the
|
||||
source :ref:`item <topics-items>` (:issue:`5504`, :issue:`5508`).
|
||||
|
||||
|
||||
Bug fixes
|
||||
~~~~~~~~~
|
||||
|
||||
- When using Google Cloud Storage with a :ref:`media pipeline
|
||||
<topics-media-pipeline>`, :setting:`FILES_EXPIRES` now also works when
|
||||
:setting:`FILES_STORE` does not point at the root of your Google Cloud
|
||||
Storage bucket (:issue:`5317`, :issue:`5318`).
|
||||
|
||||
- The :command:`parse` command now supports :ref:`asynchronous callbacks
|
||||
<topics-coroutines>` (:issue:`5424`, :issue:`5577`).
|
||||
|
||||
- When using the :command:`parse` command with a URL for which there is no
|
||||
available spider, an exception is no longer raised (:issue:`3264`,
|
||||
:issue:`3265`, :issue:`5375`, :issue:`5376`, :issue:`5497`).
|
||||
|
||||
- :class:`~scrapy.http.TextResponse` now gives higher priority to the `byte
|
||||
order mark`_ when determining the text encoding of the response body,
|
||||
following the `HTML living standard`_ (:issue:`5601`, :issue:`5611`).
|
||||
|
||||
.. _byte order mark: https://en.wikipedia.org/wiki/Byte_order_mark
|
||||
.. _HTML living standard: https://html.spec.whatwg.org/multipage/parsing.html#determining-the-character-encoding
|
||||
|
||||
- MIME sniffing takes the response body into account in FTP and HTTP/1.0
|
||||
requests, as well as in cached requests (:issue:`4873`).
|
||||
|
||||
- MIME sniffing now detects valid HTML 5 documents even if the ``html`` tag
|
||||
is missing (:issue:`4873`).
|
||||
|
||||
- An exception is now raised if :setting:`ASYNCIO_EVENT_LOOP` has a value
|
||||
that does not match the asyncio event loop actually installed
|
||||
(:issue:`5529`).
|
||||
|
||||
- Fixed :meth:`Headers.getlist <scrapy.http.headers.Headers.getlist>`
|
||||
returning only the last header (:issue:`5515`, :issue:`5526`).
|
||||
|
||||
- Fixed :class:`LinkExtractor
|
||||
<scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor>` not ignoring the
|
||||
``tar.gz`` file extension by default (:issue:`1837`, :issue:`2067`,
|
||||
:issue:`4066`)
|
||||
|
||||
|
||||
Documentation
|
||||
~~~~~~~~~~~~~
|
||||
|
||||
- Clarified the return type of :meth:`Spider.parse <scrapy.Spider.parse>`
|
||||
(:issue:`5602`, :issue:`5608`).
|
||||
|
||||
- To enable
|
||||
:class:`~scrapy.downloadermiddlewares.httpcompression.HttpCompressionMiddleware`
|
||||
to do `brotli compression`_, installing brotli_ is now recommended instead
|
||||
of installing brotlipy_, as the former provides a more recent version of
|
||||
brotli.
|
||||
|
||||
.. _brotli: https://github.com/google/brotli
|
||||
.. _brotli compression: https://www.ietf.org/rfc/rfc7932.txt
|
||||
|
||||
- :ref:`Signal documentation <topics-signals>` now mentions :ref:`coroutine
|
||||
support <topics-coroutines>` and uses it in code examples (:issue:`4852`,
|
||||
:issue:`5358`).
|
||||
|
||||
- :ref:`bans` now recommends `Common Crawl`_ instead of `Google cache`_
|
||||
(:issue:`3582`, :issue:`5432`).
|
||||
|
||||
.. _Common Crawl: https://commoncrawl.org/
|
||||
.. _Google cache: http://www.googleguide.com/cached_pages.html
|
||||
|
||||
- The new :ref:`topics-components` topic covers enforcing requirements on
|
||||
Scrapy components, like :ref:`downloader middlewares
|
||||
<topics-downloader-middleware>`, :ref:`extensions <topics-extensions>`,
|
||||
:ref:`item pipelines <topics-item-pipeline>`, :ref:`spider middlewares
|
||||
<topics-spider-middleware>`, and more; :ref:`enforce-asyncio-requirement`
|
||||
has also been added (:issue:`4978`).
|
||||
|
||||
- :ref:`topics-settings` now indicates that setting values must be
|
||||
:ref:`picklable <pickle-picklable>` (:issue:`5607`, :issue:`5629`).
|
||||
|
||||
- Removed outdated documentation (:issue:`5446`, :issue:`5373`,
|
||||
:issue:`5369`, :issue:`5370`, :issue:`5554`).
|
||||
|
||||
- Fixed typos (:issue:`5442`, :issue:`5455`, :issue:`5457`, :issue:`5461`,
|
||||
:issue:`5538`, :issue:`5553`, :issue:`5558`, :issue:`5624`, :issue:`5631`).
|
||||
|
||||
- Fixed other issues (:issue:`5283`, :issue:`5284`, :issue:`5559`,
|
||||
:issue:`5567`, :issue:`5648`, :issue:`5659`, :issue:`5665`).
|
||||
|
||||
|
||||
Quality assurance
|
||||
~~~~~~~~~~~~~~~~~
|
||||
|
||||
- Added a continuous integration job to run `twine check`_ (:issue:`5655`,
|
||||
:issue:`5656`).
|
||||
|
||||
.. _twine check: https://twine.readthedocs.io/en/stable/#twine-check
|
||||
|
||||
- Addressed test issues and warnings (:issue:`5560`, :issue:`5561`,
|
||||
:issue:`5612`, :issue:`5617`, :issue:`5639`, :issue:`5645`, :issue:`5662`,
|
||||
:issue:`5671`, :issue:`5675`).
|
||||
|
||||
- Cleaned up code (:issue:`4991`, :issue:`4995`, :issue:`5451`,
|
||||
:issue:`5487`, :issue:`5542`, :issue:`5667`, :issue:`5668`, :issue:`5672`).
|
||||
|
||||
- Applied minor code improvements (:issue:`5661`).
|
||||
|
||||
|
||||
.. _release-2.6.3:
|
||||
|
||||
Scrapy 2.6.3 (2022-09-27)
|
||||
|
|
@ -3139,7 +3380,7 @@ New Features
|
|||
~~~~~~~~~~~~
|
||||
|
||||
- Accept proxy credentials in :reqmeta:`proxy` request meta key (:issue:`2526`)
|
||||
- Support `brotli`_-compressed content; requires optional `brotlipy`_
|
||||
- Support `brotli-compressed`_ content; requires optional `brotlipy`_
|
||||
(:issue:`2535`)
|
||||
- New :ref:`response.follow <response-follow-example>` shortcut
|
||||
for creating requests (:issue:`1940`)
|
||||
|
|
@ -3176,7 +3417,7 @@ New Features
|
|||
- ``python -m scrapy`` as a more explicit alternative to ``scrapy`` command
|
||||
(:issue:`2740`)
|
||||
|
||||
.. _brotli: https://github.com/google/brotli
|
||||
.. _brotli-compressed: https://www.ietf.org/rfc/rfc7932.txt
|
||||
.. _brotlipy: https://github.com/python-hyper/brotlipy/
|
||||
|
||||
Bug fixes
|
||||
|
|
|
|||
|
|
@ -68,7 +68,7 @@ IP (:setting:`CONCURRENT_REQUESTS_PER_IP`).
|
|||
|
||||
The default global concurrency limit in Scrapy is not suitable for crawling
|
||||
many different domains in parallel, so you will want to increase it. How much
|
||||
to increase it will depend on how much CPU and memory you crawler will have
|
||||
to increase it will depend on how much CPU and memory your crawler will have
|
||||
available.
|
||||
|
||||
A good starting point is ``100``::
|
||||
|
|
|
|||
|
|
@ -271,11 +271,31 @@ crawl
|
|||
|
||||
Start crawling using a spider.
|
||||
|
||||
Supported options:
|
||||
|
||||
* ``-h, --help``: show a help message and exit
|
||||
|
||||
* ``-a NAME=VALUE``: set a spider argument (may be repeated)
|
||||
|
||||
* ``--output FILE`` or ``-o FILE``: append scraped items to the end of FILE (use - for stdout), to define format set a colon at the end of the output URI (i.e. ``-o FILE:FORMAT``)
|
||||
|
||||
* ``--overwrite-output FILE`` or ``-O FILE``: dump scraped items into FILE, overwriting any existing file, to define format set a colon at the end of the output URI (i.e. ``-O FILE:FORMAT``)
|
||||
|
||||
* ``--output-format FORMAT`` or ``-t FORMAT``: deprecated way to define format to use for dumping items, does not work in combination with ``-O``
|
||||
|
||||
Usage examples::
|
||||
|
||||
$ scrapy crawl myspider
|
||||
[ ... myspider starts crawling ... ]
|
||||
|
||||
$ scrapy -o myfile:csv myspider
|
||||
[ ... myspider starts crawling and appends the result to the file myfile in csv format ... ]
|
||||
|
||||
$ scrapy -O myfile:json myspider
|
||||
[ ... myspider starts crawling and saves the result in myfile in json format overwriting the original content... ]
|
||||
|
||||
$ scrapy -o myfile -t csv myspider
|
||||
[ ... myspider starts crawling and appends the result to the file myfile in csv format ... ]
|
||||
|
||||
.. command:: check
|
||||
|
||||
|
|
|
|||
|
|
@ -75,9 +75,9 @@ If your requirement is a minimum Scrapy version, you may use
|
|||
class MyComponent:
|
||||
|
||||
def __init__(self):
|
||||
if parse_version(scrapy.__version__) < parse_version('VERSION'):
|
||||
if parse_version(scrapy.__version__) < parse_version('2.7'):
|
||||
raise RuntimeError(
|
||||
f"{MyComponent.__qualname__} requires Scrapy VERSION or "
|
||||
f"{MyComponent.__qualname__} requires Scrapy 2.7 or "
|
||||
f"later, which allow defining the process_spider_output "
|
||||
f"method of spider middlewares as an asynchronous "
|
||||
f"generator."
|
||||
|
|
|
|||
|
|
@ -102,7 +102,7 @@ override three methods:
|
|||
.. method:: Contract.post_process(output)
|
||||
|
||||
This allows processing the output of the callback. Iterators are
|
||||
converted listified before being passed to this hook.
|
||||
converted to lists before being passed to this hook.
|
||||
|
||||
Raise :class:`~scrapy.exceptions.ContractFail` from
|
||||
:class:`~scrapy.contracts.Contract.pre_process` or
|
||||
|
|
|
|||
|
|
@ -22,7 +22,7 @@ hence use coroutine syntax (e.g. ``await``, ``async for``, ``async with``):
|
|||
If you are using any custom or third-party :ref:`spider middleware
|
||||
<topics-spider-middleware>`, see :ref:`sync-async-spider-middleware`.
|
||||
|
||||
.. versionchanged:: VERSION
|
||||
.. versionchanged:: 2.7
|
||||
Output of async callbacks is now processed asynchronously instead of
|
||||
collecting all of it first.
|
||||
|
||||
|
|
@ -49,7 +49,7 @@ hence use coroutine syntax (e.g. ``await``, ``async for``, ``async with``):
|
|||
See also :ref:`sync-async-spider-middleware` and
|
||||
:ref:`universal-spider-middleware`.
|
||||
|
||||
.. versionadded:: VERSION
|
||||
.. versionadded:: 2.7
|
||||
|
||||
General usage
|
||||
=============
|
||||
|
|
@ -129,7 +129,7 @@ Common use cases for asynchronous code include:
|
|||
Mixing synchronous and asynchronous spider middlewares
|
||||
======================================================
|
||||
|
||||
.. versionadded:: VERSION
|
||||
.. versionadded:: 2.7
|
||||
|
||||
The output of a :class:`~scrapy.Request` callback is passed as the ``result``
|
||||
parameter to the
|
||||
|
|
@ -182,10 +182,10 @@ process_spider_output_async method <universal-spider-middleware>`.
|
|||
Universal spider middlewares
|
||||
============================
|
||||
|
||||
.. versionadded:: VERSION
|
||||
.. versionadded:: 2.7
|
||||
|
||||
To allow writing a spider middleware that supports asynchronous execution of
|
||||
its ``process_spider_output`` method in Scrapy VERSION and later (avoiding
|
||||
its ``process_spider_output`` method in Scrapy 2.7 and later (avoiding
|
||||
:ref:`asynchronous-to-synchronous conversions <sync-async-spider-middleware>`)
|
||||
while maintaining support for older Scrapy versions, you may define
|
||||
``process_spider_output`` as a synchronous method and define an
|
||||
|
|
@ -206,7 +206,7 @@ For example::
|
|||
yield r
|
||||
|
||||
.. note:: This is an interim measure to allow, for a time, to write code that
|
||||
works in Scrapy VERSION and later without requiring
|
||||
works in Scrapy 2.7 and later without requiring
|
||||
asynchronous-to-synchronous conversions, and works in earlier Scrapy
|
||||
versions as well.
|
||||
|
||||
|
|
|
|||
|
|
@ -150,3 +150,31 @@ available in all future runs should they be necessary again::
|
|||
For more information, check the :ref:`topics-logging` section.
|
||||
|
||||
.. _base tag: https://www.w3schools.com/tags/tag_base.asp
|
||||
|
||||
Visual Studio Code
|
||||
==================
|
||||
|
||||
.. highlight:: json
|
||||
|
||||
To debug spiders with Visual Studio Code you can use the following ``launch.json``::
|
||||
|
||||
{
|
||||
"version": "0.1.0",
|
||||
"configurations": [
|
||||
{
|
||||
"name": "Python: Launch Scrapy Spider",
|
||||
"type": "python",
|
||||
"request": "launch",
|
||||
"module": "scrapy",
|
||||
"args": [
|
||||
"runspider",
|
||||
"${file}"
|
||||
],
|
||||
"console": "integratedTerminal"
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
|
||||
Also, make sure you enable "User Uncaught Exceptions", to catch exceptions in
|
||||
your Scrapy spider.
|
||||
|
|
|
|||
|
|
@ -17,7 +17,7 @@ settings, just like any other Scrapy code.
|
|||
|
||||
It is customary for extensions to prefix their settings with their own name, to
|
||||
avoid collision with existing (and future) extensions. For example, a
|
||||
hypothetic extension to handle `Google Sitemaps`_ would use settings like
|
||||
hypothetical extension to handle `Google Sitemaps`_ would use settings like
|
||||
``GOOGLESITEMAP_ENABLED``, ``GOOGLESITEMAP_DEPTH``, and so on.
|
||||
|
||||
.. _Google Sitemaps: https://en.wikipedia.org/wiki/Sitemaps
|
||||
|
|
|
|||
|
|
@ -154,7 +154,7 @@ Too many spiders?
|
|||
If your project has too many spiders executed in parallel,
|
||||
the output of :func:`prefs()` can be difficult to read.
|
||||
For this reason, that function has a ``ignore`` argument which can be used to
|
||||
ignore a particular class (and all its subclases). For
|
||||
ignore a particular class (and all its subclasses). For
|
||||
example, this won't show any live references to spiders:
|
||||
|
||||
>>> from scrapy.spiders import Spider
|
||||
|
|
|
|||
|
|
@ -394,7 +394,7 @@ To change how request fingerprints are built for your requests, use the
|
|||
REQUEST_FINGERPRINTER_CLASS
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
.. versionadded:: VERSION
|
||||
.. versionadded:: 2.7
|
||||
|
||||
Default: :class:`scrapy.utils.request.RequestFingerprinter`
|
||||
|
||||
|
|
@ -409,54 +409,54 @@ import path.
|
|||
REQUEST_FINGERPRINTER_IMPLEMENTATION
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
.. versionadded:: VERSION
|
||||
.. versionadded:: 2.7
|
||||
|
||||
Default: ``'PREVIOUS_VERSION'``
|
||||
Default: ``'2.6'``
|
||||
|
||||
Determines which request fingerprinting algorithm is used by the default
|
||||
request fingerprinter class (see :setting:`REQUEST_FINGERPRINTER_CLASS`).
|
||||
|
||||
Possible values are:
|
||||
|
||||
- ``'PREVIOUS_VERSION'`` (default)
|
||||
- ``'2.6'`` (default)
|
||||
|
||||
This implementation uses the same request fingerprinting algorithm as
|
||||
Scrapy PREVIOUS_VERSION and earlier versions.
|
||||
Scrapy 2.6 and earlier versions.
|
||||
|
||||
Even though this is the default value for backward compatibility reasons,
|
||||
it is a deprecated value.
|
||||
|
||||
- ``'VERSION'``
|
||||
- ``'2.7'``
|
||||
|
||||
This implementation was introduced in Scrapy VERSION to fix an issue of the
|
||||
This implementation was introduced in Scrapy 2.7 to fix an issue of the
|
||||
previous implementation.
|
||||
|
||||
New projects should use this value. The :command:`startproject` command
|
||||
sets this value in the generated ``settings.py`` file.
|
||||
|
||||
If you are using the default value (``'PREVIOUS_VERSION'``) for this setting, and you are
|
||||
If you are using the default value (``'2.6'``) for this setting, and you are
|
||||
using Scrapy components where changing the request fingerprinting algorithm
|
||||
would cause undesired results, you need to carefully decide when to change the
|
||||
value of this setting, or switch the :setting:`REQUEST_FINGERPRINTER_CLASS`
|
||||
setting to a custom request fingerprinter class that implements the PREVIOUS_VERSION request
|
||||
setting to a custom request fingerprinter class that implements the 2.6 request
|
||||
fingerprinting algorithm and does not log this warning (
|
||||
:ref:`PREVIOUS_VERSION-request-fingerprinter` includes an example implementation of such a
|
||||
:ref:`2.6-request-fingerprinter` includes an example implementation of such a
|
||||
class).
|
||||
|
||||
Scenarios where changing the request fingerprinting algorithm may cause
|
||||
undesired results include, for example, using the HTTP cache middleware (see
|
||||
:class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware`).
|
||||
Changing the request fingerprinting algorithm would invalidade the current
|
||||
Changing the request fingerprinting algorithm would invalidate the current
|
||||
cache, requiring you to redownload all requests again.
|
||||
|
||||
Otherwise, set :setting:`REQUEST_FINGERPRINTER_IMPLEMENTATION` to ``'VERSION'`` in
|
||||
Otherwise, set :setting:`REQUEST_FINGERPRINTER_IMPLEMENTATION` to ``'2.7'`` in
|
||||
your settings to switch already to the request fingerprinting implementation
|
||||
that will be the only request fingerprinting implementation available in a
|
||||
future version of Scrapy, and remove the deprecation warning triggered by using
|
||||
the default value (``'PREVIOUS_VERSION'``).
|
||||
the default value (``'2.6'``).
|
||||
|
||||
|
||||
.. _PREVIOUS_VERSION-request-fingerprinter:
|
||||
.. _2.6-request-fingerprinter:
|
||||
.. _custom-request-fingerprinter:
|
||||
|
||||
Writing your own request fingerprinter
|
||||
|
|
@ -464,6 +464,8 @@ Writing your own request fingerprinter
|
|||
|
||||
A request fingerprinter is a class that must implement the following method:
|
||||
|
||||
.. currentmodule:: None
|
||||
|
||||
.. method:: fingerprint(self, request)
|
||||
|
||||
Return a :class:`bytes` object that uniquely identifies *request*.
|
||||
|
|
@ -476,6 +478,7 @@ A request fingerprinter is a class that must implement the following method:
|
|||
Additionally, it may also implement the following methods:
|
||||
|
||||
.. classmethod:: from_crawler(cls, crawler)
|
||||
:noindex:
|
||||
|
||||
If present, this class method is called to create a request fingerprinter
|
||||
instance from a :class:`~scrapy.crawler.Crawler` object. It must return a
|
||||
|
|
@ -495,11 +498,13 @@ Additionally, it may also implement the following methods:
|
|||
:class:`~scrapy.settings.Settings` object. It must return a new instance of
|
||||
the request fingerprinter.
|
||||
|
||||
The ``fingerprint`` method of the default request fingerprinter,
|
||||
.. currentmodule:: scrapy.http
|
||||
|
||||
The :meth:`fingerprint` method of the default request fingerprinter,
|
||||
:class:`scrapy.utils.request.RequestFingerprinter`, uses
|
||||
:func:`scrapy.utils.request.fingerprint` with its default parameters. For some
|
||||
common use cases you can use :func:`~scrapy.utils.request.fingerprint` as well
|
||||
in your ``fingerprint`` method implementation:
|
||||
common use cases you can use :func:`scrapy.utils.request.fingerprint` as well
|
||||
in your :meth:`fingerprint` method implementation:
|
||||
|
||||
.. autofunction:: scrapy.utils.request.fingerprint
|
||||
|
||||
|
|
@ -519,7 +524,7 @@ account::
|
|||
|
||||
You can also write your own fingerprinting logic from scratch.
|
||||
|
||||
However, if you do not use :func:`~scrapy.utils.request.fingerprint`, make sure
|
||||
However, if you do not use :func:`scrapy.utils.request.fingerprint`, make sure
|
||||
you use :class:`~weakref.WeakKeyDictionary` to cache request fingerprints:
|
||||
|
||||
- Caching saves CPU by ensuring that fingerprints are calculated only once
|
||||
|
|
@ -553,7 +558,7 @@ If you need to be able to override the request fingerprinting for arbitrary
|
|||
requests from your spider callbacks, you may implement a request fingerprinter
|
||||
that reads fingerprints from :attr:`request.meta <scrapy.http.Request.meta>`
|
||||
when available, and then falls back to
|
||||
:func:`~scrapy.utils.request.fingerprint`. For example::
|
||||
:func:`scrapy.utils.request.fingerprint`. For example::
|
||||
|
||||
from scrapy.utils.request import fingerprint
|
||||
|
||||
|
|
@ -564,8 +569,8 @@ when available, and then falls back to
|
|||
return request.meta['fingerprint']
|
||||
return fingerprint(request)
|
||||
|
||||
If you need to reproduce the same fingerprinting algorithm as Scrapy PREVIOUS_VERSION
|
||||
without using the deprecated ``'PREVIOUS_VERSION'`` value of the
|
||||
If you need to reproduce the same fingerprinting algorithm as Scrapy 2.6
|
||||
without using the deprecated ``'2.6'`` value of the
|
||||
:setting:`REQUEST_FINGERPRINTER_IMPLEMENTATION` setting, use the following
|
||||
request fingerprinter::
|
||||
|
||||
|
|
|
|||
|
|
@ -1642,6 +1642,11 @@ install the default reactor defined by Twisted for the current platform. This
|
|||
is to maintain backward compatibility and avoid possible problems caused by
|
||||
using a non-default reactor.
|
||||
|
||||
.. versionchanged:: 2.7
|
||||
The :command:`startproject` command now sets this setting to
|
||||
``twisted.internet.asyncioreactor.AsyncioSelectorReactor`` in the generated
|
||||
``settings.py`` file.
|
||||
|
||||
For additional information, see :doc:`core/howto/choosing-reactor`.
|
||||
|
||||
|
||||
|
|
@ -1656,14 +1661,14 @@ Scope: ``spidermiddlewares.urllength``
|
|||
|
||||
The maximum URL length to allow for crawled URLs.
|
||||
|
||||
This setting can act as a stopping condition in case of URLs of ever-increasing
|
||||
length, which may be caused for example by a programming error either in the
|
||||
target server or in your code. See also :setting:`REDIRECT_MAX_TIMES` and
|
||||
This setting can act as a stopping condition in case of URLs of ever-increasing
|
||||
length, which may be caused for example by a programming error either in the
|
||||
target server or in your code. See also :setting:`REDIRECT_MAX_TIMES` and
|
||||
:setting:`DEPTH_LIMIT`.
|
||||
|
||||
Use ``0`` to allow URLs of any length.
|
||||
|
||||
The default value is copied from the `Microsoft Internet Explorer maximum URL
|
||||
The default value is copied from the `Microsoft Internet Explorer maximum URL
|
||||
length`_, even though this setting exists for different reasons.
|
||||
|
||||
.. _Microsoft Internet Explorer maximum URL length: https://support.microsoft.com/en-us/topic/maximum-url-length-is-2-083-characters-in-internet-explorer-174e7c8a-6666-f4e0-6fd6-908b53c12246
|
||||
|
|
|
|||
|
|
@ -105,17 +105,17 @@ object gives you access, for example, to the :ref:`settings <topics-settings>`.
|
|||
:class:`~scrapy.Request` objects and :ref:`item objects
|
||||
<topics-items>`.
|
||||
|
||||
.. versionchanged:: VERSION
|
||||
.. versionchanged:: 2.7
|
||||
This method may be defined as an :term:`asynchronous generator`, in
|
||||
which case ``result`` is an :term:`asynchronous iterable`.
|
||||
|
||||
Consider defining this method as an :term:`asynchronous generator`,
|
||||
which will be a requirement in a future version of Scrapy. However, if
|
||||
you plan on sharing your spider middleware with other people, consider
|
||||
either :ref:`enforcing Scrapy VERSION <enforce-component-requirements>`
|
||||
either :ref:`enforcing Scrapy 2.7 <enforce-component-requirements>`
|
||||
as a minimum requirement of your spider middleware, or :ref:`making
|
||||
your spider middleware universal <universal-spider-middleware>` so that
|
||||
it works with Scrapy versions earlier than Scrapy VERSION.
|
||||
it works with Scrapy versions earlier than Scrapy 2.7.
|
||||
|
||||
:param response: the response which generated this output from the
|
||||
spider
|
||||
|
|
@ -130,7 +130,7 @@ object gives you access, for example, to the :ref:`settings <topics-settings>`.
|
|||
|
||||
.. method:: process_spider_output_async(response, result, spider)
|
||||
|
||||
.. versionadded:: VERSION
|
||||
.. versionadded:: 2.7
|
||||
|
||||
If defined, this method must be an :term:`asynchronous generator`,
|
||||
which will be called instead of :meth:`process_spider_output` if
|
||||
|
|
|
|||
|
|
@ -24,3 +24,5 @@ markers =
|
|||
filterwarnings =
|
||||
ignore:scrapy.downloadermiddlewares.decompression is deprecated
|
||||
ignore:Module scrapy.utils.reqser is deprecated
|
||||
ignore:typing.re is deprecated
|
||||
ignore:typing.io is deprecated
|
||||
|
|
|
|||
|
|
@ -1 +1 @@
|
|||
2.6.2
|
||||
2.7.1
|
||||
|
|
|
|||
|
|
@ -7,7 +7,7 @@ import pkg_resources
|
|||
|
||||
import scrapy
|
||||
from scrapy.crawler import CrawlerProcess
|
||||
from scrapy.commands import ScrapyCommand, ScrapyHelpFormatter
|
||||
from scrapy.commands import ScrapyCommand, ScrapyHelpFormatter, BaseRunSpiderCommand
|
||||
from scrapy.exceptions import UsageError
|
||||
from scrapy.utils.misc import walk_modules
|
||||
from scrapy.utils.project import inside_project, get_project_settings
|
||||
|
|
@ -32,7 +32,7 @@ def _iter_command_classes(module_name):
|
|||
inspect.isclass(obj)
|
||||
and issubclass(obj, ScrapyCommand)
|
||||
and obj.__module__ == module.__name__
|
||||
and not obj == ScrapyCommand
|
||||
and obj not in (ScrapyCommand, BaseRunSpiderCommand)
|
||||
):
|
||||
yield obj
|
||||
|
||||
|
|
@ -78,7 +78,8 @@ def _pop_command_name(argv):
|
|||
def _print_header(settings, inproject):
|
||||
version = scrapy.__version__
|
||||
if inproject:
|
||||
print(f"Scrapy {version} - project: {settings['BOT_NAME']}\n")
|
||||
print(f"Scrapy {version} - active project: {settings['BOT_NAME']}\n")
|
||||
|
||||
else:
|
||||
print(f"Scrapy {version} - no active project\n")
|
||||
|
||||
|
|
|
|||
|
|
@ -116,9 +116,11 @@ class BaseRunSpiderCommand(ScrapyCommand):
|
|||
parser.add_argument("-a", dest="spargs", action="append", default=[], metavar="NAME=VALUE",
|
||||
help="set spider argument (may be repeated)")
|
||||
parser.add_argument("-o", "--output", metavar="FILE", action="append",
|
||||
help="append scraped items to the end of FILE (use - for stdout)")
|
||||
help="append scraped items to the end of FILE (use - for stdout),"
|
||||
" to define format set a colon at the end of the output URI (i.e. -o FILE:FORMAT)")
|
||||
parser.add_argument("-O", "--overwrite-output", metavar="FILE", action="append",
|
||||
help="dump scraped items into FILE, overwriting any existing file")
|
||||
help="dump scraped items into FILE, overwriting any existing file,"
|
||||
" to define format set a colon at the end of the output URI (i.e. -O FILE:FORMAT)")
|
||||
parser.add_argument("-t", "--output-format", metavar="FORMAT",
|
||||
help="format to use for dumping items")
|
||||
|
||||
|
|
|
|||
|
|
@ -5,6 +5,8 @@ from typing import Dict
|
|||
from itemadapter import is_item, ItemAdapter
|
||||
from w3lib.url import is_url
|
||||
|
||||
from twisted.internet.defer import maybeDeferred
|
||||
|
||||
from scrapy.commands import BaseRunSpiderCommand
|
||||
from scrapy.http import Request
|
||||
from scrapy.utils import display
|
||||
|
|
@ -110,16 +112,19 @@ class Command(BaseRunSpiderCommand):
|
|||
if not opts.nolinks:
|
||||
self.print_requests(colour=colour)
|
||||
|
||||
def run_callback(self, response, callback, cb_kwargs=None):
|
||||
cb_kwargs = cb_kwargs or {}
|
||||
def _get_items_and_requests(self, spider_output, opts, depth, spider, callback):
|
||||
items, requests = [], []
|
||||
|
||||
for x in iterate_spider_output(callback(response, **cb_kwargs)):
|
||||
for x in spider_output:
|
||||
if is_item(x):
|
||||
items.append(x)
|
||||
elif isinstance(x, Request):
|
||||
requests.append(x)
|
||||
return items, requests
|
||||
return items, requests, opts, depth, spider, callback
|
||||
|
||||
def run_callback(self, response, callback, cb_kwargs=None):
|
||||
cb_kwargs = cb_kwargs or {}
|
||||
d = maybeDeferred(iterate_spider_output, callback(response, **cb_kwargs))
|
||||
return d
|
||||
|
||||
def get_callback_from_rules(self, spider, response):
|
||||
if getattr(spider, 'rules', None):
|
||||
|
|
@ -158,6 +163,25 @@ class Command(BaseRunSpiderCommand):
|
|||
logger.error('No response downloaded for: %(url)s',
|
||||
{'url': url})
|
||||
|
||||
def scraped_data(self, args):
|
||||
items, requests, opts, depth, spider, callback = args
|
||||
if opts.pipelines:
|
||||
itemproc = self.pcrawler.engine.scraper.itemproc
|
||||
for item in items:
|
||||
itemproc.process_item(item, spider)
|
||||
self.add_items(depth, items)
|
||||
self.add_requests(depth, requests)
|
||||
|
||||
scraped_data = items if opts.output else []
|
||||
if depth < opts.depth:
|
||||
for req in requests:
|
||||
req.meta['_depth'] = depth + 1
|
||||
req.meta['_callback'] = req.callback
|
||||
req.callback = callback
|
||||
scraped_data += requests
|
||||
|
||||
return scraped_data
|
||||
|
||||
def prepare_request(self, spider, request, opts):
|
||||
def callback(response, **cb_kwargs):
|
||||
# memorize first request
|
||||
|
|
@ -191,23 +215,10 @@ class Command(BaseRunSpiderCommand):
|
|||
# parse items and requests
|
||||
depth = response.meta['_depth']
|
||||
|
||||
items, requests = self.run_callback(response, cb, cb_kwargs)
|
||||
if opts.pipelines:
|
||||
itemproc = self.pcrawler.engine.scraper.itemproc
|
||||
for item in items:
|
||||
itemproc.process_item(item, spider)
|
||||
self.add_items(depth, items)
|
||||
self.add_requests(depth, requests)
|
||||
|
||||
scraped_data = items if opts.output else []
|
||||
if depth < opts.depth:
|
||||
for req in requests:
|
||||
req.meta['_depth'] = depth + 1
|
||||
req.meta['_callback'] = req.callback
|
||||
req.callback = callback
|
||||
scraped_data += requests
|
||||
|
||||
return scraped_data
|
||||
d = self.run_callback(response, cb, cb_kwargs)
|
||||
d.addCallback(self._get_items_and_requests, opts, depth, spider, callback)
|
||||
d.addCallback(self.scraped_data)
|
||||
return d
|
||||
|
||||
# update request meta if any extra meta was passed through the --meta/-m opts.
|
||||
if opts.meta:
|
||||
|
|
|
|||
|
|
@ -3,7 +3,6 @@
|
|||
import ipaddress
|
||||
import logging
|
||||
import re
|
||||
import warnings
|
||||
from contextlib import suppress
|
||||
from io import BytesIO
|
||||
from time import time
|
||||
|
|
@ -22,7 +21,7 @@ from zope.interface import implementer
|
|||
from scrapy import signals
|
||||
from scrapy.core.downloader.contextfactory import load_context_factory_from_settings
|
||||
from scrapy.core.downloader.webclient import _parse
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning, StopDownload
|
||||
from scrapy.exceptions import StopDownload
|
||||
from scrapy.http import Headers
|
||||
from scrapy.responsetypes import responsetypes
|
||||
from scrapy.utils.python import to_bytes, to_unicode
|
||||
|
|
@ -279,17 +278,7 @@ class ScrapyAgent:
|
|||
proxyScheme, proxyNetloc, proxyHost, proxyPort, proxyParams = _parse(proxy)
|
||||
scheme = _parse(request.url)[0]
|
||||
proxyHost = to_unicode(proxyHost)
|
||||
omitConnectTunnel = b'noconnect' in proxyParams
|
||||
if omitConnectTunnel:
|
||||
warnings.warn(
|
||||
"Using HTTPS proxies in the noconnect mode is deprecated. "
|
||||
"If you use Zyte Smart Proxy Manager, it doesn't require "
|
||||
"this mode anymore, so you should update scrapy-crawlera "
|
||||
"to scrapy-zyte-smartproxy and remove '?noconnect' "
|
||||
"from the Zyte Smart Proxy Manager URL.",
|
||||
ScrapyDeprecationWarning,
|
||||
)
|
||||
if scheme == b'https' and not omitConnectTunnel:
|
||||
if scheme == b'https':
|
||||
proxyAuth = request.headers.get(b'Proxy-Authorization', None)
|
||||
proxyConf = (proxyHost, proxyPort, proxyAuth)
|
||||
return self._TunnelingAgent(
|
||||
|
|
@ -302,8 +291,6 @@ class ScrapyAgent:
|
|||
)
|
||||
else:
|
||||
proxyScheme = proxyScheme or b'http'
|
||||
proxyHost = to_bytes(proxyHost, encoding='ascii')
|
||||
proxyPort = to_bytes(str(proxyPort), encoding='ascii')
|
||||
proxyURI = urlunparse((proxyScheme, proxyNetloc, proxyParams, '', '', ''))
|
||||
return self._ProxyAgent(
|
||||
reactor=reactor,
|
||||
|
|
@ -384,8 +371,7 @@ class ScrapyAgent:
|
|||
logger.debug("Download stopped for %(request)s from signal handler %(handler)s",
|
||||
{"request": request, "handler": handler.__qualname__})
|
||||
txresponse._transport.stopProducing()
|
||||
with suppress(AttributeError):
|
||||
txresponse._transport._producer.loseConnection()
|
||||
txresponse._transport.loseConnection()
|
||||
return {
|
||||
"txresponse": txresponse,
|
||||
"body": b"",
|
||||
|
|
@ -417,7 +403,7 @@ class ScrapyAgent:
|
|||
|
||||
logger.warning(warning_msg, warning_args)
|
||||
|
||||
txresponse._transport._producer.loseConnection()
|
||||
txresponse._transport.loseConnection()
|
||||
raise defer.CancelledError(warning_msg % warning_args)
|
||||
|
||||
if warnsize and expected_size > warnsize:
|
||||
|
|
@ -543,7 +529,7 @@ class _ResponseReader(protocol.Protocol):
|
|||
logger.debug("Download stopped for %(request)s from signal handler %(handler)s",
|
||||
{"request": self._request, "handler": handler.__qualname__})
|
||||
self.transport.stopProducing()
|
||||
self.transport._producer.loseConnection()
|
||||
self.transport.loseConnection()
|
||||
failure = result if result.value.fail else None
|
||||
self._finish_response(flags=["download_stopped"], failure=failure)
|
||||
|
||||
|
|
|
|||
|
|
@ -1,4 +1,3 @@
|
|||
import warnings
|
||||
from time import time
|
||||
from typing import Optional, Type, TypeVar
|
||||
from urllib.parse import urldefrag
|
||||
|
|
@ -69,19 +68,8 @@ class ScrapyH2Agent:
|
|||
if proxy:
|
||||
_, _, proxy_host, proxy_port, proxy_params = _parse(proxy)
|
||||
scheme = _parse(request.url)[0]
|
||||
proxy_host = proxy_host.decode()
|
||||
omit_connect_tunnel = b'noconnect' in proxy_params
|
||||
if omit_connect_tunnel:
|
||||
warnings.warn(
|
||||
"Using HTTPS proxies in the noconnect mode is not "
|
||||
"supported by the downloader handler. If you use Zyte "
|
||||
"Smart Proxy Manager, it doesn't require this mode "
|
||||
"anymore, so you should update scrapy-crawlera to "
|
||||
"scrapy-zyte-smartproxy and remove '?noconnect' from the "
|
||||
"Zyte Smart Proxy Manager URL."
|
||||
)
|
||||
|
||||
if scheme == b'https' and not omit_connect_tunnel:
|
||||
if scheme == b'https':
|
||||
# ToDo
|
||||
raise NotImplementedError('Tunneling via CONNECT method using HTTP/2.0 is not yet supported')
|
||||
return self._ProxyAgent(
|
||||
|
|
|
|||
|
|
@ -257,9 +257,7 @@ class ExecutionEngine:
|
|||
|
||||
def download(self, request: Request, spider: Optional[Spider] = None) -> Deferred:
|
||||
"""Return a Deferred which fires with a Response as result, only downloader middlewares are applied"""
|
||||
if spider is None:
|
||||
spider = self.spider
|
||||
else:
|
||||
if spider is not None:
|
||||
warnings.warn(
|
||||
"Passing a 'spider' argument to ExecutionEngine.download is deprecated",
|
||||
category=ScrapyDeprecationWarning,
|
||||
|
|
@ -267,7 +265,7 @@ class ExecutionEngine:
|
|||
)
|
||||
if spider is not self.spider:
|
||||
logger.warning("The spider '%s' does not match the open spider", spider.name)
|
||||
if spider is None:
|
||||
if self.spider is None:
|
||||
raise RuntimeError(f"No open spider to crawl: {request}")
|
||||
return self._download(request, spider).addBoth(self._downloaded, request, spider)
|
||||
|
||||
|
|
@ -278,11 +276,14 @@ class ExecutionEngine:
|
|||
self.slot.remove_request(request)
|
||||
return self.download(result, spider) if isinstance(result, Request) else result
|
||||
|
||||
def _download(self, request: Request, spider: Spider) -> Deferred:
|
||||
def _download(self, request: Request, spider: Optional[Spider]) -> Deferred:
|
||||
assert self.slot is not None # typing
|
||||
|
||||
self.slot.add_request(request)
|
||||
|
||||
if spider is None:
|
||||
spider = self.spider
|
||||
|
||||
def _on_success(result: Union[Response, Request]) -> Union[Response, Request]:
|
||||
if not isinstance(result, (Response, Request)):
|
||||
raise TypeError(f"Incorrect type: expected Response or Request, got {type(result)}: {result!r}")
|
||||
|
|
|
|||
|
|
@ -151,11 +151,9 @@ class Stream:
|
|||
|
||||
self._deferred_response = Deferred(_cancel)
|
||||
|
||||
def __str__(self) -> str:
|
||||
def __repr__(self) -> str:
|
||||
return f'Stream(id={self.stream_id!r})'
|
||||
|
||||
__repr__ = __str__
|
||||
|
||||
@property
|
||||
def _log_warnsize(self) -> bool:
|
||||
"""Checks if we have received data which exceeds the download warnsize
|
||||
|
|
|
|||
|
|
@ -34,7 +34,12 @@ from scrapy.utils.log import (
|
|||
)
|
||||
from scrapy.utils.misc import create_instance, load_object
|
||||
from scrapy.utils.ossignal import install_shutdown_handlers, signal_names
|
||||
from scrapy.utils.reactor import install_reactor, verify_installed_reactor
|
||||
from scrapy.utils.reactor import (
|
||||
install_reactor,
|
||||
is_asyncio_reactor_installed,
|
||||
verify_installed_asyncio_event_loop,
|
||||
verify_installed_reactor,
|
||||
)
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from scrapy.utils.request import RequestFingerprinter
|
||||
|
|
@ -84,17 +89,20 @@ class Crawler:
|
|||
crawler=self,
|
||||
)
|
||||
|
||||
reactor_class = self.settings.get("TWISTED_REACTOR")
|
||||
reactor_class = self.settings["TWISTED_REACTOR"]
|
||||
event_loop = self.settings["ASYNCIO_EVENT_LOOP"]
|
||||
if init_reactor:
|
||||
# this needs to be done after the spider settings are merged,
|
||||
# but before something imports twisted.internet.reactor
|
||||
if reactor_class:
|
||||
install_reactor(reactor_class, self.settings["ASYNCIO_EVENT_LOOP"])
|
||||
install_reactor(reactor_class, event_loop)
|
||||
else:
|
||||
from twisted.internet import reactor # noqa: F401
|
||||
log_reactor_info()
|
||||
if reactor_class:
|
||||
verify_installed_reactor(reactor_class)
|
||||
if is_asyncio_reactor_installed() and event_loop:
|
||||
verify_installed_asyncio_event_loop(event_loop)
|
||||
|
||||
self.extensions = ExtensionManager.from_crawler(self)
|
||||
|
||||
|
|
|
|||
|
|
@ -78,4 +78,7 @@ class HttpProxyMiddleware:
|
|||
del request.headers[b'Proxy-Authorization']
|
||||
del request.meta['_auth_proxy']
|
||||
elif b'Proxy-Authorization' in request.headers:
|
||||
del request.headers[b'Proxy-Authorization']
|
||||
if proxy_url:
|
||||
request.meta['_auth_proxy'] = proxy_url
|
||||
else:
|
||||
del request.headers[b'Proxy-Authorization']
|
||||
|
|
|
|||
|
|
@ -75,15 +75,16 @@ class MemoryUsage:
|
|||
self.crawler.stats.max_value('memusage/max', self.get_virtual_size())
|
||||
|
||||
def _check_limit(self):
|
||||
if self.get_virtual_size() > self.limit:
|
||||
peak_mem_usage = self.get_virtual_size()
|
||||
if peak_mem_usage > self.limit:
|
||||
self.crawler.stats.set_value('memusage/limit_reached', 1)
|
||||
mem = self.limit / 1024 / 1024
|
||||
logger.error("Memory usage exceeded %(memusage)dM. Shutting down Scrapy...",
|
||||
logger.error("Memory usage exceeded %(memusage)dMiB. Shutting down Scrapy...",
|
||||
{'memusage': mem}, extra={'crawler': self.crawler})
|
||||
if self.notify_mails:
|
||||
subj = (
|
||||
f"{self.crawler.settings['BOT_NAME']} terminated: "
|
||||
f"memory usage exceeded {mem}M at {socket.gethostname()}"
|
||||
f"memory usage exceeded {mem}MiB at {socket.gethostname()}"
|
||||
)
|
||||
self._send_report(self.notify_mails, subj)
|
||||
self.crawler.stats.set_value('memusage/limit_notified', 1)
|
||||
|
|
@ -92,6 +93,8 @@ class MemoryUsage:
|
|||
self.crawler.engine.close_spider(self.crawler.engine.spider, 'memusage_exceeded')
|
||||
else:
|
||||
self.crawler.stop()
|
||||
else:
|
||||
logger.info("Peak memory usage is %(virtualsize)dMiB", {'virtualsize': peak_mem_usage / 1024 / 1024})
|
||||
|
||||
def _check_warning(self):
|
||||
if self.warned: # warn only once
|
||||
|
|
@ -99,12 +102,12 @@ class MemoryUsage:
|
|||
if self.get_virtual_size() > self.warning:
|
||||
self.crawler.stats.set_value('memusage/warning_reached', 1)
|
||||
mem = self.warning / 1024 / 1024
|
||||
logger.warning("Memory usage reached %(memusage)dM",
|
||||
logger.warning("Memory usage reached %(memusage)dMiB",
|
||||
{'memusage': mem}, extra={'crawler': self.crawler})
|
||||
if self.notify_mails:
|
||||
subj = (
|
||||
f"{self.crawler.settings['BOT_NAME']} warning: "
|
||||
f"memory usage reached {mem}M at {socket.gethostname()}"
|
||||
f"memory usage reached {mem}MiB at {socket.gethostname()}"
|
||||
)
|
||||
self._send_report(self.notify_mails, subj)
|
||||
self.crawler.stats.set_value('memusage/warning_notified', 1)
|
||||
|
|
|
|||
|
|
@ -121,11 +121,9 @@ class Request(object_ref):
|
|||
def encoding(self) -> str:
|
||||
return self._encoding
|
||||
|
||||
def __str__(self) -> str:
|
||||
def __repr__(self) -> str:
|
||||
return f"<{self.method} {self.url}>"
|
||||
|
||||
__repr__ = __str__
|
||||
|
||||
def copy(self) -> "Request":
|
||||
return self.replace()
|
||||
|
||||
|
|
|
|||
|
|
@ -100,11 +100,9 @@ class Response(object_ref):
|
|||
|
||||
body = property(_get_body, obsolete_setter(_set_body, 'body'))
|
||||
|
||||
def __str__(self):
|
||||
def __repr__(self):
|
||||
return f"<{self.status} {self.url}>"
|
||||
|
||||
__repr__ = __str__
|
||||
|
||||
def copy(self):
|
||||
"""Return a copy of this Response"""
|
||||
return self.replace()
|
||||
|
|
|
|||
|
|
@ -6,18 +6,6 @@ This package contains a collection of Link Extractors.
|
|||
For more info see docs/topics/link-extractors.rst
|
||||
"""
|
||||
import re
|
||||
from urllib.parse import urlparse
|
||||
from warnings import warn
|
||||
|
||||
from parsel.csstranslator import HTMLTranslator
|
||||
from w3lib.url import canonicalize_url
|
||||
|
||||
from scrapy.utils.deprecate import ScrapyDeprecationWarning
|
||||
from scrapy.utils.misc import arg_to_iter
|
||||
from scrapy.utils.url import (
|
||||
url_is_from_any_domain, url_has_any_extension,
|
||||
)
|
||||
|
||||
|
||||
# common file extensions that are not followed if they occur in links
|
||||
IGNORED_EXTENSIONS = [
|
||||
|
|
@ -55,82 +43,5 @@ def _is_valid_url(url):
|
|||
return url.split('://', 1)[0] in {'http', 'https', 'file', 'ftp'}
|
||||
|
||||
|
||||
class FilteringLinkExtractor:
|
||||
|
||||
_csstranslator = HTMLTranslator()
|
||||
|
||||
def __new__(cls, *args, **kwargs):
|
||||
from scrapy.linkextractors.lxmlhtml import LxmlLinkExtractor
|
||||
if issubclass(cls, FilteringLinkExtractor) and not issubclass(cls, LxmlLinkExtractor):
|
||||
warn('scrapy.linkextractors.FilteringLinkExtractor is deprecated, '
|
||||
'please use scrapy.linkextractors.LinkExtractor instead',
|
||||
ScrapyDeprecationWarning, stacklevel=2)
|
||||
return super().__new__(cls)
|
||||
|
||||
def __init__(self, link_extractor, allow, deny, allow_domains, deny_domains,
|
||||
restrict_xpaths, canonicalize, deny_extensions, restrict_css, restrict_text):
|
||||
|
||||
self.link_extractor = link_extractor
|
||||
|
||||
self.allow_res = [x if isinstance(x, _re_type) else re.compile(x)
|
||||
for x in arg_to_iter(allow)]
|
||||
self.deny_res = [x if isinstance(x, _re_type) else re.compile(x)
|
||||
for x in arg_to_iter(deny)]
|
||||
|
||||
self.allow_domains = set(arg_to_iter(allow_domains))
|
||||
self.deny_domains = set(arg_to_iter(deny_domains))
|
||||
|
||||
self.restrict_xpaths = tuple(arg_to_iter(restrict_xpaths))
|
||||
self.restrict_xpaths += tuple(map(self._csstranslator.css_to_xpath,
|
||||
arg_to_iter(restrict_css)))
|
||||
|
||||
self.canonicalize = canonicalize
|
||||
if deny_extensions is None:
|
||||
deny_extensions = IGNORED_EXTENSIONS
|
||||
self.deny_extensions = {'.' + e for e in arg_to_iter(deny_extensions)}
|
||||
self.restrict_text = [x if isinstance(x, _re_type) else re.compile(x)
|
||||
for x in arg_to_iter(restrict_text)]
|
||||
|
||||
def _link_allowed(self, link):
|
||||
if not _is_valid_url(link.url):
|
||||
return False
|
||||
if self.allow_res and not _matches(link.url, self.allow_res):
|
||||
return False
|
||||
if self.deny_res and _matches(link.url, self.deny_res):
|
||||
return False
|
||||
parsed_url = urlparse(link.url)
|
||||
if self.allow_domains and not url_is_from_any_domain(parsed_url, self.allow_domains):
|
||||
return False
|
||||
if self.deny_domains and url_is_from_any_domain(parsed_url, self.deny_domains):
|
||||
return False
|
||||
if self.deny_extensions and url_has_any_extension(parsed_url, self.deny_extensions):
|
||||
return False
|
||||
if self.restrict_text and not _matches(link.text, self.restrict_text):
|
||||
return False
|
||||
return True
|
||||
|
||||
def matches(self, url):
|
||||
|
||||
if self.allow_domains and not url_is_from_any_domain(url, self.allow_domains):
|
||||
return False
|
||||
if self.deny_domains and url_is_from_any_domain(url, self.deny_domains):
|
||||
return False
|
||||
|
||||
allowed = (regex.search(url) for regex in self.allow_res) if self.allow_res else [True]
|
||||
denied = (regex.search(url) for regex in self.deny_res) if self.deny_res else []
|
||||
return any(allowed) and not any(denied)
|
||||
|
||||
def _process_links(self, links):
|
||||
links = [x for x in links if self._link_allowed(x)]
|
||||
if self.canonicalize:
|
||||
for link in links:
|
||||
link.url = canonicalize_url(link.url)
|
||||
links = self.link_extractor._process_links(links)
|
||||
return links
|
||||
|
||||
def _extract_links(self, *args, **kwargs):
|
||||
return self.link_extractor._extract_links(*args, **kwargs)
|
||||
|
||||
|
||||
# Top-level imports
|
||||
from scrapy.linkextractors.lxmlhtml import LxmlLinkExtractor as LinkExtractor
|
||||
|
|
|
|||
|
|
@ -3,18 +3,20 @@ Link extractor based on lxml.html
|
|||
"""
|
||||
import operator
|
||||
from functools import partial
|
||||
from urllib.parse import urljoin
|
||||
from urllib.parse import urljoin, urlparse
|
||||
|
||||
import lxml.etree as etree
|
||||
from parsel.csstranslator import HTMLTranslator
|
||||
from w3lib.html import strip_html5_whitespace
|
||||
from w3lib.url import canonicalize_url, safe_url_string
|
||||
|
||||
from scrapy.link import Link
|
||||
from scrapy.linkextractors import FilteringLinkExtractor
|
||||
from scrapy.linkextractors import (IGNORED_EXTENSIONS, _is_valid_url, _matches,
|
||||
_re_type, re)
|
||||
from scrapy.utils.misc import arg_to_iter, rel_has_nofollow
|
||||
from scrapy.utils.python import unique as unique_list
|
||||
from scrapy.utils.response import get_base_url
|
||||
|
||||
from scrapy.utils.url import url_has_any_extension, url_is_from_any_domain
|
||||
|
||||
# from lxml/src/lxml/html/__init__.py
|
||||
XHTML_NAMESPACE = "http://www.w3.org/1999/xhtml"
|
||||
|
|
@ -98,7 +100,8 @@ class LxmlParserLinkExtractor:
|
|||
return links
|
||||
|
||||
|
||||
class LxmlLinkExtractor(FilteringLinkExtractor):
|
||||
class LxmlLinkExtractor:
|
||||
_csstranslator = HTMLTranslator()
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
|
|
@ -118,7 +121,7 @@ class LxmlLinkExtractor(FilteringLinkExtractor):
|
|||
restrict_text=None,
|
||||
):
|
||||
tags, attrs = set(arg_to_iter(tags)), set(arg_to_iter(attrs))
|
||||
lx = LxmlParserLinkExtractor(
|
||||
self.link_extractor = LxmlParserLinkExtractor(
|
||||
tag=partial(operator.contains, tags),
|
||||
attr=partial(operator.contains, attrs),
|
||||
unique=unique,
|
||||
|
|
@ -126,18 +129,64 @@ class LxmlLinkExtractor(FilteringLinkExtractor):
|
|||
strip=strip,
|
||||
canonicalized=canonicalize
|
||||
)
|
||||
super().__init__(
|
||||
link_extractor=lx,
|
||||
allow=allow,
|
||||
deny=deny,
|
||||
allow_domains=allow_domains,
|
||||
deny_domains=deny_domains,
|
||||
restrict_xpaths=restrict_xpaths,
|
||||
restrict_css=restrict_css,
|
||||
canonicalize=canonicalize,
|
||||
deny_extensions=deny_extensions,
|
||||
restrict_text=restrict_text,
|
||||
)
|
||||
self.allow_res = [x if isinstance(x, _re_type) else re.compile(x)
|
||||
for x in arg_to_iter(allow)]
|
||||
self.deny_res = [x if isinstance(x, _re_type) else re.compile(x)
|
||||
for x in arg_to_iter(deny)]
|
||||
|
||||
self.allow_domains = set(arg_to_iter(allow_domains))
|
||||
self.deny_domains = set(arg_to_iter(deny_domains))
|
||||
|
||||
self.restrict_xpaths = tuple(arg_to_iter(restrict_xpaths))
|
||||
self.restrict_xpaths += tuple(map(self._csstranslator.css_to_xpath,
|
||||
arg_to_iter(restrict_css)))
|
||||
|
||||
if deny_extensions is None:
|
||||
deny_extensions = IGNORED_EXTENSIONS
|
||||
self.canonicalize = canonicalize
|
||||
self.deny_extensions = {'.' + e for e in arg_to_iter(deny_extensions)}
|
||||
self.restrict_text = [x if isinstance(x, _re_type) else re.compile(x)
|
||||
for x in arg_to_iter(restrict_text)]
|
||||
|
||||
def _link_allowed(self, link):
|
||||
if not _is_valid_url(link.url):
|
||||
return False
|
||||
if self.allow_res and not _matches(link.url, self.allow_res):
|
||||
return False
|
||||
if self.deny_res and _matches(link.url, self.deny_res):
|
||||
return False
|
||||
parsed_url = urlparse(link.url)
|
||||
if self.allow_domains and not url_is_from_any_domain(parsed_url, self.allow_domains):
|
||||
return False
|
||||
if self.deny_domains and url_is_from_any_domain(parsed_url, self.deny_domains):
|
||||
return False
|
||||
if self.deny_extensions and url_has_any_extension(parsed_url, self.deny_extensions):
|
||||
return False
|
||||
if self.restrict_text and not _matches(link.text, self.restrict_text):
|
||||
return False
|
||||
return True
|
||||
|
||||
def matches(self, url):
|
||||
|
||||
if self.allow_domains and not url_is_from_any_domain(url, self.allow_domains):
|
||||
return False
|
||||
if self.deny_domains and url_is_from_any_domain(url, self.deny_domains):
|
||||
return False
|
||||
|
||||
allowed = (regex.search(url) for regex in self.allow_res) if self.allow_res else [True]
|
||||
denied = (regex.search(url) for regex in self.deny_res) if self.deny_res else []
|
||||
return any(allowed) and not any(denied)
|
||||
|
||||
def _process_links(self, links):
|
||||
links = [x for x in links if self._link_allowed(x)]
|
||||
if self.canonicalize:
|
||||
for link in links:
|
||||
link.url = canonicalize_url(link.url)
|
||||
links = self.link_extractor._process_links(links)
|
||||
return links
|
||||
|
||||
def _extract_links(self, *args, **kwargs):
|
||||
return self.link_extractor._extract_links(*args, **kwargs)
|
||||
|
||||
def extract_links(self, response):
|
||||
"""Returns a list of :class:`~scrapy.link.Link` objects from the
|
||||
|
|
|
|||
|
|
@ -5,23 +5,28 @@ See documentation in topics/media-pipeline.rst
|
|||
"""
|
||||
import functools
|
||||
import hashlib
|
||||
import warnings
|
||||
from contextlib import suppress
|
||||
from io import BytesIO
|
||||
|
||||
from itemadapter import ItemAdapter
|
||||
|
||||
from scrapy.exceptions import DropItem, NotConfigured
|
||||
from scrapy.exceptions import DropItem, NotConfigured, ScrapyDeprecationWarning
|
||||
from scrapy.http import Request
|
||||
from scrapy.pipelines.files import FileException, FilesPipeline
|
||||
# TODO: from scrapy.pipelines.media import MediaPipeline
|
||||
from scrapy.settings import Settings
|
||||
from scrapy.utils.misc import md5sum
|
||||
from scrapy.utils.python import to_bytes
|
||||
from scrapy.utils.python import get_func_args, to_bytes
|
||||
|
||||
|
||||
class NoimagesDrop(DropItem):
|
||||
"""Product with no images exception"""
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
warnings.warn("The NoimagesDrop class is deprecated", category=ScrapyDeprecationWarning, stacklevel=2)
|
||||
super().__init__(*args, **kwargs)
|
||||
|
||||
|
||||
class ImageException(FileException):
|
||||
"""General image error exception"""
|
||||
|
|
@ -87,6 +92,8 @@ class ImagesPipeline(FilesPipeline):
|
|||
resolve('IMAGES_THUMBS'), self.THUMBS
|
||||
)
|
||||
|
||||
self._deprecated_convert_image = None
|
||||
|
||||
@classmethod
|
||||
def from_settings(cls, settings):
|
||||
s3store = cls.STORE_SCHEMES['s3']
|
||||
|
|
@ -137,15 +144,33 @@ class ImagesPipeline(FilesPipeline):
|
|||
f"({width}x{height} < "
|
||||
f"{self.min_width}x{self.min_height})")
|
||||
|
||||
image, buf = self.convert_image(orig_image)
|
||||
if self._deprecated_convert_image is None:
|
||||
self._deprecated_convert_image = 'response_body' not in get_func_args(self.convert_image)
|
||||
if self._deprecated_convert_image:
|
||||
warnings.warn(f'{self.__class__.__name__}.convert_image() method overriden in a deprecated way, '
|
||||
'overriden method does not accept response_body argument.',
|
||||
category=ScrapyDeprecationWarning)
|
||||
|
||||
if self._deprecated_convert_image:
|
||||
image, buf = self.convert_image(orig_image)
|
||||
else:
|
||||
image, buf = self.convert_image(orig_image, response_body=BytesIO(response.body))
|
||||
yield path, image, buf
|
||||
|
||||
for thumb_id, size in self.thumbs.items():
|
||||
thumb_path = self.thumb_path(request, thumb_id, response=response, info=info, item=item)
|
||||
thumb_image, thumb_buf = self.convert_image(image, size)
|
||||
if self._deprecated_convert_image:
|
||||
thumb_image, thumb_buf = self.convert_image(image, size)
|
||||
else:
|
||||
thumb_image, thumb_buf = self.convert_image(image, size, buf)
|
||||
yield thumb_path, thumb_image, thumb_buf
|
||||
|
||||
def convert_image(self, image, size=None):
|
||||
def convert_image(self, image, size=None, response_body=None):
|
||||
if response_body is None:
|
||||
warnings.warn(f'{self.__class__.__name__}.convert_image() method called in a deprecated way, '
|
||||
'method called without response_body argument.',
|
||||
category=ScrapyDeprecationWarning, stacklevel=2)
|
||||
|
||||
if image.format == 'PNG' and image.mode == 'RGBA':
|
||||
background = self._Image.new('RGBA', image.size, (255, 255, 255))
|
||||
background.paste(image, image)
|
||||
|
|
@ -160,7 +185,16 @@ class ImagesPipeline(FilesPipeline):
|
|||
|
||||
if size:
|
||||
image = image.copy()
|
||||
image.thumbnail(size, self._Image.ANTIALIAS)
|
||||
try:
|
||||
# Image.Resampling.LANCZOS was added in Pillow 9.1.0
|
||||
# remove this try except block,
|
||||
# when updating the minimum requirements for Pillow.
|
||||
resampling_filter = self._Image.Resampling.LANCZOS
|
||||
except AttributeError:
|
||||
resampling_filter = self._Image.ANTIALIAS
|
||||
image.thumbnail(size, resampling_filter)
|
||||
elif response_body is not None and image.format == 'JPEG':
|
||||
return image, response_body
|
||||
|
||||
buf = BytesIO()
|
||||
image.save(buf, 'JPEG')
|
||||
|
|
|
|||
|
|
@ -51,11 +51,9 @@ class SettingsAttribute:
|
|||
self.value = value
|
||||
self.priority = priority
|
||||
|
||||
def __str__(self):
|
||||
def __repr__(self):
|
||||
return f"<SettingsAttribute value={self.value!r} priority={self.priority}>"
|
||||
|
||||
__repr__ = __str__
|
||||
|
||||
|
||||
class BaseSettings(MutableMapping):
|
||||
"""
|
||||
|
|
@ -437,30 +435,6 @@ class BaseSettings(MutableMapping):
|
|||
p.text(pformat(self.copy_to_dict()))
|
||||
|
||||
|
||||
class _DictProxy(MutableMapping):
|
||||
|
||||
def __init__(self, settings, priority):
|
||||
self.o = {}
|
||||
self.settings = settings
|
||||
self.priority = priority
|
||||
|
||||
def __len__(self):
|
||||
return len(self.o)
|
||||
|
||||
def __getitem__(self, k):
|
||||
return self.o[k]
|
||||
|
||||
def __setitem__(self, k, v):
|
||||
self.settings.set(k, v, priority=self.priority)
|
||||
self.o[k] = v
|
||||
|
||||
def __delitem__(self, k):
|
||||
del self.o[k]
|
||||
|
||||
def __iter__(self, k, v):
|
||||
return iter(self.o)
|
||||
|
||||
|
||||
class Settings(BaseSettings):
|
||||
"""
|
||||
This object stores Scrapy settings for the configuration of internal
|
||||
|
|
|
|||
|
|
@ -248,7 +248,7 @@ REFERER_ENABLED = True
|
|||
REFERRER_POLICY = 'scrapy.spidermiddlewares.referer.DefaultReferrerPolicy'
|
||||
|
||||
REQUEST_FINGERPRINTER_CLASS = 'scrapy.utils.request.RequestFingerprinter'
|
||||
REQUEST_FINGERPRINTER_IMPLEMENTATION = 'PREVIOUS_VERSION'
|
||||
REQUEST_FINGERPRINTER_IMPLEMENTATION = '2.6'
|
||||
|
||||
RETRY_ENABLED = True
|
||||
RETRY_TIMES = 2 # initial response + 2 retries = 3 requests
|
||||
|
|
|
|||
|
|
@ -88,11 +88,9 @@ class Spider(object_ref):
|
|||
if callable(closed):
|
||||
return closed(reason)
|
||||
|
||||
def __str__(self):
|
||||
def __repr__(self):
|
||||
return f"<{type(self).__name__} {self.name!r} at 0x{id(self):0x}>"
|
||||
|
||||
__repr__ = __str__
|
||||
|
||||
|
||||
# Top-level imports
|
||||
from scrapy.spiders.crawl import CrawlSpider, Rule
|
||||
|
|
|
|||
|
|
@ -102,9 +102,9 @@ class CrawlSpider(Spider):
|
|||
request = self._build_request(rule_index, link)
|
||||
yield rule.process_request(request, response)
|
||||
|
||||
def _callback(self, response):
|
||||
def _callback(self, response, **cb_kwargs):
|
||||
rule = self._rules[response.meta['rule']]
|
||||
return self._parse_response(response, rule.callback, rule.cb_kwargs, rule.follow)
|
||||
return self._parse_response(response, rule.callback, {**rule.cb_kwargs, **cb_kwargs}, rule.follow)
|
||||
|
||||
def _errback(self, failure):
|
||||
rule = self._rules[failure.request.meta['rule']]
|
||||
|
|
|
|||
|
|
@ -88,4 +88,5 @@ ROBOTSTXT_OBEY = True
|
|||
#HTTPCACHE_STORAGE = 'scrapy.extensions.httpcache.FilesystemCacheStorage'
|
||||
|
||||
# Set settings whose default value is deprecated to a future-proof value
|
||||
REQUEST_FINGERPRINTER_IMPLEMENTATION = 'VERSION'
|
||||
REQUEST_FINGERPRINTER_IMPLEMENTATION = '2.7'
|
||||
TWISTED_REACTOR = 'twisted.internet.asyncioreactor.AsyncioSelectorReactor'
|
||||
|
|
|
|||
|
|
@ -1,27 +1,4 @@
|
|||
"""Boto/botocore helpers"""
|
||||
import warnings
|
||||
|
||||
from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning
|
||||
|
||||
|
||||
def is_botocore():
|
||||
""" Returns True if botocore is available, otherwise raises NotConfigured. Never returns False.
|
||||
|
||||
Previously, when boto was supported in addition to botocore, this returned False if boto was available
|
||||
but botocore wasn't.
|
||||
"""
|
||||
message = (
|
||||
'is_botocore() is deprecated and always returns True or raises an Exception, '
|
||||
'so it cannot be used for checking if boto is available instead of botocore. '
|
||||
'You can use scrapy.utils.boto.is_botocore_available() to check if botocore '
|
||||
'is available.'
|
||||
)
|
||||
warnings.warn(message, ScrapyDeprecationWarning, stacklevel=2)
|
||||
try:
|
||||
import botocore # noqa: F401
|
||||
return True
|
||||
except ImportError:
|
||||
raise NotConfigured('missing botocore library')
|
||||
|
||||
|
||||
def is_botocore_available():
|
||||
|
|
|
|||
|
|
@ -46,14 +46,12 @@ def build_component_list(compdict, custom=None, convert=update_classpath):
|
|||
raise ValueError(f'Invalid value {value} for component {name}, '
|
||||
'please provide a real number or None instead')
|
||||
|
||||
# BEGIN Backward compatibility for old (base, custom) call signature
|
||||
if isinstance(custom, (list, tuple)):
|
||||
_check_components(custom)
|
||||
return type(custom)(convert(c) for c in custom)
|
||||
|
||||
if custom is not None:
|
||||
compdict.update(custom)
|
||||
# END Backward compatibility
|
||||
|
||||
_validate_values(compdict)
|
||||
compdict = without_none_values(_map_keys(compdict))
|
||||
|
|
@ -157,6 +155,15 @@ def feed_process_params_from_cli(settings, output: List[str], output_format=None
|
|||
raise UsageError(
|
||||
"Please use only one of -o/--output and -O/--overwrite-output"
|
||||
)
|
||||
if output_format:
|
||||
raise UsageError(
|
||||
"-t/--output-format is a deprecated command line option"
|
||||
" and does not work in combination with -O/--overwrite-output."
|
||||
" To specify a format please specify it after a colon at the end of the"
|
||||
" output URI (i.e. -O <URI>:<FORMAT>)."
|
||||
" Example working in the tutorial: "
|
||||
"scrapy crawl quotes -O quotes.json:json"
|
||||
)
|
||||
output = overwrite_output
|
||||
overwrite = True
|
||||
|
||||
|
|
@ -164,9 +171,13 @@ def feed_process_params_from_cli(settings, output: List[str], output_format=None
|
|||
if len(output) == 1:
|
||||
check_valid_format(output_format)
|
||||
message = (
|
||||
'The -t command line option is deprecated in favor of '
|
||||
'specifying the output format within the output URI. See the '
|
||||
'documentation of the -o and -O options for more information.'
|
||||
"The -t/--output-format command line option is deprecated in favor of "
|
||||
"specifying the output format within the output URI using the -o/--output or the"
|
||||
" -O/--overwrite-output option (i.e. -o/-O <URI>:<FORMAT>). See the documentation"
|
||||
" of the -o or -O option or the following examples for more information. "
|
||||
"Examples working in the tutorial: "
|
||||
"scrapy crawl quotes -o quotes.csv:csv or "
|
||||
"scrapy crawl quotes -O quotes.json:json"
|
||||
)
|
||||
warnings.warn(message, ScrapyDeprecationWarning, stacklevel=2)
|
||||
return {output[0]: {'format': output_format}}
|
||||
|
|
|
|||
|
|
@ -26,7 +26,7 @@ from twisted.python import failure
|
|||
from twisted.python.failure import Failure
|
||||
|
||||
from scrapy.exceptions import IgnoreRequest
|
||||
from scrapy.utils.reactor import is_asyncio_reactor_installed
|
||||
from scrapy.utils.reactor import is_asyncio_reactor_installed, get_asyncio_event_loop_policy
|
||||
|
||||
|
||||
def defer_fail(_failure: Failure) -> Deferred:
|
||||
|
|
@ -269,7 +269,8 @@ def deferred_from_coro(o) -> Any:
|
|||
return ensureDeferred(o)
|
||||
else:
|
||||
# wrapping the coroutine into a Future and then into a Deferred, this requires AsyncioSelectorReactor
|
||||
return Deferred.fromFuture(asyncio.ensure_future(o))
|
||||
event_loop = get_asyncio_event_loop_policy().get_event_loop()
|
||||
return Deferred.fromFuture(asyncio.ensure_future(o, loop=event_loop))
|
||||
return o
|
||||
|
||||
|
||||
|
|
@ -320,7 +321,8 @@ def deferred_to_future(d: Deferred) -> Future:
|
|||
d = treq.get('https://example.com/additional')
|
||||
additional_response = await deferred_to_future(d)
|
||||
"""
|
||||
return d.asFuture(asyncio.get_event_loop())
|
||||
policy = get_asyncio_event_loop_policy()
|
||||
return d.asFuture(policy.get_event_loop())
|
||||
|
||||
|
||||
def maybe_deferred_to_future(d: Deferred) -> Union[Deferred, Future]:
|
||||
|
|
|
|||
|
|
@ -2,6 +2,7 @@
|
|||
|
||||
import warnings
|
||||
import inspect
|
||||
from typing import List, Tuple
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
|
||||
|
||||
|
|
@ -126,9 +127,7 @@ def _clspath(cls, forced=None):
|
|||
return f'{cls.__module__}.{cls.__name__}'
|
||||
|
||||
|
||||
DEPRECATION_RULES = [
|
||||
('scrapy.telnet.', 'scrapy.extensions.telnet.'),
|
||||
]
|
||||
DEPRECATION_RULES: List[Tuple[str, str]] = []
|
||||
|
||||
|
||||
def update_classpath(path):
|
||||
|
|
|
|||
|
|
@ -2,17 +2,6 @@ import struct
|
|||
from gzip import GzipFile
|
||||
from io import BytesIO
|
||||
|
||||
from scrapy.utils.decorators import deprecated
|
||||
|
||||
|
||||
# - GzipFile's read() has issues returning leftover uncompressed data when
|
||||
# input is corrupted
|
||||
# - read1(), which fetches data before raising EOFError on next call
|
||||
# works here
|
||||
@deprecated('GzipFile.read1')
|
||||
def read1(gzf, size=-1):
|
||||
return gzf.read1(size)
|
||||
|
||||
|
||||
def gunzip(data):
|
||||
"""Gunzip the given data and return as much data as possible.
|
||||
|
|
|
|||
|
|
@ -9,6 +9,7 @@ from collections import deque
|
|||
from contextlib import contextmanager
|
||||
from importlib import import_module
|
||||
from pkgutil import iter_modules
|
||||
from functools import partial
|
||||
|
||||
from w3lib.html import replace_entities
|
||||
|
||||
|
|
@ -226,7 +227,18 @@ def is_generator_with_return_value(callable):
|
|||
return value is None or isinstance(value, ast.NameConstant) and value.value is None
|
||||
|
||||
if inspect.isgeneratorfunction(callable):
|
||||
code = re.sub(r"^[\t ]+", "", inspect.getsource(callable))
|
||||
func = callable
|
||||
while isinstance(func, partial):
|
||||
func = func.func
|
||||
|
||||
src = inspect.getsource(func)
|
||||
pattern = re.compile(r"(^[\t ]+)")
|
||||
code = pattern.sub("", src)
|
||||
|
||||
match = pattern.match(src) # finds indentation
|
||||
if match:
|
||||
code = re.sub(f"\n{match.group(0)}", "\n", code) # remove indentation
|
||||
|
||||
tree = ast.parse(code)
|
||||
for node in walk_callable(tree):
|
||||
if isinstance(node, ast.Return) and not returns_none(node):
|
||||
|
|
|
|||
|
|
@ -6,7 +6,7 @@ from pathlib import Path
|
|||
|
||||
from scrapy.utils.conf import closest_scrapy_cfg, get_config, init_env
|
||||
from scrapy.settings import Settings
|
||||
from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning
|
||||
from scrapy.exceptions import NotConfigured
|
||||
|
||||
|
||||
ENVVAR = 'SCRAPY_SETTINGS_MODULE'
|
||||
|
|
@ -68,23 +68,16 @@ def get_project_settings():
|
|||
if settings_module_path:
|
||||
settings.setmodule(settings_module_path, priority='project')
|
||||
|
||||
scrapy_envvars = {k[7:]: v for k, v in os.environ.items() if
|
||||
k.startswith('SCRAPY_')}
|
||||
valid_envvars = {
|
||||
'CHECK',
|
||||
'PROJECT',
|
||||
'PYTHON_SHELL',
|
||||
'SETTINGS_MODULE',
|
||||
}
|
||||
setting_envvars = {k for k in scrapy_envvars if k not in valid_envvars}
|
||||
if setting_envvars:
|
||||
setting_envvar_list = ', '.join(sorted(setting_envvars))
|
||||
warnings.warn(
|
||||
'Use of environment variables prefixed with SCRAPY_ to override '
|
||||
'settings is deprecated. The following environment variables are '
|
||||
f'currently defined: {setting_envvar_list}',
|
||||
ScrapyDeprecationWarning
|
||||
)
|
||||
|
||||
scrapy_envvars = {k[7:]: v for k, v in os.environ.items() if
|
||||
k.startswith('SCRAPY_') and k.replace('SCRAPY_', '') in valid_envvars}
|
||||
|
||||
settings.setdict(scrapy_envvars, priority='project')
|
||||
|
||||
return settings
|
||||
|
|
|
|||
|
|
@ -1,20 +1,16 @@
|
|||
"""
|
||||
This module contains essential stuff that should've come with Python itself ;)
|
||||
"""
|
||||
import errno
|
||||
import gc
|
||||
import inspect
|
||||
import re
|
||||
import sys
|
||||
import warnings
|
||||
import weakref
|
||||
from functools import partial, wraps
|
||||
from itertools import chain
|
||||
from typing import AsyncGenerator, AsyncIterable, Iterable, Union
|
||||
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
from scrapy.utils.asyncgen import as_async_generator
|
||||
from scrapy.utils.decorators import deprecated
|
||||
|
||||
|
||||
def flatten(x):
|
||||
|
|
@ -112,12 +108,6 @@ def to_bytes(text, encoding=None, errors='strict'):
|
|||
return text.encode(encoding, errors)
|
||||
|
||||
|
||||
@deprecated('to_unicode')
|
||||
def to_native_str(text, encoding=None, errors='strict'):
|
||||
""" Return str representation of ``text``. """
|
||||
return to_unicode(text, encoding, errors)
|
||||
|
||||
|
||||
def re_rsearch(pattern, text, chunk_size=1024):
|
||||
"""
|
||||
This function does a reverse search in a text using a regular expression
|
||||
|
|
@ -263,30 +253,6 @@ def equal_attributes(obj1, obj2, attributes):
|
|||
return True
|
||||
|
||||
|
||||
class WeakKeyCache:
|
||||
|
||||
def __init__(self, default_factory):
|
||||
warnings.warn("The WeakKeyCache class is deprecated", category=ScrapyDeprecationWarning, stacklevel=2)
|
||||
self.default_factory = default_factory
|
||||
self._weakdict = weakref.WeakKeyDictionary()
|
||||
|
||||
def __getitem__(self, key):
|
||||
if key not in self._weakdict:
|
||||
self._weakdict[key] = self.default_factory(key)
|
||||
return self._weakdict[key]
|
||||
|
||||
|
||||
@deprecated
|
||||
def retry_on_eintr(function, *args, **kw):
|
||||
"""Run a function and retry it while getting EINTR errors"""
|
||||
while True:
|
||||
try:
|
||||
return function(*args, **kw)
|
||||
except IOError as e:
|
||||
if e.errno != errno.EINTR:
|
||||
raise
|
||||
|
||||
|
||||
def without_none_values(iterable):
|
||||
"""Return a copy of ``iterable`` with all ``None`` entries removed.
|
||||
|
||||
|
|
@ -337,10 +303,6 @@ class MutableChain(Iterable):
|
|||
def __next__(self):
|
||||
return next(self.data)
|
||||
|
||||
@deprecated("scrapy.utils.python.MutableChain.__next__")
|
||||
def next(self):
|
||||
return self.__next__()
|
||||
|
||||
|
||||
async def _async_chain(*iterables: Union[Iterable, AsyncIterable]) -> AsyncGenerator:
|
||||
for it in iterables:
|
||||
|
|
|
|||
|
|
@ -51,6 +51,19 @@ class CallLaterOnce:
|
|||
return self._func(*self._a, **self._kw)
|
||||
|
||||
|
||||
def get_asyncio_event_loop_policy():
|
||||
policy = asyncio.get_event_loop_policy()
|
||||
if (
|
||||
sys.version_info >= (3, 8)
|
||||
and sys.platform == "win32"
|
||||
and not isinstance(policy, asyncio.WindowsSelectorEventLoopPolicy)
|
||||
):
|
||||
policy = asyncio.WindowsSelectorEventLoopPolicy()
|
||||
asyncio.set_event_loop_policy(policy)
|
||||
|
||||
return policy
|
||||
|
||||
|
||||
def install_reactor(reactor_path, event_loop_path=None):
|
||||
"""Installs the :mod:`~twisted.internet.reactor` with the specified
|
||||
import path. Also installs the asyncio event loop with the specified import
|
||||
|
|
@ -58,16 +71,14 @@ def install_reactor(reactor_path, event_loop_path=None):
|
|||
reactor_class = load_object(reactor_path)
|
||||
if reactor_class is asyncioreactor.AsyncioSelectorReactor:
|
||||
with suppress(error.ReactorAlreadyInstalledError):
|
||||
if sys.version_info >= (3, 8) and sys.platform == "win32":
|
||||
policy = asyncio.get_event_loop_policy()
|
||||
if not isinstance(policy, asyncio.WindowsSelectorEventLoopPolicy):
|
||||
asyncio.set_event_loop_policy(asyncio.WindowsSelectorEventLoopPolicy())
|
||||
policy = get_asyncio_event_loop_policy()
|
||||
if event_loop_path is not None:
|
||||
event_loop_class = load_object(event_loop_path)
|
||||
event_loop = event_loop_class()
|
||||
asyncio.set_event_loop(event_loop)
|
||||
else:
|
||||
event_loop = asyncio.get_event_loop()
|
||||
event_loop = policy.get_event_loop()
|
||||
|
||||
asyncioreactor.install(eventloop=event_loop)
|
||||
else:
|
||||
*module, _ = reactor_path.split(".")
|
||||
|
|
@ -90,6 +101,24 @@ def verify_installed_reactor(reactor_path):
|
|||
raise Exception(msg)
|
||||
|
||||
|
||||
def verify_installed_asyncio_event_loop(loop_path):
|
||||
from twisted.internet import reactor
|
||||
loop_class = load_object(loop_path)
|
||||
if isinstance(reactor._asyncioEventloop, loop_class):
|
||||
return
|
||||
installed = (
|
||||
f"{reactor._asyncioEventloop.__class__.__module__}"
|
||||
f".{reactor._asyncioEventloop.__class__.__qualname__}"
|
||||
)
|
||||
specified = f"{loop_class.__module__}.{loop_class.__qualname__}"
|
||||
raise Exception(
|
||||
"Scrapy found an asyncio Twisted reactor already "
|
||||
f"installed, and its event loop class ({installed}) does "
|
||||
"not match the one specified in the ASYNCIO_EVENT_LOOP "
|
||||
f"setting ({specified})"
|
||||
)
|
||||
|
||||
|
||||
def is_asyncio_reactor_installed():
|
||||
from twisted.internet import reactor
|
||||
return isinstance(reactor, asyncioreactor.AsyncioSelectorReactor)
|
||||
|
|
|
|||
|
|
@ -236,10 +236,10 @@ class RequestFingerprinter:
|
|||
'REQUEST_FINGERPRINTER_IMPLEMENTATION'
|
||||
)
|
||||
else:
|
||||
implementation = 'PREVIOUS_VERSION'
|
||||
if implementation == 'PREVIOUS_VERSION':
|
||||
implementation = '2.6'
|
||||
if implementation == '2.6':
|
||||
message = (
|
||||
'\'PREVIOUS_VERSION\' is a deprecated value for the '
|
||||
'\'2.6\' is a deprecated value for the '
|
||||
'\'REQUEST_FINGERPRINTER_IMPLEMENTATION\' setting.\n'
|
||||
'\n'
|
||||
'It is also the default value. In other words, it is normal '
|
||||
|
|
@ -254,14 +254,14 @@ class RequestFingerprinter:
|
|||
)
|
||||
warnings.warn(message, category=ScrapyDeprecationWarning, stacklevel=2)
|
||||
self._fingerprint = _request_fingerprint_as_bytes
|
||||
elif implementation == 'VERSION':
|
||||
elif implementation == '2.7':
|
||||
self._fingerprint = fingerprint
|
||||
else:
|
||||
raise ValueError(
|
||||
f'Got an invalid value on setting '
|
||||
f'\'REQUEST_FINGERPRINTER_IMPLEMENTATION\': '
|
||||
f'{implementation!r}. Valid values are \'PREVIOUS_VERSION\' (deprecated) '
|
||||
f'and \'VERSION\'.'
|
||||
f'{implementation!r}. Valid values are \'2.6\' (deprecated) '
|
||||
f'and \'2.7\'.'
|
||||
)
|
||||
|
||||
def fingerprint(self, request: Request):
|
||||
|
|
|
|||
|
|
@ -66,7 +66,7 @@ def get_crawler(spidercls=None, settings_dict=None, prevent_warnings=True):
|
|||
# Set by default settings that prevent deprecation warnings.
|
||||
settings = {}
|
||||
if prevent_warnings:
|
||||
settings['REQUEST_FINGERPRINTER_IMPLEMENTATION'] = 'VERSION'
|
||||
settings['REQUEST_FINGERPRINTER_IMPLEMENTATION'] = '2.7'
|
||||
settings.update(settings_dict or {})
|
||||
runner = CrawlerRunner(settings)
|
||||
return runner.create_crawler(spidercls or Spider)
|
||||
|
|
|
|||
|
|
@ -34,6 +34,7 @@ def url_has_any_extension(url, extensions):
|
|||
lowercase_path = parse_url(url).path.lower()
|
||||
return any(lowercase_path.endswith(ext) for ext in extensions)
|
||||
|
||||
|
||||
def parse_url(url, encoding=None):
|
||||
"""Return urlparsed url from the given argument (which could be an already
|
||||
parsed url)
|
||||
|
|
|
|||
|
|
@ -0,0 +1,25 @@
|
|||
import asyncio
|
||||
import sys
|
||||
|
||||
from twisted.internet import asyncioreactor
|
||||
if sys.version_info >= (3, 8) and sys.platform == "win32":
|
||||
asyncio.set_event_loop_policy(asyncio.WindowsSelectorEventLoopPolicy())
|
||||
asyncioreactor.install(asyncio.get_event_loop())
|
||||
|
||||
import scrapy # noqa: E402
|
||||
from scrapy.crawler import CrawlerProcess # noqa: E402
|
||||
|
||||
|
||||
class NoRequestsSpider(scrapy.Spider):
|
||||
name = 'no_request'
|
||||
|
||||
def start_requests(self):
|
||||
return []
|
||||
|
||||
|
||||
process = CrawlerProcess(settings={
|
||||
"TWISTED_REACTOR": "twisted.internet.asyncioreactor.AsyncioSelectorReactor",
|
||||
"ASYNCIO_EVENT_LOOP": "uvloop.Loop",
|
||||
})
|
||||
process.crawl(NoRequestsSpider)
|
||||
process.start()
|
||||
|
|
@ -0,0 +1,28 @@
|
|||
import asyncio
|
||||
import sys
|
||||
|
||||
from uvloop import Loop
|
||||
|
||||
from twisted.internet import asyncioreactor
|
||||
if sys.version_info >= (3, 8) and sys.platform == "win32":
|
||||
asyncio.set_event_loop_policy(asyncio.WindowsSelectorEventLoopPolicy())
|
||||
asyncio.set_event_loop(Loop())
|
||||
asyncioreactor.install(asyncio.get_event_loop())
|
||||
|
||||
import scrapy # noqa: E402
|
||||
from scrapy.crawler import CrawlerProcess # noqa: E402
|
||||
|
||||
|
||||
class NoRequestsSpider(scrapy.Spider):
|
||||
name = 'no_request'
|
||||
|
||||
def start_requests(self):
|
||||
return []
|
||||
|
||||
|
||||
process = CrawlerProcess(settings={
|
||||
"TWISTED_REACTOR": "twisted.internet.asyncioreactor.AsyncioSelectorReactor",
|
||||
"ASYNCIO_EVENT_LOOP": "uvloop.Loop",
|
||||
})
|
||||
process.crawl(NoRequestsSpider)
|
||||
process.start()
|
||||
|
|
@ -419,6 +419,17 @@ class CrawlSpiderWithErrback(CrawlSpiderWithParseMethod):
|
|||
self.logger.info('[errback] status %i', failure.value.response.status)
|
||||
|
||||
|
||||
class CrawlSpiderWithProcessRequestCallbackKeywordArguments(CrawlSpiderWithParseMethod):
|
||||
name = 'crawl_spider_with_process_request_cb_kwargs'
|
||||
rules = (
|
||||
Rule(LinkExtractor(), callback='parse', follow=True, process_request="process_request"),
|
||||
)
|
||||
|
||||
def process_request(self, request, response):
|
||||
request.cb_kwargs["foo"] = "process_request"
|
||||
return request
|
||||
|
||||
|
||||
class BytesReceivedCallbackSpider(MetaSpider):
|
||||
|
||||
full_response_length = 2**18
|
||||
|
|
|
|||
|
|
@ -31,10 +31,6 @@ class CmdlineTest(unittest.TestCase):
|
|||
self.assertEqual(self._execute('settings', '--get', 'TEST1', '-s',
|
||||
'TEST1=override'), 'override')
|
||||
|
||||
def test_override_settings_using_envvar(self):
|
||||
self.env['SCRAPY_TEST1'] = 'override'
|
||||
self.assertEqual(self._execute('settings', '--get', 'TEST1'), 'override')
|
||||
|
||||
def test_profiling(self):
|
||||
path = Path(tempfile.mkdtemp())
|
||||
filename = path / 'res.prof'
|
||||
|
|
|
|||
|
|
@ -27,7 +27,15 @@ class ParseCommandTest(ProcessTest, SiteTest, CommandTest):
|
|||
import scrapy
|
||||
from scrapy.linkextractors import LinkExtractor
|
||||
from scrapy.spiders import CrawlSpider, Rule
|
||||
from scrapy.utils.test import get_from_asyncio_queue
|
||||
|
||||
class AsyncDefAsyncioSpider(scrapy.Spider):
|
||||
|
||||
name = 'asyncdef{self.spider_name}'
|
||||
|
||||
async def parse(self, response):
|
||||
status = await get_from_asyncio_queue(response.status)
|
||||
return [scrapy.Item(), dict(foo='bar')]
|
||||
|
||||
class MySpider(scrapy.Spider):
|
||||
name = '{self.spider_name}'
|
||||
|
|
@ -155,6 +163,13 @@ ITEM_PIPELINES = {{'{self.project_name}.pipelines.MyPipeline': 1}}
|
|||
self.url('/html')])
|
||||
self.assertIn("INFO: It Works!", _textmode(stderr))
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_asyncio_parse_items(self):
|
||||
status, out, stderr = yield self.execute(
|
||||
['--spider', 'asyncdef' + self.spider_name, '-c', 'parse', self.url('/html')]
|
||||
)
|
||||
self.assertIn("""[{}, {'foo': 'bar'}]""", _textmode(out))
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_parse_items(self):
|
||||
status, out, stderr = yield self.execute(
|
||||
|
|
|
|||
|
|
@ -689,8 +689,15 @@ class MySpider(scrapy.Spider):
|
|||
])
|
||||
self.assertIn("Using reactor: twisted.internet.asyncioreactor.AsyncioSelectorReactor", log)
|
||||
|
||||
def test_asyncio_enabled_false(self):
|
||||
def test_asyncio_enabled_default(self):
|
||||
log = self.get_log(self.debug_log_spider, args=[])
|
||||
self.assertIn("Using reactor: twisted.internet.asyncioreactor.AsyncioSelectorReactor", log)
|
||||
|
||||
def test_asyncio_enabled_false(self):
|
||||
log = self.get_log(self.debug_log_spider, args=[
|
||||
'-s', 'TWISTED_REACTOR=twisted.internet.selectreactor.SelectReactor'
|
||||
])
|
||||
self.assertIn("Using reactor: twisted.internet.selectreactor.SelectReactor", log)
|
||||
self.assertNotIn("Using reactor: twisted.internet.asyncioreactor.AsyncioSelectorReactor", log)
|
||||
|
||||
@mark.skipif(sys.implementation.name == 'pypy', reason='uvloop does not support pypy properly')
|
||||
|
|
|
|||
|
|
@ -41,6 +41,7 @@ from tests.spiders import (
|
|||
CrawlSpiderWithAsyncGeneratorCallback,
|
||||
CrawlSpiderWithErrback,
|
||||
CrawlSpiderWithParseMethod,
|
||||
CrawlSpiderWithProcessRequestCallbackKeywordArguments,
|
||||
DelaySpider,
|
||||
DuplicateStartRequestsSpider,
|
||||
FollowAllSpider,
|
||||
|
|
@ -350,7 +351,7 @@ with multiples lines
|
|||
|
||||
@defer.inlineCallbacks
|
||||
def test_crawl_multiple(self):
|
||||
runner = CrawlerRunner({'REQUEST_FINGERPRINTER_IMPLEMENTATION': 'VERSION'})
|
||||
runner = CrawlerRunner({'REQUEST_FINGERPRINTER_IMPLEMENTATION': '2.7'})
|
||||
runner.crawl(SimpleSpider, self.mockserver.url("/status?n=200"), mockserver=self.mockserver)
|
||||
runner.crawl(SimpleSpider, self.mockserver.url("/status?n=503"), mockserver=self.mockserver)
|
||||
|
||||
|
|
@ -426,6 +427,16 @@ class CrawlSpiderTestCase(TestCase):
|
|||
self.assertIn("[errback] status 500", str(log))
|
||||
self.assertIn("[errback] status 501", str(log))
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_crawlspider_process_request_cb_kwargs(self):
|
||||
crawler = get_crawler(CrawlSpiderWithProcessRequestCallbackKeywordArguments)
|
||||
with LogCapture() as log:
|
||||
yield crawler.crawl(mockserver=self.mockserver)
|
||||
|
||||
self.assertIn("[parse] status 200 (foo: process_request)", str(log))
|
||||
self.assertIn("[parse] status 201 (foo: process_request)", str(log))
|
||||
self.assertIn("[parse] status 202 (foo: bar)", str(log))
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_async_def_parse(self):
|
||||
crawler = get_crawler(AsyncDefSpider)
|
||||
|
|
|
|||
|
|
@ -109,7 +109,7 @@ class CrawlerLoggingTestCase(unittest.TestCase):
|
|||
'LOG_LEVEL': 'INFO',
|
||||
'LOG_FILE': str(log_file),
|
||||
# settings to avoid extra warnings
|
||||
'REQUEST_FINGERPRINTER_IMPLEMENTATION': 'VERSION',
|
||||
'REQUEST_FINGERPRINTER_IMPLEMENTATION': '2.7',
|
||||
'TELNETCONSOLE_ENABLED': telnet.TWISTED_CONCH_AVAILABLE,
|
||||
}
|
||||
|
||||
|
|
@ -231,7 +231,7 @@ class NoRequestsSpider(scrapy.Spider):
|
|||
class CrawlerRunnerHasSpider(unittest.TestCase):
|
||||
|
||||
def _runner(self):
|
||||
return CrawlerRunner({'REQUEST_FINGERPRINTER_IMPLEMENTATION': 'VERSION'})
|
||||
return CrawlerRunner({'REQUEST_FINGERPRINTER_IMPLEMENTATION': '2.7'})
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_crawler_runner_bootstrap_successful(self):
|
||||
|
|
@ -279,14 +279,14 @@ class CrawlerRunnerHasSpider(unittest.TestCase):
|
|||
if self.reactor_pytest == 'asyncio':
|
||||
CrawlerRunner(settings={
|
||||
"TWISTED_REACTOR": "twisted.internet.asyncioreactor.AsyncioSelectorReactor",
|
||||
"REQUEST_FINGERPRINTER_IMPLEMENTATION": "VERSION",
|
||||
"REQUEST_FINGERPRINTER_IMPLEMENTATION": "2.7",
|
||||
})
|
||||
else:
|
||||
msg = r"The installed reactor \(.*?\) does not match the requested one \(.*?\)"
|
||||
with self.assertRaisesRegex(Exception, msg):
|
||||
runner = CrawlerRunner(settings={
|
||||
"TWISTED_REACTOR": "twisted.internet.asyncioreactor.AsyncioSelectorReactor",
|
||||
"REQUEST_FINGERPRINTER_IMPLEMENTATION": "VERSION",
|
||||
"REQUEST_FINGERPRINTER_IMPLEMENTATION": "2.7",
|
||||
})
|
||||
yield runner.crawl(NoRequestsSpider)
|
||||
|
||||
|
|
@ -452,6 +452,29 @@ class CrawlerProcessSubprocess(ScriptRunnerMixin, unittest.TestCase):
|
|||
self.assertIn("Using asyncio event loop: uvloop.Loop", log)
|
||||
self.assertIn("async pipeline opened!", log)
|
||||
|
||||
@mark.skipif(sys.implementation.name == 'pypy', reason='uvloop does not support pypy properly')
|
||||
@mark.skipif(platform.system() == 'Windows', reason='uvloop does not support Windows')
|
||||
@mark.skipif(twisted_version == Version('twisted', 21, 2, 0), reason='https://twistedmatrix.com/trac/ticket/10106')
|
||||
def test_asyncio_enabled_reactor_same_loop(self):
|
||||
log = self.run_script("asyncio_enabled_reactor_same_loop.py")
|
||||
self.assertIn("Spider closed (finished)", log)
|
||||
self.assertIn("Using reactor: twisted.internet.asyncioreactor.AsyncioSelectorReactor", log)
|
||||
self.assertIn("Using asyncio event loop: uvloop.Loop", log)
|
||||
|
||||
@mark.skipif(sys.implementation.name == 'pypy', reason='uvloop does not support pypy properly')
|
||||
@mark.skipif(platform.system() == 'Windows', reason='uvloop does not support Windows')
|
||||
@mark.skipif(twisted_version == Version('twisted', 21, 2, 0), reason='https://twistedmatrix.com/trac/ticket/10106')
|
||||
def test_asyncio_enabled_reactor_different_loop(self):
|
||||
log = self.run_script("asyncio_enabled_reactor_different_loop.py")
|
||||
self.assertNotIn("Spider closed (finished)", log)
|
||||
self.assertIn(
|
||||
(
|
||||
"does not match the one specified in the ASYNCIO_EVENT_LOOP "
|
||||
"setting (uvloop.Loop)"
|
||||
),
|
||||
log,
|
||||
)
|
||||
|
||||
def test_default_loop_asyncio_deferred_signal(self):
|
||||
log = self.run_script("asyncio_deferred_signal.py")
|
||||
self.assertIn("Spider closed (finished)", log)
|
||||
|
|
|
|||
|
|
@ -24,7 +24,7 @@ from scrapy.core.downloader.handlers.http import HTTPDownloadHandler
|
|||
from scrapy.core.downloader.handlers.http10 import HTTP10DownloadHandler
|
||||
from scrapy.core.downloader.handlers.http11 import HTTP11DownloadHandler
|
||||
from scrapy.core.downloader.handlers.s3 import S3DownloadHandler
|
||||
from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning
|
||||
from scrapy.exceptions import NotConfigured
|
||||
from scrapy.http import Headers, HtmlResponse, Request
|
||||
from scrapy.http.response.text import TextResponse
|
||||
from scrapy.responsetypes import responsetypes
|
||||
|
|
@ -108,6 +108,7 @@ class LoadTestCase(unittest.TestCase):
|
|||
class FileTestCase(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
# add a special char to check that they are handled correctly
|
||||
self.tmpname = Path(self.mktemp() + '^')
|
||||
Path(self.tmpname).write_text('0123456789')
|
||||
handler = create_instance(FileDownloadHandler, None, get_crawler())
|
||||
|
|
@ -756,18 +757,6 @@ class HttpProxyTestCase(unittest.TestCase):
|
|||
request = Request('http://example.com', meta={'proxy': http_proxy})
|
||||
return self.download_request(request, Spider('foo')).addCallback(_test)
|
||||
|
||||
def test_download_with_proxy_https_noconnect(self):
|
||||
def _test(response):
|
||||
self.assertEqual(response.status, 200)
|
||||
self.assertEqual(response.url, request.url)
|
||||
self.assertEqual(response.body, b'https://example.com')
|
||||
|
||||
http_proxy = f'{self.getURL("")}?noconnect'
|
||||
request = Request('https://example.com', meta={'proxy': http_proxy})
|
||||
with self.assertWarnsRegex(ScrapyDeprecationWarning,
|
||||
r'Using HTTPS proxies in the noconnect mode is deprecated'):
|
||||
return self.download_request(request, Spider('foo')).addCallback(_test)
|
||||
|
||||
def test_download_without_proxy(self):
|
||||
def _test(response):
|
||||
self.assertEqual(response.status, 200)
|
||||
|
|
|
|||
|
|
@ -242,21 +242,6 @@ class Https2ProxyTestCase(Http11ProxyTestCase):
|
|||
def getURL(self, path):
|
||||
return f"{self.scheme}://{self.host}:{self.portno}/{path}"
|
||||
|
||||
def test_download_with_proxy_https_noconnect(self):
|
||||
def _test(response):
|
||||
self.assertEqual(response.status, 200)
|
||||
self.assertEqual(response.url, request.url)
|
||||
self.assertEqual(response.body, b'/')
|
||||
|
||||
http_proxy = f"{self.getURL('')}?noconnect"
|
||||
request = Request('https://example.com', meta={'proxy': http_proxy})
|
||||
with self.assertWarnsRegex(
|
||||
Warning,
|
||||
r'Using HTTPS proxies in the noconnect mode is not supported by the '
|
||||
r'downloader handler.'
|
||||
):
|
||||
return self.download_request(request, Spider('foo')).addCallback(_test)
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_download_with_proxy_https_timeout(self):
|
||||
with self.assertRaises(NotImplementedError):
|
||||
|
|
|
|||
|
|
@ -400,6 +400,9 @@ class TestHttpProxyMiddleware(TestCase):
|
|||
self.assertNotIn(b'Proxy-Authorization', request.headers)
|
||||
|
||||
def test_proxy_authentication_header_proxy_without_credentials(self):
|
||||
"""As long as the proxy URL in request metadata remains the same, the
|
||||
Proxy-Authorization header is used and kept, and may even be
|
||||
changed."""
|
||||
middleware = HttpProxyMiddleware()
|
||||
request = Request(
|
||||
'https://example.com',
|
||||
|
|
@ -408,7 +411,16 @@ class TestHttpProxyMiddleware(TestCase):
|
|||
)
|
||||
assert middleware.process_request(request, spider) is None
|
||||
self.assertEqual(request.meta['proxy'], 'https://example.com')
|
||||
self.assertNotIn(b'Proxy-Authorization', request.headers)
|
||||
self.assertEqual(request.headers['Proxy-Authorization'], b'Basic foo')
|
||||
|
||||
assert middleware.process_request(request, spider) is None
|
||||
self.assertEqual(request.meta['proxy'], 'https://example.com')
|
||||
self.assertEqual(request.headers['Proxy-Authorization'], b'Basic foo')
|
||||
|
||||
request.headers['Proxy-Authorization'] = b'Basic bar'
|
||||
assert middleware.process_request(request, spider) is None
|
||||
self.assertEqual(request.meta['proxy'], 'https://example.com')
|
||||
self.assertEqual(request.headers['Proxy-Authorization'], b'Basic bar')
|
||||
|
||||
def test_proxy_authentication_header_proxy_with_same_credentials(self):
|
||||
middleware = HttpProxyMiddleware()
|
||||
|
|
|
|||
|
|
@ -51,7 +51,7 @@ class RFPDupeFilterTest(unittest.TestCase):
|
|||
def test_df_from_crawler_scheduler(self):
|
||||
settings = {'DUPEFILTER_DEBUG': True,
|
||||
'DUPEFILTER_CLASS': FromCrawlerRFPDupeFilter,
|
||||
'REQUEST_FINGERPRINTER_IMPLEMENTATION': 'VERSION'}
|
||||
'REQUEST_FINGERPRINTER_IMPLEMENTATION': '2.7'}
|
||||
crawler = get_crawler(settings_dict=settings)
|
||||
scheduler = Scheduler.from_crawler(crawler)
|
||||
self.assertTrue(scheduler.df.debug)
|
||||
|
|
@ -60,7 +60,7 @@ class RFPDupeFilterTest(unittest.TestCase):
|
|||
def test_df_from_settings_scheduler(self):
|
||||
settings = {'DUPEFILTER_DEBUG': True,
|
||||
'DUPEFILTER_CLASS': FromSettingsRFPDupeFilter,
|
||||
'REQUEST_FINGERPRINTER_IMPLEMENTATION': 'VERSION'}
|
||||
'REQUEST_FINGERPRINTER_IMPLEMENTATION': '2.7'}
|
||||
crawler = get_crawler(settings_dict=settings)
|
||||
scheduler = Scheduler.from_crawler(crawler)
|
||||
self.assertTrue(scheduler.df.debug)
|
||||
|
|
@ -68,7 +68,7 @@ class RFPDupeFilterTest(unittest.TestCase):
|
|||
|
||||
def test_df_direct_scheduler(self):
|
||||
settings = {'DUPEFILTER_CLASS': DirectDupeFilter,
|
||||
'REQUEST_FINGERPRINTER_IMPLEMENTATION': 'VERSION'}
|
||||
'REQUEST_FINGERPRINTER_IMPLEMENTATION': '2.7'}
|
||||
crawler = get_crawler(settings_dict=settings)
|
||||
scheduler = Scheduler.from_crawler(crawler)
|
||||
self.assertEqual(scheduler.df.method, 'n/a')
|
||||
|
|
@ -172,7 +172,7 @@ class RFPDupeFilterTest(unittest.TestCase):
|
|||
with LogCapture() as log:
|
||||
settings = {'DUPEFILTER_DEBUG': False,
|
||||
'DUPEFILTER_CLASS': FromCrawlerRFPDupeFilter,
|
||||
'REQUEST_FINGERPRINTER_IMPLEMENTATION': 'VERSION'}
|
||||
'REQUEST_FINGERPRINTER_IMPLEMENTATION': '2.7'}
|
||||
crawler = get_crawler(SimpleSpider, settings_dict=settings)
|
||||
spider = SimpleSpider.from_crawler(crawler)
|
||||
dupefilter = _get_dupefilter(crawler=crawler)
|
||||
|
|
@ -199,7 +199,7 @@ class RFPDupeFilterTest(unittest.TestCase):
|
|||
with LogCapture() as log:
|
||||
settings = {'DUPEFILTER_DEBUG': True,
|
||||
'DUPEFILTER_CLASS': FromCrawlerRFPDupeFilter,
|
||||
'REQUEST_FINGERPRINTER_IMPLEMENTATION': 'VERSION'}
|
||||
'REQUEST_FINGERPRINTER_IMPLEMENTATION': '2.7'}
|
||||
crawler = get_crawler(SimpleSpider, settings_dict=settings)
|
||||
spider = SimpleSpider.from_crawler(crawler)
|
||||
dupefilter = _get_dupefilter(crawler=crawler)
|
||||
|
|
@ -233,7 +233,7 @@ class RFPDupeFilterTest(unittest.TestCase):
|
|||
def test_log_debug_default_dupefilter(self):
|
||||
with LogCapture() as log:
|
||||
settings = {'DUPEFILTER_DEBUG': True,
|
||||
'REQUEST_FINGERPRINTER_IMPLEMENTATION': 'VERSION'}
|
||||
'REQUEST_FINGERPRINTER_IMPLEMENTATION': '2.7'}
|
||||
crawler = get_crawler(SimpleSpider, settings_dict=settings)
|
||||
spider = SimpleSpider.from_crawler(crawler)
|
||||
dupefilter = _get_dupefilter(crawler=crawler)
|
||||
|
|
|
|||
|
|
@ -1,12 +1,9 @@
|
|||
import pickle
|
||||
import re
|
||||
import unittest
|
||||
from warnings import catch_warnings
|
||||
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
from scrapy.http import HtmlResponse, XmlResponse
|
||||
from scrapy.link import Link
|
||||
from scrapy.linkextractors import FilteringLinkExtractor
|
||||
from scrapy.linkextractors.lxmlhtml import LxmlLinkExtractor
|
||||
from tests import get_testdata
|
||||
|
||||
|
|
@ -517,32 +514,3 @@ class LxmlLinkExtractorTestCase(Base.LinkExtractorTestCase):
|
|||
|
||||
def test_restrict_xpaths_with_html_entities(self):
|
||||
super().test_restrict_xpaths_with_html_entities()
|
||||
|
||||
def test_filteringlinkextractor_deprecation_warning(self):
|
||||
"""Make sure the FilteringLinkExtractor deprecation warning is not
|
||||
issued for LxmlLinkExtractor"""
|
||||
with catch_warnings(record=True) as warnings:
|
||||
LxmlLinkExtractor()
|
||||
self.assertEqual(len(warnings), 0)
|
||||
|
||||
class SubclassedLxmlLinkExtractor(LxmlLinkExtractor):
|
||||
pass
|
||||
|
||||
SubclassedLxmlLinkExtractor()
|
||||
self.assertEqual(len(warnings), 0)
|
||||
|
||||
|
||||
class FilteringLinkExtractorTest(unittest.TestCase):
|
||||
|
||||
def test_deprecation_warning(self):
|
||||
args = [None] * 10
|
||||
with catch_warnings(record=True) as warnings:
|
||||
FilteringLinkExtractor(*args)
|
||||
self.assertEqual(len(warnings), 1)
|
||||
self.assertEqual(warnings[0].category, ScrapyDeprecationWarning)
|
||||
with catch_warnings(record=True) as warnings:
|
||||
class SubclassedFilteringLinkExtractor(FilteringLinkExtractor):
|
||||
pass
|
||||
SubclassedFilteringLinkExtractor(*args)
|
||||
self.assertEqual(len(warnings), 1)
|
||||
self.assertEqual(warnings[0].category, ScrapyDeprecationWarning)
|
||||
|
|
|
|||
|
|
@ -295,7 +295,7 @@ class SelectortemLoaderTest(unittest.TestCase):
|
|||
|
||||
l.add_css('name', 'div::text')
|
||||
self.assertEqual(l.get_output_value('name'), ['Marta'])
|
||||
|
||||
|
||||
def test_init_method_with_base_response(self):
|
||||
"""Selector should be None after initialization"""
|
||||
response = Response("https://scrapy.org")
|
||||
|
|
|
|||
|
|
@ -64,7 +64,7 @@ class FileDownloadCrawlTestCase(TestCase):
|
|||
self.tmpmediastore = Path(self.mktemp())
|
||||
self.tmpmediastore.mkdir()
|
||||
self.settings = {
|
||||
'REQUEST_FINGERPRINTER_IMPLEMENTATION': 'VERSION',
|
||||
'REQUEST_FINGERPRINTER_IMPLEMENTATION': '2.7',
|
||||
'ITEM_PIPELINES': {self.pipeline_class: 1},
|
||||
self.store_setting_key: str(self.tmpmediastore),
|
||||
}
|
||||
|
|
|
|||
|
|
@ -1,17 +1,20 @@
|
|||
import dataclasses
|
||||
import hashlib
|
||||
import io
|
||||
import random
|
||||
import warnings
|
||||
from shutil import rmtree
|
||||
from tempfile import mkdtemp
|
||||
import dataclasses
|
||||
from unittest.mock import patch
|
||||
|
||||
import attr
|
||||
from itemadapter import ItemAdapter
|
||||
from twisted.trial import unittest
|
||||
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
from scrapy.http import Request, Response
|
||||
from scrapy.item import Field, Item
|
||||
from scrapy.pipelines.images import ImagesPipeline
|
||||
from scrapy.pipelines.images import ImageException, ImagesPipeline, NoimagesDrop
|
||||
from scrapy.settings import Settings
|
||||
from scrapy.utils.python import to_bytes
|
||||
|
||||
|
|
@ -102,32 +105,143 @@ class ImagesPipelineTestCase(unittest.TestCase):
|
|||
request = Request("http://example.com")
|
||||
self.assertEqual(thumb_path(request, 'small', item=item), 'thumb/small/path-to-store-file')
|
||||
|
||||
def test_convert_image(self):
|
||||
def test_get_images_exception(self):
|
||||
self.pipeline.min_width = 100
|
||||
self.pipeline.min_height = 100
|
||||
|
||||
_, buf1 = _create_image('JPEG', 'RGB', (50, 50), (0, 0, 0))
|
||||
_, buf2 = _create_image('JPEG', 'RGB', (150, 50), (0, 0, 0))
|
||||
_, buf3 = _create_image('JPEG', 'RGB', (50, 150), (0, 0, 0))
|
||||
|
||||
resp1 = Response(url="https://dev.mydeco.com/mydeco.gif", body=buf1.getvalue())
|
||||
resp2 = Response(url="https://dev.mydeco.com/mydeco.gif", body=buf2.getvalue())
|
||||
resp3 = Response(url="https://dev.mydeco.com/mydeco.gif", body=buf3.getvalue())
|
||||
req = Request(url="https://dev.mydeco.com/mydeco.gif")
|
||||
|
||||
with self.assertRaises(ImageException):
|
||||
next(self.pipeline.get_images(response=resp1, request=req, info=object()))
|
||||
with self.assertRaises(ImageException):
|
||||
next(self.pipeline.get_images(response=resp2, request=req, info=object()))
|
||||
with self.assertRaises(ImageException):
|
||||
next(self.pipeline.get_images(response=resp3, request=req, info=object()))
|
||||
|
||||
def test_get_images_new(self):
|
||||
self.pipeline.min_width = 0
|
||||
self.pipeline.min_height = 0
|
||||
self.pipeline.thumbs = {'small': (20, 20)}
|
||||
|
||||
orig_im, buf = _create_image('JPEG', 'RGB', (50, 50), (0, 0, 0))
|
||||
orig_thumb, orig_thumb_buf = _create_image('JPEG', 'RGB', (20, 20), (0, 0, 0))
|
||||
resp = Response(url="https://dev.mydeco.com/mydeco.gif", body=buf.getvalue())
|
||||
req = Request(url="https://dev.mydeco.com/mydeco.gif")
|
||||
|
||||
get_images_gen = self.pipeline.get_images(response=resp, request=req, info=object())
|
||||
|
||||
path, new_im, new_buf = next(get_images_gen)
|
||||
self.assertEqual(path, 'full/3fd165099d8e71b8a48b2683946e64dbfad8b52d.jpg')
|
||||
self.assertEqual(orig_im, new_im)
|
||||
self.assertEqual(buf.getvalue(), new_buf.getvalue())
|
||||
|
||||
thumb_path, thumb_img, thumb_buf = next(get_images_gen)
|
||||
self.assertEqual(thumb_path, 'thumbs/small/3fd165099d8e71b8a48b2683946e64dbfad8b52d.jpg')
|
||||
self.assertEqual(thumb_img, thumb_img)
|
||||
self.assertEqual(orig_thumb_buf.getvalue(), thumb_buf.getvalue())
|
||||
|
||||
def test_get_images_old(self):
|
||||
self.pipeline.thumbs = {'small': (20, 20)}
|
||||
orig_im, buf = _create_image('JPEG', 'RGB', (50, 50), (0, 0, 0))
|
||||
resp = Response(url="https://dev.mydeco.com/mydeco.gif", body=buf.getvalue())
|
||||
req = Request(url="https://dev.mydeco.com/mydeco.gif")
|
||||
|
||||
def overridden_convert_image(image, size=None):
|
||||
im, buf = _create_image('JPEG', 'RGB', (50, 50), (0, 0, 0))
|
||||
return im, buf
|
||||
|
||||
with patch.object(self.pipeline, 'convert_image', overridden_convert_image):
|
||||
with warnings.catch_warnings(record=True) as w:
|
||||
warnings.simplefilter('always')
|
||||
get_images_gen = self.pipeline.get_images(response=resp, request=req, info=object())
|
||||
path, new_im, new_buf = next(get_images_gen)
|
||||
self.assertEqual(path, 'full/3fd165099d8e71b8a48b2683946e64dbfad8b52d.jpg')
|
||||
self.assertEqual(orig_im.mode, new_im.mode)
|
||||
self.assertEqual(orig_im.getcolors(), new_im.getcolors())
|
||||
self.assertEqual(buf.getvalue(), new_buf.getvalue())
|
||||
|
||||
thumb_path, thumb_img, thumb_buf = next(get_images_gen)
|
||||
self.assertEqual(thumb_path, 'thumbs/small/3fd165099d8e71b8a48b2683946e64dbfad8b52d.jpg')
|
||||
self.assertEqual(orig_im.mode, thumb_img.mode)
|
||||
self.assertEqual(orig_im.getcolors(), thumb_img.getcolors())
|
||||
self.assertEqual(buf.getvalue(), thumb_buf.getvalue())
|
||||
|
||||
expected_warning_msg = ('.convert_image() method overriden in a deprecated way, '
|
||||
'overriden method does not accept response_body argument.')
|
||||
self.assertEqual(len([warning for warning in w if expected_warning_msg in str(warning.message)]), 1)
|
||||
|
||||
def test_convert_image_old(self):
|
||||
# tests for old API
|
||||
with warnings.catch_warnings(record=True) as w:
|
||||
warnings.simplefilter('always')
|
||||
SIZE = (100, 100)
|
||||
# straigh forward case: RGB and JPEG
|
||||
COLOUR = (0, 127, 255)
|
||||
im, _ = _create_image('JPEG', 'RGB', SIZE, COLOUR)
|
||||
converted, _ = self.pipeline.convert_image(im)
|
||||
self.assertEqual(converted.mode, 'RGB')
|
||||
self.assertEqual(converted.getcolors(), [(10000, COLOUR)])
|
||||
|
||||
# check that thumbnail keep image ratio
|
||||
thumbnail, _ = self.pipeline.convert_image(converted, size=(10, 25))
|
||||
self.assertEqual(thumbnail.mode, 'RGB')
|
||||
self.assertEqual(thumbnail.size, (10, 10))
|
||||
|
||||
# transparency case: RGBA and PNG
|
||||
COLOUR = (0, 127, 255, 50)
|
||||
im, _ = _create_image('PNG', 'RGBA', SIZE, COLOUR)
|
||||
converted, _ = self.pipeline.convert_image(im)
|
||||
self.assertEqual(converted.mode, 'RGB')
|
||||
self.assertEqual(converted.getcolors(), [(10000, (205, 230, 255))])
|
||||
|
||||
# transparency case with palette: P and PNG
|
||||
COLOUR = (0, 127, 255, 50)
|
||||
im, _ = _create_image('PNG', 'RGBA', SIZE, COLOUR)
|
||||
im = im.convert('P')
|
||||
converted, _ = self.pipeline.convert_image(im)
|
||||
self.assertEqual(converted.mode, 'RGB')
|
||||
self.assertEqual(converted.getcolors(), [(10000, (205, 230, 255))])
|
||||
|
||||
# ensure that we recieved deprecation warnings
|
||||
expected_warning_msg = '.convert_image() method called in a deprecated way'
|
||||
self.assertTrue(len([warning for warning in w if expected_warning_msg in str(warning.message)]) == 4)
|
||||
|
||||
def test_convert_image_new(self):
|
||||
# tests for new API
|
||||
SIZE = (100, 100)
|
||||
# straigh forward case: RGB and JPEG
|
||||
COLOUR = (0, 127, 255)
|
||||
im = _create_image('JPEG', 'RGB', SIZE, COLOUR)
|
||||
converted, _ = self.pipeline.convert_image(im)
|
||||
im, buf = _create_image('JPEG', 'RGB', SIZE, COLOUR)
|
||||
converted, converted_buf = self.pipeline.convert_image(im, response_body=buf)
|
||||
self.assertEqual(converted.mode, 'RGB')
|
||||
self.assertEqual(converted.getcolors(), [(10000, COLOUR)])
|
||||
# check that we don't convert JPEGs again
|
||||
self.assertEqual(converted_buf, buf)
|
||||
|
||||
# check that thumbnail keep image ratio
|
||||
thumbnail, _ = self.pipeline.convert_image(converted, size=(10, 25))
|
||||
thumbnail, _ = self.pipeline.convert_image(converted, size=(10, 25), response_body=converted_buf)
|
||||
self.assertEqual(thumbnail.mode, 'RGB')
|
||||
self.assertEqual(thumbnail.size, (10, 10))
|
||||
|
||||
# transparency case: RGBA and PNG
|
||||
COLOUR = (0, 127, 255, 50)
|
||||
im = _create_image('PNG', 'RGBA', SIZE, COLOUR)
|
||||
converted, _ = self.pipeline.convert_image(im)
|
||||
im, buf = _create_image('PNG', 'RGBA', SIZE, COLOUR)
|
||||
converted, _ = self.pipeline.convert_image(im, response_body=buf)
|
||||
self.assertEqual(converted.mode, 'RGB')
|
||||
self.assertEqual(converted.getcolors(), [(10000, (205, 230, 255))])
|
||||
|
||||
# transparency case with palette: P and PNG
|
||||
COLOUR = (0, 127, 255, 50)
|
||||
im = _create_image('PNG', 'RGBA', SIZE, COLOUR)
|
||||
im, buf = _create_image('PNG', 'RGBA', SIZE, COLOUR)
|
||||
im = im.convert('P')
|
||||
converted, _ = self.pipeline.convert_image(im)
|
||||
converted, _ = self.pipeline.convert_image(im, response_body=buf)
|
||||
self.assertEqual(converted.mode, 'RGB')
|
||||
self.assertEqual(converted.getcolors(), [(10000, (205, 230, 255))])
|
||||
|
||||
|
|
@ -416,11 +530,27 @@ class ImagesPipelineTestCaseCustomSettings(unittest.TestCase):
|
|||
expected_value)
|
||||
|
||||
|
||||
class NoimagesDropTestCase(unittest.TestCase):
|
||||
|
||||
def test_deprecation_warning(self):
|
||||
arg = str()
|
||||
with warnings.catch_warnings(record=True) as w:
|
||||
NoimagesDrop(arg)
|
||||
self.assertEqual(len(w), 1)
|
||||
self.assertEqual(w[0].category, ScrapyDeprecationWarning)
|
||||
with warnings.catch_warnings(record=True) as w:
|
||||
class SubclassedNoimagesDrop(NoimagesDrop):
|
||||
pass
|
||||
SubclassedNoimagesDrop(arg)
|
||||
self.assertEqual(len(w), 1)
|
||||
self.assertEqual(w[0].category, ScrapyDeprecationWarning)
|
||||
|
||||
|
||||
def _create_image(format, *a, **kw):
|
||||
buf = io.BytesIO()
|
||||
Image.new(*a, **kw).save(buf, format)
|
||||
buf.seek(0)
|
||||
return Image.open(buf)
|
||||
return Image.open(buf), buf
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
|
|
|||
|
|
@ -52,7 +52,7 @@ class MockCrawler(Crawler):
|
|||
SCHEDULER_PRIORITY_QUEUE=priority_queue_cls,
|
||||
JOBDIR=jobdir,
|
||||
DUPEFILTER_CLASS='scrapy.dupefilters.BaseDupeFilter',
|
||||
REQUEST_FINGERPRINTER_IMPLEMENTATION='VERSION',
|
||||
REQUEST_FINGERPRINTER_IMPLEMENTATION='2.7',
|
||||
)
|
||||
super().__init__(Spider, settings)
|
||||
self.engine = MockEngine(downloader=MockDownloader())
|
||||
|
|
|
|||
|
|
@ -98,7 +98,7 @@ class SpiderLoaderTest(unittest.TestCase):
|
|||
module = 'tests.test_spiderloader.test_spiders.spider1'
|
||||
runner = CrawlerRunner({
|
||||
'SPIDER_MODULES': [module],
|
||||
'REQUEST_FINGERPRINTER_IMPLEMENTATION': 'VERSION',
|
||||
'REQUEST_FINGERPRINTER_IMPLEMENTATION': '2.7',
|
||||
})
|
||||
|
||||
self.assertRaisesRegex(KeyError, 'Spider not found',
|
||||
|
|
|
|||
|
|
@ -1,3 +1,4 @@
|
|||
import warnings
|
||||
from unittest import TestCase
|
||||
|
||||
from pytest import mark
|
||||
|
|
@ -13,5 +14,6 @@ class AsyncioTest(TestCase):
|
|||
self.assertEqual(is_asyncio_reactor_installed(), self.reactor_pytest == 'asyncio')
|
||||
|
||||
def test_install_asyncio_reactor(self):
|
||||
# this should do nothing
|
||||
install_reactor("twisted.internet.asyncioreactor.AsyncioSelectorReactor")
|
||||
with warnings.catch_warnings(record=True) as w:
|
||||
install_reactor("twisted.internet.asyncioreactor.AsyncioSelectorReactor")
|
||||
self.assertEqual(len(w), 0)
|
||||
|
|
|
|||
|
|
@ -1,5 +1,6 @@
|
|||
import unittest
|
||||
import warnings
|
||||
from functools import partial
|
||||
from unittest import mock
|
||||
|
||||
from scrapy.utils.misc import is_generator_with_return_value, warn_on_generator_with_return_value
|
||||
|
|
@ -165,9 +166,99 @@ https://example.org
|
|||
warn_on_generator_with_return_value(None, l2)
|
||||
self.assertEqual(len(w), 0)
|
||||
|
||||
def test_generators_return_none_with_decorator(self):
|
||||
def decorator(func):
|
||||
def inner_func():
|
||||
func()
|
||||
return inner_func
|
||||
|
||||
@decorator
|
||||
def f3():
|
||||
yield 1
|
||||
return None
|
||||
|
||||
@decorator
|
||||
def g3():
|
||||
yield 1
|
||||
return
|
||||
|
||||
@decorator
|
||||
def h3():
|
||||
yield 1
|
||||
|
||||
@decorator
|
||||
def i3():
|
||||
yield 1
|
||||
yield from generator_that_returns_stuff()
|
||||
|
||||
@decorator
|
||||
def j3():
|
||||
yield 1
|
||||
|
||||
def helper():
|
||||
return 0
|
||||
|
||||
yield helper()
|
||||
|
||||
@decorator
|
||||
def k3():
|
||||
"""
|
||||
docstring
|
||||
"""
|
||||
url = """
|
||||
https://example.org
|
||||
"""
|
||||
yield url
|
||||
return
|
||||
|
||||
@decorator
|
||||
def l3():
|
||||
return
|
||||
|
||||
assert not is_generator_with_return_value(top_level_return_none)
|
||||
assert not is_generator_with_return_value(f3)
|
||||
assert not is_generator_with_return_value(g3)
|
||||
assert not is_generator_with_return_value(h3)
|
||||
assert not is_generator_with_return_value(i3)
|
||||
assert not is_generator_with_return_value(j3) # not recursive
|
||||
assert not is_generator_with_return_value(k3) # not recursive
|
||||
assert not is_generator_with_return_value(l3)
|
||||
|
||||
with warnings.catch_warnings(record=True) as w:
|
||||
warn_on_generator_with_return_value(None, top_level_return_none)
|
||||
self.assertEqual(len(w), 0)
|
||||
with warnings.catch_warnings(record=True) as w:
|
||||
warn_on_generator_with_return_value(None, f3)
|
||||
self.assertEqual(len(w), 0)
|
||||
with warnings.catch_warnings(record=True) as w:
|
||||
warn_on_generator_with_return_value(None, g3)
|
||||
self.assertEqual(len(w), 0)
|
||||
with warnings.catch_warnings(record=True) as w:
|
||||
warn_on_generator_with_return_value(None, h3)
|
||||
self.assertEqual(len(w), 0)
|
||||
with warnings.catch_warnings(record=True) as w:
|
||||
warn_on_generator_with_return_value(None, i3)
|
||||
self.assertEqual(len(w), 0)
|
||||
with warnings.catch_warnings(record=True) as w:
|
||||
warn_on_generator_with_return_value(None, j3)
|
||||
self.assertEqual(len(w), 0)
|
||||
with warnings.catch_warnings(record=True) as w:
|
||||
warn_on_generator_with_return_value(None, k3)
|
||||
self.assertEqual(len(w), 0)
|
||||
with warnings.catch_warnings(record=True) as w:
|
||||
warn_on_generator_with_return_value(None, l3)
|
||||
self.assertEqual(len(w), 0)
|
||||
|
||||
@mock.patch("scrapy.utils.misc.is_generator_with_return_value", new=_indentation_error)
|
||||
def test_indentation_error(self):
|
||||
with warnings.catch_warnings(record=True) as w:
|
||||
warn_on_generator_with_return_value(None, top_level_return_none)
|
||||
self.assertEqual(len(w), 1)
|
||||
self.assertIn('Unable to determine', str(w[0].message))
|
||||
|
||||
def test_partial(self):
|
||||
def cb(arg1, arg2):
|
||||
yield {}
|
||||
|
||||
partial_cb = partial(cb, arg1=42)
|
||||
assert not is_generator_with_return_value(partial_cb)
|
||||
|
|
|
|||
|
|
@ -6,9 +6,6 @@ import contextlib
|
|||
import warnings
|
||||
from pathlib import Path
|
||||
|
||||
from pytest import warns
|
||||
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
from scrapy.utils.project import data_path, get_project_settings
|
||||
|
||||
|
||||
|
|
@ -79,10 +76,10 @@ class GetProjectSettingsTestCase(unittest.TestCase):
|
|||
envvars = {
|
||||
'SCRAPY_FOO': 'bar',
|
||||
}
|
||||
with warns(ScrapyDeprecationWarning, match=': FOO') as record:
|
||||
with set_env(**envvars):
|
||||
get_project_settings()
|
||||
assert len(record) == 1
|
||||
with set_env(**envvars):
|
||||
settings = get_project_settings()
|
||||
|
||||
assert settings.get("SCRAPY_FOO") is None
|
||||
|
||||
def test_valid_and_invalid_envvars(self):
|
||||
value = 'tests.test_cmdline.settings'
|
||||
|
|
@ -90,8 +87,7 @@ class GetProjectSettingsTestCase(unittest.TestCase):
|
|||
'SCRAPY_FOO': 'bar',
|
||||
'SCRAPY_SETTINGS_MODULE': value,
|
||||
}
|
||||
with warns(ScrapyDeprecationWarning, match=': FOO') as record:
|
||||
with set_env(**envvars):
|
||||
settings = get_project_settings()
|
||||
assert len(record) == 1
|
||||
with set_env(**envvars):
|
||||
settings = get_project_settings()
|
||||
assert settings.get('SETTINGS_MODULE') == value
|
||||
assert settings.get('SCRAPY_FOO') is None
|
||||
|
|
|
|||
|
|
@ -1,18 +1,14 @@
|
|||
import functools
|
||||
import gc
|
||||
import operator
|
||||
import platform
|
||||
from itertools import count
|
||||
from warnings import catch_warnings, filterwarnings
|
||||
|
||||
from twisted.trial import unittest
|
||||
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
from scrapy.utils.asyncgen import as_async_generator, collect_asyncgen
|
||||
from scrapy.utils.defer import deferred_f_from_coro_f, aiter_errback
|
||||
from scrapy.utils.python import (
|
||||
memoizemethod_noargs, binary_is_text, equal_attributes,
|
||||
WeakKeyCache, get_func_args, to_bytes, to_unicode,
|
||||
get_func_args, to_bytes, to_unicode,
|
||||
without_none_values, MutableChain, MutableAsyncChain)
|
||||
|
||||
|
||||
|
|
@ -27,12 +23,7 @@ class MutableChainTest(unittest.TestCase):
|
|||
m.extend([9, 10], (11, 12))
|
||||
self.assertEqual(next(m), 0)
|
||||
self.assertEqual(m.__next__(), 1)
|
||||
with catch_warnings(record=True) as warnings:
|
||||
self.assertEqual(m.next(), 2)
|
||||
self.assertEqual(len(warnings), 1)
|
||||
self.assertIn('scrapy.utils.python.MutableChain.__next__',
|
||||
str(warnings[0].message))
|
||||
self.assertEqual(list(m), list(range(3, 13)))
|
||||
self.assertEqual(list(m), list(range(2, 13)))
|
||||
|
||||
|
||||
class MutableAsyncChainTest(unittest.TestCase):
|
||||
|
|
@ -209,27 +200,6 @@ class UtilsPythonTestCase(unittest.TestCase):
|
|||
a.meta['z'] = 2
|
||||
self.assertFalse(equal_attributes(a, b, [compare_z, 'x']))
|
||||
|
||||
def test_weakkeycache(self):
|
||||
class _Weakme:
|
||||
pass
|
||||
|
||||
_values = count()
|
||||
|
||||
with catch_warnings():
|
||||
filterwarnings("ignore", category=ScrapyDeprecationWarning)
|
||||
wk = WeakKeyCache(lambda k: next(_values))
|
||||
|
||||
k = _Weakme()
|
||||
v = wk[k]
|
||||
self.assertEqual(v, wk[k])
|
||||
self.assertNotEqual(v, wk[_Weakme()])
|
||||
self.assertEqual(v, wk[k])
|
||||
del k
|
||||
for _ in range(100):
|
||||
if wk._weakdict:
|
||||
gc.collect()
|
||||
self.assertFalse(len(wk._weakdict))
|
||||
|
||||
def test_get_func_args(self):
|
||||
def f1(a, b, c):
|
||||
pass
|
||||
|
|
|
|||
|
|
@ -505,7 +505,7 @@ class RequestFingerprinterTestCase(unittest.TestCase):
|
|||
|
||||
def test_deprecated_implementation(self):
|
||||
settings = {
|
||||
'REQUEST_FINGERPRINTER_IMPLEMENTATION': 'PREVIOUS_VERSION',
|
||||
'REQUEST_FINGERPRINTER_IMPLEMENTATION': '2.6',
|
||||
}
|
||||
with warnings.catch_warnings(record=True) as logged_warnings:
|
||||
crawler = get_crawler(settings_dict=settings)
|
||||
|
|
@ -518,7 +518,7 @@ class RequestFingerprinterTestCase(unittest.TestCase):
|
|||
|
||||
def test_recommended_implementation(self):
|
||||
settings = {
|
||||
'REQUEST_FINGERPRINTER_IMPLEMENTATION': 'VERSION',
|
||||
'REQUEST_FINGERPRINTER_IMPLEMENTATION': '2.7',
|
||||
}
|
||||
with warnings.catch_warnings(record=True) as logged_warnings:
|
||||
crawler = get_crawler(settings_dict=settings)
|
||||
|
|
|
|||
Loading…
Reference in New Issue