Merge branch 'master' into feat-685

This commit is contained in:
Eugenio Lacuesta 2020-03-18 21:19:34 -03:00
commit 34d607194b
No known key found for this signature in database
GPG Key ID: DA3EF2D0913E9810
213 changed files with 2674 additions and 1407 deletions

View File

@ -1,8 +1,7 @@
[bumpversion]
current_version = 1.8.0
current_version = 2.0.0
commit = True
tag = True
tag_name = {new_version}
[bumpversion:file:scrapy/VERSION]

View File

@ -1,9 +1,11 @@
version: 2
sphinx:
configuration: docs/conf.py
fail_on_warning: true
python:
# For available versions, see:
# https://docs.readthedocs.io/en/stable/config-file/v2.html#build-image
version: 3.7 # Keep in sync with .travis.yml
install:
- requirements: docs/requirements.txt
- path: .

View File

@ -1,24 +0,0 @@
TRIAL := $(shell which trial)
BRANCH := $(shell git rev-parse --abbrev-ref HEAD)
export PYTHONPATH=$(PWD)
test:
coverage run --branch $(TRIAL) --reporter=text tests
rm -rf htmlcov && coverage html
-s3cmd sync -P htmlcov/ s3://static.scrapy.org/coverage-scrapy-$(BRANCH)/
build:
git describe --tags --match '[0-9]*' |sed 's/-/.post/;s/-g/+g/' >scrapy/VERSION
debchange -m -D unstable --force-distribution -v \
$$(python setup.py --version |sed -r 's/([0-9]+.[0-9]+.[0-9]+)(a|b|rc|dev)([0-9]*)/\1~\2\3/')-$$(date +%s) \
"Automatic build"
debuild -us -uc -b
clean:
git checkout debian scrapy/VERSION
git clean -dfq
pypi:
umask 0022 && chmod -R a+rX . && python setup.py sdist upload
.PHONY: clean test build

View File

@ -41,7 +41,7 @@ Requirements
============
* Python 3.5+
* Works on Linux, Windows, Mac OSX, BSD
* Works on Linux, Windows, macOS, BSD
Install
=======

View File

@ -11,7 +11,9 @@ collect_ignore = [
# not a test, but looks like a test
"scrapy/utils/testsite.py",
# contains scripts to be run by tests/test_crawler.py::CrawlerProcessSubprocess
*_py_files("tests/CrawlerProcess")
*_py_files("tests/CrawlerProcess"),
# Py36-only parts of respective tests
*_py_files("tests/py36"),
]
for line in open('tests/ignores.txt'):

5
debian/changelog vendored
View File

@ -1,5 +0,0 @@
scrapy (0.11) unstable; urgency=low
* Initial release.
-- Scrapinghub Team <info@scrapinghub.com> Thu, 10 Jun 2010 17:24:02 -0300

1
debian/compat vendored
View File

@ -1 +0,0 @@
7

20
debian/control vendored
View File

@ -1,20 +0,0 @@
Source: scrapy
Section: python
Priority: optional
Maintainer: Scrapinghub Team <info@scrapinghub.com>
Build-Depends: debhelper (>= 7.0.50), python (>=2.7), python-twisted, python-w3lib, python-lxml, python-six (>=1.5.2)
Standards-Version: 3.8.4
Homepage: https://scrapy.org/
Package: scrapy
Architecture: all
Depends: ${python:Depends}, python-lxml, python-twisted, python-openssl,
python-w3lib (>= 1.8.0), python-queuelib, python-cssselect (>= 0.9), python-six (>=1.5.2)
Recommends: python-setuptools
Conflicts: python-scrapy, scrapy-0.25
Provides: python-scrapy, scrapy-0.25
Description: Python web crawling and web scraping framework
Scrapy is a fast high-level web crawling and web scraping framework,
used to crawl websites and extract structured data from their pages.
It can be used for a wide range of purposes, from data mining to
monitoring and automated testing.

40
debian/copyright vendored
View File

@ -1,40 +0,0 @@
This package was debianized by the Scrapinghub team <info@scrapinghub.com>.
It was downloaded from https://scrapy.org
Upstream Author: Scrapy Developers
Copyright: 2007-2013 Scrapy Developers
License: bsd
Copyright (c) Scrapy developers.
All rights reserved.
Redistribution and use in source and binary forms, with or without modification,
are permitted provided that the following conditions are met:
1. Redistributions of source code must retain the above copyright notice,
this list of conditions and the following disclaimer.
2. Redistributions in binary form must reproduce the above copyright
notice, this list of conditions and the following disclaimer in the
documentation and/or other materials provided with the distribution.
3. Neither the name of Scrapy nor the names of its contributors may be used
to endorse or promote products derived from this software without
specific prior written permission.
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR
ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
(INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON
ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
(INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS
SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
The Debian packaging is (C) 2010-2013, Scrapinghub <info@scrapinghub.com> and
is licensed under the BSD, see `/usr/share/common-licenses/BSD'.

1
debian/pyversions vendored
View File

@ -1 +0,0 @@
2.7

5
debian/rules vendored
View File

@ -1,5 +0,0 @@
#!/usr/bin/make -f
# -*- makefile -*-
%:
dh $@

2
debian/scrapy.docs vendored
View File

@ -1,2 +0,0 @@
README.rst
AUTHORS

View File

@ -1,2 +0,0 @@
extras/scrapy_bash_completion etc/bash_completion.d/
extras/scrapy_zsh_completion /usr/share/zsh/vendor-completions/_scrapy

View File

@ -1 +0,0 @@
new-package-should-close-itp-bug

View File

@ -1 +0,0 @@
extras/scrapy.1

View File

@ -281,6 +281,7 @@ coverage_ignore_pyobjects = [
intersphinx_mapping = {
'coverage': ('https://coverage.readthedocs.io/en/stable', None),
'cssselect': ('https://cssselect.readthedocs.io/en/latest', None),
'pytest': ('https://docs.pytest.org/en/latest', None),
'python': ('https://docs.python.org/3', None),
'sphinx': ('https://www.sphinx-doc.org/en/master', None),

View File

@ -143,7 +143,7 @@ by running ``git fetch upstream pull/$PR_NUMBER/head:$BRANCH_NAME_TO_CREATE``
(replace 'upstream' with a remote name for scrapy repository,
``$PR_NUMBER`` with an ID of the pull request, and ``$BRANCH_NAME_TO_CREATE``
with a name of the branch you want to create locally).
See also: https://help.github.com/articles/checking-out-pull-requests-locally/#modifying-an-inactive-pull-request-locally.
See also: https://help.github.com/en/github/collaborating-with-issues-and-pull-requests/checking-out-pull-requests-locally#modifying-an-inactive-pull-request-locally.
When writing GitHub pull requests, try to keep titles short but descriptive.
E.g. For bug #411: "Scrapy hangs if an exception raises in start_requests"
@ -168,7 +168,7 @@ Scrapy:
* Don't put your name in the code you contribute; git provides enough
metadata to identify author of the code.
See https://help.github.com/articles/setting-your-username-in-git/ for
See https://help.github.com/en/github/using-git/setting-your-username-in-git for
setup instructions.
.. _documentation-policies:
@ -266,5 +266,5 @@ And their unit-tests are in::
.. _tests/: https://github.com/scrapy/scrapy/tree/master/tests
.. _open issues: https://github.com/scrapy/scrapy/issues
.. _PEP 257: https://www.python.org/dev/peps/pep-0257/
.. _pull request: https://help.github.com/en/articles/creating-a-pull-request
.. _pull request: https://help.github.com/en/github/collaborating-with-issues-and-pull-requests/creating-a-pull-request
.. _pytest-xdist: https://github.com/pytest-dev/pytest-xdist

View File

@ -22,8 +22,8 @@ In other words, comparing `BeautifulSoup`_ (or `lxml`_) to Scrapy is like
comparing `jinja2`_ to `Django`_.
.. _BeautifulSoup: https://www.crummy.com/software/BeautifulSoup/
.. _lxml: http://lxml.de/
.. _jinja2: http://jinja.pocoo.org/
.. _lxml: https://lxml.de/
.. _jinja2: https://palletsprojects.com/p/jinja/
.. _Django: https://www.djangoproject.com/
Can I use Scrapy with BeautifulSoup?
@ -269,7 +269,7 @@ The ``__VIEWSTATE`` parameter is used in sites built with ASP.NET/VB.NET. For
more info on how it works see `this page`_. Also, here's an `example spider`_
which scrapes one of these sites.
.. _this page: http://search.cpan.org/~ecarroll/HTML-TreeBuilderX-ASP_NET-0.09/lib/HTML/TreeBuilderX/ASP_NET.pm
.. _this page: https://metacpan.org/pod/release/ECARROLL/HTML-TreeBuilderX-ASP_NET-0.09/lib/HTML/TreeBuilderX/ASP_NET.pm
.. _example spider: https://github.com/AmbientLighter/rpn-fas/blob/master/fas/spiders/rnp.py
What's the best way to parse big XML/CSV data feeds?

View File

@ -165,6 +165,8 @@ Solving specific problems
topics/autothrottle
topics/benchmarking
topics/jobs
topics/coroutines
topics/asyncio
:doc:`faq`
Get answers to most frequently asked questions.
@ -205,6 +207,12 @@ Solving specific problems
:doc:`topics/jobs`
Learn how to pause and resume crawls for large spiders.
:doc:`topics/coroutines`
Use the :ref:`coroutine syntax <async>`.
:doc:`topics/asyncio`
Use :mod:`asyncio` and :mod:`asyncio`-powered libraries.
.. _extending-scrapy:
Extending Scrapy

View File

@ -7,12 +7,12 @@ Installation guide
Installing Scrapy
=================
Scrapy runs on Python 3.5 or above
under CPython (default Python implementation) and PyPy (starting with PyPy 5.9).
Scrapy runs on Python 3.5 or above under CPython (default Python
implementation) and PyPy (starting with PyPy 5.9).
If you're using `Anaconda`_ or `Miniconda`_, you can install the package from
the `conda-forge`_ channel, which has up-to-date packages for Linux, Windows
and OS X.
and macOS.
To install Scrapy using ``conda``, run::
@ -65,7 +65,7 @@ please refer to their respective installation instructions:
* `lxml installation`_
* `cryptography installation`_
.. _lxml installation: http://lxml.de/installation.html
.. _lxml installation: https://lxml.de/installation.html
.. _cryptography installation: https://cryptography.io/en/latest/installation/
@ -81,35 +81,18 @@ Python packages can be installed either globally (a.k.a system wide),
or in user-space. We do not recommend installing Scrapy system wide.
Instead, we recommend that you install Scrapy within a so-called
"virtual environment" (`virtualenv`_).
Virtualenvs allow you to not conflict with already-installed Python
"virtual environment" (:mod:`venv`).
Virtual environments allow you to not conflict with already-installed Python
system packages (which could break some of your system tools and scripts),
and still install packages normally with ``pip`` (without ``sudo`` and the likes).
To get started with virtual environments, see `virtualenv installation instructions`_.
To install it globally (having it globally installed actually helps here),
it should be a matter of running::
See :ref:`tut-venv` on how to create your virtual environment.
$ [sudo] pip install virtualenv
Check this `user guide`_ on how to create your virtualenv.
.. note::
If you use Linux or OS X, `virtualenvwrapper`_ is a handy tool to create virtualenvs.
Once you have created a virtualenv, you can install Scrapy inside it with ``pip``,
Once you have created a virtual environment, you can install Scrapy inside it with ``pip``,
just like any other Python package.
(See :ref:`platform-specific guides <intro-install-platform-notes>`
below for non-Python dependencies that you may need to install beforehand).
Python virtualenvs can be created to use Python 2 by default, or Python 3 by default. As Scrapy
only supports Python 3, make sure you created a Python 3 virtualenv.
.. _virtualenv: https://virtualenv.pypa.io
.. _virtualenv installation instructions: https://virtualenv.pypa.io/en/stable/installation/
.. _virtualenvwrapper: https://virtualenvwrapper.readthedocs.io/en/latest/install.html
.. _user guide: https://virtualenv.pypa.io/en/stable/userguide/
.. _intro-install-platform-notes:
@ -165,11 +148,11 @@ you can install Scrapy with ``pip`` after that::
.. _intro-install-macos:
Mac OS X
--------
macOS
-----
Building Scrapy's dependencies requires the presence of a C compiler and
development headers. On OS X this is typically provided by Apples Xcode
development headers. On macOS this is typically provided by Apples Xcode
development tools. To install the Xcode command line tools open a terminal
window and run::
@ -205,15 +188,12 @@ solutions:
brew update; brew upgrade python
* *(Optional)* Install Scrapy inside an isolated python environment.
* *(Optional)* :ref:`Install Scrapy inside a Python virtual environment
<intro-using-virtualenv>`.
This method is a workaround for the above OS X issue, but it's an overall
This method is a workaround for the above macOS issue, but it's an overall
good practice for managing dependencies and can complement the first method.
`virtualenv`_ is a tool you can use to create virtual environments in python.
We recommended reading a tutorial like
http://docs.python-guide.org/en/latest/dev/virtualenvs/ to get started.
After any of these workarounds you should be able to install Scrapy::
pip install Scrapy
@ -227,7 +207,7 @@ For PyPy3, only Linux installation was tested.
Most Scrapy dependencides now have binary wheels for CPython, but not for PyPy.
This means that these dependecies will be built during installation.
On OS X, you are likely to face an issue with building Cryptography dependency,
On macOS, you are likely to face an issue with building Cryptography dependency,
solution to this problem is described
`here <https://github.com/pyca/cryptography/issues/2692#issuecomment-272773481>`_,
that is to ``brew install openssl`` and then export the flags that this command
@ -273,11 +253,11 @@ For details, see `Issue #2473 <https://github.com/scrapy/scrapy/issues/2473>`_.
.. _Python: https://www.python.org/
.. _pip: https://pip.pypa.io/en/latest/installing/
.. _lxml: https://lxml.de/index.html
.. _parsel: https://pypi.python.org/pypi/parsel
.. _w3lib: https://pypi.python.org/pypi/w3lib
.. _twisted: https://twistedmatrix.com/
.. _cryptography: https://cryptography.io/
.. _pyOpenSSL: https://pypi.python.org/pypi/pyOpenSSL
.. _parsel: https://pypi.org/project/parsel/
.. _w3lib: https://pypi.org/project/w3lib/
.. _twisted: https://twistedmatrix.com/trac/
.. _cryptography: https://cryptography.io/en/latest/
.. _pyOpenSSL: https://pypi.org/project/pyOpenSSL/
.. _setuptools: https://pypi.python.org/pypi/setuptools
.. _AUR Scrapy package: https://aur.archlinux.org/packages/scrapy/
.. _homebrew: https://brew.sh/

View File

@ -212,7 +212,7 @@ using the :ref:`Scrapy shell <topics-shell>`. Run::
.. note::
Remember to always enclose urls in quotes when running Scrapy shell from
command-line, otherwise urls containing arguments (ie. ``&`` character)
command-line, otherwise urls containing arguments (i.e. ``&`` character)
will not work.
On Windows, use double quotes instead::
@ -306,7 +306,7 @@ with a selector (see :ref:`topics-developer-tools`).
visually selected elements, which works in many browsers.
.. _regular expressions: https://docs.python.org/3/library/re.html
.. _Selector Gadget: http://selectorgadget.com/
.. _Selector Gadget: https://selectorgadget.com/
XPath: a brief intro
@ -337,7 +337,7 @@ recommend `this tutorial to learn XPath through examples
<http://zvon.org/comp/r/tut-XPath_1.html>`_, and `this tutorial to learn "how
to think in XPath" <http://plasmasturm.org/log/xpath101/>`_.
.. _XPath: https://www.w3.org/TR/xpath
.. _XPath: https://www.w3.org/TR/xpath/all/
.. _CSS: https://www.w3.org/TR/selectors
Extracting quotes and authors

View File

@ -3,8 +3,468 @@
Release notes
=============
.. note:: Scrapy 1.x will be the last series supporting Python 2. Scrapy 2.0,
planned for Q4 2019 or Q1 2020, will support **Python 3 only**.
.. _release-2.0.1:
Scrapy 2.0.1 (2020-03-18)
-------------------------
* :meth:`Response.follow_all <scrapy.http.Response.follow_all>` now supports
an empty URL iterable as input (:issue:`4408`, :issue:`4420`)
* Removed top-level :mod:`~twisted.internet.reactor` imports to prevent
errors about the wrong Twisted reactor being installed when setting a
different Twisted reactor using :setting:`TWISTED_REACTOR` (:issue:`4401`,
:issue:`4406`)
* Fixed tests (:issue:`4422`)
.. _release-2.0.0:
Scrapy 2.0.0 (2020-03-03)
-------------------------
Highlights:
* Python 2 support has been removed
* :doc:`Partial <topics/coroutines>` :ref:`coroutine syntax <async>` support
and :doc:`experimental <topics/asyncio>` :mod:`asyncio` support
* New :meth:`Response.follow_all <scrapy.http.Response.follow_all>` method
* :ref:`FTP support <media-pipeline-ftp>` for media pipelines
* New :attr:`Response.certificate <scrapy.http.Response.certificate>`
attribute
* IPv6 support through :setting:`DNS_RESOLVER`
Backward-incompatible changes
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
* Python 2 support has been removed, following `Python 2 end-of-life on
January 1, 2020`_ (:issue:`4091`, :issue:`4114`, :issue:`4115`,
:issue:`4121`, :issue:`4138`, :issue:`4231`, :issue:`4242`, :issue:`4304`,
:issue:`4309`, :issue:`4373`)
* Retry gaveups (see :setting:`RETRY_TIMES`) are now logged as errors instead
of as debug information (:issue:`3171`, :issue:`3566`)
* File extensions that
:class:`LinkExtractor <scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor>`
ignores by default now also include ``7z``, ``7zip``, ``apk``, ``bz2``,
``cdr``, ``dmg``, ``ico``, ``iso``, ``tar``, ``tar.gz``, ``webm``, and
``xz`` (:issue:`1837`, :issue:`2067`, :issue:`4066`)
* The :setting:`METAREFRESH_IGNORE_TAGS` setting is now an empty list by
default, following web browser behavior (:issue:`3844`, :issue:`4311`)
* The
:class:`~scrapy.downloadermiddlewares.httpcompression.HttpCompressionMiddleware`
now includes spaces after commas in the value of the ``Accept-Encoding``
header that it sets, following web browser behavior (:issue:`4293`)
* The ``__init__`` method of custom download handlers (see
:setting:`DOWNLOAD_HANDLERS`) or subclasses of the following downloader
handlers no longer receives a ``settings`` parameter:
* :class:`scrapy.core.downloader.handlers.datauri.DataURIDownloadHandler`
* :class:`scrapy.core.downloader.handlers.file.FileDownloadHandler`
Use the ``from_settings`` or ``from_crawler`` class methods to expose such
a parameter to your custom download handlers.
(:issue:`4126`)
* We have refactored the :class:`scrapy.core.scheduler.Scheduler` class and
related queue classes (see :setting:`SCHEDULER_PRIORITY_QUEUE`,
:setting:`SCHEDULER_DISK_QUEUE` and :setting:`SCHEDULER_MEMORY_QUEUE`) to
make it easier to implement custom scheduler queue classes. See
:ref:`2-0-0-scheduler-queue-changes` below for details.
* Overridden settings are now logged in a different format. This is more in
line with similar information logged at startup (:issue:`4199`)
.. _Python 2 end-of-life on January 1, 2020: https://www.python.org/doc/sunset-python-2/
Deprecation removals
~~~~~~~~~~~~~~~~~~~~
* The :ref:`Scrapy shell <topics-shell>` no longer provides a `sel` proxy
object, use :meth:`response.selector <scrapy.http.Response.selector>`
instead (:issue:`4347`)
* LevelDB support has been removed (:issue:`4112`)
* The following functions have been removed from :mod:`scrapy.utils.python`:
``isbinarytext``, ``is_writable``, ``setattr_default``, ``stringify_dict``
(:issue:`4362`)
Deprecations
~~~~~~~~~~~~
* Using environment variables prefixed with ``SCRAPY_`` to override settings
is deprecated (:issue:`4300`, :issue:`4374`, :issue:`4375`)
* :class:`scrapy.linkextractors.FilteringLinkExtractor` is deprecated, use
:class:`scrapy.linkextractors.LinkExtractor
<scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor>` instead (:issue:`4045`)
* The ``noconnect`` query string argument of proxy URLs is deprecated and
should be removed from proxy URLs (:issue:`4198`)
* The :meth:`next <scrapy.utils.python.MutableChain.next>` method of
:class:`scrapy.utils.python.MutableChain` is deprecated, use the global
:func:`next` function or :meth:`MutableChain.__next__
<scrapy.utils.python.MutableChain.__next__>` instead (:issue:`4153`)
New features
~~~~~~~~~~~~
* Added :doc:`partial support <topics/coroutines>` for Pythons
:ref:`coroutine syntax <async>` and :doc:`experimental support
<topics/asyncio>` for :mod:`asyncio` and :mod:`asyncio`-powered libraries
(:issue:`4010`, :issue:`4259`, :issue:`4269`, :issue:`4270`, :issue:`4271`,
:issue:`4316`, :issue:`4318`)
* The new :meth:`Response.follow_all <scrapy.http.Response.follow_all>`
method offers the same functionality as
:meth:`Response.follow <scrapy.http.Response.follow>` but supports an
iterable of URLs as input and returns an iterable of requests
(:issue:`2582`, :issue:`4057`, :issue:`4286`)
* :ref:`Media pipelines <topics-media-pipeline>` now support :ref:`FTP
storage <media-pipeline-ftp>` (:issue:`3928`, :issue:`3961`)
* The new :attr:`Response.certificate <scrapy.http.Response.certificate>`
attribute exposes the SSL certificate of the server as a
:class:`twisted.internet.ssl.Certificate` object for HTTPS responses
(:issue:`2726`, :issue:`4054`)
* A new :setting:`DNS_RESOLVER` setting allows enabling IPv6 support
(:issue:`1031`, :issue:`4227`)
* A new :setting:`SCRAPER_SLOT_MAX_ACTIVE_SIZE` setting allows configuring
the existing soft limit that pauses request downloads when the total
response data being processed is too high (:issue:`1410`, :issue:`3551`)
* A new :setting:`TWISTED_REACTOR` setting allows customizing the
:mod:`~twisted.internet.reactor` that Scrapy uses, allowing to
:doc:`enable asyncio support <topics/asyncio>` or deal with a
:ref:`common macOS issue <faq-specific-reactor>` (:issue:`2905`,
:issue:`4294`)
* Scheduler disk and memory queues may now use the class methods
``from_crawler`` or ``from_settings`` (:issue:`3884`)
* The new :attr:`Response.cb_kwargs <scrapy.http.Response.cb_kwargs>`
attribute serves as a shortcut for :attr:`Response.request.cb_kwargs
<scrapy.http.Request.cb_kwargs>` (:issue:`4331`)
* :meth:`Response.follow <scrapy.http.Response.follow>` now supports a
``flags`` parameter, for consistency with :class:`~scrapy.http.Request`
(:issue:`4277`, :issue:`4279`)
* :ref:`Item loader processors <topics-loaders-processors>` can now be
regular functions, they no longer need to be methods (:issue:`3899`)
* :class:`~scrapy.spiders.Rule` now accepts an ``errback`` parameter
(:issue:`4000`)
* :class:`~scrapy.http.Request` no longer requires a ``callback`` parameter
when an ``errback`` parameter is specified (:issue:`3586`, :issue:`4008`)
* :class:`~scrapy.logformatter.LogFormatter` now supports some additional
methods:
* :class:`~scrapy.logformatter.LogFormatter.download_error` for
download errors
* :class:`~scrapy.logformatter.LogFormatter.item_error` for exceptions
raised during item processing by :ref:`item pipelines
<topics-item-pipeline>`
* :class:`~scrapy.logformatter.LogFormatter.spider_error` for exceptions
raised from :ref:`spider callbacks <topics-spiders>`
(:issue:`374`, :issue:`3986`, :issue:`3989`, :issue:`4176`, :issue:`4188`)
* The :setting:`FEED_URI` setting now supports :class:`pathlib.Path` values
(:issue:`3731`, :issue:`4074`)
* A new :signal:`request_left_downloader` signal is sent when a request
leaves the downloader (:issue:`4303`)
* Scrapy logs a warning when it detects a request callback or errback that
uses ``yield`` but also returns a value, since the returned value would be
lost (:issue:`3484`, :issue:`3869`)
* :class:`~scrapy.spiders.Spider` objects now raise an :exc:`AttributeError`
exception if they do not have a :class:`~scrapy.spiders.Spider.start_urls`
attribute nor reimplement :class:`~scrapy.spiders.Spider.start_requests`,
but have a ``start_url`` attribute (:issue:`4133`, :issue:`4170`)
* :class:`~scrapy.exporters.BaseItemExporter` subclasses may now use
``super().__init__(**kwargs)`` instead of ``self._configure(kwargs)`` in
their ``__init__`` method, passing ``dont_fail=True`` to the parent
``__init__`` method if needed, and accessing ``kwargs`` at ``self._kwargs``
after calling their parent ``__init__`` method (:issue:`4193`,
:issue:`4370`)
* A new ``keep_fragments`` parameter of
:func:`scrapy.utils.request.request_fingerprint` allows to generate
different fingerprints for requests with different fragments in their URL
(:issue:`4104`)
* Download handlers (see :setting:`DOWNLOAD_HANDLERS`) may now use the
``from_settings`` and ``from_crawler`` class methods that other Scrapy
components already supported (:issue:`4126`)
* :class:`scrapy.utils.python.MutableChain.__iter__` now returns ``self``,
`allowing it to be used as a sequence <https://lgtm.com/rules/4850080/>`_
(:issue:`4153`)
Bug fixes
~~~~~~~~~
* The :command:`crawl` command now also exits with exit code 1 when an
exception happens before the crawling starts (:issue:`4175`, :issue:`4207`)
* :class:`LinkExtractor.extract_links
<scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor.extract_links>` no longer
re-encodes the query string or URLs from non-UTF-8 responses in UTF-8
(:issue:`998`, :issue:`1403`, :issue:`1949`, :issue:`4321`)
* The first spider middleware (see :setting:`SPIDER_MIDDLEWARES`) now also
processes exceptions raised from callbacks that are generators
(:issue:`4260`, :issue:`4272`)
* Redirects to URLs starting with 3 slashes (``///``) are now supported
(:issue:`4032`, :issue:`4042`)
* :class:`~scrapy.http.Request` no longer accepts strings as ``url`` simply
because they have a colon (:issue:`2552`, :issue:`4094`)
* The correct encoding is now used for attach names in
:class:`~scrapy.mail.MailSender` (:issue:`4229`, :issue:`4239`)
* :class:`~scrapy.dupefilters.RFPDupeFilter`, the default
:setting:`DUPEFILTER_CLASS`, no longer writes an extra ``\r`` character on
each line in Windows, which made the size of the ``requests.seen`` file
unnecessarily large on that platform (:issue:`4283`)
* Z shell auto-completion now looks for ``.html`` files, not ``.http`` files,
and covers the ``-h`` command-line switch (:issue:`4122`, :issue:`4291`)
* Adding items to a :class:`scrapy.utils.datatypes.LocalCache` object
without a ``limit`` defined no longer raises a :exc:`TypeError` exception
(:issue:`4123`)
* Fixed a typo in the message of the :exc:`ValueError` exception raised when
:func:`scrapy.utils.misc.create_instance` gets both ``settings`` and
``crawler`` set to ``None`` (:issue:`4128`)
Documentation
~~~~~~~~~~~~~
* API documentation now links to an online, syntax-highlighted view of the
corresponding source code (:issue:`4148`)
* Links to unexisting documentation pages now allow access to the sidebar
(:issue:`4152`, :issue:`4169`)
* Cross-references within our documentation now display a tooltip when
hovered (:issue:`4173`, :issue:`4183`)
* Improved the documentation about :meth:`LinkExtractor.extract_links
<scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor.extract_links>` and
simplified :ref:`topics-link-extractors` (:issue:`4045`)
* Clarified how :class:`ItemLoader.item <scrapy.loader.ItemLoader.item>`
works (:issue:`3574`, :issue:`4099`)
* Clarified that :func:`logging.basicConfig` should not be used when also
using :class:`~scrapy.crawler.CrawlerProcess` (:issue:`2149`,
:issue:`2352`, :issue:`3146`, :issue:`3960`)
* Clarified the requirements for :class:`~scrapy.http.Request` objects
:ref:`when using persistence <request-serialization>` (:issue:`4124`,
:issue:`4139`)
* Clarified how to install a :ref:`custom image pipeline
<media-pipeline-example>` (:issue:`4034`, :issue:`4252`)
* Fixed the signatures of the ``file_path`` method in :ref:`media pipeline
<topics-media-pipeline>` examples (:issue:`4290`)
* Covered a backward-incompatible change in Scrapy 1.7.0 affecting custom
:class:`scrapy.core.scheduler.Scheduler` subclasses (:issue:`4274`)
* Improved the ``README.rst`` and ``CODE_OF_CONDUCT.md`` files
(:issue:`4059`)
* Documentation examples are now checked as part of our test suite and we
have fixed some of the issues detected (:issue:`4142`, :issue:`4146`,
:issue:`4171`, :issue:`4184`, :issue:`4190`)
* Fixed logic issues, broken links and typos (:issue:`4247`, :issue:`4258`,
:issue:`4282`, :issue:`4288`, :issue:`4305`, :issue:`4308`, :issue:`4323`,
:issue:`4338`, :issue:`4359`, :issue:`4361`)
* Improved consistency when referring to the ``__init__`` method of an object
(:issue:`4086`, :issue:`4088`)
* Fixed an inconsistency between code and output in :ref:`intro-overview`
(:issue:`4213`)
* Extended :mod:`~sphinx.ext.intersphinx` usage (:issue:`4147`,
:issue:`4172`, :issue:`4185`, :issue:`4194`, :issue:`4197`)
* We now use a recent version of Python to build the documentation
(:issue:`4140`, :issue:`4249`)
* Cleaned up documentation (:issue:`4143`, :issue:`4275`)
Quality assurance
~~~~~~~~~~~~~~~~~
* Re-enabled proxy ``CONNECT`` tests (:issue:`2545`, :issue:`4114`)
* Added Bandit_ security checks to our test suite (:issue:`4162`,
:issue:`4181`)
* Added Flake8_ style checks to our test suite and applied many of the
corresponding changes (:issue:`3944`, :issue:`3945`, :issue:`4137`,
:issue:`4157`, :issue:`4167`, :issue:`4174`, :issue:`4186`, :issue:`4195`,
:issue:`4238`, :issue:`4246`, :issue:`4355`, :issue:`4360`, :issue:`4365`)
* Improved test coverage (:issue:`4097`, :issue:`4218`, :issue:`4236`)
* Started reporting slowest tests, and improved the performance of some of
them (:issue:`4163`, :issue:`4164`)
* Fixed broken tests and refactored some tests (:issue:`4014`, :issue:`4095`,
:issue:`4244`, :issue:`4268`, :issue:`4372`)
* Modified the :doc:`tox <tox:index>` configuration to allow running tests
with any Python version, run Bandit_ and Flake8_ tests by default, and
enforce a minimum tox version programmatically (:issue:`4179`)
* Cleaned up code (:issue:`3937`, :issue:`4208`, :issue:`4209`,
:issue:`4210`, :issue:`4212`, :issue:`4369`, :issue:`4376`, :issue:`4378`)
.. _Bandit: https://bandit.readthedocs.io/
.. _Flake8: https://flake8.pycqa.org/en/latest/
.. _2-0-0-scheduler-queue-changes:
Changes to scheduler queue classes
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
The following changes may impact any custom queue classes of all types:
* The ``push`` method no longer receives a second positional parameter
containing ``request.priority * -1``. If you need that value, get it
from the first positional parameter, ``request``, instead, or use
the new :meth:`~scrapy.core.scheduler.ScrapyPriorityQueue.priority`
method in :class:`scrapy.core.scheduler.ScrapyPriorityQueue`
subclasses.
The following changes may impact custom priority queue classes:
* In the ``__init__`` method or the ``from_crawler`` or ``from_settings``
class methods:
* The parameter that used to contain a factory function,
``qfactory``, is now passed as a keyword parameter named
``downstream_queue_cls``.
* A new keyword parameter has been added: ``key``. It is a string
that is always an empty string for memory queues and indicates the
:setting:`JOB_DIR` value for disk queues.
* The parameter for disk queues that contains data from the previous
crawl, ``startprios`` or ``slot_startprios``, is now passed as a
keyword parameter named ``startprios``.
* The ``serialize`` parameter is no longer passed. The disk queue
class must take care of request serialization on its own before
writing to disk, using the
:func:`~scrapy.utils.reqser.request_to_dict` and
:func:`~scrapy.utils.reqser.request_from_dict` functions from the
:mod:`scrapy.utils.reqser` module.
The following changes may impact custom disk and memory queue classes:
* The signature of the ``__init__`` method is now
``__init__(self, crawler, key)``.
The following changes affect specifically the
:class:`~scrapy.core.scheduler.ScrapyPriorityQueue` and
:class:`~scrapy.core.scheduler.DownloaderAwarePriorityQueue` classes from
:mod:`scrapy.core.scheduler` and may affect subclasses:
* In the ``__init__`` method, most of the changes described above apply.
``__init__`` may still receive all parameters as positional parameters,
however:
* ``downstream_queue_cls``, which replaced ``qfactory``, must be
instantiated differently.
``qfactory`` was instantiated with a priority value (integer).
Instances of ``downstream_queue_cls`` should be created using
the new
:meth:`ScrapyPriorityQueue.qfactory <scrapy.core.scheduler.ScrapyPriorityQueue.qfactory>`
or
:meth:`DownloaderAwarePriorityQueue.pqfactory <scrapy.core.scheduler.DownloaderAwarePriorityQueue.pqfactory>`
methods.
* The new ``key`` parameter displaced the ``startprios``
parameter 1 position to the right.
* The following class attributes have been added:
* :attr:`~scrapy.core.scheduler.ScrapyPriorityQueue.crawler`
* :attr:`~scrapy.core.scheduler.ScrapyPriorityQueue.downstream_queue_cls`
(details above)
* :attr:`~scrapy.core.scheduler.ScrapyPriorityQueue.key` (details above)
* The ``serialize`` attribute has been removed (details above)
The following changes affect specifically the
:class:`~scrapy.core.scheduler.ScrapyPriorityQueue` class and may affect
subclasses:
* A new :meth:`~scrapy.core.scheduler.ScrapyPriorityQueue.priority`
method has been added which, given a request, returns
``request.priority * -1``.
It is used in :meth:`~scrapy.core.scheduler.ScrapyPriorityQueue.push`
to make up for the removal of its ``priority`` parameter.
* The ``spider`` attribute has been removed. Use
:attr:`crawler.spider <scrapy.core.scheduler.ScrapyPriorityQueue.crawler>`
instead.
The following changes affect specifically the
:class:`~scrapy.core.scheduler.DownloaderAwarePriorityQueue` class and may
affect subclasses:
* A new :attr:`~scrapy.core.scheduler.DownloaderAwarePriorityQueue.pqueues`
attribute offers a mapping of downloader slot names to the
corresponding instances of
:attr:`~scrapy.core.scheduler.DownloaderAwarePriorityQueue.downstream_queue_cls`.
(:issue:`3884`)
.. _release-1.8.0:
@ -26,7 +486,7 @@ Backward-incompatible changes
* Python 3.4 is no longer supported, and some of the minimum requirements of
Scrapy have also changed:
* cssselect_ 0.9.1
* :doc:`cssselect <cssselect:index>` 0.9.1
* cryptography_ 2.0
* lxml_ 3.5.0
* pyOpenSSL_ 16.2.0
@ -288,12 +748,12 @@ Backward-incompatible changes
:class:`~scrapy.http.Request` objects instead of arbitrary Python data
structures.
* An additional ``crawler`` parameter has been added to the ``__init__`` method
of the :class:`scrapy.core.scheduler.Scheduler` class.
Custom scheduler subclasses which don't accept arbitrary parameters in
their ``__init__`` method might break because of this change.
* An additional ``crawler`` parameter has been added to the ``__init__``
method of the :class:`~scrapy.core.scheduler.Scheduler` class. Custom
scheduler subclasses which don't accept arbitrary parameters in their
``__init__`` method might break because of this change.
For more information, refer to the documentation for the :setting:`SCHEDULER` setting.
For more information, see :setting:`SCHEDULER`.
See also :ref:`1.7-deprecation-removals` below.
@ -1076,7 +1536,7 @@ Cleanups & Refactoring
~~~~~~~~~~~~~~~~~~~~~~
- Tests: remove temp files and folders (:issue:`2570`),
fixed ProjectUtilsTest on OS X (:issue:`2569`),
fixed ProjectUtilsTest on macOS (:issue:`2569`),
use portable pypy for Linux on Travis CI (:issue:`2710`)
- Separate building request from ``_requests_to_follow`` in CrawlSpider (:issue:`2562`)
- Remove “Python 3 progress” badge (:issue:`2567`)
@ -1616,7 +2076,7 @@ Deprecations and Removals
+ ``scrapy.utils.datatypes.SiteNode``
- The previously bundled ``scrapy.xlib.pydispatch`` library was deprecated and
replaced by `pydispatcher <https://pypi.python.org/pypi/PyDispatcher>`_.
replaced by `pydispatcher <https://pypi.org/project/PyDispatcher/>`_.
Relocations
@ -1645,7 +2105,7 @@ Bugfixes
- Makes ``_monkeypatches`` more robust (:issue:`1634`).
- Fixed bug on ``XMLItemExporter`` with non-string fields in
items (:issue:`1738`).
- Fixed startproject command in OS X (:issue:`1635`).
- Fixed startproject command in macOS (:issue:`1635`).
- Fixed :class:`~scrapy.exporters.PythonItemExporter` and CSVExporter for
non-string item types (:issue:`1737`).
- Various logging related fixes (:issue:`1294`, :issue:`1419`, :issue:`1263`,
@ -1713,12 +2173,12 @@ Scrapy 1.0.4 (2015-12-30)
- Typos corrections (:commit:`7067117`)
- fix typos in downloader-middleware.rst and exceptions.rst, middlware -> middleware (:commit:`32f115c`)
- Add note to Ubuntu install section about Debian compatibility (:commit:`23fda69`)
- Replace alternative OSX install workaround with virtualenv (:commit:`98b63ee`)
- Replace alternative macOS install workaround with virtualenv (:commit:`98b63ee`)
- Reference Homebrew's homepage for installation instructions (:commit:`1925db1`)
- Add oldest supported tox version to contributing docs (:commit:`5d10d6d`)
- Note in install docs about pip being already included in python>=2.7.9 (:commit:`85c980e`)
- Add non-python dependencies to Ubuntu install section in the docs (:commit:`fbd010d`)
- Add OS X installation section to docs (:commit:`d8f4cba`)
- Add macOS installation section to docs (:commit:`d8f4cba`)
- DOC(ENH): specify path to rtd theme explicitly (:commit:`de73b1a`)
- minor: scrapy.Spider docs grammar (:commit:`1ddcc7b`)
- Make common practices sample code match the comments (:commit:`1b85bcf`)
@ -2450,7 +2910,7 @@ Other
~~~~~
- Dropped Python 2.6 support (:issue:`448`)
- Add `cssselect`_ python package as install dependency
- Add :doc:`cssselect <cssselect:index>` python package as install dependency
- Drop libxml2 and multi selector's backend support, `lxml`_ is required from now on.
- Minimum Twisted version increased to 10.0.0, dropped Twisted 8.0 support.
- Running test suite now requires ``mock`` python library (:issue:`390`)
@ -2571,7 +3031,7 @@ Scrapy 0.18.0 (released 2013-08-09)
- MetaRefreshMiddldeware and RedirectMiddleware have different priorities to address #62
- added from_crawler method to spiders
- added system tests with mock server
- more improvements to Mac OS compatibility (thanks Alex Cepoi)
- more improvements to macOS compatibility (thanks Alex Cepoi)
- several more cleanups to singletons and multi-spider support (thanks Nicolas Ramirez)
- support custom download slots
- added --spider option to "shell" command.
@ -2647,7 +3107,7 @@ Scrapy 0.16.3 (released 2012-12-07)
- Remove concurrency limitation when using download delays and still ensure inter-request delays are enforced (:commit:`487b9b5`)
- add error details when image pipeline fails (:commit:`8232569`)
- improve mac os compatibility (:commit:`8dcf8aa`)
- improve macOS compatibility (:commit:`8dcf8aa`)
- setup.py: use README.rst to populate long_description (:commit:`7b5310d`)
- doc: removed obsolete references to ClientForm (:commit:`80f9bb6`)
- correct docs for default storage backend (:commit:`2aa491b`)
@ -3047,17 +3507,16 @@ Scrapy 0.7
First release of Scrapy.
.. _AJAX crawleable urls: https://developers.google.com/webmasters/ajax-crawling/docs/getting-started?csw=1
.. _AJAX crawleable urls: https://developers.google.com/search/docs/ajax-crawling/docs/getting-started?csw=1
.. _botocore: https://github.com/boto/botocore
.. _chunked transfer encoding: https://en.wikipedia.org/wiki/Chunked_transfer_encoding
.. _ClientForm: http://wwwsearch.sourceforge.net/old/ClientForm/
.. _Creating a pull request: https://help.github.com/en/articles/creating-a-pull-request
.. _cryptography: https://cryptography.io/en/latest/
.. _cssselect: https://github.com/scrapy/cssselect/
.. _docstrings: https://docs.python.org/glossary.html#term-docstring
.. _KeyboardInterrupt: https://docs.python.org/library/exceptions.html#KeyboardInterrupt
.. _docstrings: https://docs.python.org/3/glossary.html#term-docstring
.. _KeyboardInterrupt: https://docs.python.org/3/library/exceptions.html#KeyboardInterrupt
.. _LevelDB: https://github.com/google/leveldb
.. _lxml: http://lxml.de/
.. _lxml: https://lxml.de/
.. _marshal: https://docs.python.org/2/library/marshal.html
.. _parsel.csstranslator.GenericTranslator: https://parsel.readthedocs.io/en/latest/parsel.html#parsel.csstranslator.GenericTranslator
.. _parsel.csstranslator.HTMLTranslator: https://parsel.readthedocs.io/en/latest/parsel.html#parsel.csstranslator.HTMLTranslator
@ -3068,11 +3527,11 @@ First release of Scrapy.
.. _queuelib: https://github.com/scrapy/queuelib
.. _registered with IANA: https://www.iana.org/assignments/media-types/media-types.xhtml
.. _resource: https://docs.python.org/2/library/resource.html
.. _robots.txt: http://www.robotstxt.org/
.. _robots.txt: https://www.robotstxt.org/
.. _scrapely: https://github.com/scrapy/scrapely
.. _service_identity: https://service-identity.readthedocs.io/en/stable/
.. _six: https://six.readthedocs.io/
.. _tox: https://pypi.python.org/pypi/tox
.. _tox: https://pypi.org/project/tox/
.. _Twisted: https://twistedmatrix.com/trac/
.. _Twisted - hello, asynchronous programming: http://jessenoller.com/blog/2009/02/11/twisted-hello-asynchronous-programming/
.. _w3lib: https://github.com/scrapy/w3lib

28
docs/topics/asyncio.rst Normal file
View File

@ -0,0 +1,28 @@
=======
asyncio
=======
.. versionadded:: 2.0
Scrapy has partial support :mod:`asyncio`. After you :ref:`install the asyncio
reactor <install-asyncio>`, you may use :mod:`asyncio` and
:mod:`asyncio`-powered libraries in any :doc:`coroutine <coroutines>`.
.. warning:: :mod:`asyncio` support in Scrapy is experimental. Future Scrapy
versions may introduce related changes without a deprecation
period or warning.
.. _install-asyncio:
Installing the asyncio reactor
==============================
To enable :mod:`asyncio` support, set the :setting:`TWISTED_REACTOR` setting to
``'twisted.internet.asyncioreactor.AsyncioSelectorReactor'``.
If you are using :class:`~scrapy.crawler.CrawlerRunner`, you also need to
install the :class:`~twisted.internet.asyncioreactor.AsyncioSelectorReactor`
reactor manually. You can do that using
:func:`~scrapy.utils.reactor.install_reactor`::
install_reactor('twisted.internet.asyncioreactor.AsyncioSelectorReactor')

View File

@ -188,7 +188,7 @@ AjaxCrawlMiddleware helps to crawl them correctly.
It is turned OFF by default because it has some performance overhead,
and enabling it for focused crawls doesn't make much sense.
.. _ajax crawlable: https://developers.google.com/webmasters/ajax-crawling/docs/getting-started
.. _ajax crawlable: https://developers.google.com/search/docs/ajax-crawling/docs/getting-started
.. _broad-crawls-bfo:

110
docs/topics/coroutines.rst Normal file
View File

@ -0,0 +1,110 @@
==========
Coroutines
==========
.. versionadded:: 2.0
Scrapy has :ref:`partial support <coroutine-support>` for the
:ref:`coroutine syntax <async>`.
.. warning:: :mod:`asyncio` support in Scrapy is experimental. Future Scrapy
versions may introduce related API and behavior changes without a
deprecation period or warning.
.. _coroutine-support:
Supported callables
===================
The following callables may be defined as coroutines using ``async def``, and
hence use coroutine syntax (e.g. ``await``, ``async for``, ``async with``):
- :class:`~scrapy.http.Request` callbacks.
The following are known caveats of the current implementation that we aim
to address in future versions of Scrapy:
- The callback output is not processed until the whole callback finishes.
As a side effect, if the callback raises an exception, none of its
output is processed.
- Because `asynchronous generators were introduced in Python 3.6`_, you
can only use ``yield`` if you are using Python 3.6 or later.
If you need to output multiple items or requests and you are using
Python 3.5, return an iterable (e.g. a list) instead.
- The :meth:`process_item` method of
:ref:`item pipelines <topics-item-pipeline>`.
- The
:meth:`~scrapy.downloadermiddlewares.DownloaderMiddleware.process_request`,
:meth:`~scrapy.downloadermiddlewares.DownloaderMiddleware.process_response`,
and
:meth:`~scrapy.downloadermiddlewares.DownloaderMiddleware.process_exception`
methods of
:ref:`downloader middlewares <topics-downloader-middleware-custom>`.
- :ref:`Signal handlers that support deferreds <signal-deferred>`.
.. _asynchronous generators were introduced in Python 3.6: https://www.python.org/dev/peps/pep-0525/
Usage
=====
There are several use cases for coroutines in Scrapy. Code that would
return Deferreds when written for previous Scrapy versions, such as downloader
middlewares and signal handlers, can be rewritten to be shorter and cleaner::
class DbPipeline:
def _update_item(self, data, item):
item['field'] = data
return item
def process_item(self, item, spider):
dfd = db.get_some_data(item['id'])
dfd.addCallback(self._update_item, item)
return dfd
becomes::
class DbPipeline:
async def process_item(self, item, spider):
item['field'] = await db.get_some_data(item['id'])
return item
Coroutines may be used to call asynchronous code. This includes other
coroutines, functions that return Deferreds and functions that return
`awaitable objects`_ such as :class:`~asyncio.Future`. This means you can use
many useful Python libraries providing such code::
class MySpider(Spider):
# ...
async def parse_with_deferred(self, response):
additional_response = await treq.get('https://additional.url')
additional_data = await treq.content(additional_response)
# ... use response and additional_data to yield items and requests
async def parse_with_asyncio(self, response):
async with aiohttp.ClientSession() as session:
async with session.get('https://additional.url') as additional_response:
additional_data = await r.text()
# ... use response and additional_data to yield items and requests
.. note:: Many libraries that use coroutines, such as `aio-libs`_, require the
:mod:`asyncio` loop and to use them you need to
:doc:`enable asyncio support in Scrapy<asyncio>`.
Common use cases for asynchronous code include:
* requesting data from websites, databases and other services (in callbacks,
pipelines and middlewares);
* storing data in databases (in pipelines and middlewares);
* delaying the spider initialization until some external event (in the
:signal:`spider_opened` handler);
* calling asynchronous Scrapy methods like ``ExecutionEngine.download`` (see
:ref:`the screenshot pipeline example<ScreenshotPipeline>`).
.. _aio-libs: https://github.com/aio-libs
.. _awaitable objects: https://docs.python.org/3/glossary.html#term-awaitable

View File

@ -259,8 +259,8 @@ COOKIES_DEBUG
Default: ``False``
If enabled, Scrapy will log all cookies sent in requests (ie. ``Cookie``
header) and all cookies received in responses (ie. ``Set-Cookie`` header).
If enabled, Scrapy will log all cookies sent in requests (i.e. ``Cookie``
header) and all cookies received in responses (i.e. ``Set-Cookie`` header).
Here's an example of a log with :setting:`COOKIES_DEBUG` enabled::
@ -709,7 +709,7 @@ HttpCompressionMiddleware
provided `brotlipy`_ is installed.
.. _brotli-compressed: https://www.ietf.org/rfc/rfc7932.txt
.. _brotlipy: https://pypi.python.org/pypi/brotlipy
.. _brotlipy: https://pypi.org/project/brotlipy/
HttpCompressionMiddleware Settings
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
@ -872,6 +872,10 @@ Default: ``[]``
Meta tags within these tags are ignored.
.. versionchanged:: 2.0
The default value of :setting:`METAREFRESH_IGNORE_TAGS` changed from
``['script', 'noscript']`` to ``[]``.
.. setting:: METAREFRESH_MAXDELAY
METAREFRESH_MAXDELAY
@ -1038,7 +1042,7 @@ Based on `RobotFileParser
* is Python's built-in robots.txt_ parser
* is compliant with `Martijn Koster's 1996 draft specification
<http://www.robotstxt.org/norobots-rfc.txt>`_
<https://www.robotstxt.org/norobots-rfc.txt>`_
* lacks support for wildcard matching
@ -1061,7 +1065,7 @@ Based on `Reppy <https://github.com/seomoz/reppy/>`_:
<https://github.com/seomoz/rep-cpp>`_
* is compliant with `Martijn Koster's 1996 draft specification
<http://www.robotstxt.org/norobots-rfc.txt>`_
<https://www.robotstxt.org/norobots-rfc.txt>`_
* supports wildcard matching
@ -1086,7 +1090,7 @@ Based on `Robotexclusionrulesparser <http://nikitathespider.com/python/rerp/>`_:
* implemented in Python
* is compliant with `Martijn Koster's 1996 draft specification
<http://www.robotstxt.org/norobots-rfc.txt>`_
<https://www.robotstxt.org/norobots-rfc.txt>`_
* supports wildcard matching
@ -1115,7 +1119,7 @@ implementing the methods described below.
.. autoclass:: RobotParser
:members:
.. _robots.txt: http://www.robotstxt.org/
.. _robots.txt: https://www.robotstxt.org/
DownloaderStats
---------------
@ -1155,7 +1159,7 @@ AjaxCrawlMiddleware
Middleware that finds 'AJAX crawlable' page variants based
on meta-fragment html tag. See
https://developers.google.com/webmasters/ajax-crawling/docs/getting-started
https://developers.google.com/search/docs/ajax-crawling/docs/getting-started
for more info.
.. note::

View File

@ -241,12 +241,12 @@ along with `scrapy-selenium`_ for seamless integration.
.. _headless browser: https://en.wikipedia.org/wiki/Headless_browser
.. _JavaScript: https://en.wikipedia.org/wiki/JavaScript
.. _js2xml: https://github.com/scrapinghub/js2xml
.. _json.loads: https://docs.python.org/library/json.html#json.loads
.. _json.loads: https://docs.python.org/3/library/json.html#json.loads
.. _pytesseract: https://github.com/madmaze/pytesseract
.. _regular expression: https://docs.python.org/library/re.html
.. _regular expression: https://docs.python.org/3/library/re.html
.. _scrapy-selenium: https://github.com/clemfromspace/scrapy-selenium
.. _scrapy-splash: https://github.com/scrapy-plugins/scrapy-splash
.. _Selenium: https://www.seleniumhq.org/
.. _Selenium: https://www.selenium.dev/
.. _Splash: https://github.com/scrapinghub/splash
.. _tabula-py: https://github.com/chezou/tabula-py
.. _wget: https://www.gnu.org/software/wget/

View File

@ -42,7 +42,7 @@ value of one of their fields::
from scrapy.exporters import XmlItemExporter
class PerYearXmlExportPipeline(object):
class PerYearXmlExportPipeline:
"""Distribute items across multiple XML files according to their 'year' field"""
def open_spider(self, spider):
@ -137,7 +137,7 @@ output examples, which assume you're exporting these two items::
BaseItemExporter
----------------
.. class:: BaseItemExporter(fields_to_export=None, export_empty_fields=False, encoding='utf-8', indent=0)
.. class:: BaseItemExporter(fields_to_export=None, export_empty_fields=False, encoding='utf-8', indent=0, dont_fail=False)
This is the (abstract) base class for all Item Exporters. It provides
support for common features used by all (concrete) Item Exporters, such as
@ -148,6 +148,9 @@ BaseItemExporter
populate their respective instance attributes: :attr:`fields_to_export`,
:attr:`export_empty_fields`, :attr:`encoding`, :attr:`indent`.
.. versionadded:: 2.0
The *dont_fail* parameter.
.. method:: export_item(item)
Exports the given item. This method must be implemented in subclasses.

View File

@ -63,7 +63,7 @@ but disabled unless the :setting:`HTTPCACHE_ENABLED` setting is set.
Disabling an extension
======================
In order to disable an extension that comes enabled by default (ie. those
In order to disable an extension that comes enabled by default (i.e. those
included in the :setting:`EXTENSIONS_BASE` setting) you must set its order to
``None``. For example::
@ -107,7 +107,7 @@ Here is the code of such extension::
logger = logging.getLogger(__name__)
class SpiderOpenCloseLogging(object):
class SpiderOpenCloseLogging:
def __init__(self, item_count):
self.item_count = item_count
@ -345,7 +345,7 @@ signal is received. The information dumped is the following:
After the stack trace and engine status is dumped, the Scrapy process continues
running normally.
This extension only works on POSIX-compliant platforms (ie. not Windows),
This extension only works on POSIX-compliant platforms (i.e. not Windows),
because the `SIGQUIT`_ and `SIGUSR2`_ signals are not available on Windows.
There are at least two ways to send Scrapy the `SIGQUIT`_ signal:
@ -370,7 +370,7 @@ running normally.
For more info see `Debugging in Python`_.
This extension only works on POSIX-compliant platforms (ie. not Windows).
This extension only works on POSIX-compliant platforms (i.e. not Windows).
.. _Python debugger: https://docs.python.org/2/library/pdb.html
.. _Debugging in Python: https://pythonconquerstheuniverse.wordpress.com/2009/09/10/debugging-in-python/

View File

@ -12,7 +12,7 @@ generating an "export file" with the scraped data (commonly called "export
feed") to be consumed by other systems.
Scrapy provides this functionality out of the box with the Feed Exports, which
allows you to generate a feed with the scraped items, using multiple
allows you to generate feeds with the scraped items, using multiple
serialization formats and storage backends.
.. _topics-feed-format:
@ -36,7 +36,7 @@ But you can also extend the supported format through the
JSON
----
* :setting:`FEED_FORMAT`: ``json``
* Value for the ``format`` key in the :setting:`FEEDS` setting: ``json``
* Exporter used: :class:`~scrapy.exporters.JsonItemExporter`
* See :ref:`this warning <json-with-large-data>` if you're using JSON with
large feeds.
@ -46,7 +46,7 @@ JSON
JSON lines
----------
* :setting:`FEED_FORMAT`: ``jsonlines``
* Value for the ``format`` key in the :setting:`FEEDS` setting: ``jsonlines``
* Exporter used: :class:`~scrapy.exporters.JsonLinesItemExporter`
.. _topics-feed-format-csv:
@ -54,7 +54,7 @@ JSON lines
CSV
---
* :setting:`FEED_FORMAT`: ``csv``
* Value for the ``format`` key in the :setting:`FEEDS` setting: ``csv``
* Exporter used: :class:`~scrapy.exporters.CsvItemExporter`
* To specify columns to export and their order use
:setting:`FEED_EXPORT_FIELDS`. Other feed exporters can also use this
@ -66,7 +66,7 @@ CSV
XML
---
* :setting:`FEED_FORMAT`: ``xml``
* Value for the ``format`` key in the :setting:`FEEDS` setting: ``xml``
* Exporter used: :class:`~scrapy.exporters.XmlItemExporter`
.. _topics-feed-format-pickle:
@ -74,7 +74,7 @@ XML
Pickle
------
* :setting:`FEED_FORMAT`: ``pickle``
* Value for the ``format`` key in the :setting:`FEEDS` setting: ``pickle``
* Exporter used: :class:`~scrapy.exporters.PickleItemExporter`
.. _topics-feed-format-marshal:
@ -82,7 +82,7 @@ Pickle
Marshal
-------
* :setting:`FEED_FORMAT`: ``marshal``
* Value for the ``format`` key in the :setting:`FEEDS` setting: ``marshal``
* Exporter used: :class:`~scrapy.exporters.MarshalItemExporter`
@ -91,8 +91,8 @@ Marshal
Storages
========
When using the feed exports you define where to store the feed using a URI_
(through the :setting:`FEED_URI` setting). The feed exports supports multiple
When using the feed exports you define where to store the feed using one or multiple URIs_
(through the :setting:`FEEDS` setting). The feed exports supports multiple
storage backend types which are defined by the URI scheme.
The storages backends supported out of the box are:
@ -232,38 +232,66 @@ Settings
These are the settings used for configuring the feed exports:
* :setting:`FEED_URI` (mandatory)
* :setting:`FEED_FORMAT`
* :setting:`FEEDS` (mandatory)
* :setting:`FEED_EXPORT_ENCODING`
* :setting:`FEED_STORE_EMPTY`
* :setting:`FEED_EXPORT_FIELDS`
* :setting:`FEED_EXPORT_INDENT`
* :setting:`FEED_STORAGES`
* :setting:`FEED_STORAGE_FTP_ACTIVE`
* :setting:`FEED_STORAGE_S3_ACL`
* :setting:`FEED_EXPORTERS`
* :setting:`FEED_STORE_EMPTY`
* :setting:`FEED_EXPORT_ENCODING`
* :setting:`FEED_EXPORT_FIELDS`
* :setting:`FEED_EXPORT_INDENT`
.. currentmodule:: scrapy.extensions.feedexport
.. setting:: FEED_URI
.. setting:: FEEDS
FEED_URI
--------
FEEDS
-----
Default: ``None``
.. versionadded:: 2.1
The URI of the export feed. See :ref:`topics-feed-storage-backends` for
supported URI schemes.
Default: ``{}``
This setting is required for enabling the feed exports.
A dictionary in which every key is a feed URI (or a :class:`pathlib.Path`
object) and each value is a nested dictionary containing configuration
parameters for the specific feed.
This setting is required for enabling the feed export feature.
.. setting:: FEED_FORMAT
See :ref:`topics-feed-storage-backends` for supported URI schemes.
FEED_FORMAT
-----------
For instance::
The serialization format to be used for the feed. See
:ref:`topics-feed-format` for possible values.
{
'items.json': {
'format': 'json',
'encoding': 'utf8',
'store_empty': False,
'fields': None,
'indent': 4,
},
'/home/user/documents/items.xml': {
'format': 'xml',
'fields': ['name', 'price'],
'encoding': 'latin1',
'indent': 8,
},
pathlib.Path('items.csv'): {
'format': 'csv',
'fields': ['price', 'name'],
},
}
The following is a list of the accepted keys and the setting that is used
as a fallback value if that key is not provided for a specific feed definition.
* ``format``: the serialization format to be used for the feed.
See :ref:`topics-feed-format` for possible values.
Mandatory, no fallback setting
* ``encoding``: falls back to :setting:`FEED_EXPORT_ENCODING`
* ``fields``: falls back to :setting:`FEED_EXPORT_FIELDS`
* ``indent``: falls back to :setting:`FEED_EXPORT_INDENT`
* ``store_empty``: falls back to :setting:`FEED_STORE_EMPTY`
.. setting:: FEED_EXPORT_ENCODING
@ -322,7 +350,7 @@ FEED_STORE_EMPTY
Default: ``False``
Whether to export empty feeds (ie. feeds with no items).
Whether to export empty feeds (i.e. feeds with no items).
.. setting:: FEED_STORAGES
@ -418,7 +446,7 @@ format in :setting:`FEED_EXPORTERS`. E.g., to disable the built-in CSV exporter
'csv': None,
}
.. _URI: https://en.wikipedia.org/wiki/Uniform_Resource_Identifier
.. _URIs: https://en.wikipedia.org/wiki/Uniform_Resource_Identifier
.. _Amazon S3: https://aws.amazon.com/s3/
.. _botocore: https://github.com/boto/botocore
.. _Canned ACL: https://docs.aws.amazon.com/AmazonS3/latest/dev/acl-overview.html#canned-acl

View File

@ -81,7 +81,7 @@ contain a price::
from scrapy.exceptions import DropItem
class PricePipeline(object):
class PricePipeline:
vat_factor = 1.15
@ -103,7 +103,7 @@ format::
import json
class JsonWriterPipeline(object):
class JsonWriterPipeline:
def open_spider(self, spider):
self.file = open('items.jl', 'w')
@ -132,7 +132,7 @@ method and how to clean up the resources properly.::
import pymongo
class MongoPipeline(object):
class MongoPipeline:
collection_name = 'scrapy_items'
@ -158,18 +158,20 @@ method and how to clean up the resources properly.::
self.db[self.collection_name].insert_one(dict(item))
return item
.. _MongoDB: https://www.mongodb.org/
.. _pymongo: https://api.mongodb.org/python/current/
.. _MongoDB: https://www.mongodb.com/
.. _pymongo: https://api.mongodb.com/python/current/
.. _ScreenshotPipeline:
Take screenshot of item
-----------------------
This example demonstrates how to return a
:class:`~twisted.internet.defer.Deferred` from the :meth:`process_item` method.
It uses Splash_ to render screenshot of item url. Pipeline
makes request to locally running instance of Splash_. After request is downloaded
and Deferred callback fires, it saves item to a file and adds filename to an item.
makes request to locally running instance of Splash_. After request is downloaded,
it saves the screenshot to a file and adds filename to the item.
::
@ -178,21 +180,18 @@ and Deferred callback fires, it saves item to a file and adds filename to an ite
from urllib.parse import quote
class ScreenshotPipeline(object):
class ScreenshotPipeline:
"""Pipeline that uses Splash to render screenshot of
every Scrapy item."""
SPLASH_URL = "http://localhost:8050/render.png?url={}"
def process_item(self, item, spider):
async def process_item(self, item, spider):
encoded_item_url = quote(item["url"])
screenshot_url = self.SPLASH_URL.format(encoded_item_url)
request = scrapy.Request(screenshot_url)
dfd = spider.crawler.engine.download(request, spider)
dfd.addBoth(self.return_item, item)
return dfd
response = await spider.crawler.engine.download(request, spider)
def return_item(self, response, item):
if response.status != 200:
# Error happened, return item.
return item
@ -220,7 +219,7 @@ returns multiples items with the same id::
from scrapy.exceptions import DropItem
class DuplicatesPipeline(object):
class DuplicatesPipeline:
def __init__(self):
self.ids_seen = set()

View File

@ -166,7 +166,7 @@ If your item contains mutable_ values like lists or dictionaries, a shallow
copy will keep references to the same mutable values across all different
copies.
.. _mutable: https://docs.python.org/glossary.html#term-mutable
.. _mutable: https://docs.python.org/3/glossary.html#term-mutable
For example, if you have an item with a list of tags, and you create a shallow
copy of that item, both the original item and the copy have the same list of
@ -177,7 +177,7 @@ If that is not the desired behavior, use a deep copy instead.
See the `documentation of the copy module`_ for more information.
.. _documentation of the copy module: https://docs.python.org/library/copy.html
.. _documentation of the copy module: https://docs.python.org/3/library/copy.html
To create a shallow copy of an item, you can either call
:meth:`~scrapy.item.Item.copy` on an existing item

View File

@ -22,7 +22,7 @@ Job directory
To enable persistence support you just need to define a *job directory* through
the ``JOBDIR`` setting. This directory will be for storing all required data to
keep the state of a single job (ie. a spider run). It's important to note that
keep the state of a single job (i.e. a spider run). It's important to note that
this directory must not be shared by different spiders, or even different
jobs/runs of the same spider, as it's meant to be used for storing the state of
a *single* job.
@ -68,6 +68,9 @@ Cookies may expire. So, if you don't resume your spider quickly the requests
scheduled may no longer work. This won't be an issue if you spider doesn't rely
on cookies.
.. _request-serialization:
Request serialization
---------------------

View File

@ -17,8 +17,8 @@ what is known as a "memory leak".
To help debugging memory leaks, Scrapy provides a built-in mechanism for
tracking objects references called :ref:`trackref <topics-leaks-trackrefs>`,
and you can also use a third-party library called :ref:`Guppy
<topics-leaks-guppy>` for more advanced memory debugging (see below for more
and you can also use a third-party library called :ref:`muppy
<topics-leaks-muppy>` for more advanced memory debugging (see below for more
info). Both mechanisms must be used from the :ref:`Telnet Console
<topics-telnetconsole>`.
@ -170,7 +170,7 @@ Here are the functions available in the :mod:`~scrapy.utils.trackref` module.
.. class:: object_ref
Inherit from this class (instead of object) if you want to track live
Inherit from this class if you want to track live
instances with the ``trackref`` module.
.. function:: print_live_refs(class_name, ignore=NoneType)
@ -193,9 +193,9 @@ Here are the functions available in the :mod:`~scrapy.utils.trackref` module.
``None`` if none is found. Use :func:`print_live_refs` first to get a list
of all tracked live objects per class name.
.. _topics-leaks-guppy:
.. _topics-leaks-muppy:
Debugging memory leaks with Guppy
Debugging memory leaks with muppy
=================================
``trackref`` provides a very convenient mechanism for tracking down memory
@ -203,63 +203,9 @@ leaks, but it only keeps track of the objects that are more likely to cause
memory leaks (Requests, Responses, Items, and Selectors). However, there are
other cases where the memory leaks could come from other (more or less obscure)
objects. If this is your case, and you can't find your leaks using ``trackref``,
you still have another resource: the `Guppy library`_.
If you're using Python3, see :ref:`topics-leaks-muppy`.
you still have another resource: the muppy library.
.. _Guppy library: https://pypi.python.org/pypi/guppy
If you use ``pip``, you can install Guppy with the following command::
pip install guppy
The telnet console also comes with a built-in shortcut (``hpy``) for accessing
Guppy heap objects. Here's an example to view all Python objects available in
the heap using Guppy:
>>> x = hpy.heap()
>>> x.bytype
Partition of a set of 297033 objects. Total size = 52587824 bytes.
Index Count % Size % Cumulative % Type
0 22307 8 16423880 31 16423880 31 dict
1 122285 41 12441544 24 28865424 55 str
2 68346 23 5966696 11 34832120 66 tuple
3 227 0 5836528 11 40668648 77 unicode
4 2461 1 2222272 4 42890920 82 type
5 16870 6 2024400 4 44915320 85 function
6 13949 5 1673880 3 46589200 89 types.CodeType
7 13422 5 1653104 3 48242304 92 list
8 3735 1 1173680 2 49415984 94 _sre.SRE_Pattern
9 1209 0 456936 1 49872920 95 scrapy.http.headers.Headers
<1676 more rows. Type e.g. '_.more' to view.>
You can see that most space is used by dicts. Then, if you want to see from
which attribute those dicts are referenced, you could do:
>>> x.bytype[0].byvia
Partition of a set of 22307 objects. Total size = 16423880 bytes.
Index Count % Size % Cumulative % Referred Via:
0 10982 49 9416336 57 9416336 57 '.__dict__'
1 1820 8 2681504 16 12097840 74 '.__dict__', '.func_globals'
2 3097 14 1122904 7 13220744 80
3 990 4 277200 2 13497944 82 "['cookies']"
4 987 4 276360 2 13774304 84 "['cache']"
5 985 4 275800 2 14050104 86 "['meta']"
6 897 4 251160 2 14301264 87 '[2]'
7 1 0 196888 1 14498152 88 "['moduleDict']", "['modules']"
8 672 3 188160 1 14686312 89 "['cb_kwargs']"
9 27 0 155016 1 14841328 90 '[1]'
<333 more rows. Type e.g. '_.more' to view.>
As you can see, the Guppy module is very powerful but also requires some deep
knowledge about Python internals. For more info about Guppy, refer to the
`Guppy documentation`_.
.. _Guppy documentation: http://guppy-pe.sourceforge.net/
.. _topics-leaks-muppy:
Debugging memory leaks with muppy
=================================
You can use muppy from `Pympler`_.
.. _Pympler: https://pypi.org/project/Pympler/
@ -311,9 +257,9 @@ though neither Scrapy nor your project are leaking memory. This is due to a
(not so well) known problem of Python, which may not return released memory to
the operating system in some cases. For more information on this issue see:
* `Python Memory Management <http://www.evanjones.ca/python-memory.html>`_
* `Python Memory Management Part 2 <http://www.evanjones.ca/python-memory-part2.html>`_
* `Python Memory Management Part 3 <http://www.evanjones.ca/python-memory-part3.html>`_
* `Python Memory Management <https://www.evanjones.ca/python-memory.html>`_
* `Python Memory Management Part 2 <https://www.evanjones.ca/python-memory-part2.html>`_
* `Python Memory Management Part 3 <https://www.evanjones.ca/python-memory-part3.html>`_
The improvements proposed by Evan Jones, which are detailed in `this paper`_,
got merged in Python 2.5, but this only reduces the problem, it doesn't fix it
@ -327,7 +273,7 @@ completely. To quote the paper:
to move to a compacting garbage collector, which is able to move objects in
memory. This would require significant changes to the Python interpreter.*
.. _this paper: http://www.evanjones.ca/memoryallocator/
.. _this paper: https://www.evanjones.ca/memoryallocator/
To keep memory consumption reasonable you can split the job into several
smaller jobs or enable :ref:`persistent job queue <topics-jobs>`

View File

@ -49,7 +49,7 @@ LxmlLinkExtractor
:type allow: a regular expression (or list of)
:param deny: a single regular expression (or list of regular expressions)
that the (absolute) urls must match in order to be excluded (ie. not
that the (absolute) urls must match in order to be excluded (i.e. not
extracted). It has precedence over the ``allow`` parameter. If not
given (or empty) it won't exclude any links.
:type deny: a regular expression (or list of)
@ -64,9 +64,13 @@ LxmlLinkExtractor
:param deny_extensions: a single value or list of strings containing
extensions that should be ignored when extracting links.
If not given, it will default to the
``IGNORED_EXTENSIONS`` list defined in the
`scrapy.linkextractors`_ package.
If not given, it will default to
:data:`scrapy.linkextractors.IGNORED_EXTENSIONS`.
.. versionchanged:: 2.0
:data:`~scrapy.linkextractors.IGNORED_EXTENSIONS` now includes
``7z``, ``7zip``, ``apk``, ``bz2``, ``cdr``, ``dmg``, ``ico``,
``iso``, ``tar``, ``tar.gz``, ``webm``, and ``xz``.
:type deny_extensions: list
:param restrict_xpaths: is an XPath (or list of XPath's) which defines

View File

@ -136,6 +136,9 @@ with the data to be parsed, and return a parsed value. So you can use any
function as input or output processor. The only requirement is that they must
accept one (and only one) positional argument, which will be an iterable.
.. versionchanged:: 2.0
Processors no longer need to be methods.
.. note:: Both input and output processors must receive an iterable as their
first argument. The output of those functions can be anything. The result of
input processors will be appended to an internal list (in the Loader)

View File

@ -116,12 +116,6 @@ For the Images Pipeline, set the :setting:`IMAGES_STORE` setting::
Supported Storage
=================
File system is currently the only officially supported storage, but there are
also support for storing files in `Amazon S3`_ and `Google Cloud Storage`_.
.. _Amazon S3: https://aws.amazon.com/s3/
.. _Google Cloud Storage: https://cloud.google.com/storage/
File system storage
-------------------
@ -147,9 +141,13 @@ Where:
* ``full`` is a sub-directory to separate full images from thumbnails (if
used). For more info see :ref:`topics-images-thumbnails`.
.. _media-pipeline-ftp:
FTP server storage
------------------
.. versionadded:: 2.0
:setting:`FILES_STORE` and :setting:`IMAGES_STORE` can point to an FTP server.
Scrapy will automatically upload the files to the server.
@ -573,6 +571,8 @@ See here the methods that you can override in your custom Images Pipeline:
By default, the :meth:`item_completed` method returns the item.
.. _media-pipeline-example:
Custom Images pipeline example
==============================

View File

@ -31,6 +31,8 @@ Request objects
a :class:`Response`.
:param url: the URL of this request
If the URL is invalid, a :exc:`ValueError` exception is raised.
:type url: string
:param callback: the function that will be called with the response of this
@ -125,6 +127,10 @@ Request objects
:exc:`~twisted.python.failure.Failure` as first parameter.
For more information,
see :ref:`topics-request-response-ref-errbacks` below.
.. versionchanged:: 2.0
The *callback* parameter is no longer required when the *errback*
parameter is specified.
:type errback: callable
:param flags: Flags sent to the request, can be used for logging or similar purposes.
@ -396,7 +402,7 @@ The FormRequest class extends the base :class:`Request` with functionality for
dealing with HTML forms. It uses `lxml.html forms`_ to pre-populate form
fields with form data from :class:`Response` objects.
.. _lxml.html forms: http://lxml.de/lxmlhtml.html#forms
.. _lxml.html forms: https://lxml.de/lxmlhtml.html#forms
.. class:: FormRequest(url, [formdata, ...])
@ -609,7 +615,10 @@ Response objects
:param request: the initial value of the :attr:`Response.request` attribute.
This represents the :class:`Request` that generated this response.
:type request: :class:`Request` object
:type request: scrapy.http.Request
:param certificate: an object representing the server's SSL certificate.
:type certificate: twisted.internet.ssl.Certificate
.. attribute:: Response.url
@ -664,7 +673,7 @@ Response objects
.. attribute:: Response.meta
A shortcut to the :attr:`Request.meta` attribute of the
:attr:`Response.request` object (ie. ``self.request.meta``).
:attr:`Response.request` object (i.e. ``self.request.meta``).
Unlike the :attr:`Response.request` attribute, the :attr:`Response.meta`
attribute is propagated along redirects and retries, so you will get
@ -672,6 +681,20 @@ Response objects
.. seealso:: :attr:`Request.meta` attribute
.. attribute:: Response.cb_kwargs
.. versionadded:: 2.0
A shortcut to the :attr:`Request.cb_kwargs` attribute of the
:attr:`Response.request` object (i.e. ``self.request.cb_kwargs``).
Unlike the :attr:`Response.request` attribute, the
:attr:`Response.cb_kwargs` attribute is propagated along redirects and
retries, so you will get the original :attr:`Request.cb_kwargs` sent
from your spider.
.. seealso:: :attr:`Request.cb_kwargs` attribute
.. attribute:: Response.flags
A list that contains flags for this response. Flags are labels used for
@ -679,6 +702,13 @@ Response objects
they're shown on the string representation of the Response (`__str__`
method) which is used by the engine for logging.
.. attribute:: Response.certificate
A :class:`twisted.internet.ssl.Certificate` object representing
the server's SSL certificate.
Only populated for ``https`` responses, ``None`` otherwise.
.. method:: Response.copy()
Returns a new Response which is a copy of this Response.
@ -760,7 +790,7 @@ TextResponse objects
1. the encoding passed in the ``__init__`` method ``encoding`` argument
2. the encoding declared in the Content-Type HTTP header. If this
encoding is not valid (ie. unknown), it is ignored and the next
encoding is not valid (i.e. unknown), it is ignored and the next
resolution mechanism is tried.
3. the encoding declared in the response body. The TextResponse class

View File

@ -35,12 +35,11 @@ defines selectors to associate those styles with specific HTML elements.
in speed and parsing accuracy to lxml.
.. _BeautifulSoup: https://www.crummy.com/software/BeautifulSoup/
.. _lxml: http://lxml.de/
.. _lxml: https://lxml.de/
.. _ElementTree: https://docs.python.org/2/library/xml.etree.elementtree.html
.. _cssselect: https://pypi.python.org/pypi/cssselect/
.. _XPath: https://www.w3.org/TR/xpath
.. _XPath: https://www.w3.org/TR/xpath/all/
.. _CSS: https://www.w3.org/TR/selectors
.. _parsel: https://parsel.readthedocs.io/
.. _parsel: https://parsel.readthedocs.io/en/latest/
Using selectors
===============
@ -255,7 +254,7 @@ that Scrapy (parsel) implements a couple of **non-standard pseudo-elements**:
They will most probably not work with other libraries like
`lxml`_ or `PyQuery`_.
.. _PyQuery: https://pypi.python.org/pypi/pyquery
.. _PyQuery: https://pypi.org/project/pyquery/
Examples:
@ -309,7 +308,7 @@ Examples:
make much sense: text nodes do not have attributes, and attribute values
are string values already and do not have children nodes.
.. _CSS Selectors: https://www.w3.org/TR/css3-selectors/#selectors
.. _CSS Selectors: https://www.w3.org/TR/selectors-3/#selectors
.. _topics-selectors-nesting-selectors:
@ -504,7 +503,7 @@ Another common case would be to extract all direct ``<p>`` children:
For more details about relative XPaths see the `Location Paths`_ section in the
XPath specification.
.. _Location Paths: https://www.w3.org/TR/xpath#location-paths
.. _Location Paths: https://www.w3.org/TR/xpath/all/#location-paths
When querying by class, consider using CSS
------------------------------------------
@ -612,7 +611,7 @@ But using the ``.`` to mean the node, works:
>>> sel.xpath("//a[contains(., 'Next Page')]").getall()
['<a href="#">Click here to go to the <strong>Next Page</strong></a>']
.. _`XPath string function`: https://www.w3.org/TR/xpath/#section-String-Functions
.. _`XPath string function`: https://www.w3.org/TR/xpath/all/#section-String-Functions
.. _topics-selectors-xpath-variables:
@ -764,7 +763,7 @@ Set operations
These can be handy for excluding parts of a document tree before
extracting text elements for example.
Example extracting microdata (sample content taken from http://schema.org/Product)
Example extracting microdata (sample content taken from https://schema.org/Product)
with groups of itemscopes and corresponding itemprops::
>>> doc = u"""
@ -986,7 +985,7 @@ a :class:`~scrapy.http.HtmlResponse` object like this::
sel = Selector(html_response)
1. Select all ``<h1>`` elements from an HTML response body, returning a list of
:class:`Selector` objects (ie. a :class:`SelectorList` object)::
:class:`Selector` objects (i.e. a :class:`SelectorList` object)::
sel.xpath("//h1")
@ -1013,7 +1012,7 @@ instantiated with an :class:`~scrapy.http.XmlResponse` object::
sel = Selector(xml_response)
1. Select all ``<product>`` elements from an XML response body, returning a list
of :class:`Selector` objects (ie. a :class:`SelectorList` object)::
of :class:`Selector` objects (i.e. a :class:`SelectorList` object)::
sel.xpath("//product")

View File

@ -124,7 +124,7 @@ Settings can be accessed through the :attr:`scrapy.crawler.Crawler.settings`
attribute of the Crawler that is passed to ``from_crawler`` method in
extensions, middlewares and item pipelines::
class MyExtension(object):
class MyExtension:
def __init__(self, log_is_enabled=False):
if log_is_enabled:
print("log is enabled!")
@ -248,7 +248,7 @@ CONCURRENT_REQUESTS
Default: ``16``
The maximum number of concurrent (ie. simultaneous) requests that will be
The maximum number of concurrent (i.e. simultaneous) requests that will be
performed by the Scrapy downloader.
.. setting:: CONCURRENT_REQUESTS_PER_DOMAIN
@ -258,7 +258,7 @@ CONCURRENT_REQUESTS_PER_DOMAIN
Default: ``8``
The maximum number of concurrent (ie. simultaneous) requests that will be
The maximum number of concurrent (i.e. simultaneous) requests that will be
performed to any single domain.
See also: :ref:`topics-autothrottle` and its
@ -272,7 +272,7 @@ CONCURRENT_REQUESTS_PER_IP
Default: ``0``
The maximum number of concurrent (ie. simultaneous) requests that will be
The maximum number of concurrent (i.e. simultaneous) requests that will be
performed to any single IP. If non-zero, the
:setting:`CONCURRENT_REQUESTS_PER_DOMAIN` setting is ignored, and this one is
used instead. In other words, concurrency limits will be applied per IP, not
@ -381,6 +381,8 @@ DNS in-memory cache size.
DNS_RESOLVER
------------
.. versionadded:: 2.0
Default: ``'scrapy.resolver.CachingThreadedResolver'``
The class to be used to resolve DNS names. The default ``scrapy.resolver.CachingThreadedResolver``
@ -1275,6 +1277,9 @@ does not work together with :setting:`CONCURRENT_REQUESTS_PER_IP`.
SCRAPER_SLOT_MAX_ACTIVE_SIZE
----------------------------
.. versionadded:: 2.0
Default: ``5_000_000``
Soft limit (in bytes) for response data being processed.
@ -1464,24 +1469,95 @@ in the ``project`` subdirectory.
TWISTED_REACTOR
---------------
.. versionadded:: 2.0
Default: ``None``
Import path of a given Twisted reactor, for instance:
:class:`twisted.internet.asyncioreactor.AsyncioSelectorReactor`.
Import path of a given :mod:`~twisted.internet.reactor`.
Scrapy will install this reactor if no other is installed yet, such as when
the ``scrapy`` CLI program is invoked or when using the
:class:`~scrapy.crawler.CrawlerProcess` class. If you are using the
:class:`~scrapy.crawler.CrawlerRunner` class, you need to install the correct
reactor manually. An exception will be raised if the installation fails.
Scrapy will install this reactor if no other reactor is installed yet, such as
when the ``scrapy`` CLI program is invoked or when using the
:class:`~scrapy.crawler.CrawlerProcess` class.
The default value for this option is currently ``None``, which means that Scrapy
will not attempt to install any specific reactor, and the default one defined by
Twisted for the current platform will be used. This is to maintain backward
compatibility and avoid possible problems caused by using a non-default reactor.
If you are using the :class:`~scrapy.crawler.CrawlerRunner` class, you also
need to install the correct reactor manually. You can do that using
:func:`~scrapy.utils.reactor.install_reactor`:
For additional information, please see
:doc:`core/howto/choosing-reactor`.
.. autofunction:: scrapy.utils.reactor.install_reactor
If a reactor is already installed,
:func:`~scrapy.utils.reactor.install_reactor` has no effect.
:meth:`CrawlerRunner.__init__ <scrapy.crawler.CrawlerRunner.__init__>` raises
:exc:`Exception` if the installed reactor does not match the
:setting:`TWISTED_REACTOR` setting; therfore, having top-level
:mod:`~twisted.internet.reactor` imports in project files and imported
third-party libraries will make Scrapy raise :exc:`Exception` when
it checks which reactor is installed.
In order to use the reactor installed by Scrapy::
import scrapy
from twisted.internet import reactor
class QuotesSpider(scrapy.Spider):
name = 'quotes'
def __init__(self, *args, **kwargs):
self.timeout = int(kwargs.pop('timeout', '60'))
super(QuotesSpider, self).__init__(*args, **kwargs)
def start_requests(self):
reactor.callLater(self.timeout, self.stop)
urls = ['http://quotes.toscrape.com/page/1']
for url in urls:
yield scrapy.Request(url=url, callback=self.parse)
def parse(self, response):
for quote in response.css('div.quote'):
yield {'text': quote.css('span.text::text').get()}
def stop(self):
self.crawler.engine.close_spider(self, 'timeout')
which raises :exc:`Exception`, becomes::
import scrapy
class QuotesSpider(scrapy.Spider):
name = 'quotes'
def __init__(self, *args, **kwargs):
self.timeout = int(kwargs.pop('timeout', '60'))
super(QuotesSpider, self).__init__(*args, **kwargs)
def start_requests(self):
from twisted.internet import reactor
reactor.callLater(self.timeout, self.stop)
urls = ['http://quotes.toscrape.com/page/1']
for url in urls:
yield scrapy.Request(url=url, callback=self.parse)
def parse(self, response):
for quote in response.css('div.quote'):
yield {'text': quote.css('span.text::text').get()}
def stop(self):
self.crawler.engine.close_spider(self, 'timeout')
The default value of the :setting:`TWISTED_REACTOR` setting is ``None``, which
means that Scrapy will not attempt to install any specific reactor, and the
default reactor defined by Twisted for the current platform will be used. This
is to maintain backward compatibility and avoid possible problems caused by
using a non-default reactor.
For additional information, see :doc:`core/howto/choosing-reactor`.
.. setting:: URLLENGTH_LIMIT

View File

@ -41,7 +41,7 @@ variable; or by defining it in your :ref:`scrapy.cfg <topics-config-settings>`::
.. _IPython: https://ipython.org/
.. _IPython installation guide: https://ipython.org/install.html
.. _bpython: https://www.bpython-interpreter.org/
.. _bpython: https://bpython-interpreter.org/
Launch the shell
================
@ -142,7 +142,7 @@ Example of shell session
========================
Here's an example of a typical shell session where we start by scraping the
https://scrapy.org page, and then proceed to scrape the https://reddit.com
https://scrapy.org page, and then proceed to scrape the https://old.reddit.com/
page. Finally, we modify the (Reddit) request method to POST and re-fetch it
getting an error. We end the session by typing Ctrl-D (in Unix systems) or
Ctrl-Z in Windows.
@ -182,7 +182,7 @@ After that, we can start playing with the objects:
>>> response.xpath('//title/text()').get()
'Scrapy | A Fast and Powerful Scraping and Web Crawling Framework'
>>> fetch("https://reddit.com")
>>> fetch("https://old.reddit.com/")
>>> response.xpath('//title/text()').get()
'reddit: the front page of the internet'

View File

@ -46,6 +46,7 @@ Here is a simple example showing how you can catch signals and perform some acti
def parse(self, response):
pass
.. _signal-deferred:
Deferred signal handlers
========================
@ -141,7 +142,7 @@ item_error
.. signal:: item_error
.. function:: item_error(item, response, spider, failure)
Sent when a :ref:`topics-item-pipeline` generates an error (ie. raises
Sent when a :ref:`topics-item-pipeline` generates an error (i.e. raises
an exception), except :exc:`~scrapy.exceptions.DropItem` exception.
This signal supports returning deferreds from their handlers.
@ -232,7 +233,7 @@ spider_error
.. signal:: spider_error
.. function:: spider_error(failure, response, spider)
Sent when a spider callback generates an error (ie. raises an exception).
Sent when a spider callback generates an error (i.e. raises an exception).
This signal does not support returning deferreds from their handlers.
@ -301,6 +302,8 @@ request_left_downloader
.. signal:: request_left_downloader
.. function:: request_left_downloader(request, spider)
.. versionadded:: 2.0
Sent when a :class:`~scrapy.http.Request` leaves the downloader, even in case of
failure.

View File

@ -299,8 +299,8 @@ The spider will not do any parsing on its own.
If you were to set the ``start_urls`` attribute from the command line,
you would have to parse it on your own into a list
using something like
`ast.literal_eval <https://docs.python.org/library/ast.html#ast.literal_eval>`_
or `json.loads <https://docs.python.org/library/json.html#json.loads>`_
`ast.literal_eval <https://docs.python.org/3/library/ast.html#ast.literal_eval>`_
or `json.loads <https://docs.python.org/3/library/json.html#json.loads>`_
and then set it as an attribute.
Otherwise, you would cause iteration over a ``start_urls`` string
(a very common python pitfall)
@ -420,6 +420,9 @@ Crawling rules
It receives a :class:`Twisted Failure <twisted.python.failure.Failure>`
instance as first parameter.
.. versionadded:: 2.0
The *errback* parameter.
CrawlSpider example
~~~~~~~~~~~~~~~~~~~
@ -811,6 +814,6 @@ Combine SitemapSpider with other sources of urls::
.. _Sitemaps: https://www.sitemaps.org/index.html
.. _Sitemap index files: https://www.sitemaps.org/protocol.html#index
.. _robots.txt: http://www.robotstxt.org/
.. _robots.txt: https://www.robotstxt.org/
.. _TLD: https://en.wikipedia.org/wiki/Top-level_domain
.. _Scrapyd documentation: https://scrapyd.readthedocs.io/en/latest/

View File

@ -32,7 +32,7 @@ Common Stats Collector uses
Access the stats collector through the :attr:`~scrapy.crawler.Crawler.stats`
attribute. Here is an example of an extension that access stats::
class ExtensionThatAccessStats(object):
class ExtensionThatAccessStats:
def __init__(self, stats):
self.stats = stats

View File

@ -37,7 +37,7 @@ flake8-ignore =
scrapy/commands/edit.py E501
scrapy/commands/fetch.py E401 E501 E128 E731
scrapy/commands/genspider.py E128 E501 E502
scrapy/commands/parse.py E128 E501 E731 E226
scrapy/commands/parse.py E128 E501 E731
scrapy/commands/runspider.py E501
scrapy/commands/settings.py E128
scrapy/commands/shell.py E128 E501 E502
@ -47,22 +47,22 @@ flake8-ignore =
scrapy/contracts/__init__.py E501 W504
scrapy/contracts/default.py E128
# scrapy/core
scrapy/core/engine.py E501 E128 E127 E306 E502
scrapy/core/engine.py E501 E128 E127 E502
scrapy/core/scheduler.py E501
scrapy/core/scraper.py E501 E306 E128 W504
scrapy/core/spidermw.py E501 E731 E126 E226
scrapy/core/scraper.py E501 E128 W504
scrapy/core/spidermw.py E501 E731 E126
scrapy/core/downloader/__init__.py E501
scrapy/core/downloader/contextfactory.py E501 E128 E126
scrapy/core/downloader/middleware.py E501 E502
scrapy/core/downloader/tls.py E501 E305 E241
scrapy/core/downloader/webclient.py E731 E501 E128 E126 E226
scrapy/core/downloader/tls.py E501 E241
scrapy/core/downloader/webclient.py E731 E501 E128 E126
scrapy/core/downloader/handlers/__init__.py E501
scrapy/core/downloader/handlers/ftp.py E501 E305 E128 E127
scrapy/core/downloader/handlers/ftp.py E501 E128 E127
scrapy/core/downloader/handlers/http10.py E501
scrapy/core/downloader/handlers/http11.py E501
scrapy/core/downloader/handlers/s3.py E501 E128 E126
# scrapy/downloadermiddlewares
scrapy/downloadermiddlewares/ajaxcrawl.py E501 E226
scrapy/downloadermiddlewares/ajaxcrawl.py E501
scrapy/downloadermiddlewares/decompression.py E501
scrapy/downloadermiddlewares/defaultheaders.py E501
scrapy/downloadermiddlewares/httpcache.py E501 E126
@ -76,7 +76,7 @@ flake8-ignore =
scrapy/extensions/closespider.py E501 E128 E123
scrapy/extensions/corestats.py E501
scrapy/extensions/feedexport.py E128 E501
scrapy/extensions/httpcache.py E128 E501 E303
scrapy/extensions/httpcache.py E128 E501
scrapy/extensions/memdebug.py E501
scrapy/extensions/spiderstate.py E501
scrapy/extensions/telnet.py E501 W504
@ -91,7 +91,7 @@ flake8-ignore =
scrapy/http/response/text.py E501 E128 E124
# scrapy/linkextractors
scrapy/linkextractors/__init__.py E731 E501 E402 W504
scrapy/linkextractors/lxmlhtml.py E501 E731 E226
scrapy/linkextractors/lxmlhtml.py E501 E731
# scrapy/loader
scrapy/loader/__init__.py E501 E128
scrapy/loader/processors.py E501
@ -105,7 +105,7 @@ flake8-ignore =
scrapy/selector/unified.py E501 E111
# scrapy/settings
scrapy/settings/__init__.py E501
scrapy/settings/default_settings.py E501 E114 E116 E226
scrapy/settings/default_settings.py E501 E114 E116
scrapy/settings/deprecated.py E501
# scrapy/spidermiddlewares
scrapy/spidermiddlewares/httperror.py E501
@ -121,28 +121,27 @@ flake8-ignore =
scrapy/utils/asyncio.py E501
scrapy/utils/benchserver.py E501
scrapy/utils/conf.py E402 E501
scrapy/utils/console.py E306 E305
scrapy/utils/datatypes.py E501 E226
scrapy/utils/datatypes.py E501
scrapy/utils/decorators.py E501
scrapy/utils/defer.py E501 E128
scrapy/utils/deprecate.py E128 E501 E127 E502
scrapy/utils/gz.py E305 E501 W504
scrapy/utils/http.py F403 E226
scrapy/utils/gz.py E501 W504
scrapy/utils/http.py F403
scrapy/utils/httpobj.py E501
scrapy/utils/iterators.py E501 E701
scrapy/utils/iterators.py E501
scrapy/utils/log.py E128 E501
scrapy/utils/markup.py F403
scrapy/utils/misc.py E501 E226
scrapy/utils/misc.py E501
scrapy/utils/multipart.py F403
scrapy/utils/project.py E501
scrapy/utils/python.py E501
scrapy/utils/reactor.py E226 E501
scrapy/utils/reactor.py E501
scrapy/utils/reqser.py E501
scrapy/utils/request.py E127 E501
scrapy/utils/response.py E501 E128
scrapy/utils/signal.py E501 E128
scrapy/utils/sitemap.py E501
scrapy/utils/spider.py E271 E501
scrapy/utils/spider.py E501
scrapy/utils/ssl.py E501
scrapy/utils/test.py E501
scrapy/utils/url.py E501 F403 E128 F405
@ -152,7 +151,7 @@ flake8-ignore =
scrapy/crawler.py E501
scrapy/dupefilters.py E501 E202
scrapy/exceptions.py E501
scrapy/exporters.py E501 E226
scrapy/exporters.py E501
scrapy/interfaces.py E501
scrapy/item.py E501 E128
scrapy/link.py E501
@ -161,93 +160,91 @@ flake8-ignore =
scrapy/middleware.py E128 E501
scrapy/pqueues.py E501
scrapy/resolver.py E501
scrapy/responsetypes.py E128 E501 E305
scrapy/responsetypes.py E128 E501
scrapy/robotstxt.py E501
scrapy/shell.py E501
scrapy/signalmanager.py E501
scrapy/spiderloader.py E225 F841 E501 E126
scrapy/spiderloader.py F841 E501 E126
scrapy/squeues.py E128
scrapy/statscollectors.py E501
# tests
tests/__init__.py E402 E501
tests/mockserver.py E401 E501 E126 E123
tests/pipelines.py F841 E226
tests/pipelines.py F841
tests/spiders.py E501 E127
tests/test_closespider.py E501 E127
tests/test_command_fetch.py E501
tests/test_command_parse.py E501 E128 E303 E226
tests/test_command_parse.py E501 E128
tests/test_command_shell.py E501 E128
tests/test_commands.py E128 E501
tests/test_contracts.py E501 E128
tests/test_crawl.py E501 E741 E265
tests/test_crawler.py F841 E306 E501
tests/test_dependencies.py F841 E501 E305
tests/test_downloader_handlers.py E124 E127 E128 E225 E265 E501 E701 E126 E226 E123
tests/test_crawler.py F841 E501
tests/test_dependencies.py F841 E501
tests/test_downloader_handlers.py E124 E127 E128 E265 E501 E126 E123
tests/test_downloadermiddleware.py E501
tests/test_downloadermiddleware_ajaxcrawlable.py E501
tests/test_downloadermiddleware_cookies.py E731 E741 E501 E128 E303 E265 E126
tests/test_downloadermiddleware_cookies.py E731 E741 E501 E128 E265 E126
tests/test_downloadermiddleware_decompression.py E127
tests/test_downloadermiddleware_defaultheaders.py E501
tests/test_downloadermiddleware_downloadtimeout.py E501
tests/test_downloadermiddleware_httpcache.py E501 E305
tests/test_downloadermiddleware_httpcompression.py E501 E251 E126 E123
tests/test_downloadermiddleware_httpcache.py E501
tests/test_downloadermiddleware_httpcompression.py E501 E126 E123
tests/test_downloadermiddleware_httpproxy.py E501 E128
tests/test_downloadermiddleware_redirect.py E501 E303 E128 E306 E127 E305
tests/test_downloadermiddleware_retry.py E501 E128 E251 E303 E126
tests/test_downloadermiddleware_redirect.py E501 E128 E127
tests/test_downloadermiddleware_retry.py E501 E128 E126
tests/test_downloadermiddleware_robotstxt.py E501
tests/test_downloadermiddleware_stats.py E501
tests/test_dupefilters.py E221 E501 E741 E128 E124
tests/test_dupefilters.py E501 E741 E128 E124
tests/test_engine.py E401 E501 E128
tests/test_exporters.py E501 E731 E306 E128 E124
tests/test_exporters.py E501 E731 E128 E124
tests/test_extension_telnet.py F841
tests/test_feedexport.py E501 F841 E241
tests/test_http_cookies.py E501
tests/test_http_headers.py E501
tests/test_http_request.py E402 E501 E127 E128 E128 E126 E123
tests/test_http_response.py E501 E301 E128 E265
tests/test_item.py E701 E128 F841 E306
tests/test_http_response.py E501 E128 E265
tests/test_item.py E128 F841
tests/test_link.py E501
tests/test_linkextractors.py E501 E128 E124
tests/test_loader.py E501 E731 E303 E741 E128 E117 E241
tests/test_loader.py E501 E731 E741 E128 E117 E241
tests/test_logformatter.py E128 E501 E122
tests/test_mail.py E128 E501 E305
tests/test_mail.py E128 E501
tests/test_middleware.py E501 E128
tests/test_pipeline_crawl.py E131 E501 E128 E126
tests/test_pipeline_files.py E501 E303 E272 E226
tests/test_pipeline_images.py F841 E501 E303
tests/test_pipeline_media.py E501 E741 E731 E128 E306 E502
tests/test_pipeline_crawl.py E501 E128 E126
tests/test_pipeline_files.py E501
tests/test_pipeline_images.py F841 E501
tests/test_pipeline_media.py E501 E741 E731 E128 E502
tests/test_proxy_connect.py E501 E741
tests/test_request_cb_kwargs.py E501
tests/test_responsetypes.py E501 E305
tests/test_responsetypes.py E501
tests/test_robotstxt_interface.py E501 E501
tests/test_scheduler.py E501 E126 E123
tests/test_selector.py E501 E127
tests/test_spider.py E501
tests/test_spidermiddleware.py E501 E226
tests/test_spidermiddleware.py E501
tests/test_spidermiddleware_httperror.py E128 E501 E127 E121
tests/test_spidermiddleware_offsite.py E501 E128 E111
tests/test_spidermiddleware_output_chain.py E501 E226
tests/test_spidermiddleware_output_chain.py E501
tests/test_spidermiddleware_referer.py E501 F841 E125 E201 E124 E501 E241 E121
tests/test_squeues.py E501 E701 E741
tests/test_squeues.py E501 E741
tests/test_utils_asyncio.py E501
tests/test_utils_conf.py E501 E303 E128
tests/test_utils_conf.py E501 E128
tests/test_utils_curl.py E501
tests/test_utils_datatypes.py E402 E501 E305
tests/test_utils_defer.py E306 E501 F841 E226
tests/test_utils_deprecate.py F841 E306 E501
tests/test_utils_datatypes.py E402 E501
tests/test_utils_defer.py E501 F841
tests/test_utils_deprecate.py F841 E501
tests/test_utils_http.py E501 E128 W504
tests/test_utils_iterators.py E501 E128 E129 E303 E241
tests/test_utils_log.py E741 E226
tests/test_utils_python.py E501 E303 E731 E701 E305
tests/test_utils_iterators.py E501 E128 E129 E241
tests/test_utils_log.py E741
tests/test_utils_python.py E501 E731
tests/test_utils_reqser.py E501 E128
tests/test_utils_request.py E501 E128 E305
tests/test_utils_request.py E501 E128
tests/test_utils_response.py E501
tests/test_utils_signal.py E741 F841 E731 E226
tests/test_utils_signal.py E741 F841 E731
tests/test_utils_sitemap.py E128 E501 E124
tests/test_utils_spider.py E305
tests/test_utils_template.py E305
tests/test_utils_url.py E501 E127 E305 E211 E125 E501 E226 E241 E126 E123
tests/test_webclient.py E501 E128 E122 E303 E402 E306 E226 E241 E123 E126
tests/test_utils_url.py E501 E127 E125 E501 E241 E126 E123
tests/test_webclient.py E501 E128 E122 E402 E241 E123 E126
tests/test_cmdline/__init__.py E501
tests/test_settings/__init__.py E501 E128
tests/test_spiderloader/__init__.py E128 E501

View File

@ -1 +1 @@
1.8.0
2.0.0

View File

@ -12,7 +12,6 @@ from scrapy.exceptions import UsageError
from scrapy.utils.misc import walk_modules
from scrapy.utils.project import inside_project, get_project_settings
from scrapy.utils.python import garbage_collect
from scrapy.settings.deprecated import check_deprecated_settings
def _iter_command_classes(module_name):
@ -118,7 +117,6 @@ def execute(argv=None, settings=None):
pass
else:
settings['EDITOR'] = editor
check_deprecated_settings(settings)
inproject = inside_project()
cmds = _get_commands_dict(settings, inproject)

View File

@ -9,7 +9,7 @@ from scrapy.utils.conf import arglist_to_dict
from scrapy.exceptions import UsageError
class ScrapyCommand(object):
class ScrapyCommand:
requires_project = False
crawler_process = None

View File

@ -25,7 +25,7 @@ class Command(ScrapyCommand):
self.crawler_process.start()
class _BenchServer(object):
class _BenchServer:
def __enter__(self):
from scrapy.utils.test import get_testenv

View File

@ -1,7 +1,5 @@
import os
from scrapy.commands import ScrapyCommand
from scrapy.utils.conf import arglist_to_dict
from scrapy.utils.python import without_none_values
from scrapy.utils.conf import arglist_to_dict, feed_process_params_from_cli
from scrapy.exceptions import UsageError
@ -19,7 +17,7 @@ class Command(ScrapyCommand):
ScrapyCommand.add_options(self, parser)
parser.add_option("-a", dest="spargs", action="append", default=[], metavar="NAME=VALUE",
help="set spider argument (may be repeated)")
parser.add_option("-o", "--output", metavar="FILE",
parser.add_option("-o", "--output", metavar="FILE", action="append",
help="dump scraped items into FILE (use - for stdout)")
parser.add_option("-t", "--output-format", metavar="FORMAT",
help="format to use for dumping items with -o")
@ -31,21 +29,8 @@ class Command(ScrapyCommand):
except ValueError:
raise UsageError("Invalid -a value, use -a NAME=VALUE", print_help=False)
if opts.output:
if opts.output == '-':
self.settings.set('FEED_URI', 'stdout:', priority='cmdline')
else:
self.settings.set('FEED_URI', opts.output, priority='cmdline')
feed_exporters = without_none_values(
self.settings.getwithbase('FEED_EXPORTERS'))
valid_output_formats = feed_exporters.keys()
if not opts.output_format:
opts.output_format = os.path.splitext(opts.output)[1].replace(".", "")
if opts.output_format not in valid_output_formats:
raise UsageError("Unrecognized output format '%s', set one"
" using the '-t' switch or as a file extension"
" from the supported list %s" % (opts.output_format,
tuple(valid_output_formats)))
self.settings.set('FEED_FORMAT', opts.output_format, priority='cmdline')
feeds = feed_process_params_from_cli(self.settings, opts.output, opts.output_format)
self.settings.set('FEEDS', feeds, priority='cmdline')
def run(self, args, opts):
if len(args) < 1:
@ -54,8 +39,13 @@ class Command(ScrapyCommand):
raise UsageError("running 'scrapy crawl' with more than one spider is no longer supported")
spname = args[0]
self.crawler_process.crawl(spname, **opts.spargs)
self.crawler_process.start()
crawl_defer = self.crawler_process.crawl(spname, **opts.spargs)
if self.crawler_process.bootstrap_failed:
if getattr(crawl_defer, 'result', None) is not None and issubclass(crawl_defer.result.type, Exception):
self.exitcode = 1
else:
self.crawler_process.start()
if self.crawler_process.bootstrap_failed or \
(hasattr(self.crawler_process, 'has_exception') and self.crawler_process.has_exception):
self.exitcode = 1

View File

@ -80,7 +80,7 @@ class Command(ScrapyCommand):
else:
items = self.items.get(lvl, [])
print("# Scraped Items ", "-"*60)
print("# Scraped Items ", "-" * 60)
display.pprint([dict(x) for x in items], colorize=colour)
def print_requests(self, lvl=None, colour=True):
@ -92,14 +92,14 @@ class Command(ScrapyCommand):
else:
requests = self.requests.get(lvl, [])
print("# Requests ", "-"*65)
print("# Requests ", "-" * 65)
display.pprint(requests, colorize=colour)
def print_results(self, opts):
colour = not opts.nocolour
if opts.verbose:
for level in range(1, self.max_level+1):
for level in range(1, self.max_level + 1):
print('\n>>> DEPTH LEVEL: %s <<<' % level)
if not opts.noitems:
self.print_items(level, colour)

View File

@ -5,8 +5,7 @@ from importlib import import_module
from scrapy.utils.spider import iter_spider_classes
from scrapy.commands import ScrapyCommand
from scrapy.exceptions import UsageError
from scrapy.utils.conf import arglist_to_dict
from scrapy.utils.python import without_none_values
from scrapy.utils.conf import arglist_to_dict, feed_process_params_from_cli
def _import_file(filepath):
@ -43,7 +42,7 @@ class Command(ScrapyCommand):
ScrapyCommand.add_options(self, parser)
parser.add_option("-a", dest="spargs", action="append", default=[], metavar="NAME=VALUE",
help="set spider argument (may be repeated)")
parser.add_option("-o", "--output", metavar="FILE",
parser.add_option("-o", "--output", metavar="FILE", action="append",
help="dump scraped items into FILE (use - for stdout)")
parser.add_option("-t", "--output-format", metavar="FORMAT",
help="format to use for dumping items with -o")
@ -55,19 +54,8 @@ class Command(ScrapyCommand):
except ValueError:
raise UsageError("Invalid -a value, use -a NAME=VALUE", print_help=False)
if opts.output:
if opts.output == '-':
self.settings.set('FEED_URI', 'stdout:', priority='cmdline')
else:
self.settings.set('FEED_URI', opts.output, priority='cmdline')
feed_exporters = without_none_values(self.settings.getwithbase('FEED_EXPORTERS'))
if not opts.output_format:
opts.output_format = os.path.splitext(opts.output)[1].replace(".", "")
if opts.output_format not in feed_exporters:
raise UsageError("Unrecognized output format '%s', set one"
" using the '-t' switch or as a file extension"
" from the supported list %s" % (opts.output_format,
tuple(feed_exporters)))
self.settings.set('FEED_FORMAT', opts.output_format, priority='cmdline')
feeds = feed_process_params_from_cli(self.settings, opts.output, opts.output_format)
self.settings.set('FEEDS', feeds, priority='cmdline')
def run(self, args, opts):
if len(args) != 1:

View File

@ -9,7 +9,7 @@ from scrapy.utils.spider import iterate_spider_output
from scrapy.utils.python import get_spec
class ContractsManager(object):
class ContractsManager:
contracts = {}
def __init__(self, contracts):
@ -107,7 +107,7 @@ class ContractsManager(object):
request.errback = eb_wrapper
class Contract(object):
class Contract:
""" Abstract class for contracts """
request_cls = None

View File

@ -3,7 +3,7 @@ from time import time
from datetime import datetime
from collections import deque
from twisted.internet import reactor, defer, task
from twisted.internet import defer, task
from scrapy.utils.defer import mustbe_deferred
from scrapy.utils.httpobj import urlparse_cached
@ -13,7 +13,7 @@ from scrapy.core.downloader.middleware import DownloaderMiddlewareManager
from scrapy.core.downloader.handlers import DownloadHandlers
class Slot(object):
class Slot:
"""Downloader slot"""
def __init__(self, concurrency, delay, randomize_delay):
@ -66,7 +66,7 @@ def _get_concurrency_delay(concurrency, spider, settings):
return concurrency, delay
class Downloader(object):
class Downloader:
DOWNLOAD_SLOT = 'download_slot'
@ -133,6 +133,7 @@ class Downloader(object):
return deferred
def _process_queue(self, spider, slot):
from twisted.internet import reactor
if slot.latercall and slot.latercall.active():
return

View File

@ -32,7 +32,6 @@ import re
from io import BytesIO
from urllib.parse import unquote
from twisted.internet import reactor
from twisted.internet.protocol import ClientCreator, Protocol
from twisted.protocols.ftp import CommandFailed, FTPClient
@ -81,6 +80,7 @@ class FTPDownloadHandler:
return cls(crawler.settings)
def download_request(self, request, spider):
from twisted.internet import reactor
parsed_url = urlparse_cached(request)
user = request.meta.get("ftp_user", self.default_user)
password = request.meta.get("ftp_password", self.default_password)

View File

@ -1,7 +1,5 @@
"""Download handlers for http and https schemes
"""
from twisted.internet import reactor
from scrapy.utils.misc import create_instance, load_object
from scrapy.utils.python import to_unicode
@ -26,6 +24,7 @@ class HTTP10DownloadHandler:
return factory.deferred
def _connect(self, factory):
from twisted.internet import reactor
host, port = to_unicode(factory.host), factory.port
if factory.scheme == b'https':
client_context_factory = create_instance(

View File

@ -3,11 +3,12 @@
import logging
import re
import warnings
from contextlib import suppress
from io import BytesIO
from time import time
from urllib.parse import urldefrag
from twisted.internet import defer, protocol, reactor
from twisted.internet import defer, protocol, ssl
from twisted.internet.endpoints import TCP4ClientEndpoint
from twisted.internet.error import TimeoutError
from twisted.web.client import Agent, HTTPConnectionPool, ResponseDone, ResponseFailed, URI
@ -32,6 +33,7 @@ class HTTP11DownloadHandler:
lazy = False
def __init__(self, settings, crawler=None):
from twisted.internet import reactor
self._pool = HTTPConnectionPool(reactor, persistent=True)
self._pool.maxPersistentPerHost = settings.getint('CONCURRENT_REQUESTS_PER_DOMAIN')
self._pool._factory.noisy = False
@ -80,6 +82,7 @@ class HTTP11DownloadHandler:
return agent.download_request(request)
def close(self):
from twisted.internet import reactor
d = self._pool.closeCachedConnections()
# closeCachedConnections will hang on network or server issues, so
# we'll manually timeout the deferred.
@ -265,7 +268,7 @@ class ScrapyProxyAgent(Agent):
)
class ScrapyAgent(object):
class ScrapyAgent:
_Agent = Agent
_ProxyAgent = ScrapyProxyAgent
@ -283,6 +286,7 @@ class ScrapyAgent(object):
self._txresponse = None
def _get_agent(self, request, timeout):
from twisted.internet import reactor
bindaddress = request.meta.get('bindaddress') or self._bindAddress
proxy = request.meta.get('proxy')
if proxy:
@ -325,6 +329,7 @@ class ScrapyAgent(object):
)
def download_request(self, request):
from twisted.internet import reactor
timeout = request.meta.get('download_timeout') or self._connectTimeout
agent = self._get_agent(request, timeout)
@ -382,7 +387,7 @@ class ScrapyAgent(object):
def _cb_bodyready(self, txresponse, request):
# deliverBody hangs for responses without body
if txresponse.length == 0:
return txresponse, b'', None
return txresponse, b'', None, None
maxsize = request.meta.get('download_maxsize', self._maxsize)
warnsize = request.meta.get('download_warnsize', self._warnsize)
@ -418,15 +423,16 @@ class ScrapyAgent(object):
return d
def _cb_bodydone(self, result, request, url):
txresponse, body, flags = result
txresponse, body, flags, certificate = result
status = int(txresponse.code)
headers = Headers(txresponse.headers.getAllRawHeaders())
respcls = responsetypes.from_args(headers=headers, url=url, body=body)
return respcls(url=url, status=status, headers=headers, body=body, flags=flags)
return respcls(url=url, status=status, headers=headers, body=body,
flags=flags, certificate=certificate)
@implementer(IBodyProducer)
class _RequestBodyProducer(object):
class _RequestBodyProducer:
def __init__(self, body):
self.body = body
@ -456,6 +462,12 @@ class _ResponseReader(protocol.Protocol):
self._fail_on_dataloss_warned = False
self._reached_warnsize = False
self._bytes_received = 0
self._certificate = None
def connectionMade(self):
if self._certificate is None:
with suppress(AttributeError):
self._certificate = ssl.Certificate(self.transport._producer.getPeerCertificate())
def dataReceived(self, bodyBytes):
# This maybe called several times after cancel was called with buffered data.
@ -488,16 +500,16 @@ class _ResponseReader(protocol.Protocol):
body = self._bodybuf.getvalue()
if reason.check(ResponseDone):
self._finished.callback((self._txresponse, body, None))
self._finished.callback((self._txresponse, body, None, self._certificate))
return
if reason.check(PotentialDataLoss):
self._finished.callback((self._txresponse, body, ['partial']))
self._finished.callback((self._txresponse, body, ['partial'], self._certificate))
return
if reason.check(ResponseFailed) and any(r.check(_DataLoss) for r in reason.value.reasons):
if not self._fail_on_dataloss:
self._finished.callback((self._txresponse, body, ['dataloss']))
self._finished.callback((self._txresponse, body, ['dataloss'], self._certificate))
return
elif not self._fail_on_dataloss_warned:

View File

@ -89,4 +89,5 @@ class ScrapyClientTLSOptions(ClientTLSOptions):
'from host "{}" (exception: {})'.format(
self._hostnameASCII, repr(e)))
DEFAULT_CIPHERS = AcceptableCiphers.fromOpenSSLCipherString('DEFAULT')

View File

@ -140,7 +140,7 @@ class ScrapyHTTPClientFactory(HTTPClientFactory):
self.headers['Content-Length'] = 0
def _build_response(self, body, request):
request.meta['download_latency'] = self.headers_time-self.start_time
request.meta['download_latency'] = self.headers_time - self.start_time
status = int(self.status)
headers = Headers(self.response_headers)
respcls = responsetypes.from_args(headers=headers, url=self._url)

View File

@ -21,7 +21,7 @@ from scrapy.utils.log import logformatter_adapter, failure_to_exc_info
logger = logging.getLogger(__name__)
class Slot(object):
class Slot:
def __init__(self, start_requests, close_if_idle, nextcall, scheduler):
self.closing = False
@ -53,7 +53,7 @@ class Slot(object):
self.closing.callback(None)
class ExecutionEngine(object):
class ExecutionEngine:
def __init__(self, crawler, spider_closed_callback):
self.crawler = crawler
@ -230,6 +230,7 @@ class ExecutionEngine(object):
def _download(self, request, spider):
slot = self.slot
slot.add_request(request)
def _on_success(response):
assert isinstance(response, (Response, Request))
if isinstance(response, Response):

View File

@ -14,7 +14,7 @@ from scrapy.utils.deprecate import ScrapyDeprecationWarning
logger = logging.getLogger(__name__)
class Scheduler(object):
class Scheduler:
"""
Scrapy Scheduler. It allows to enqueue requests and then get
a next request to download. Scheduler is also handling duplication
@ -66,8 +66,7 @@ class Scheduler(object):
dqclass = load_object(settings['SCHEDULER_DISK_QUEUE'])
mqclass = load_object(settings['SCHEDULER_MEMORY_QUEUE'])
logunser = settings.getbool('LOG_UNSERIALIZABLE_REQUESTS',
settings.getbool('SCHEDULER_DEBUG'))
logunser = settings.getbool('SCHEDULER_DEBUG')
return cls(dupefilter, jobdir=job_dir(settings), logunser=logunser,
stats=crawler.stats, pqclass=pqclass, dqclass=dqclass,
mqclass=mqclass, crawler=crawler)
@ -119,7 +118,7 @@ class Scheduler(object):
if self.dqs is None:
return
try:
self.dqs.push(request, -request.priority)
self.dqs.push(request)
except ValueError as e: # non serializable request
if self.logunser:
msg = ("Unable to serialize request: %(request)s - reason:"
@ -135,35 +134,29 @@ class Scheduler(object):
return True
def _mqpush(self, request):
self.mqs.push(request, -request.priority)
self.mqs.push(request)
def _dqpop(self):
if self.dqs:
return self.dqs.pop()
def _newmq(self, priority):
""" Factory for creating memory queues. """
return self.mqclass()
def _newdq(self, priority):
""" Factory for creating disk queues. """
path = join(self.dqdir, 'p%s' % (priority, ))
return self.dqclass(path)
def _mq(self):
""" Create a new priority queue instance, with in-memory storage """
return create_instance(self.pqclass, None, self.crawler, self._newmq,
serialize=False)
return create_instance(self.pqclass,
settings=None,
crawler=self.crawler,
downstream_queue_cls=self.mqclass,
key='')
def _dq(self):
""" Create a new priority queue instance, with disk storage """
state = self._read_dqs_state(self.dqdir)
q = create_instance(self.pqclass,
None,
self.crawler,
self._newdq,
state,
serialize=True)
settings=None,
crawler=self.crawler,
downstream_queue_cls=self.dqclass,
key=self.dqdir,
startprios=state)
if q:
logger.info("Resuming crawl (%(queuesize)d requests scheduled)",
{'queuesize': len(q)}, extra={'spider': self.spider})

View File

@ -16,13 +16,12 @@ from scrapy import signals
from scrapy.http import Request, Response
from scrapy.item import BaseItem
from scrapy.core.spidermw import SpiderMiddlewareManager
from scrapy.utils.request import referer_str
logger = logging.getLogger(__name__)
class Slot(object):
class Slot:
"""Scraper slot (one per running spider)"""
MIN_RESPONSE_SIZE = 1024
@ -63,7 +62,7 @@ class Slot(object):
return self.active_size > self.max_active_size
class Scraper(object):
class Scraper:
def __init__(self, crawler):
self.slot = None
@ -158,9 +157,9 @@ class Scraper(object):
if isinstance(exc, CloseSpider):
self.crawler.engine.close_spider(spider, exc.reason or 'cancelled')
return
logger.error(
"Spider error processing %(request)s (referer: %(referer)s)",
{'request': request, 'referer': referer_str(request)},
logkws = self.logformatter.spider_error(_failure, request, response, spider)
logger.log(
*logformatter_adapter(logkws),
exc_info=failure_to_exc_info(_failure),
extra={'spider': spider}
)
@ -208,16 +207,21 @@ class Scraper(object):
"""
if isinstance(download_failure, Failure) and not download_failure.check(IgnoreRequest):
if download_failure.frames:
logger.error('Error downloading %(request)s',
{'request': request},
exc_info=failure_to_exc_info(download_failure),
extra={'spider': spider})
logkws = self.logformatter.download_error(download_failure, request, spider)
logger.log(
*logformatter_adapter(logkws),
extra={'spider': spider},
exc_info=failure_to_exc_info(download_failure),
)
else:
errmsg = download_failure.getErrorMessage()
if errmsg:
logger.error('Error downloading %(request)s: %(errmsg)s',
{'request': request, 'errmsg': errmsg},
extra={'spider': spider})
logkws = self.logformatter.download_error(
download_failure, request, spider, errmsg)
logger.log(
*logformatter_adapter(logkws),
extra={'spider': spider},
)
if spider_failure is not download_failure:
return spider_failure
@ -236,7 +240,7 @@ class Scraper(object):
signal=signals.item_dropped, item=item, response=response,
spider=spider, exception=output.value)
else:
logkws = self.logformatter.error(item, ex, response, spider)
logkws = self.logformatter.item_error(item, ex, response, spider)
logger.log(*logformatter_adapter(logkws), extra={'spider': spider},
exc_info=failure_to_exc_info(output))
return self.signals.send_catch_log_deferred(

View File

@ -82,7 +82,7 @@ class SpiderMiddlewareManager(MiddlewareManager):
if _isiterable(result):
# stop exception handling by handing control over to the
# process_spider_output chain if an iterable has been returned
return process_spider_output(result, method_index+1)
return process_spider_output(result, method_index + 1)
elif result is None:
continue
else:
@ -103,12 +103,12 @@ class SpiderMiddlewareManager(MiddlewareManager):
# might fail directly if the output value is not a generator
result = method(response=response, result=result, spider=spider)
except Exception as ex:
exception_result = process_spider_exception(Failure(ex), method_index+1)
exception_result = process_spider_exception(Failure(ex), method_index + 1)
if isinstance(exception_result, Failure):
raise
return exception_result
if _isiterable(result):
result = _evaluate_iterable(result, method_index+1, recovered)
result = _evaluate_iterable(result, method_index + 1, recovered)
else:
msg = "Middleware {} must return an iterable, got {}"
raise _InvalidOutput(msg.format(_fname(method), type(result)))

View File

@ -68,17 +68,6 @@ class Crawler:
self.spider = None
self.engine = None
@property
def spiders(self):
if not hasattr(self, '_spiders'):
warnings.warn("Crawler.spiders is deprecated, use "
"CrawlerRunner.spider_loader or instantiate "
"scrapy.spiderloader.SpiderLoader with your "
"settings.",
category=ScrapyDeprecationWarning, stacklevel=2)
self._spiders = _get_spider_loader(self.settings.frozencopy())
return self._spiders
@defer.inlineCallbacks
def crawl(self, *args, **kwargs):
assert not self.crawling, "Crawling already taking place"
@ -130,11 +119,27 @@ class CrawlerRunner:
":meth:`crawl` and managed by this class."
)
@staticmethod
def _get_spider_loader(settings):
""" Get SpiderLoader instance from settings """
cls_path = settings.get('SPIDER_LOADER_CLASS')
loader_cls = load_object(cls_path)
try:
verifyClass(ISpiderLoader, loader_cls)
except DoesNotImplement:
warnings.warn(
'SPIDER_LOADER_CLASS (previously named SPIDER_MANAGER_CLASS) does '
'not fully implement scrapy.interfaces.ISpiderLoader interface. '
'Please add all missing methods to avoid unexpected runtime errors.',
category=ScrapyDeprecationWarning, stacklevel=2
)
return loader_cls.from_settings(settings.frozencopy())
def __init__(self, settings=None):
if isinstance(settings, dict) or settings is None:
settings = Settings(settings)
self.settings = settings
self.spider_loader = _get_spider_loader(settings)
self.spider_loader = self._get_spider_loader(settings)
self._crawlers = set()
self._active = set()
self.bootstrap_failed = False
@ -327,19 +332,3 @@ class CrawlerProcess(CrawlerRunner):
if self.settings.get("TWISTED_REACTOR"):
install_reactor(self.settings["TWISTED_REACTOR"])
super()._handle_twisted_reactor()
def _get_spider_loader(settings):
""" Get SpiderLoader instance from settings """
cls_path = settings.get('SPIDER_LOADER_CLASS')
loader_cls = load_object(cls_path)
try:
verifyClass(ISpiderLoader, loader_cls)
except DoesNotImplement:
warnings.warn(
'SPIDER_LOADER_CLASS (previously named SPIDER_MANAGER_CLASS) does '
'not fully implement scrapy.interfaces.ISpiderLoader interface. '
'Please add all missing methods to avoid unexpected runtime errors.',
category=ScrapyDeprecationWarning, stacklevel=2
)
return loader_cls.from_settings(settings.frozencopy())

View File

@ -11,7 +11,7 @@ from scrapy.http import HtmlResponse
logger = logging.getLogger(__name__)
class AjaxCrawlMiddleware(object):
class AjaxCrawlMiddleware:
"""
Handle 'AJAX crawlable' pages marked as crawlable via meta tag.
For more info see https://developers.google.com/webmasters/ajax-crawling/docs/getting-started.
@ -47,7 +47,7 @@ class AjaxCrawlMiddleware(object):
return response
# scrapy already handles #! links properly
ajax_crawl_request = request.replace(url=request.url+'#!')
ajax_crawl_request = request.replace(url=request.url + '#!')
logger.debug("Downloading AJAX crawlable %(ajax_crawl_request)s instead of %(request)s",
{'ajax_crawl_request': ajax_crawl_request, 'request': request},
extra={'spider': spider})

View File

@ -1,21 +0,0 @@
import warnings
from scrapy.exceptions import ScrapyDeprecationWarning
from scrapy.utils.http import decode_chunked_transfer
warnings.warn("Module `scrapy.downloadermiddlewares.chunked` is deprecated, "
"chunked transfers are supported by default.",
ScrapyDeprecationWarning, stacklevel=2)
class ChunkedTransferMiddleware(object):
"""This middleware adds support for chunked transfer encoding, as
documented in: https://en.wikipedia.org/wiki/Chunked_transfer_encoding
"""
def process_response(self, request, response, spider):
if response.headers.get('Transfer-Encoding') == 'chunked':
body = decode_chunked_transfer(response.body)
return response.replace(body=body)
return response

View File

@ -10,7 +10,7 @@ from scrapy.utils.python import to_unicode
logger = logging.getLogger(__name__)
class CookiesMiddleware(object):
class CookiesMiddleware:
"""This middleware enables working with sites that need cookies"""
def __init__(self, debug=False):

View File

@ -16,7 +16,7 @@ from scrapy.responsetypes import responsetypes
logger = logging.getLogger(__name__)
class DecompressionMiddleware(object):
class DecompressionMiddleware:
""" This middleware tries to recognise and extract the possibly compressed
responses that may arrive. """

View File

@ -7,7 +7,7 @@ See documentation in docs/topics/downloader-middleware.rst
from scrapy.utils.python import without_none_values
class DefaultHeadersMiddleware(object):
class DefaultHeadersMiddleware:
def __init__(self, headers):
self._headers = headers

View File

@ -7,7 +7,7 @@ See documentation in docs/topics/downloader-middleware.rst
from scrapy import signals
class DownloadTimeoutMiddleware(object):
class DownloadTimeoutMiddleware:
def __init__(self, timeout=180):
self._timeout = timeout

View File

@ -9,7 +9,7 @@ from w3lib.http import basic_auth_header
from scrapy import signals
class HttpAuthMiddleware(object):
class HttpAuthMiddleware:
"""Set Basic HTTP Authorization header
(http_user and http_pass spider class attributes)"""

View File

@ -17,7 +17,7 @@ from scrapy.exceptions import IgnoreRequest, NotConfigured
from scrapy.utils.misc import load_object
class HttpCacheMiddleware(object):
class HttpCacheMiddleware:
DOWNLOAD_EXCEPTIONS = (defer.TimeoutError, TimeoutError, DNSLookupError,
ConnectionRefusedError, ConnectionDone, ConnectError,

View File

@ -15,7 +15,7 @@ except ImportError:
pass
class HttpCompressionMiddleware(object):
class HttpCompressionMiddleware:
"""This middleware allows compressed (gzip, deflate) traffic to be
sent/received from web sites"""
@classmethod

View File

@ -7,7 +7,7 @@ from scrapy.utils.httpobj import urlparse_cached
from scrapy.utils.python import to_bytes
class HttpProxyMiddleware(object):
class HttpProxyMiddleware:
def __init__(self, auth_encoding='latin-1'):
self.auth_encoding = auth_encoding

View File

@ -11,7 +11,7 @@ from scrapy.exceptions import IgnoreRequest, NotConfigured
logger = logging.getLogger(__name__)
class BaseRedirectMiddleware(object):
class BaseRedirectMiddleware:
enabled_setting = 'REDIRECT_ENABLED'
@ -93,8 +93,7 @@ class MetaRefreshMiddleware(BaseRedirectMiddleware):
def __init__(self, settings):
super(MetaRefreshMiddleware, self).__init__(settings)
self._ignore_tags = settings.getlist('METAREFRESH_IGNORE_TAGS')
self._maxdelay = settings.getint('REDIRECT_MAX_METAREFRESH_DELAY',
settings.getint('METAREFRESH_MAXDELAY'))
self._maxdelay = settings.getint('METAREFRESH_MAXDELAY')
def process_response(self, request, response, spider):
if request.meta.get('dont_redirect', False) or request.method == 'HEAD' or \

View File

@ -25,7 +25,7 @@ from scrapy.utils.python import global_object_name
logger = logging.getLogger(__name__)
class RetryMiddleware(object):
class RetryMiddleware:
# IOError is raised by the HttpCompression middleware when trying to
# decompress an empty response

View File

@ -16,7 +16,7 @@ from scrapy.utils.misc import load_object
logger = logging.getLogger(__name__)
class RobotsTxtMiddleware(object):
class RobotsTxtMiddleware:
DOWNLOAD_PRIORITY = 1000
def __init__(self, crawler):

View File

@ -4,7 +4,7 @@ from scrapy.utils.response import response_httprepr
from scrapy.utils.python import global_object_name
class DownloaderStats(object):
class DownloaderStats:
def __init__(self, stats):
self.stats = stats

View File

@ -3,7 +3,7 @@
from scrapy import signals
class UserAgentMiddleware(object):
class UserAgentMiddleware:
"""This middleware allows spiders to override the user_agent"""
def __init__(self, user_agent='Scrapy'):

View File

@ -5,7 +5,7 @@ from scrapy.utils.job import job_dir
from scrapy.utils.request import referer_str, request_fingerprint
class BaseDupeFilter(object):
class BaseDupeFilter:
@classmethod
def from_settings(cls, settings):

View File

@ -21,9 +21,9 @@ __all__ = ['BaseItemExporter', 'PprintItemExporter', 'PickleItemExporter',
'JsonItemExporter', 'MarshalItemExporter']
class BaseItemExporter(object):
class BaseItemExporter:
def __init__(self, dont_fail=False, **kwargs):
def __init__(self, *, dont_fail=False, **kwargs):
self._kwargs = kwargs
self._configure(kwargs, dont_fail=dont_fail)
@ -173,12 +173,12 @@ class XmlItemExporter(BaseItemExporter):
if hasattr(serialized_value, 'items'):
self._beautify_newline()
for subname, value in serialized_value.items():
self._export_xml_field(subname, value, depth=depth+1)
self._export_xml_field(subname, value, depth=depth + 1)
self._beautify_indent(depth=depth)
elif is_listlike(serialized_value):
self._beautify_newline()
for value in serialized_value:
self._export_xml_field('value', value, depth=depth+1)
self._export_xml_field('value', value, depth=depth + 1)
self._beautify_indent(depth=depth)
elif isinstance(serialized_value, str):
self.xg.characters(serialized_value)

View File

@ -6,13 +6,11 @@ See documentation in docs/topics/extensions.rst
from collections import defaultdict
from twisted.internet import reactor
from scrapy import signals
from scrapy.exceptions import NotConfigured
class CloseSpider(object):
class CloseSpider:
def __init__(self, crawler):
self.crawler = crawler
@ -54,6 +52,7 @@ class CloseSpider(object):
self.crawler.engine.close_spider(spider, 'closespider_pagecount')
def spider_opened(self, spider):
from twisted.internet import reactor
self.task = reactor.callLater(self.close_on['timeout'],
self.crawler.engine.close_spider, spider,
reason='closespider_timeout')

View File

@ -6,7 +6,7 @@ from datetime import datetime
from scrapy import signals
class CoreStats(object):
class CoreStats:
def __init__(self, stats):
self.stats = stats

View File

@ -17,7 +17,7 @@ from scrapy.utils.trackref import format_live_refs
logger = logging.getLogger(__name__)
class StackTraceDump(object):
class StackTraceDump:
def __init__(self, crawler=None):
self.crawler = crawler
@ -52,7 +52,7 @@ class StackTraceDump(object):
return dumps
class Debugger(object):
class Debugger:
def __init__(self):
try:
signal.signal(signal.SIGUSR2, self._enter_debugger)

View File

@ -4,24 +4,27 @@ Feed Exports extension
See documentation in docs/topics/feed-exports.rst
"""
import logging
import os
import sys
import logging
from tempfile import NamedTemporaryFile
import warnings
from datetime import datetime
from urllib.parse import urlparse, unquote
from tempfile import NamedTemporaryFile
from urllib.parse import unquote, urlparse
from zope.interface import Interface, implementer
from twisted.internet import defer, threads
from w3lib.url import file_uri_to_path
from zope.interface import implementer, Interface
from scrapy import signals
from scrapy.utils.ftp import ftp_store_file
from scrapy.exceptions import NotConfigured
from scrapy.utils.misc import create_instance, load_object
from scrapy.utils.log import failure_to_exc_info
from scrapy.utils.python import without_none_values
from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning
from scrapy.utils.boto import is_botocore
from scrapy.utils.conf import feed_complete_default_values_from_settings
from scrapy.utils.ftp import ftp_store_file
from scrapy.utils.log import failure_to_exc_info
from scrapy.utils.misc import create_instance, load_object
from scrapy.utils.python import without_none_values
logger = logging.getLogger(__name__)
@ -41,7 +44,7 @@ class IFeedStorage(Interface):
@implementer(IFeedStorage)
class BlockingFeedStorage(object):
class BlockingFeedStorage:
def open(self, spider):
path = spider.crawler.settings['FEED_TEMPDIR']
@ -58,7 +61,7 @@ class BlockingFeedStorage(object):
@implementer(IFeedStorage)
class StdoutFeedStorage(object):
class StdoutFeedStorage:
def __init__(self, uri, _stdout=None):
if not _stdout:
@ -73,7 +76,7 @@ class StdoutFeedStorage(object):
@implementer(IFeedStorage)
class FileFeedStorage(object):
class FileFeedStorage:
def __init__(self, uri):
self.path = file_uri_to_path(uri)
@ -98,8 +101,6 @@ class S3FeedStorage(BlockingFeedStorage):
from scrapy.utils.project import get_project_settings
settings = get_project_settings()
if 'AWS_ACCESS_KEY_ID' in settings or 'AWS_SECRET_ACCESS_KEY' in settings:
import warnings
from scrapy.exceptions import ScrapyDeprecationWarning
warnings.warn(
"Initialising `scrapy.extensions.feedexport.S3FeedStorage` "
"without AWS keys is deprecated. Please supply credentials or "
@ -204,88 +205,117 @@ class FTPFeedStorage(BlockingFeedStorage):
)
class SpiderSlot(object):
def __init__(self, file, exporter, storage, uri):
class _FeedSlot:
def __init__(self, file, exporter, storage, uri, format, store_empty):
self.file = file
self.exporter = exporter
self.storage = storage
# feed params
self.uri = uri
self.format = format
self.store_empty = store_empty
# flags
self.itemcount = 0
class FeedExporter(object):
def __init__(self, settings):
self.settings = settings
if not settings['FEED_URI']:
raise NotConfigured
self.urifmt = str(settings['FEED_URI'])
self.format = settings['FEED_FORMAT'].lower()
self.export_encoding = settings['FEED_EXPORT_ENCODING']
self.storages = self._load_components('FEED_STORAGES')
self.exporters = self._load_components('FEED_EXPORTERS')
if not self._storage_supported(self.urifmt):
raise NotConfigured
if not self._exporter_supported(self.format):
raise NotConfigured
self.store_empty = settings.getbool('FEED_STORE_EMPTY')
self._exporting = False
self.export_fields = settings.getlist('FEED_EXPORT_FIELDS') or None
self.indent = None
if settings.get('FEED_EXPORT_INDENT') is not None:
self.indent = settings.getint('FEED_EXPORT_INDENT')
uripar = settings['FEED_URI_PARAMS']
self._uripar = load_object(uripar) if uripar else lambda x, y: None
def start_exporting(self):
if not self._exporting:
self.exporter.start_exporting()
self._exporting = True
def finish_exporting(self):
if self._exporting:
self.exporter.finish_exporting()
self._exporting = False
class FeedExporter:
@classmethod
def from_crawler(cls, crawler):
o = cls(crawler.settings)
o.crawler = crawler
crawler.signals.connect(o.open_spider, signals.spider_opened)
crawler.signals.connect(o.close_spider, signals.spider_closed)
crawler.signals.connect(o.item_scraped, signals.item_scraped)
return o
exporter = cls(crawler)
crawler.signals.connect(exporter.open_spider, signals.spider_opened)
crawler.signals.connect(exporter.close_spider, signals.spider_closed)
crawler.signals.connect(exporter.item_scraped, signals.item_scraped)
return exporter
def __init__(self, crawler):
self.crawler = crawler
self.settings = crawler.settings
self.feeds = {}
self.slots = []
if not self.settings['FEEDS'] and not self.settings['FEED_URI']:
raise NotConfigured
# Begin: Backward compatibility for FEED_URI and FEED_FORMAT settings
if self.settings['FEED_URI']:
warnings.warn(
'The `FEED_URI` and `FEED_FORMAT` settings have been deprecated in favor of '
'the `FEEDS` setting. Please see the `FEEDS` setting docs for more details',
category=ScrapyDeprecationWarning, stacklevel=2,
)
uri = str(self.settings['FEED_URI']) # handle pathlib.Path objects
feed = {'format': self.settings.get('FEED_FORMAT', 'jsonlines')}
self.feeds[uri] = feed_complete_default_values_from_settings(feed, self.settings)
# End: Backward compatibility for FEED_URI and FEED_FORMAT settings
# 'FEEDS' setting takes precedence over 'FEED_URI'
for uri, feed in self.settings.getdict('FEEDS').items():
uri = str(uri) # handle pathlib.Path objects
self.feeds[uri] = feed_complete_default_values_from_settings(feed, self.settings)
self.storages = self._load_components('FEED_STORAGES')
self.exporters = self._load_components('FEED_EXPORTERS')
for uri, feed in self.feeds.items():
if not self._storage_supported(uri):
raise NotConfigured
if not self._exporter_supported(feed['format']):
raise NotConfigured
def open_spider(self, spider):
uri = self.urifmt % self._get_uri_params(spider)
storage = self._get_storage(uri)
file = storage.open(spider)
exporter = self._get_exporter(file, fields_to_export=self.export_fields,
encoding=self.export_encoding, indent=self.indent)
if self.store_empty:
exporter.start_exporting()
self._exporting = True
self.slot = SpiderSlot(file, exporter, storage, uri)
for uri, feed in self.feeds.items():
uri = uri % self._get_uri_params(spider, feed['uri_params'])
storage = self._get_storage(uri)
file = storage.open(spider)
exporter = self._get_exporter(
file=file,
format=feed['format'],
fields_to_export=feed['fields'],
encoding=feed['encoding'],
indent=feed['indent'],
)
slot = _FeedSlot(file, exporter, storage, uri, feed['format'], feed['store_empty'])
self.slots.append(slot)
if slot.store_empty:
slot.start_exporting()
def close_spider(self, spider):
slot = self.slot
if not slot.itemcount and not self.store_empty:
# We need to call slot.storage.store nonetheless to get the file
# properly closed.
return defer.maybeDeferred(slot.storage.store, slot.file)
if self._exporting:
slot.exporter.finish_exporting()
self._exporting = False
logfmt = "%s %%(format)s feed (%%(itemcount)d items) in: %%(uri)s"
log_args = {'format': self.format,
'itemcount': slot.itemcount,
'uri': slot.uri}
d = defer.maybeDeferred(slot.storage.store, slot.file)
d.addCallback(lambda _: logger.info(logfmt % "Stored", log_args,
extra={'spider': spider}))
d.addErrback(lambda f: logger.error(logfmt % "Error storing", log_args,
exc_info=failure_to_exc_info(f),
extra={'spider': spider}))
return d
deferred_list = []
for slot in self.slots:
if not slot.itemcount and not slot.store_empty:
# We need to call slot.storage.store nonetheless to get the file
# properly closed.
return defer.maybeDeferred(slot.storage.store, slot.file)
slot.finish_exporting()
logfmt = "%s %%(format)s feed (%%(itemcount)d items) in: %%(uri)s"
log_args = {'format': slot.format,
'itemcount': slot.itemcount,
'uri': slot.uri}
d = defer.maybeDeferred(slot.storage.store, slot.file)
d.addCallback(lambda _: logger.info(logfmt % "Stored", log_args,
extra={'spider': spider}))
d.addErrback(lambda f: logger.error(logfmt % "Error storing", log_args,
exc_info=failure_to_exc_info(f),
extra={'spider': spider}))
deferred_list.append(d)
return defer.DeferredList(deferred_list) if deferred_list else None
def item_scraped(self, item, spider):
slot = self.slot
if not self._exporting:
slot.exporter.start_exporting()
self._exporting = True
slot.exporter.export_item(item)
slot.itemcount += 1
return item
for slot in self.slots:
slot.start_exporting()
slot.exporter.export_item(item)
slot.itemcount += 1
def _load_components(self, setting_prefix):
conf = without_none_values(self.settings.getwithbase(setting_prefix))
@ -321,17 +351,18 @@ class FeedExporter(object):
objcls, self.settings, getattr(self, 'crawler', None),
*args, **kwargs)
def _get_exporter(self, *args, **kwargs):
return self._get_instance(self.exporters[self.format], *args, **kwargs)
def _get_exporter(self, file, format, *args, **kwargs):
return self._get_instance(self.exporters[format], file, *args, **kwargs)
def _get_storage(self, uri):
return self._get_instance(self.storages[urlparse(uri).scheme], uri)
def _get_uri_params(self, spider):
def _get_uri_params(self, spider, uri_params):
params = {}
for k in dir(spider):
params[k] = getattr(spider, k)
ts = datetime.utcnow().replace(microsecond=0).isoformat().replace(':', '-')
params['time'] = ts
self._uripar(params, spider)
uripar_function = load_object(uri_params) if uri_params else lambda x, y: None
uripar_function(params, spider)
return params

View File

@ -20,7 +20,7 @@ from scrapy.utils.request import request_fingerprint
logger = logging.getLogger(__name__)
class DummyPolicy(object):
class DummyPolicy:
def __init__(self, settings):
self.ignore_schemes = settings.getlist('HTTPCACHE_IGNORE_SCHEMES')
@ -39,7 +39,7 @@ class DummyPolicy(object):
return True
class RFC2616Policy(object):
class RFC2616Policy:
MAXAGE = 3600 * 24 * 365 # one year
@ -213,7 +213,7 @@ class RFC2616Policy(object):
return currentage
class DbmCacheStorage(object):
class DbmCacheStorage:
def __init__(self, settings):
self.cachedir = data_path(settings['HTTPCACHE_DIR'], createdir=True)
@ -270,7 +270,7 @@ class DbmCacheStorage(object):
return request_fingerprint(request)
class FilesystemCacheStorage(object):
class FilesystemCacheStorage:
def __init__(self, settings):
self.cachedir = data_path(settings['HTTPCACHE_DIR'])

View File

@ -8,7 +8,7 @@ from scrapy import signals
logger = logging.getLogger(__name__)
class LogStats(object):
class LogStats:
"""Log basic scraping stats periodically"""
def __init__(self, stats, interval=60.0):

View File

@ -11,7 +11,7 @@ from scrapy.exceptions import NotConfigured
from scrapy.utils.trackref import live_refs
class MemoryDebugger(object):
class MemoryDebugger:
def __init__(self, stats):
self.stats = stats

View File

@ -19,7 +19,7 @@ from scrapy.utils.engine import get_engine_status
logger = logging.getLogger(__name__)
class MemoryUsage(object):
class MemoryUsage:
def __init__(self, crawler):
if not crawler.settings.getbool('MEMUSAGE_ENABLED'):
@ -47,7 +47,7 @@ class MemoryUsage(object):
def get_virtual_size(self):
size = self.resource.getrusage(self.resource.RUSAGE_SELF).ru_maxrss
if sys.platform != 'darwin':
# on Mac OS X ru_maxrss is in bytes, on Linux it is in KB
# on macOS ru_maxrss is in bytes, on Linux it is in KB
size *= 1024
return size

View File

@ -6,7 +6,7 @@ from scrapy.exceptions import NotConfigured
from scrapy.utils.job import job_dir
class SpiderState(object):
class SpiderState:
"""Store and load spider state during a scraping job"""
def __init__(self, jobdir=None):

View File

@ -8,7 +8,7 @@ from scrapy import signals
from scrapy.mail import MailSender
from scrapy.exceptions import NotConfigured
class StatsMailer(object):
class StatsMailer:
def __init__(self, stats, recipients, mail):
self.stats = stats

View File

@ -26,11 +26,6 @@ from scrapy.utils.engine import print_engine_status
from scrapy.utils.reactor import listen_tcp
from scrapy.utils.decorators import defers
try:
import guppy
hpy = guppy.hpy()
except ImportError:
hpy = None
logger = logging.getLogger(__name__)
@ -110,7 +105,6 @@ class TelnetConsole(protocol.ServerFactory):
'est': lambda: print_engine_status(self.crawler.engine),
'p': pprint.pprint,
'prefs': print_live_refs,
'hpy': hpy,
'help': "This is Scrapy telnet console. For more info see: "
"https://docs.scrapy.org/en/latest/topics/telnetconsole.html",
}

View File

@ -6,7 +6,7 @@ from scrapy import signals
logger = logging.getLogger(__name__)
class AutoThrottle(object):
class AutoThrottle:
def __init__(self, crawler):
self.crawler = crawler

View File

@ -5,7 +5,7 @@ from scrapy.utils.httpobj import urlparse_cached
from scrapy.utils.python import to_unicode
class CookieJar(object):
class CookieJar:
def __init__(self, policy=None, check_expired_frequency=10000):
self.policy = policy or DefaultCookiePolicy()
self.jar = _CookieJar(self.policy)
@ -100,7 +100,7 @@ def potential_domain_matches(domain):
return matches + ['.' + d for d in matches]
class _DummyLock(object):
class _DummyLock:
def acquire(self):
pass
@ -108,7 +108,7 @@ class _DummyLock(object):
pass
class WrappedRequest(object):
class WrappedRequest:
"""Wraps a scrapy Request class with methods defined by urllib2.Request class to interact with CookieJar class
see http://docs.python.org/library/urllib2.html#urllib2.Request
@ -178,7 +178,7 @@ class WrappedRequest(object):
self.request.headers.appendlist(name, value)
class WrappedResponse(object):
class WrappedResponse:
def __init__(self, response):
self.response = response

View File

@ -17,13 +17,24 @@ from scrapy.utils.trackref import object_ref
class Response(object_ref):
def __init__(self, url, status=200, headers=None, body=b'', flags=None, request=None):
def __init__(self, url, status=200, headers=None, body=b'', flags=None, request=None, certificate=None):
self.headers = Headers(headers or {})
self.status = int(status)
self._set_body(body)
self._set_url(url)
self.request = request
self.flags = [] if flags is None else list(flags)
self.certificate = certificate
@property
def cb_kwargs(self):
try:
return self.request.cb_kwargs
except AttributeError:
raise AttributeError(
"Response.cb_kwargs not available, this response "
"is not tied to any request"
)
@property
def meta(self):
@ -76,7 +87,7 @@ class Response(object_ref):
"""Create a new Response with the same attributes except for those
given new values.
"""
for x in ['url', 'status', 'headers', 'body', 'request', 'flags']:
for x in ['url', 'status', 'headers', 'body', 'request', 'flags', 'certificate']:
kwargs.setdefault(x, getattr(self, x))
cls = kwargs.pop('cls', self.__class__)
return cls(*args, **kwargs)
@ -107,7 +118,7 @@ class Response(object_ref):
def follow(self, url, callback=None, method='GET', headers=None, body=None,
cookies=None, meta=None, encoding='utf-8', priority=0,
dont_filter=False, errback=None, cb_kwargs=None):
dont_filter=False, errback=None, cb_kwargs=None, flags=None):
# type: (...) -> Request
"""
Return a :class:`~.Request` instance to follow a link ``url``.
@ -118,12 +129,16 @@ class Response(object_ref):
:class:`~.TextResponse` provides a :meth:`~.TextResponse.follow`
method which supports selectors in addition to absolute/relative URLs
and Link objects.
.. versionadded:: 2.0
The *flags* parameter.
"""
if isinstance(url, Link):
url = url.url
elif url is None:
raise ValueError("url can't be None")
url = self.urljoin(url)
return Request(
url=url,
callback=callback,
@ -137,13 +152,16 @@ class Response(object_ref):
dont_filter=dont_filter,
errback=errback,
cb_kwargs=cb_kwargs,
flags=flags,
)
def follow_all(self, urls, callback=None, method='GET', headers=None, body=None,
cookies=None, meta=None, encoding='utf-8', priority=0,
dont_filter=False, errback=None, cb_kwargs=None):
dont_filter=False, errback=None, cb_kwargs=None, flags=None):
# type: (...) -> Generator[Request, None, None]
"""
.. versionadded:: 2.0
Return an iterable of :class:`~.Request` instances to follow all links
in ``urls``. It accepts the same arguments as ``Request.__init__`` method,
but elements of ``urls`` can be relative URLs or :class:`~scrapy.link.Link` objects,
@ -169,6 +187,7 @@ class Response(object_ref):
dont_filter=dont_filter,
errback=errback,
cb_kwargs=cb_kwargs,
flags=flags,
)
for url in urls
)

View File

@ -121,7 +121,7 @@ class TextResponse(Response):
def follow(self, url, callback=None, method='GET', headers=None, body=None,
cookies=None, meta=None, encoding=None, priority=0,
dont_filter=False, errback=None, cb_kwargs=None):
dont_filter=False, errback=None, cb_kwargs=None, flags=None):
# type: (...) -> Request
"""
Return a :class:`~.Request` instance to follow a link ``url``.
@ -157,11 +157,12 @@ class TextResponse(Response):
dont_filter=dont_filter,
errback=errback,
cb_kwargs=cb_kwargs,
flags=flags,
)
def follow_all(self, urls=None, callback=None, method='GET', headers=None, body=None,
cookies=None, meta=None, encoding=None, priority=0,
dont_filter=False, errback=None, cb_kwargs=None,
dont_filter=False, errback=None, cb_kwargs=None, flags=None,
css=None, xpath=None):
# type: (...) -> Generator[Request, None, None]
"""
@ -187,9 +188,11 @@ class TextResponse(Response):
selectors from which links cannot be obtained (for instance, anchor tags without an
``href`` attribute)
"""
arg_count = len(list(filter(None, (urls, css, xpath))))
if arg_count != 1:
raise ValueError('Please supply exactly one of the following arguments: urls, css, xpath')
arguments = [x for x in (urls, css, xpath) if x is not None]
if len(arguments) != 1:
raise ValueError(
"Please supply exactly one of the following arguments: urls, css, xpath"
)
if not urls:
if css:
urls = self.css(css)
@ -214,6 +217,7 @@ class TextResponse(Response):
dont_filter=dont_filter,
errback=errback,
cb_kwargs=cb_kwargs,
flags=flags,
)

View File

@ -6,7 +6,7 @@ its documentation in: docs/topics/link-extractors.rst
"""
class Link(object):
class Link:
"""Link objects represent an extracted link by the LinkExtractor."""
__slots__ = ['url', 'text', 'fragment', 'nofollow']

View File

@ -49,7 +49,7 @@ _matches = lambda url, regexs: any(r.search(url) for r in regexs)
_is_valid_url = lambda url: url.split('://', 1)[0] in {'http', 'https', 'file', 'ftp'}
class FilteringLinkExtractor(object):
class FilteringLinkExtractor:
_csstranslator = HTMLTranslator()

View File

@ -5,11 +5,11 @@ from urllib.parse import urljoin
import lxml.etree as etree
from w3lib.html import strip_html5_whitespace
from w3lib.url import canonicalize_url
from w3lib.url import canonicalize_url, safe_url_string
from scrapy.link import Link
from scrapy.utils.misc import arg_to_iter, rel_has_nofollow
from scrapy.utils.python import unique as unique_list, to_unicode
from scrapy.utils.python import unique as unique_list
from scrapy.utils.response import get_base_url
from scrapy.linkextractors import FilteringLinkExtractor
@ -22,12 +22,12 @@ _collect_string_content = etree.XPath("string()")
def _nons(tag):
if isinstance(tag, str):
if tag[0] == '{' and tag[1:len(XHTML_NAMESPACE)+1] == XHTML_NAMESPACE:
if tag[0] == '{' and tag[1:len(XHTML_NAMESPACE) + 1] == XHTML_NAMESPACE:
return tag.split('}')[-1]
return tag
class LxmlParserLinkExtractor(object):
class LxmlParserLinkExtractor:
def __init__(self, tag="a", attr="href", process=None, unique=False,
strip=True, canonicalized=False):
self.scan_tag = tag if callable(tag) else lambda t: t == tag
@ -66,7 +66,7 @@ class LxmlParserLinkExtractor(object):
url = self.process_attr(attr_val)
if url is None:
continue
url = to_unicode(url, encoding=response_encoding)
url = safe_url_string(url, encoding=response_encoding)
# to fix relative links after process_value
url = urljoin(response_url, url)
link = Link(url, _collect_string_content(el) or u'',

View File

@ -25,7 +25,7 @@ def unbound_method(method):
return method
class ItemLoader(object):
class ItemLoader:
default_item_class = Item
default_input_processor = Identity()

Some files were not shown because too many files have changed in this diff Show More