mirror of https://github.com/scrapy/scrapy.git
Merge branch 'master' into feat-685
This commit is contained in:
commit
34d607194b
|
|
@ -1,8 +1,7 @@
|
|||
[bumpversion]
|
||||
current_version = 1.8.0
|
||||
current_version = 2.0.0
|
||||
commit = True
|
||||
tag = True
|
||||
tag_name = {new_version}
|
||||
|
||||
[bumpversion:file:scrapy/VERSION]
|
||||
|
||||
|
|
|
|||
|
|
@ -1,9 +1,11 @@
|
|||
version: 2
|
||||
sphinx:
|
||||
configuration: docs/conf.py
|
||||
fail_on_warning: true
|
||||
python:
|
||||
# For available versions, see:
|
||||
# https://docs.readthedocs.io/en/stable/config-file/v2.html#build-image
|
||||
version: 3.7 # Keep in sync with .travis.yml
|
||||
install:
|
||||
- requirements: docs/requirements.txt
|
||||
- path: .
|
||||
|
|
|
|||
|
|
@ -1,24 +0,0 @@
|
|||
TRIAL := $(shell which trial)
|
||||
BRANCH := $(shell git rev-parse --abbrev-ref HEAD)
|
||||
export PYTHONPATH=$(PWD)
|
||||
|
||||
test:
|
||||
coverage run --branch $(TRIAL) --reporter=text tests
|
||||
rm -rf htmlcov && coverage html
|
||||
-s3cmd sync -P htmlcov/ s3://static.scrapy.org/coverage-scrapy-$(BRANCH)/
|
||||
|
||||
build:
|
||||
git describe --tags --match '[0-9]*' |sed 's/-/.post/;s/-g/+g/' >scrapy/VERSION
|
||||
debchange -m -D unstable --force-distribution -v \
|
||||
$$(python setup.py --version |sed -r 's/([0-9]+.[0-9]+.[0-9]+)(a|b|rc|dev)([0-9]*)/\1~\2\3/')-$$(date +%s) \
|
||||
"Automatic build"
|
||||
debuild -us -uc -b
|
||||
|
||||
clean:
|
||||
git checkout debian scrapy/VERSION
|
||||
git clean -dfq
|
||||
|
||||
pypi:
|
||||
umask 0022 && chmod -R a+rX . && python setup.py sdist upload
|
||||
|
||||
.PHONY: clean test build
|
||||
|
|
@ -41,7 +41,7 @@ Requirements
|
|||
============
|
||||
|
||||
* Python 3.5+
|
||||
* Works on Linux, Windows, Mac OSX, BSD
|
||||
* Works on Linux, Windows, macOS, BSD
|
||||
|
||||
Install
|
||||
=======
|
||||
|
|
|
|||
|
|
@ -11,7 +11,9 @@ collect_ignore = [
|
|||
# not a test, but looks like a test
|
||||
"scrapy/utils/testsite.py",
|
||||
# contains scripts to be run by tests/test_crawler.py::CrawlerProcessSubprocess
|
||||
*_py_files("tests/CrawlerProcess")
|
||||
*_py_files("tests/CrawlerProcess"),
|
||||
# Py36-only parts of respective tests
|
||||
*_py_files("tests/py36"),
|
||||
]
|
||||
|
||||
for line in open('tests/ignores.txt'):
|
||||
|
|
|
|||
|
|
@ -1,5 +0,0 @@
|
|||
scrapy (0.11) unstable; urgency=low
|
||||
|
||||
* Initial release.
|
||||
|
||||
-- Scrapinghub Team <info@scrapinghub.com> Thu, 10 Jun 2010 17:24:02 -0300
|
||||
|
|
@ -1 +0,0 @@
|
|||
7
|
||||
|
|
@ -1,20 +0,0 @@
|
|||
Source: scrapy
|
||||
Section: python
|
||||
Priority: optional
|
||||
Maintainer: Scrapinghub Team <info@scrapinghub.com>
|
||||
Build-Depends: debhelper (>= 7.0.50), python (>=2.7), python-twisted, python-w3lib, python-lxml, python-six (>=1.5.2)
|
||||
Standards-Version: 3.8.4
|
||||
Homepage: https://scrapy.org/
|
||||
|
||||
Package: scrapy
|
||||
Architecture: all
|
||||
Depends: ${python:Depends}, python-lxml, python-twisted, python-openssl,
|
||||
python-w3lib (>= 1.8.0), python-queuelib, python-cssselect (>= 0.9), python-six (>=1.5.2)
|
||||
Recommends: python-setuptools
|
||||
Conflicts: python-scrapy, scrapy-0.25
|
||||
Provides: python-scrapy, scrapy-0.25
|
||||
Description: Python web crawling and web scraping framework
|
||||
Scrapy is a fast high-level web crawling and web scraping framework,
|
||||
used to crawl websites and extract structured data from their pages.
|
||||
It can be used for a wide range of purposes, from data mining to
|
||||
monitoring and automated testing.
|
||||
|
|
@ -1,40 +0,0 @@
|
|||
This package was debianized by the Scrapinghub team <info@scrapinghub.com>.
|
||||
|
||||
It was downloaded from https://scrapy.org
|
||||
|
||||
Upstream Author: Scrapy Developers
|
||||
|
||||
Copyright: 2007-2013 Scrapy Developers
|
||||
|
||||
License: bsd
|
||||
|
||||
Copyright (c) Scrapy developers.
|
||||
All rights reserved.
|
||||
|
||||
Redistribution and use in source and binary forms, with or without modification,
|
||||
are permitted provided that the following conditions are met:
|
||||
|
||||
1. Redistributions of source code must retain the above copyright notice,
|
||||
this list of conditions and the following disclaimer.
|
||||
|
||||
2. Redistributions in binary form must reproduce the above copyright
|
||||
notice, this list of conditions and the following disclaimer in the
|
||||
documentation and/or other materials provided with the distribution.
|
||||
|
||||
3. Neither the name of Scrapy nor the names of its contributors may be used
|
||||
to endorse or promote products derived from this software without
|
||||
specific prior written permission.
|
||||
|
||||
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
|
||||
ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
|
||||
WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR
|
||||
ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
|
||||
(INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
|
||||
LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON
|
||||
ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
(INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS
|
||||
SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
|
||||
The Debian packaging is (C) 2010-2013, Scrapinghub <info@scrapinghub.com> and
|
||||
is licensed under the BSD, see `/usr/share/common-licenses/BSD'.
|
||||
|
|
@ -1 +0,0 @@
|
|||
2.7
|
||||
|
|
@ -1,5 +0,0 @@
|
|||
#!/usr/bin/make -f
|
||||
# -*- makefile -*-
|
||||
|
||||
%:
|
||||
dh $@
|
||||
|
|
@ -1,2 +0,0 @@
|
|||
README.rst
|
||||
AUTHORS
|
||||
|
|
@ -1,2 +0,0 @@
|
|||
extras/scrapy_bash_completion etc/bash_completion.d/
|
||||
extras/scrapy_zsh_completion /usr/share/zsh/vendor-completions/_scrapy
|
||||
|
|
@ -1 +0,0 @@
|
|||
new-package-should-close-itp-bug
|
||||
|
|
@ -1 +0,0 @@
|
|||
extras/scrapy.1
|
||||
|
|
@ -281,6 +281,7 @@ coverage_ignore_pyobjects = [
|
|||
|
||||
intersphinx_mapping = {
|
||||
'coverage': ('https://coverage.readthedocs.io/en/stable', None),
|
||||
'cssselect': ('https://cssselect.readthedocs.io/en/latest', None),
|
||||
'pytest': ('https://docs.pytest.org/en/latest', None),
|
||||
'python': ('https://docs.python.org/3', None),
|
||||
'sphinx': ('https://www.sphinx-doc.org/en/master', None),
|
||||
|
|
|
|||
|
|
@ -143,7 +143,7 @@ by running ``git fetch upstream pull/$PR_NUMBER/head:$BRANCH_NAME_TO_CREATE``
|
|||
(replace 'upstream' with a remote name for scrapy repository,
|
||||
``$PR_NUMBER`` with an ID of the pull request, and ``$BRANCH_NAME_TO_CREATE``
|
||||
with a name of the branch you want to create locally).
|
||||
See also: https://help.github.com/articles/checking-out-pull-requests-locally/#modifying-an-inactive-pull-request-locally.
|
||||
See also: https://help.github.com/en/github/collaborating-with-issues-and-pull-requests/checking-out-pull-requests-locally#modifying-an-inactive-pull-request-locally.
|
||||
|
||||
When writing GitHub pull requests, try to keep titles short but descriptive.
|
||||
E.g. For bug #411: "Scrapy hangs if an exception raises in start_requests"
|
||||
|
|
@ -168,7 +168,7 @@ Scrapy:
|
|||
|
||||
* Don't put your name in the code you contribute; git provides enough
|
||||
metadata to identify author of the code.
|
||||
See https://help.github.com/articles/setting-your-username-in-git/ for
|
||||
See https://help.github.com/en/github/using-git/setting-your-username-in-git for
|
||||
setup instructions.
|
||||
|
||||
.. _documentation-policies:
|
||||
|
|
@ -266,5 +266,5 @@ And their unit-tests are in::
|
|||
.. _tests/: https://github.com/scrapy/scrapy/tree/master/tests
|
||||
.. _open issues: https://github.com/scrapy/scrapy/issues
|
||||
.. _PEP 257: https://www.python.org/dev/peps/pep-0257/
|
||||
.. _pull request: https://help.github.com/en/articles/creating-a-pull-request
|
||||
.. _pull request: https://help.github.com/en/github/collaborating-with-issues-and-pull-requests/creating-a-pull-request
|
||||
.. _pytest-xdist: https://github.com/pytest-dev/pytest-xdist
|
||||
|
|
|
|||
|
|
@ -22,8 +22,8 @@ In other words, comparing `BeautifulSoup`_ (or `lxml`_) to Scrapy is like
|
|||
comparing `jinja2`_ to `Django`_.
|
||||
|
||||
.. _BeautifulSoup: https://www.crummy.com/software/BeautifulSoup/
|
||||
.. _lxml: http://lxml.de/
|
||||
.. _jinja2: http://jinja.pocoo.org/
|
||||
.. _lxml: https://lxml.de/
|
||||
.. _jinja2: https://palletsprojects.com/p/jinja/
|
||||
.. _Django: https://www.djangoproject.com/
|
||||
|
||||
Can I use Scrapy with BeautifulSoup?
|
||||
|
|
@ -269,7 +269,7 @@ The ``__VIEWSTATE`` parameter is used in sites built with ASP.NET/VB.NET. For
|
|||
more info on how it works see `this page`_. Also, here's an `example spider`_
|
||||
which scrapes one of these sites.
|
||||
|
||||
.. _this page: http://search.cpan.org/~ecarroll/HTML-TreeBuilderX-ASP_NET-0.09/lib/HTML/TreeBuilderX/ASP_NET.pm
|
||||
.. _this page: https://metacpan.org/pod/release/ECARROLL/HTML-TreeBuilderX-ASP_NET-0.09/lib/HTML/TreeBuilderX/ASP_NET.pm
|
||||
.. _example spider: https://github.com/AmbientLighter/rpn-fas/blob/master/fas/spiders/rnp.py
|
||||
|
||||
What's the best way to parse big XML/CSV data feeds?
|
||||
|
|
|
|||
|
|
@ -165,6 +165,8 @@ Solving specific problems
|
|||
topics/autothrottle
|
||||
topics/benchmarking
|
||||
topics/jobs
|
||||
topics/coroutines
|
||||
topics/asyncio
|
||||
|
||||
:doc:`faq`
|
||||
Get answers to most frequently asked questions.
|
||||
|
|
@ -205,6 +207,12 @@ Solving specific problems
|
|||
:doc:`topics/jobs`
|
||||
Learn how to pause and resume crawls for large spiders.
|
||||
|
||||
:doc:`topics/coroutines`
|
||||
Use the :ref:`coroutine syntax <async>`.
|
||||
|
||||
:doc:`topics/asyncio`
|
||||
Use :mod:`asyncio` and :mod:`asyncio`-powered libraries.
|
||||
|
||||
.. _extending-scrapy:
|
||||
|
||||
Extending Scrapy
|
||||
|
|
|
|||
|
|
@ -7,12 +7,12 @@ Installation guide
|
|||
Installing Scrapy
|
||||
=================
|
||||
|
||||
Scrapy runs on Python 3.5 or above
|
||||
under CPython (default Python implementation) and PyPy (starting with PyPy 5.9).
|
||||
Scrapy runs on Python 3.5 or above under CPython (default Python
|
||||
implementation) and PyPy (starting with PyPy 5.9).
|
||||
|
||||
If you're using `Anaconda`_ or `Miniconda`_, you can install the package from
|
||||
the `conda-forge`_ channel, which has up-to-date packages for Linux, Windows
|
||||
and OS X.
|
||||
and macOS.
|
||||
|
||||
To install Scrapy using ``conda``, run::
|
||||
|
||||
|
|
@ -65,7 +65,7 @@ please refer to their respective installation instructions:
|
|||
* `lxml installation`_
|
||||
* `cryptography installation`_
|
||||
|
||||
.. _lxml installation: http://lxml.de/installation.html
|
||||
.. _lxml installation: https://lxml.de/installation.html
|
||||
.. _cryptography installation: https://cryptography.io/en/latest/installation/
|
||||
|
||||
|
||||
|
|
@ -81,35 +81,18 @@ Python packages can be installed either globally (a.k.a system wide),
|
|||
or in user-space. We do not recommend installing Scrapy system wide.
|
||||
|
||||
Instead, we recommend that you install Scrapy within a so-called
|
||||
"virtual environment" (`virtualenv`_).
|
||||
Virtualenvs allow you to not conflict with already-installed Python
|
||||
"virtual environment" (:mod:`venv`).
|
||||
Virtual environments allow you to not conflict with already-installed Python
|
||||
system packages (which could break some of your system tools and scripts),
|
||||
and still install packages normally with ``pip`` (without ``sudo`` and the likes).
|
||||
|
||||
To get started with virtual environments, see `virtualenv installation instructions`_.
|
||||
To install it globally (having it globally installed actually helps here),
|
||||
it should be a matter of running::
|
||||
See :ref:`tut-venv` on how to create your virtual environment.
|
||||
|
||||
$ [sudo] pip install virtualenv
|
||||
|
||||
Check this `user guide`_ on how to create your virtualenv.
|
||||
|
||||
.. note::
|
||||
If you use Linux or OS X, `virtualenvwrapper`_ is a handy tool to create virtualenvs.
|
||||
|
||||
Once you have created a virtualenv, you can install Scrapy inside it with ``pip``,
|
||||
Once you have created a virtual environment, you can install Scrapy inside it with ``pip``,
|
||||
just like any other Python package.
|
||||
(See :ref:`platform-specific guides <intro-install-platform-notes>`
|
||||
below for non-Python dependencies that you may need to install beforehand).
|
||||
|
||||
Python virtualenvs can be created to use Python 2 by default, or Python 3 by default. As Scrapy
|
||||
only supports Python 3, make sure you created a Python 3 virtualenv.
|
||||
|
||||
.. _virtualenv: https://virtualenv.pypa.io
|
||||
.. _virtualenv installation instructions: https://virtualenv.pypa.io/en/stable/installation/
|
||||
.. _virtualenvwrapper: https://virtualenvwrapper.readthedocs.io/en/latest/install.html
|
||||
.. _user guide: https://virtualenv.pypa.io/en/stable/userguide/
|
||||
|
||||
|
||||
.. _intro-install-platform-notes:
|
||||
|
||||
|
|
@ -165,11 +148,11 @@ you can install Scrapy with ``pip`` after that::
|
|||
|
||||
.. _intro-install-macos:
|
||||
|
||||
Mac OS X
|
||||
--------
|
||||
macOS
|
||||
-----
|
||||
|
||||
Building Scrapy's dependencies requires the presence of a C compiler and
|
||||
development headers. On OS X this is typically provided by Apple’s Xcode
|
||||
development headers. On macOS this is typically provided by Apple’s Xcode
|
||||
development tools. To install the Xcode command line tools open a terminal
|
||||
window and run::
|
||||
|
||||
|
|
@ -205,15 +188,12 @@ solutions:
|
|||
|
||||
brew update; brew upgrade python
|
||||
|
||||
* *(Optional)* Install Scrapy inside an isolated python environment.
|
||||
* *(Optional)* :ref:`Install Scrapy inside a Python virtual environment
|
||||
<intro-using-virtualenv>`.
|
||||
|
||||
This method is a workaround for the above OS X issue, but it's an overall
|
||||
This method is a workaround for the above macOS issue, but it's an overall
|
||||
good practice for managing dependencies and can complement the first method.
|
||||
|
||||
`virtualenv`_ is a tool you can use to create virtual environments in python.
|
||||
We recommended reading a tutorial like
|
||||
http://docs.python-guide.org/en/latest/dev/virtualenvs/ to get started.
|
||||
|
||||
After any of these workarounds you should be able to install Scrapy::
|
||||
|
||||
pip install Scrapy
|
||||
|
|
@ -227,7 +207,7 @@ For PyPy3, only Linux installation was tested.
|
|||
|
||||
Most Scrapy dependencides now have binary wheels for CPython, but not for PyPy.
|
||||
This means that these dependecies will be built during installation.
|
||||
On OS X, you are likely to face an issue with building Cryptography dependency,
|
||||
On macOS, you are likely to face an issue with building Cryptography dependency,
|
||||
solution to this problem is described
|
||||
`here <https://github.com/pyca/cryptography/issues/2692#issuecomment-272773481>`_,
|
||||
that is to ``brew install openssl`` and then export the flags that this command
|
||||
|
|
@ -273,11 +253,11 @@ For details, see `Issue #2473 <https://github.com/scrapy/scrapy/issues/2473>`_.
|
|||
.. _Python: https://www.python.org/
|
||||
.. _pip: https://pip.pypa.io/en/latest/installing/
|
||||
.. _lxml: https://lxml.de/index.html
|
||||
.. _parsel: https://pypi.python.org/pypi/parsel
|
||||
.. _w3lib: https://pypi.python.org/pypi/w3lib
|
||||
.. _twisted: https://twistedmatrix.com/
|
||||
.. _cryptography: https://cryptography.io/
|
||||
.. _pyOpenSSL: https://pypi.python.org/pypi/pyOpenSSL
|
||||
.. _parsel: https://pypi.org/project/parsel/
|
||||
.. _w3lib: https://pypi.org/project/w3lib/
|
||||
.. _twisted: https://twistedmatrix.com/trac/
|
||||
.. _cryptography: https://cryptography.io/en/latest/
|
||||
.. _pyOpenSSL: https://pypi.org/project/pyOpenSSL/
|
||||
.. _setuptools: https://pypi.python.org/pypi/setuptools
|
||||
.. _AUR Scrapy package: https://aur.archlinux.org/packages/scrapy/
|
||||
.. _homebrew: https://brew.sh/
|
||||
|
|
|
|||
|
|
@ -212,7 +212,7 @@ using the :ref:`Scrapy shell <topics-shell>`. Run::
|
|||
.. note::
|
||||
|
||||
Remember to always enclose urls in quotes when running Scrapy shell from
|
||||
command-line, otherwise urls containing arguments (ie. ``&`` character)
|
||||
command-line, otherwise urls containing arguments (i.e. ``&`` character)
|
||||
will not work.
|
||||
|
||||
On Windows, use double quotes instead::
|
||||
|
|
@ -306,7 +306,7 @@ with a selector (see :ref:`topics-developer-tools`).
|
|||
visually selected elements, which works in many browsers.
|
||||
|
||||
.. _regular expressions: https://docs.python.org/3/library/re.html
|
||||
.. _Selector Gadget: http://selectorgadget.com/
|
||||
.. _Selector Gadget: https://selectorgadget.com/
|
||||
|
||||
|
||||
XPath: a brief intro
|
||||
|
|
@ -337,7 +337,7 @@ recommend `this tutorial to learn XPath through examples
|
|||
<http://zvon.org/comp/r/tut-XPath_1.html>`_, and `this tutorial to learn "how
|
||||
to think in XPath" <http://plasmasturm.org/log/xpath101/>`_.
|
||||
|
||||
.. _XPath: https://www.w3.org/TR/xpath
|
||||
.. _XPath: https://www.w3.org/TR/xpath/all/
|
||||
.. _CSS: https://www.w3.org/TR/selectors
|
||||
|
||||
Extracting quotes and authors
|
||||
|
|
|
|||
505
docs/news.rst
505
docs/news.rst
|
|
@ -3,8 +3,468 @@
|
|||
Release notes
|
||||
=============
|
||||
|
||||
.. note:: Scrapy 1.x will be the last series supporting Python 2. Scrapy 2.0,
|
||||
planned for Q4 2019 or Q1 2020, will support **Python 3 only**.
|
||||
.. _release-2.0.1:
|
||||
|
||||
Scrapy 2.0.1 (2020-03-18)
|
||||
-------------------------
|
||||
|
||||
* :meth:`Response.follow_all <scrapy.http.Response.follow_all>` now supports
|
||||
an empty URL iterable as input (:issue:`4408`, :issue:`4420`)
|
||||
|
||||
* Removed top-level :mod:`~twisted.internet.reactor` imports to prevent
|
||||
errors about the wrong Twisted reactor being installed when setting a
|
||||
different Twisted reactor using :setting:`TWISTED_REACTOR` (:issue:`4401`,
|
||||
:issue:`4406`)
|
||||
|
||||
* Fixed tests (:issue:`4422`)
|
||||
|
||||
|
||||
.. _release-2.0.0:
|
||||
|
||||
Scrapy 2.0.0 (2020-03-03)
|
||||
-------------------------
|
||||
|
||||
Highlights:
|
||||
|
||||
* Python 2 support has been removed
|
||||
* :doc:`Partial <topics/coroutines>` :ref:`coroutine syntax <async>` support
|
||||
and :doc:`experimental <topics/asyncio>` :mod:`asyncio` support
|
||||
* New :meth:`Response.follow_all <scrapy.http.Response.follow_all>` method
|
||||
* :ref:`FTP support <media-pipeline-ftp>` for media pipelines
|
||||
* New :attr:`Response.certificate <scrapy.http.Response.certificate>`
|
||||
attribute
|
||||
* IPv6 support through :setting:`DNS_RESOLVER`
|
||||
|
||||
Backward-incompatible changes
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
* Python 2 support has been removed, following `Python 2 end-of-life on
|
||||
January 1, 2020`_ (:issue:`4091`, :issue:`4114`, :issue:`4115`,
|
||||
:issue:`4121`, :issue:`4138`, :issue:`4231`, :issue:`4242`, :issue:`4304`,
|
||||
:issue:`4309`, :issue:`4373`)
|
||||
|
||||
* Retry gaveups (see :setting:`RETRY_TIMES`) are now logged as errors instead
|
||||
of as debug information (:issue:`3171`, :issue:`3566`)
|
||||
|
||||
* File extensions that
|
||||
:class:`LinkExtractor <scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor>`
|
||||
ignores by default now also include ``7z``, ``7zip``, ``apk``, ``bz2``,
|
||||
``cdr``, ``dmg``, ``ico``, ``iso``, ``tar``, ``tar.gz``, ``webm``, and
|
||||
``xz`` (:issue:`1837`, :issue:`2067`, :issue:`4066`)
|
||||
|
||||
* The :setting:`METAREFRESH_IGNORE_TAGS` setting is now an empty list by
|
||||
default, following web browser behavior (:issue:`3844`, :issue:`4311`)
|
||||
|
||||
* The
|
||||
:class:`~scrapy.downloadermiddlewares.httpcompression.HttpCompressionMiddleware`
|
||||
now includes spaces after commas in the value of the ``Accept-Encoding``
|
||||
header that it sets, following web browser behavior (:issue:`4293`)
|
||||
|
||||
* The ``__init__`` method of custom download handlers (see
|
||||
:setting:`DOWNLOAD_HANDLERS`) or subclasses of the following downloader
|
||||
handlers no longer receives a ``settings`` parameter:
|
||||
|
||||
* :class:`scrapy.core.downloader.handlers.datauri.DataURIDownloadHandler`
|
||||
|
||||
* :class:`scrapy.core.downloader.handlers.file.FileDownloadHandler`
|
||||
|
||||
Use the ``from_settings`` or ``from_crawler`` class methods to expose such
|
||||
a parameter to your custom download handlers.
|
||||
|
||||
(:issue:`4126`)
|
||||
|
||||
* We have refactored the :class:`scrapy.core.scheduler.Scheduler` class and
|
||||
related queue classes (see :setting:`SCHEDULER_PRIORITY_QUEUE`,
|
||||
:setting:`SCHEDULER_DISK_QUEUE` and :setting:`SCHEDULER_MEMORY_QUEUE`) to
|
||||
make it easier to implement custom scheduler queue classes. See
|
||||
:ref:`2-0-0-scheduler-queue-changes` below for details.
|
||||
|
||||
* Overridden settings are now logged in a different format. This is more in
|
||||
line with similar information logged at startup (:issue:`4199`)
|
||||
|
||||
.. _Python 2 end-of-life on January 1, 2020: https://www.python.org/doc/sunset-python-2/
|
||||
|
||||
|
||||
Deprecation removals
|
||||
~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
* The :ref:`Scrapy shell <topics-shell>` no longer provides a `sel` proxy
|
||||
object, use :meth:`response.selector <scrapy.http.Response.selector>`
|
||||
instead (:issue:`4347`)
|
||||
|
||||
* LevelDB support has been removed (:issue:`4112`)
|
||||
|
||||
* The following functions have been removed from :mod:`scrapy.utils.python`:
|
||||
``isbinarytext``, ``is_writable``, ``setattr_default``, ``stringify_dict``
|
||||
(:issue:`4362`)
|
||||
|
||||
|
||||
Deprecations
|
||||
~~~~~~~~~~~~
|
||||
|
||||
* Using environment variables prefixed with ``SCRAPY_`` to override settings
|
||||
is deprecated (:issue:`4300`, :issue:`4374`, :issue:`4375`)
|
||||
|
||||
* :class:`scrapy.linkextractors.FilteringLinkExtractor` is deprecated, use
|
||||
:class:`scrapy.linkextractors.LinkExtractor
|
||||
<scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor>` instead (:issue:`4045`)
|
||||
|
||||
* The ``noconnect`` query string argument of proxy URLs is deprecated and
|
||||
should be removed from proxy URLs (:issue:`4198`)
|
||||
|
||||
* The :meth:`next <scrapy.utils.python.MutableChain.next>` method of
|
||||
:class:`scrapy.utils.python.MutableChain` is deprecated, use the global
|
||||
:func:`next` function or :meth:`MutableChain.__next__
|
||||
<scrapy.utils.python.MutableChain.__next__>` instead (:issue:`4153`)
|
||||
|
||||
|
||||
New features
|
||||
~~~~~~~~~~~~
|
||||
|
||||
* Added :doc:`partial support <topics/coroutines>` for Python’s
|
||||
:ref:`coroutine syntax <async>` and :doc:`experimental support
|
||||
<topics/asyncio>` for :mod:`asyncio` and :mod:`asyncio`-powered libraries
|
||||
(:issue:`4010`, :issue:`4259`, :issue:`4269`, :issue:`4270`, :issue:`4271`,
|
||||
:issue:`4316`, :issue:`4318`)
|
||||
|
||||
* The new :meth:`Response.follow_all <scrapy.http.Response.follow_all>`
|
||||
method offers the same functionality as
|
||||
:meth:`Response.follow <scrapy.http.Response.follow>` but supports an
|
||||
iterable of URLs as input and returns an iterable of requests
|
||||
(:issue:`2582`, :issue:`4057`, :issue:`4286`)
|
||||
|
||||
* :ref:`Media pipelines <topics-media-pipeline>` now support :ref:`FTP
|
||||
storage <media-pipeline-ftp>` (:issue:`3928`, :issue:`3961`)
|
||||
|
||||
* The new :attr:`Response.certificate <scrapy.http.Response.certificate>`
|
||||
attribute exposes the SSL certificate of the server as a
|
||||
:class:`twisted.internet.ssl.Certificate` object for HTTPS responses
|
||||
(:issue:`2726`, :issue:`4054`)
|
||||
|
||||
* A new :setting:`DNS_RESOLVER` setting allows enabling IPv6 support
|
||||
(:issue:`1031`, :issue:`4227`)
|
||||
|
||||
* A new :setting:`SCRAPER_SLOT_MAX_ACTIVE_SIZE` setting allows configuring
|
||||
the existing soft limit that pauses request downloads when the total
|
||||
response data being processed is too high (:issue:`1410`, :issue:`3551`)
|
||||
|
||||
* A new :setting:`TWISTED_REACTOR` setting allows customizing the
|
||||
:mod:`~twisted.internet.reactor` that Scrapy uses, allowing to
|
||||
:doc:`enable asyncio support <topics/asyncio>` or deal with a
|
||||
:ref:`common macOS issue <faq-specific-reactor>` (:issue:`2905`,
|
||||
:issue:`4294`)
|
||||
|
||||
* Scheduler disk and memory queues may now use the class methods
|
||||
``from_crawler`` or ``from_settings`` (:issue:`3884`)
|
||||
|
||||
* The new :attr:`Response.cb_kwargs <scrapy.http.Response.cb_kwargs>`
|
||||
attribute serves as a shortcut for :attr:`Response.request.cb_kwargs
|
||||
<scrapy.http.Request.cb_kwargs>` (:issue:`4331`)
|
||||
|
||||
* :meth:`Response.follow <scrapy.http.Response.follow>` now supports a
|
||||
``flags`` parameter, for consistency with :class:`~scrapy.http.Request`
|
||||
(:issue:`4277`, :issue:`4279`)
|
||||
|
||||
* :ref:`Item loader processors <topics-loaders-processors>` can now be
|
||||
regular functions, they no longer need to be methods (:issue:`3899`)
|
||||
|
||||
* :class:`~scrapy.spiders.Rule` now accepts an ``errback`` parameter
|
||||
(:issue:`4000`)
|
||||
|
||||
* :class:`~scrapy.http.Request` no longer requires a ``callback`` parameter
|
||||
when an ``errback`` parameter is specified (:issue:`3586`, :issue:`4008`)
|
||||
|
||||
* :class:`~scrapy.logformatter.LogFormatter` now supports some additional
|
||||
methods:
|
||||
|
||||
* :class:`~scrapy.logformatter.LogFormatter.download_error` for
|
||||
download errors
|
||||
|
||||
* :class:`~scrapy.logformatter.LogFormatter.item_error` for exceptions
|
||||
raised during item processing by :ref:`item pipelines
|
||||
<topics-item-pipeline>`
|
||||
|
||||
* :class:`~scrapy.logformatter.LogFormatter.spider_error` for exceptions
|
||||
raised from :ref:`spider callbacks <topics-spiders>`
|
||||
|
||||
(:issue:`374`, :issue:`3986`, :issue:`3989`, :issue:`4176`, :issue:`4188`)
|
||||
|
||||
* The :setting:`FEED_URI` setting now supports :class:`pathlib.Path` values
|
||||
(:issue:`3731`, :issue:`4074`)
|
||||
|
||||
* A new :signal:`request_left_downloader` signal is sent when a request
|
||||
leaves the downloader (:issue:`4303`)
|
||||
|
||||
* Scrapy logs a warning when it detects a request callback or errback that
|
||||
uses ``yield`` but also returns a value, since the returned value would be
|
||||
lost (:issue:`3484`, :issue:`3869`)
|
||||
|
||||
* :class:`~scrapy.spiders.Spider` objects now raise an :exc:`AttributeError`
|
||||
exception if they do not have a :class:`~scrapy.spiders.Spider.start_urls`
|
||||
attribute nor reimplement :class:`~scrapy.spiders.Spider.start_requests`,
|
||||
but have a ``start_url`` attribute (:issue:`4133`, :issue:`4170`)
|
||||
|
||||
* :class:`~scrapy.exporters.BaseItemExporter` subclasses may now use
|
||||
``super().__init__(**kwargs)`` instead of ``self._configure(kwargs)`` in
|
||||
their ``__init__`` method, passing ``dont_fail=True`` to the parent
|
||||
``__init__`` method if needed, and accessing ``kwargs`` at ``self._kwargs``
|
||||
after calling their parent ``__init__`` method (:issue:`4193`,
|
||||
:issue:`4370`)
|
||||
|
||||
* A new ``keep_fragments`` parameter of
|
||||
:func:`scrapy.utils.request.request_fingerprint` allows to generate
|
||||
different fingerprints for requests with different fragments in their URL
|
||||
(:issue:`4104`)
|
||||
|
||||
* Download handlers (see :setting:`DOWNLOAD_HANDLERS`) may now use the
|
||||
``from_settings`` and ``from_crawler`` class methods that other Scrapy
|
||||
components already supported (:issue:`4126`)
|
||||
|
||||
* :class:`scrapy.utils.python.MutableChain.__iter__` now returns ``self``,
|
||||
`allowing it to be used as a sequence <https://lgtm.com/rules/4850080/>`_
|
||||
(:issue:`4153`)
|
||||
|
||||
|
||||
Bug fixes
|
||||
~~~~~~~~~
|
||||
|
||||
* The :command:`crawl` command now also exits with exit code 1 when an
|
||||
exception happens before the crawling starts (:issue:`4175`, :issue:`4207`)
|
||||
|
||||
* :class:`LinkExtractor.extract_links
|
||||
<scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor.extract_links>` no longer
|
||||
re-encodes the query string or URLs from non-UTF-8 responses in UTF-8
|
||||
(:issue:`998`, :issue:`1403`, :issue:`1949`, :issue:`4321`)
|
||||
|
||||
* The first spider middleware (see :setting:`SPIDER_MIDDLEWARES`) now also
|
||||
processes exceptions raised from callbacks that are generators
|
||||
(:issue:`4260`, :issue:`4272`)
|
||||
|
||||
* Redirects to URLs starting with 3 slashes (``///``) are now supported
|
||||
(:issue:`4032`, :issue:`4042`)
|
||||
|
||||
* :class:`~scrapy.http.Request` no longer accepts strings as ``url`` simply
|
||||
because they have a colon (:issue:`2552`, :issue:`4094`)
|
||||
|
||||
* The correct encoding is now used for attach names in
|
||||
:class:`~scrapy.mail.MailSender` (:issue:`4229`, :issue:`4239`)
|
||||
|
||||
* :class:`~scrapy.dupefilters.RFPDupeFilter`, the default
|
||||
:setting:`DUPEFILTER_CLASS`, no longer writes an extra ``\r`` character on
|
||||
each line in Windows, which made the size of the ``requests.seen`` file
|
||||
unnecessarily large on that platform (:issue:`4283`)
|
||||
|
||||
* Z shell auto-completion now looks for ``.html`` files, not ``.http`` files,
|
||||
and covers the ``-h`` command-line switch (:issue:`4122`, :issue:`4291`)
|
||||
|
||||
* Adding items to a :class:`scrapy.utils.datatypes.LocalCache` object
|
||||
without a ``limit`` defined no longer raises a :exc:`TypeError` exception
|
||||
(:issue:`4123`)
|
||||
|
||||
* Fixed a typo in the message of the :exc:`ValueError` exception raised when
|
||||
:func:`scrapy.utils.misc.create_instance` gets both ``settings`` and
|
||||
``crawler`` set to ``None`` (:issue:`4128`)
|
||||
|
||||
|
||||
Documentation
|
||||
~~~~~~~~~~~~~
|
||||
|
||||
* API documentation now links to an online, syntax-highlighted view of the
|
||||
corresponding source code (:issue:`4148`)
|
||||
|
||||
* Links to unexisting documentation pages now allow access to the sidebar
|
||||
(:issue:`4152`, :issue:`4169`)
|
||||
|
||||
* Cross-references within our documentation now display a tooltip when
|
||||
hovered (:issue:`4173`, :issue:`4183`)
|
||||
|
||||
* Improved the documentation about :meth:`LinkExtractor.extract_links
|
||||
<scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor.extract_links>` and
|
||||
simplified :ref:`topics-link-extractors` (:issue:`4045`)
|
||||
|
||||
* Clarified how :class:`ItemLoader.item <scrapy.loader.ItemLoader.item>`
|
||||
works (:issue:`3574`, :issue:`4099`)
|
||||
|
||||
* Clarified that :func:`logging.basicConfig` should not be used when also
|
||||
using :class:`~scrapy.crawler.CrawlerProcess` (:issue:`2149`,
|
||||
:issue:`2352`, :issue:`3146`, :issue:`3960`)
|
||||
|
||||
* Clarified the requirements for :class:`~scrapy.http.Request` objects
|
||||
:ref:`when using persistence <request-serialization>` (:issue:`4124`,
|
||||
:issue:`4139`)
|
||||
|
||||
* Clarified how to install a :ref:`custom image pipeline
|
||||
<media-pipeline-example>` (:issue:`4034`, :issue:`4252`)
|
||||
|
||||
* Fixed the signatures of the ``file_path`` method in :ref:`media pipeline
|
||||
<topics-media-pipeline>` examples (:issue:`4290`)
|
||||
|
||||
* Covered a backward-incompatible change in Scrapy 1.7.0 affecting custom
|
||||
:class:`scrapy.core.scheduler.Scheduler` subclasses (:issue:`4274`)
|
||||
|
||||
* Improved the ``README.rst`` and ``CODE_OF_CONDUCT.md`` files
|
||||
(:issue:`4059`)
|
||||
|
||||
* Documentation examples are now checked as part of our test suite and we
|
||||
have fixed some of the issues detected (:issue:`4142`, :issue:`4146`,
|
||||
:issue:`4171`, :issue:`4184`, :issue:`4190`)
|
||||
|
||||
* Fixed logic issues, broken links and typos (:issue:`4247`, :issue:`4258`,
|
||||
:issue:`4282`, :issue:`4288`, :issue:`4305`, :issue:`4308`, :issue:`4323`,
|
||||
:issue:`4338`, :issue:`4359`, :issue:`4361`)
|
||||
|
||||
* Improved consistency when referring to the ``__init__`` method of an object
|
||||
(:issue:`4086`, :issue:`4088`)
|
||||
|
||||
* Fixed an inconsistency between code and output in :ref:`intro-overview`
|
||||
(:issue:`4213`)
|
||||
|
||||
* Extended :mod:`~sphinx.ext.intersphinx` usage (:issue:`4147`,
|
||||
:issue:`4172`, :issue:`4185`, :issue:`4194`, :issue:`4197`)
|
||||
|
||||
* We now use a recent version of Python to build the documentation
|
||||
(:issue:`4140`, :issue:`4249`)
|
||||
|
||||
* Cleaned up documentation (:issue:`4143`, :issue:`4275`)
|
||||
|
||||
|
||||
Quality assurance
|
||||
~~~~~~~~~~~~~~~~~
|
||||
|
||||
* Re-enabled proxy ``CONNECT`` tests (:issue:`2545`, :issue:`4114`)
|
||||
|
||||
* Added Bandit_ security checks to our test suite (:issue:`4162`,
|
||||
:issue:`4181`)
|
||||
|
||||
* Added Flake8_ style checks to our test suite and applied many of the
|
||||
corresponding changes (:issue:`3944`, :issue:`3945`, :issue:`4137`,
|
||||
:issue:`4157`, :issue:`4167`, :issue:`4174`, :issue:`4186`, :issue:`4195`,
|
||||
:issue:`4238`, :issue:`4246`, :issue:`4355`, :issue:`4360`, :issue:`4365`)
|
||||
|
||||
* Improved test coverage (:issue:`4097`, :issue:`4218`, :issue:`4236`)
|
||||
|
||||
* Started reporting slowest tests, and improved the performance of some of
|
||||
them (:issue:`4163`, :issue:`4164`)
|
||||
|
||||
* Fixed broken tests and refactored some tests (:issue:`4014`, :issue:`4095`,
|
||||
:issue:`4244`, :issue:`4268`, :issue:`4372`)
|
||||
|
||||
* Modified the :doc:`tox <tox:index>` configuration to allow running tests
|
||||
with any Python version, run Bandit_ and Flake8_ tests by default, and
|
||||
enforce a minimum tox version programmatically (:issue:`4179`)
|
||||
|
||||
* Cleaned up code (:issue:`3937`, :issue:`4208`, :issue:`4209`,
|
||||
:issue:`4210`, :issue:`4212`, :issue:`4369`, :issue:`4376`, :issue:`4378`)
|
||||
|
||||
.. _Bandit: https://bandit.readthedocs.io/
|
||||
.. _Flake8: https://flake8.pycqa.org/en/latest/
|
||||
|
||||
|
||||
.. _2-0-0-scheduler-queue-changes:
|
||||
|
||||
Changes to scheduler queue classes
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
The following changes may impact any custom queue classes of all types:
|
||||
|
||||
* The ``push`` method no longer receives a second positional parameter
|
||||
containing ``request.priority * -1``. If you need that value, get it
|
||||
from the first positional parameter, ``request``, instead, or use
|
||||
the new :meth:`~scrapy.core.scheduler.ScrapyPriorityQueue.priority`
|
||||
method in :class:`scrapy.core.scheduler.ScrapyPriorityQueue`
|
||||
subclasses.
|
||||
|
||||
The following changes may impact custom priority queue classes:
|
||||
|
||||
* In the ``__init__`` method or the ``from_crawler`` or ``from_settings``
|
||||
class methods:
|
||||
|
||||
* The parameter that used to contain a factory function,
|
||||
``qfactory``, is now passed as a keyword parameter named
|
||||
``downstream_queue_cls``.
|
||||
|
||||
* A new keyword parameter has been added: ``key``. It is a string
|
||||
that is always an empty string for memory queues and indicates the
|
||||
:setting:`JOB_DIR` value for disk queues.
|
||||
|
||||
* The parameter for disk queues that contains data from the previous
|
||||
crawl, ``startprios`` or ``slot_startprios``, is now passed as a
|
||||
keyword parameter named ``startprios``.
|
||||
|
||||
* The ``serialize`` parameter is no longer passed. The disk queue
|
||||
class must take care of request serialization on its own before
|
||||
writing to disk, using the
|
||||
:func:`~scrapy.utils.reqser.request_to_dict` and
|
||||
:func:`~scrapy.utils.reqser.request_from_dict` functions from the
|
||||
:mod:`scrapy.utils.reqser` module.
|
||||
|
||||
The following changes may impact custom disk and memory queue classes:
|
||||
|
||||
* The signature of the ``__init__`` method is now
|
||||
``__init__(self, crawler, key)``.
|
||||
|
||||
The following changes affect specifically the
|
||||
:class:`~scrapy.core.scheduler.ScrapyPriorityQueue` and
|
||||
:class:`~scrapy.core.scheduler.DownloaderAwarePriorityQueue` classes from
|
||||
:mod:`scrapy.core.scheduler` and may affect subclasses:
|
||||
|
||||
* In the ``__init__`` method, most of the changes described above apply.
|
||||
|
||||
``__init__`` may still receive all parameters as positional parameters,
|
||||
however:
|
||||
|
||||
* ``downstream_queue_cls``, which replaced ``qfactory``, must be
|
||||
instantiated differently.
|
||||
|
||||
``qfactory`` was instantiated with a priority value (integer).
|
||||
|
||||
Instances of ``downstream_queue_cls`` should be created using
|
||||
the new
|
||||
:meth:`ScrapyPriorityQueue.qfactory <scrapy.core.scheduler.ScrapyPriorityQueue.qfactory>`
|
||||
or
|
||||
:meth:`DownloaderAwarePriorityQueue.pqfactory <scrapy.core.scheduler.DownloaderAwarePriorityQueue.pqfactory>`
|
||||
methods.
|
||||
|
||||
* The new ``key`` parameter displaced the ``startprios``
|
||||
parameter 1 position to the right.
|
||||
|
||||
* The following class attributes have been added:
|
||||
|
||||
* :attr:`~scrapy.core.scheduler.ScrapyPriorityQueue.crawler`
|
||||
|
||||
* :attr:`~scrapy.core.scheduler.ScrapyPriorityQueue.downstream_queue_cls`
|
||||
(details above)
|
||||
|
||||
* :attr:`~scrapy.core.scheduler.ScrapyPriorityQueue.key` (details above)
|
||||
|
||||
* The ``serialize`` attribute has been removed (details above)
|
||||
|
||||
The following changes affect specifically the
|
||||
:class:`~scrapy.core.scheduler.ScrapyPriorityQueue` class and may affect
|
||||
subclasses:
|
||||
|
||||
* A new :meth:`~scrapy.core.scheduler.ScrapyPriorityQueue.priority`
|
||||
method has been added which, given a request, returns
|
||||
``request.priority * -1``.
|
||||
|
||||
It is used in :meth:`~scrapy.core.scheduler.ScrapyPriorityQueue.push`
|
||||
to make up for the removal of its ``priority`` parameter.
|
||||
|
||||
* The ``spider`` attribute has been removed. Use
|
||||
:attr:`crawler.spider <scrapy.core.scheduler.ScrapyPriorityQueue.crawler>`
|
||||
instead.
|
||||
|
||||
The following changes affect specifically the
|
||||
:class:`~scrapy.core.scheduler.DownloaderAwarePriorityQueue` class and may
|
||||
affect subclasses:
|
||||
|
||||
* A new :attr:`~scrapy.core.scheduler.DownloaderAwarePriorityQueue.pqueues`
|
||||
attribute offers a mapping of downloader slot names to the
|
||||
corresponding instances of
|
||||
:attr:`~scrapy.core.scheduler.DownloaderAwarePriorityQueue.downstream_queue_cls`.
|
||||
|
||||
(:issue:`3884`)
|
||||
|
||||
|
||||
.. _release-1.8.0:
|
||||
|
||||
|
|
@ -26,7 +486,7 @@ Backward-incompatible changes
|
|||
* Python 3.4 is no longer supported, and some of the minimum requirements of
|
||||
Scrapy have also changed:
|
||||
|
||||
* cssselect_ 0.9.1
|
||||
* :doc:`cssselect <cssselect:index>` 0.9.1
|
||||
* cryptography_ 2.0
|
||||
* lxml_ 3.5.0
|
||||
* pyOpenSSL_ 16.2.0
|
||||
|
|
@ -288,12 +748,12 @@ Backward-incompatible changes
|
|||
:class:`~scrapy.http.Request` objects instead of arbitrary Python data
|
||||
structures.
|
||||
|
||||
* An additional ``crawler`` parameter has been added to the ``__init__`` method
|
||||
of the :class:`scrapy.core.scheduler.Scheduler` class.
|
||||
Custom scheduler subclasses which don't accept arbitrary parameters in
|
||||
their ``__init__`` method might break because of this change.
|
||||
* An additional ``crawler`` parameter has been added to the ``__init__``
|
||||
method of the :class:`~scrapy.core.scheduler.Scheduler` class. Custom
|
||||
scheduler subclasses which don't accept arbitrary parameters in their
|
||||
``__init__`` method might break because of this change.
|
||||
|
||||
For more information, refer to the documentation for the :setting:`SCHEDULER` setting.
|
||||
For more information, see :setting:`SCHEDULER`.
|
||||
|
||||
See also :ref:`1.7-deprecation-removals` below.
|
||||
|
||||
|
|
@ -1076,7 +1536,7 @@ Cleanups & Refactoring
|
|||
~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
- Tests: remove temp files and folders (:issue:`2570`),
|
||||
fixed ProjectUtilsTest on OS X (:issue:`2569`),
|
||||
fixed ProjectUtilsTest on macOS (:issue:`2569`),
|
||||
use portable pypy for Linux on Travis CI (:issue:`2710`)
|
||||
- Separate building request from ``_requests_to_follow`` in CrawlSpider (:issue:`2562`)
|
||||
- Remove “Python 3 progress” badge (:issue:`2567`)
|
||||
|
|
@ -1616,7 +2076,7 @@ Deprecations and Removals
|
|||
+ ``scrapy.utils.datatypes.SiteNode``
|
||||
|
||||
- The previously bundled ``scrapy.xlib.pydispatch`` library was deprecated and
|
||||
replaced by `pydispatcher <https://pypi.python.org/pypi/PyDispatcher>`_.
|
||||
replaced by `pydispatcher <https://pypi.org/project/PyDispatcher/>`_.
|
||||
|
||||
|
||||
Relocations
|
||||
|
|
@ -1645,7 +2105,7 @@ Bugfixes
|
|||
- Makes ``_monkeypatches`` more robust (:issue:`1634`).
|
||||
- Fixed bug on ``XMLItemExporter`` with non-string fields in
|
||||
items (:issue:`1738`).
|
||||
- Fixed startproject command in OS X (:issue:`1635`).
|
||||
- Fixed startproject command in macOS (:issue:`1635`).
|
||||
- Fixed :class:`~scrapy.exporters.PythonItemExporter` and CSVExporter for
|
||||
non-string item types (:issue:`1737`).
|
||||
- Various logging related fixes (:issue:`1294`, :issue:`1419`, :issue:`1263`,
|
||||
|
|
@ -1713,12 +2173,12 @@ Scrapy 1.0.4 (2015-12-30)
|
|||
- Typos corrections (:commit:`7067117`)
|
||||
- fix typos in downloader-middleware.rst and exceptions.rst, middlware -> middleware (:commit:`32f115c`)
|
||||
- Add note to Ubuntu install section about Debian compatibility (:commit:`23fda69`)
|
||||
- Replace alternative OSX install workaround with virtualenv (:commit:`98b63ee`)
|
||||
- Replace alternative macOS install workaround with virtualenv (:commit:`98b63ee`)
|
||||
- Reference Homebrew's homepage for installation instructions (:commit:`1925db1`)
|
||||
- Add oldest supported tox version to contributing docs (:commit:`5d10d6d`)
|
||||
- Note in install docs about pip being already included in python>=2.7.9 (:commit:`85c980e`)
|
||||
- Add non-python dependencies to Ubuntu install section in the docs (:commit:`fbd010d`)
|
||||
- Add OS X installation section to docs (:commit:`d8f4cba`)
|
||||
- Add macOS installation section to docs (:commit:`d8f4cba`)
|
||||
- DOC(ENH): specify path to rtd theme explicitly (:commit:`de73b1a`)
|
||||
- minor: scrapy.Spider docs grammar (:commit:`1ddcc7b`)
|
||||
- Make common practices sample code match the comments (:commit:`1b85bcf`)
|
||||
|
|
@ -2450,7 +2910,7 @@ Other
|
|||
~~~~~
|
||||
|
||||
- Dropped Python 2.6 support (:issue:`448`)
|
||||
- Add `cssselect`_ python package as install dependency
|
||||
- Add :doc:`cssselect <cssselect:index>` python package as install dependency
|
||||
- Drop libxml2 and multi selector's backend support, `lxml`_ is required from now on.
|
||||
- Minimum Twisted version increased to 10.0.0, dropped Twisted 8.0 support.
|
||||
- Running test suite now requires ``mock`` python library (:issue:`390`)
|
||||
|
|
@ -2571,7 +3031,7 @@ Scrapy 0.18.0 (released 2013-08-09)
|
|||
- MetaRefreshMiddldeware and RedirectMiddleware have different priorities to address #62
|
||||
- added from_crawler method to spiders
|
||||
- added system tests with mock server
|
||||
- more improvements to Mac OS compatibility (thanks Alex Cepoi)
|
||||
- more improvements to macOS compatibility (thanks Alex Cepoi)
|
||||
- several more cleanups to singletons and multi-spider support (thanks Nicolas Ramirez)
|
||||
- support custom download slots
|
||||
- added --spider option to "shell" command.
|
||||
|
|
@ -2647,7 +3107,7 @@ Scrapy 0.16.3 (released 2012-12-07)
|
|||
|
||||
- Remove concurrency limitation when using download delays and still ensure inter-request delays are enforced (:commit:`487b9b5`)
|
||||
- add error details when image pipeline fails (:commit:`8232569`)
|
||||
- improve mac os compatibility (:commit:`8dcf8aa`)
|
||||
- improve macOS compatibility (:commit:`8dcf8aa`)
|
||||
- setup.py: use README.rst to populate long_description (:commit:`7b5310d`)
|
||||
- doc: removed obsolete references to ClientForm (:commit:`80f9bb6`)
|
||||
- correct docs for default storage backend (:commit:`2aa491b`)
|
||||
|
|
@ -3047,17 +3507,16 @@ Scrapy 0.7
|
|||
First release of Scrapy.
|
||||
|
||||
|
||||
.. _AJAX crawleable urls: https://developers.google.com/webmasters/ajax-crawling/docs/getting-started?csw=1
|
||||
.. _AJAX crawleable urls: https://developers.google.com/search/docs/ajax-crawling/docs/getting-started?csw=1
|
||||
.. _botocore: https://github.com/boto/botocore
|
||||
.. _chunked transfer encoding: https://en.wikipedia.org/wiki/Chunked_transfer_encoding
|
||||
.. _ClientForm: http://wwwsearch.sourceforge.net/old/ClientForm/
|
||||
.. _Creating a pull request: https://help.github.com/en/articles/creating-a-pull-request
|
||||
.. _cryptography: https://cryptography.io/en/latest/
|
||||
.. _cssselect: https://github.com/scrapy/cssselect/
|
||||
.. _docstrings: https://docs.python.org/glossary.html#term-docstring
|
||||
.. _KeyboardInterrupt: https://docs.python.org/library/exceptions.html#KeyboardInterrupt
|
||||
.. _docstrings: https://docs.python.org/3/glossary.html#term-docstring
|
||||
.. _KeyboardInterrupt: https://docs.python.org/3/library/exceptions.html#KeyboardInterrupt
|
||||
.. _LevelDB: https://github.com/google/leveldb
|
||||
.. _lxml: http://lxml.de/
|
||||
.. _lxml: https://lxml.de/
|
||||
.. _marshal: https://docs.python.org/2/library/marshal.html
|
||||
.. _parsel.csstranslator.GenericTranslator: https://parsel.readthedocs.io/en/latest/parsel.html#parsel.csstranslator.GenericTranslator
|
||||
.. _parsel.csstranslator.HTMLTranslator: https://parsel.readthedocs.io/en/latest/parsel.html#parsel.csstranslator.HTMLTranslator
|
||||
|
|
@ -3068,11 +3527,11 @@ First release of Scrapy.
|
|||
.. _queuelib: https://github.com/scrapy/queuelib
|
||||
.. _registered with IANA: https://www.iana.org/assignments/media-types/media-types.xhtml
|
||||
.. _resource: https://docs.python.org/2/library/resource.html
|
||||
.. _robots.txt: http://www.robotstxt.org/
|
||||
.. _robots.txt: https://www.robotstxt.org/
|
||||
.. _scrapely: https://github.com/scrapy/scrapely
|
||||
.. _service_identity: https://service-identity.readthedocs.io/en/stable/
|
||||
.. _six: https://six.readthedocs.io/
|
||||
.. _tox: https://pypi.python.org/pypi/tox
|
||||
.. _tox: https://pypi.org/project/tox/
|
||||
.. _Twisted: https://twistedmatrix.com/trac/
|
||||
.. _Twisted - hello, asynchronous programming: http://jessenoller.com/blog/2009/02/11/twisted-hello-asynchronous-programming/
|
||||
.. _w3lib: https://github.com/scrapy/w3lib
|
||||
|
|
|
|||
|
|
@ -0,0 +1,28 @@
|
|||
=======
|
||||
asyncio
|
||||
=======
|
||||
|
||||
.. versionadded:: 2.0
|
||||
|
||||
Scrapy has partial support :mod:`asyncio`. After you :ref:`install the asyncio
|
||||
reactor <install-asyncio>`, you may use :mod:`asyncio` and
|
||||
:mod:`asyncio`-powered libraries in any :doc:`coroutine <coroutines>`.
|
||||
|
||||
.. warning:: :mod:`asyncio` support in Scrapy is experimental. Future Scrapy
|
||||
versions may introduce related changes without a deprecation
|
||||
period or warning.
|
||||
|
||||
.. _install-asyncio:
|
||||
|
||||
Installing the asyncio reactor
|
||||
==============================
|
||||
|
||||
To enable :mod:`asyncio` support, set the :setting:`TWISTED_REACTOR` setting to
|
||||
``'twisted.internet.asyncioreactor.AsyncioSelectorReactor'``.
|
||||
|
||||
If you are using :class:`~scrapy.crawler.CrawlerRunner`, you also need to
|
||||
install the :class:`~twisted.internet.asyncioreactor.AsyncioSelectorReactor`
|
||||
reactor manually. You can do that using
|
||||
:func:`~scrapy.utils.reactor.install_reactor`::
|
||||
|
||||
install_reactor('twisted.internet.asyncioreactor.AsyncioSelectorReactor')
|
||||
|
|
@ -188,7 +188,7 @@ AjaxCrawlMiddleware helps to crawl them correctly.
|
|||
It is turned OFF by default because it has some performance overhead,
|
||||
and enabling it for focused crawls doesn't make much sense.
|
||||
|
||||
.. _ajax crawlable: https://developers.google.com/webmasters/ajax-crawling/docs/getting-started
|
||||
.. _ajax crawlable: https://developers.google.com/search/docs/ajax-crawling/docs/getting-started
|
||||
|
||||
.. _broad-crawls-bfo:
|
||||
|
||||
|
|
|
|||
|
|
@ -0,0 +1,110 @@
|
|||
==========
|
||||
Coroutines
|
||||
==========
|
||||
|
||||
.. versionadded:: 2.0
|
||||
|
||||
Scrapy has :ref:`partial support <coroutine-support>` for the
|
||||
:ref:`coroutine syntax <async>`.
|
||||
|
||||
.. warning:: :mod:`asyncio` support in Scrapy is experimental. Future Scrapy
|
||||
versions may introduce related API and behavior changes without a
|
||||
deprecation period or warning.
|
||||
|
||||
.. _coroutine-support:
|
||||
|
||||
Supported callables
|
||||
===================
|
||||
|
||||
The following callables may be defined as coroutines using ``async def``, and
|
||||
hence use coroutine syntax (e.g. ``await``, ``async for``, ``async with``):
|
||||
|
||||
- :class:`~scrapy.http.Request` callbacks.
|
||||
|
||||
The following are known caveats of the current implementation that we aim
|
||||
to address in future versions of Scrapy:
|
||||
|
||||
- The callback output is not processed until the whole callback finishes.
|
||||
|
||||
As a side effect, if the callback raises an exception, none of its
|
||||
output is processed.
|
||||
|
||||
- Because `asynchronous generators were introduced in Python 3.6`_, you
|
||||
can only use ``yield`` if you are using Python 3.6 or later.
|
||||
|
||||
If you need to output multiple items or requests and you are using
|
||||
Python 3.5, return an iterable (e.g. a list) instead.
|
||||
|
||||
- The :meth:`process_item` method of
|
||||
:ref:`item pipelines <topics-item-pipeline>`.
|
||||
|
||||
- The
|
||||
:meth:`~scrapy.downloadermiddlewares.DownloaderMiddleware.process_request`,
|
||||
:meth:`~scrapy.downloadermiddlewares.DownloaderMiddleware.process_response`,
|
||||
and
|
||||
:meth:`~scrapy.downloadermiddlewares.DownloaderMiddleware.process_exception`
|
||||
methods of
|
||||
:ref:`downloader middlewares <topics-downloader-middleware-custom>`.
|
||||
|
||||
- :ref:`Signal handlers that support deferreds <signal-deferred>`.
|
||||
|
||||
.. _asynchronous generators were introduced in Python 3.6: https://www.python.org/dev/peps/pep-0525/
|
||||
|
||||
Usage
|
||||
=====
|
||||
|
||||
There are several use cases for coroutines in Scrapy. Code that would
|
||||
return Deferreds when written for previous Scrapy versions, such as downloader
|
||||
middlewares and signal handlers, can be rewritten to be shorter and cleaner::
|
||||
|
||||
class DbPipeline:
|
||||
def _update_item(self, data, item):
|
||||
item['field'] = data
|
||||
return item
|
||||
|
||||
def process_item(self, item, spider):
|
||||
dfd = db.get_some_data(item['id'])
|
||||
dfd.addCallback(self._update_item, item)
|
||||
return dfd
|
||||
|
||||
becomes::
|
||||
|
||||
class DbPipeline:
|
||||
async def process_item(self, item, spider):
|
||||
item['field'] = await db.get_some_data(item['id'])
|
||||
return item
|
||||
|
||||
Coroutines may be used to call asynchronous code. This includes other
|
||||
coroutines, functions that return Deferreds and functions that return
|
||||
`awaitable objects`_ such as :class:`~asyncio.Future`. This means you can use
|
||||
many useful Python libraries providing such code::
|
||||
|
||||
class MySpider(Spider):
|
||||
# ...
|
||||
async def parse_with_deferred(self, response):
|
||||
additional_response = await treq.get('https://additional.url')
|
||||
additional_data = await treq.content(additional_response)
|
||||
# ... use response and additional_data to yield items and requests
|
||||
|
||||
async def parse_with_asyncio(self, response):
|
||||
async with aiohttp.ClientSession() as session:
|
||||
async with session.get('https://additional.url') as additional_response:
|
||||
additional_data = await r.text()
|
||||
# ... use response and additional_data to yield items and requests
|
||||
|
||||
.. note:: Many libraries that use coroutines, such as `aio-libs`_, require the
|
||||
:mod:`asyncio` loop and to use them you need to
|
||||
:doc:`enable asyncio support in Scrapy<asyncio>`.
|
||||
|
||||
Common use cases for asynchronous code include:
|
||||
|
||||
* requesting data from websites, databases and other services (in callbacks,
|
||||
pipelines and middlewares);
|
||||
* storing data in databases (in pipelines and middlewares);
|
||||
* delaying the spider initialization until some external event (in the
|
||||
:signal:`spider_opened` handler);
|
||||
* calling asynchronous Scrapy methods like ``ExecutionEngine.download`` (see
|
||||
:ref:`the screenshot pipeline example<ScreenshotPipeline>`).
|
||||
|
||||
.. _aio-libs: https://github.com/aio-libs
|
||||
.. _awaitable objects: https://docs.python.org/3/glossary.html#term-awaitable
|
||||
|
|
@ -259,8 +259,8 @@ COOKIES_DEBUG
|
|||
|
||||
Default: ``False``
|
||||
|
||||
If enabled, Scrapy will log all cookies sent in requests (ie. ``Cookie``
|
||||
header) and all cookies received in responses (ie. ``Set-Cookie`` header).
|
||||
If enabled, Scrapy will log all cookies sent in requests (i.e. ``Cookie``
|
||||
header) and all cookies received in responses (i.e. ``Set-Cookie`` header).
|
||||
|
||||
Here's an example of a log with :setting:`COOKIES_DEBUG` enabled::
|
||||
|
||||
|
|
@ -709,7 +709,7 @@ HttpCompressionMiddleware
|
|||
provided `brotlipy`_ is installed.
|
||||
|
||||
.. _brotli-compressed: https://www.ietf.org/rfc/rfc7932.txt
|
||||
.. _brotlipy: https://pypi.python.org/pypi/brotlipy
|
||||
.. _brotlipy: https://pypi.org/project/brotlipy/
|
||||
|
||||
HttpCompressionMiddleware Settings
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
|
@ -872,6 +872,10 @@ Default: ``[]``
|
|||
|
||||
Meta tags within these tags are ignored.
|
||||
|
||||
.. versionchanged:: 2.0
|
||||
The default value of :setting:`METAREFRESH_IGNORE_TAGS` changed from
|
||||
``['script', 'noscript']`` to ``[]``.
|
||||
|
||||
.. setting:: METAREFRESH_MAXDELAY
|
||||
|
||||
METAREFRESH_MAXDELAY
|
||||
|
|
@ -1038,7 +1042,7 @@ Based on `RobotFileParser
|
|||
* is Python's built-in robots.txt_ parser
|
||||
|
||||
* is compliant with `Martijn Koster's 1996 draft specification
|
||||
<http://www.robotstxt.org/norobots-rfc.txt>`_
|
||||
<https://www.robotstxt.org/norobots-rfc.txt>`_
|
||||
|
||||
* lacks support for wildcard matching
|
||||
|
||||
|
|
@ -1061,7 +1065,7 @@ Based on `Reppy <https://github.com/seomoz/reppy/>`_:
|
|||
<https://github.com/seomoz/rep-cpp>`_
|
||||
|
||||
* is compliant with `Martijn Koster's 1996 draft specification
|
||||
<http://www.robotstxt.org/norobots-rfc.txt>`_
|
||||
<https://www.robotstxt.org/norobots-rfc.txt>`_
|
||||
|
||||
* supports wildcard matching
|
||||
|
||||
|
|
@ -1086,7 +1090,7 @@ Based on `Robotexclusionrulesparser <http://nikitathespider.com/python/rerp/>`_:
|
|||
* implemented in Python
|
||||
|
||||
* is compliant with `Martijn Koster's 1996 draft specification
|
||||
<http://www.robotstxt.org/norobots-rfc.txt>`_
|
||||
<https://www.robotstxt.org/norobots-rfc.txt>`_
|
||||
|
||||
* supports wildcard matching
|
||||
|
||||
|
|
@ -1115,7 +1119,7 @@ implementing the methods described below.
|
|||
.. autoclass:: RobotParser
|
||||
:members:
|
||||
|
||||
.. _robots.txt: http://www.robotstxt.org/
|
||||
.. _robots.txt: https://www.robotstxt.org/
|
||||
|
||||
DownloaderStats
|
||||
---------------
|
||||
|
|
@ -1155,7 +1159,7 @@ AjaxCrawlMiddleware
|
|||
|
||||
Middleware that finds 'AJAX crawlable' page variants based
|
||||
on meta-fragment html tag. See
|
||||
https://developers.google.com/webmasters/ajax-crawling/docs/getting-started
|
||||
https://developers.google.com/search/docs/ajax-crawling/docs/getting-started
|
||||
for more info.
|
||||
|
||||
.. note::
|
||||
|
|
|
|||
|
|
@ -241,12 +241,12 @@ along with `scrapy-selenium`_ for seamless integration.
|
|||
.. _headless browser: https://en.wikipedia.org/wiki/Headless_browser
|
||||
.. _JavaScript: https://en.wikipedia.org/wiki/JavaScript
|
||||
.. _js2xml: https://github.com/scrapinghub/js2xml
|
||||
.. _json.loads: https://docs.python.org/library/json.html#json.loads
|
||||
.. _json.loads: https://docs.python.org/3/library/json.html#json.loads
|
||||
.. _pytesseract: https://github.com/madmaze/pytesseract
|
||||
.. _regular expression: https://docs.python.org/library/re.html
|
||||
.. _regular expression: https://docs.python.org/3/library/re.html
|
||||
.. _scrapy-selenium: https://github.com/clemfromspace/scrapy-selenium
|
||||
.. _scrapy-splash: https://github.com/scrapy-plugins/scrapy-splash
|
||||
.. _Selenium: https://www.seleniumhq.org/
|
||||
.. _Selenium: https://www.selenium.dev/
|
||||
.. _Splash: https://github.com/scrapinghub/splash
|
||||
.. _tabula-py: https://github.com/chezou/tabula-py
|
||||
.. _wget: https://www.gnu.org/software/wget/
|
||||
|
|
|
|||
|
|
@ -42,7 +42,7 @@ value of one of their fields::
|
|||
|
||||
from scrapy.exporters import XmlItemExporter
|
||||
|
||||
class PerYearXmlExportPipeline(object):
|
||||
class PerYearXmlExportPipeline:
|
||||
"""Distribute items across multiple XML files according to their 'year' field"""
|
||||
|
||||
def open_spider(self, spider):
|
||||
|
|
@ -137,7 +137,7 @@ output examples, which assume you're exporting these two items::
|
|||
BaseItemExporter
|
||||
----------------
|
||||
|
||||
.. class:: BaseItemExporter(fields_to_export=None, export_empty_fields=False, encoding='utf-8', indent=0)
|
||||
.. class:: BaseItemExporter(fields_to_export=None, export_empty_fields=False, encoding='utf-8', indent=0, dont_fail=False)
|
||||
|
||||
This is the (abstract) base class for all Item Exporters. It provides
|
||||
support for common features used by all (concrete) Item Exporters, such as
|
||||
|
|
@ -148,6 +148,9 @@ BaseItemExporter
|
|||
populate their respective instance attributes: :attr:`fields_to_export`,
|
||||
:attr:`export_empty_fields`, :attr:`encoding`, :attr:`indent`.
|
||||
|
||||
.. versionadded:: 2.0
|
||||
The *dont_fail* parameter.
|
||||
|
||||
.. method:: export_item(item)
|
||||
|
||||
Exports the given item. This method must be implemented in subclasses.
|
||||
|
|
|
|||
|
|
@ -63,7 +63,7 @@ but disabled unless the :setting:`HTTPCACHE_ENABLED` setting is set.
|
|||
Disabling an extension
|
||||
======================
|
||||
|
||||
In order to disable an extension that comes enabled by default (ie. those
|
||||
In order to disable an extension that comes enabled by default (i.e. those
|
||||
included in the :setting:`EXTENSIONS_BASE` setting) you must set its order to
|
||||
``None``. For example::
|
||||
|
||||
|
|
@ -107,7 +107,7 @@ Here is the code of such extension::
|
|||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
class SpiderOpenCloseLogging(object):
|
||||
class SpiderOpenCloseLogging:
|
||||
|
||||
def __init__(self, item_count):
|
||||
self.item_count = item_count
|
||||
|
|
@ -345,7 +345,7 @@ signal is received. The information dumped is the following:
|
|||
After the stack trace and engine status is dumped, the Scrapy process continues
|
||||
running normally.
|
||||
|
||||
This extension only works on POSIX-compliant platforms (ie. not Windows),
|
||||
This extension only works on POSIX-compliant platforms (i.e. not Windows),
|
||||
because the `SIGQUIT`_ and `SIGUSR2`_ signals are not available on Windows.
|
||||
|
||||
There are at least two ways to send Scrapy the `SIGQUIT`_ signal:
|
||||
|
|
@ -370,7 +370,7 @@ running normally.
|
|||
|
||||
For more info see `Debugging in Python`_.
|
||||
|
||||
This extension only works on POSIX-compliant platforms (ie. not Windows).
|
||||
This extension only works on POSIX-compliant platforms (i.e. not Windows).
|
||||
|
||||
.. _Python debugger: https://docs.python.org/2/library/pdb.html
|
||||
.. _Debugging in Python: https://pythonconquerstheuniverse.wordpress.com/2009/09/10/debugging-in-python/
|
||||
|
|
|
|||
|
|
@ -12,7 +12,7 @@ generating an "export file" with the scraped data (commonly called "export
|
|||
feed") to be consumed by other systems.
|
||||
|
||||
Scrapy provides this functionality out of the box with the Feed Exports, which
|
||||
allows you to generate a feed with the scraped items, using multiple
|
||||
allows you to generate feeds with the scraped items, using multiple
|
||||
serialization formats and storage backends.
|
||||
|
||||
.. _topics-feed-format:
|
||||
|
|
@ -36,7 +36,7 @@ But you can also extend the supported format through the
|
|||
JSON
|
||||
----
|
||||
|
||||
* :setting:`FEED_FORMAT`: ``json``
|
||||
* Value for the ``format`` key in the :setting:`FEEDS` setting: ``json``
|
||||
* Exporter used: :class:`~scrapy.exporters.JsonItemExporter`
|
||||
* See :ref:`this warning <json-with-large-data>` if you're using JSON with
|
||||
large feeds.
|
||||
|
|
@ -46,7 +46,7 @@ JSON
|
|||
JSON lines
|
||||
----------
|
||||
|
||||
* :setting:`FEED_FORMAT`: ``jsonlines``
|
||||
* Value for the ``format`` key in the :setting:`FEEDS` setting: ``jsonlines``
|
||||
* Exporter used: :class:`~scrapy.exporters.JsonLinesItemExporter`
|
||||
|
||||
.. _topics-feed-format-csv:
|
||||
|
|
@ -54,7 +54,7 @@ JSON lines
|
|||
CSV
|
||||
---
|
||||
|
||||
* :setting:`FEED_FORMAT`: ``csv``
|
||||
* Value for the ``format`` key in the :setting:`FEEDS` setting: ``csv``
|
||||
* Exporter used: :class:`~scrapy.exporters.CsvItemExporter`
|
||||
* To specify columns to export and their order use
|
||||
:setting:`FEED_EXPORT_FIELDS`. Other feed exporters can also use this
|
||||
|
|
@ -66,7 +66,7 @@ CSV
|
|||
XML
|
||||
---
|
||||
|
||||
* :setting:`FEED_FORMAT`: ``xml``
|
||||
* Value for the ``format`` key in the :setting:`FEEDS` setting: ``xml``
|
||||
* Exporter used: :class:`~scrapy.exporters.XmlItemExporter`
|
||||
|
||||
.. _topics-feed-format-pickle:
|
||||
|
|
@ -74,7 +74,7 @@ XML
|
|||
Pickle
|
||||
------
|
||||
|
||||
* :setting:`FEED_FORMAT`: ``pickle``
|
||||
* Value for the ``format`` key in the :setting:`FEEDS` setting: ``pickle``
|
||||
* Exporter used: :class:`~scrapy.exporters.PickleItemExporter`
|
||||
|
||||
.. _topics-feed-format-marshal:
|
||||
|
|
@ -82,7 +82,7 @@ Pickle
|
|||
Marshal
|
||||
-------
|
||||
|
||||
* :setting:`FEED_FORMAT`: ``marshal``
|
||||
* Value for the ``format`` key in the :setting:`FEEDS` setting: ``marshal``
|
||||
* Exporter used: :class:`~scrapy.exporters.MarshalItemExporter`
|
||||
|
||||
|
||||
|
|
@ -91,8 +91,8 @@ Marshal
|
|||
Storages
|
||||
========
|
||||
|
||||
When using the feed exports you define where to store the feed using a URI_
|
||||
(through the :setting:`FEED_URI` setting). The feed exports supports multiple
|
||||
When using the feed exports you define where to store the feed using one or multiple URIs_
|
||||
(through the :setting:`FEEDS` setting). The feed exports supports multiple
|
||||
storage backend types which are defined by the URI scheme.
|
||||
|
||||
The storages backends supported out of the box are:
|
||||
|
|
@ -232,38 +232,66 @@ Settings
|
|||
|
||||
These are the settings used for configuring the feed exports:
|
||||
|
||||
* :setting:`FEED_URI` (mandatory)
|
||||
* :setting:`FEED_FORMAT`
|
||||
* :setting:`FEEDS` (mandatory)
|
||||
* :setting:`FEED_EXPORT_ENCODING`
|
||||
* :setting:`FEED_STORE_EMPTY`
|
||||
* :setting:`FEED_EXPORT_FIELDS`
|
||||
* :setting:`FEED_EXPORT_INDENT`
|
||||
* :setting:`FEED_STORAGES`
|
||||
* :setting:`FEED_STORAGE_FTP_ACTIVE`
|
||||
* :setting:`FEED_STORAGE_S3_ACL`
|
||||
* :setting:`FEED_EXPORTERS`
|
||||
* :setting:`FEED_STORE_EMPTY`
|
||||
* :setting:`FEED_EXPORT_ENCODING`
|
||||
* :setting:`FEED_EXPORT_FIELDS`
|
||||
* :setting:`FEED_EXPORT_INDENT`
|
||||
|
||||
.. currentmodule:: scrapy.extensions.feedexport
|
||||
|
||||
.. setting:: FEED_URI
|
||||
.. setting:: FEEDS
|
||||
|
||||
FEED_URI
|
||||
--------
|
||||
FEEDS
|
||||
-----
|
||||
|
||||
Default: ``None``
|
||||
.. versionadded:: 2.1
|
||||
|
||||
The URI of the export feed. See :ref:`topics-feed-storage-backends` for
|
||||
supported URI schemes.
|
||||
Default: ``{}``
|
||||
|
||||
This setting is required for enabling the feed exports.
|
||||
A dictionary in which every key is a feed URI (or a :class:`pathlib.Path`
|
||||
object) and each value is a nested dictionary containing configuration
|
||||
parameters for the specific feed.
|
||||
This setting is required for enabling the feed export feature.
|
||||
|
||||
.. setting:: FEED_FORMAT
|
||||
See :ref:`topics-feed-storage-backends` for supported URI schemes.
|
||||
|
||||
FEED_FORMAT
|
||||
-----------
|
||||
For instance::
|
||||
|
||||
The serialization format to be used for the feed. See
|
||||
:ref:`topics-feed-format` for possible values.
|
||||
{
|
||||
'items.json': {
|
||||
'format': 'json',
|
||||
'encoding': 'utf8',
|
||||
'store_empty': False,
|
||||
'fields': None,
|
||||
'indent': 4,
|
||||
},
|
||||
'/home/user/documents/items.xml': {
|
||||
'format': 'xml',
|
||||
'fields': ['name', 'price'],
|
||||
'encoding': 'latin1',
|
||||
'indent': 8,
|
||||
},
|
||||
pathlib.Path('items.csv'): {
|
||||
'format': 'csv',
|
||||
'fields': ['price', 'name'],
|
||||
},
|
||||
}
|
||||
|
||||
The following is a list of the accepted keys and the setting that is used
|
||||
as a fallback value if that key is not provided for a specific feed definition.
|
||||
|
||||
* ``format``: the serialization format to be used for the feed.
|
||||
See :ref:`topics-feed-format` for possible values.
|
||||
Mandatory, no fallback setting
|
||||
* ``encoding``: falls back to :setting:`FEED_EXPORT_ENCODING`
|
||||
* ``fields``: falls back to :setting:`FEED_EXPORT_FIELDS`
|
||||
* ``indent``: falls back to :setting:`FEED_EXPORT_INDENT`
|
||||
* ``store_empty``: falls back to :setting:`FEED_STORE_EMPTY`
|
||||
|
||||
.. setting:: FEED_EXPORT_ENCODING
|
||||
|
||||
|
|
@ -322,7 +350,7 @@ FEED_STORE_EMPTY
|
|||
|
||||
Default: ``False``
|
||||
|
||||
Whether to export empty feeds (ie. feeds with no items).
|
||||
Whether to export empty feeds (i.e. feeds with no items).
|
||||
|
||||
.. setting:: FEED_STORAGES
|
||||
|
||||
|
|
@ -418,7 +446,7 @@ format in :setting:`FEED_EXPORTERS`. E.g., to disable the built-in CSV exporter
|
|||
'csv': None,
|
||||
}
|
||||
|
||||
.. _URI: https://en.wikipedia.org/wiki/Uniform_Resource_Identifier
|
||||
.. _URIs: https://en.wikipedia.org/wiki/Uniform_Resource_Identifier
|
||||
.. _Amazon S3: https://aws.amazon.com/s3/
|
||||
.. _botocore: https://github.com/boto/botocore
|
||||
.. _Canned ACL: https://docs.aws.amazon.com/AmazonS3/latest/dev/acl-overview.html#canned-acl
|
||||
|
|
|
|||
|
|
@ -81,7 +81,7 @@ contain a price::
|
|||
|
||||
from scrapy.exceptions import DropItem
|
||||
|
||||
class PricePipeline(object):
|
||||
class PricePipeline:
|
||||
|
||||
vat_factor = 1.15
|
||||
|
||||
|
|
@ -103,7 +103,7 @@ format::
|
|||
|
||||
import json
|
||||
|
||||
class JsonWriterPipeline(object):
|
||||
class JsonWriterPipeline:
|
||||
|
||||
def open_spider(self, spider):
|
||||
self.file = open('items.jl', 'w')
|
||||
|
|
@ -132,7 +132,7 @@ method and how to clean up the resources properly.::
|
|||
|
||||
import pymongo
|
||||
|
||||
class MongoPipeline(object):
|
||||
class MongoPipeline:
|
||||
|
||||
collection_name = 'scrapy_items'
|
||||
|
||||
|
|
@ -158,18 +158,20 @@ method and how to clean up the resources properly.::
|
|||
self.db[self.collection_name].insert_one(dict(item))
|
||||
return item
|
||||
|
||||
.. _MongoDB: https://www.mongodb.org/
|
||||
.. _pymongo: https://api.mongodb.org/python/current/
|
||||
.. _MongoDB: https://www.mongodb.com/
|
||||
.. _pymongo: https://api.mongodb.com/python/current/
|
||||
|
||||
|
||||
.. _ScreenshotPipeline:
|
||||
|
||||
Take screenshot of item
|
||||
-----------------------
|
||||
|
||||
This example demonstrates how to return a
|
||||
:class:`~twisted.internet.defer.Deferred` from the :meth:`process_item` method.
|
||||
It uses Splash_ to render screenshot of item url. Pipeline
|
||||
makes request to locally running instance of Splash_. After request is downloaded
|
||||
and Deferred callback fires, it saves item to a file and adds filename to an item.
|
||||
makes request to locally running instance of Splash_. After request is downloaded,
|
||||
it saves the screenshot to a file and adds filename to the item.
|
||||
|
||||
::
|
||||
|
||||
|
|
@ -178,21 +180,18 @@ and Deferred callback fires, it saves item to a file and adds filename to an ite
|
|||
from urllib.parse import quote
|
||||
|
||||
|
||||
class ScreenshotPipeline(object):
|
||||
class ScreenshotPipeline:
|
||||
"""Pipeline that uses Splash to render screenshot of
|
||||
every Scrapy item."""
|
||||
|
||||
SPLASH_URL = "http://localhost:8050/render.png?url={}"
|
||||
|
||||
def process_item(self, item, spider):
|
||||
async def process_item(self, item, spider):
|
||||
encoded_item_url = quote(item["url"])
|
||||
screenshot_url = self.SPLASH_URL.format(encoded_item_url)
|
||||
request = scrapy.Request(screenshot_url)
|
||||
dfd = spider.crawler.engine.download(request, spider)
|
||||
dfd.addBoth(self.return_item, item)
|
||||
return dfd
|
||||
response = await spider.crawler.engine.download(request, spider)
|
||||
|
||||
def return_item(self, response, item):
|
||||
if response.status != 200:
|
||||
# Error happened, return item.
|
||||
return item
|
||||
|
|
@ -220,7 +219,7 @@ returns multiples items with the same id::
|
|||
|
||||
from scrapy.exceptions import DropItem
|
||||
|
||||
class DuplicatesPipeline(object):
|
||||
class DuplicatesPipeline:
|
||||
|
||||
def __init__(self):
|
||||
self.ids_seen = set()
|
||||
|
|
|
|||
|
|
@ -166,7 +166,7 @@ If your item contains mutable_ values like lists or dictionaries, a shallow
|
|||
copy will keep references to the same mutable values across all different
|
||||
copies.
|
||||
|
||||
.. _mutable: https://docs.python.org/glossary.html#term-mutable
|
||||
.. _mutable: https://docs.python.org/3/glossary.html#term-mutable
|
||||
|
||||
For example, if you have an item with a list of tags, and you create a shallow
|
||||
copy of that item, both the original item and the copy have the same list of
|
||||
|
|
@ -177,7 +177,7 @@ If that is not the desired behavior, use a deep copy instead.
|
|||
|
||||
See the `documentation of the copy module`_ for more information.
|
||||
|
||||
.. _documentation of the copy module: https://docs.python.org/library/copy.html
|
||||
.. _documentation of the copy module: https://docs.python.org/3/library/copy.html
|
||||
|
||||
To create a shallow copy of an item, you can either call
|
||||
:meth:`~scrapy.item.Item.copy` on an existing item
|
||||
|
|
|
|||
|
|
@ -22,7 +22,7 @@ Job directory
|
|||
|
||||
To enable persistence support you just need to define a *job directory* through
|
||||
the ``JOBDIR`` setting. This directory will be for storing all required data to
|
||||
keep the state of a single job (ie. a spider run). It's important to note that
|
||||
keep the state of a single job (i.e. a spider run). It's important to note that
|
||||
this directory must not be shared by different spiders, or even different
|
||||
jobs/runs of the same spider, as it's meant to be used for storing the state of
|
||||
a *single* job.
|
||||
|
|
@ -68,6 +68,9 @@ Cookies may expire. So, if you don't resume your spider quickly the requests
|
|||
scheduled may no longer work. This won't be an issue if you spider doesn't rely
|
||||
on cookies.
|
||||
|
||||
|
||||
.. _request-serialization:
|
||||
|
||||
Request serialization
|
||||
---------------------
|
||||
|
||||
|
|
|
|||
|
|
@ -17,8 +17,8 @@ what is known as a "memory leak".
|
|||
|
||||
To help debugging memory leaks, Scrapy provides a built-in mechanism for
|
||||
tracking objects references called :ref:`trackref <topics-leaks-trackrefs>`,
|
||||
and you can also use a third-party library called :ref:`Guppy
|
||||
<topics-leaks-guppy>` for more advanced memory debugging (see below for more
|
||||
and you can also use a third-party library called :ref:`muppy
|
||||
<topics-leaks-muppy>` for more advanced memory debugging (see below for more
|
||||
info). Both mechanisms must be used from the :ref:`Telnet Console
|
||||
<topics-telnetconsole>`.
|
||||
|
||||
|
|
@ -170,7 +170,7 @@ Here are the functions available in the :mod:`~scrapy.utils.trackref` module.
|
|||
|
||||
.. class:: object_ref
|
||||
|
||||
Inherit from this class (instead of object) if you want to track live
|
||||
Inherit from this class if you want to track live
|
||||
instances with the ``trackref`` module.
|
||||
|
||||
.. function:: print_live_refs(class_name, ignore=NoneType)
|
||||
|
|
@ -193,9 +193,9 @@ Here are the functions available in the :mod:`~scrapy.utils.trackref` module.
|
|||
``None`` if none is found. Use :func:`print_live_refs` first to get a list
|
||||
of all tracked live objects per class name.
|
||||
|
||||
.. _topics-leaks-guppy:
|
||||
.. _topics-leaks-muppy:
|
||||
|
||||
Debugging memory leaks with Guppy
|
||||
Debugging memory leaks with muppy
|
||||
=================================
|
||||
|
||||
``trackref`` provides a very convenient mechanism for tracking down memory
|
||||
|
|
@ -203,63 +203,9 @@ leaks, but it only keeps track of the objects that are more likely to cause
|
|||
memory leaks (Requests, Responses, Items, and Selectors). However, there are
|
||||
other cases where the memory leaks could come from other (more or less obscure)
|
||||
objects. If this is your case, and you can't find your leaks using ``trackref``,
|
||||
you still have another resource: the `Guppy library`_.
|
||||
If you're using Python3, see :ref:`topics-leaks-muppy`.
|
||||
you still have another resource: the muppy library.
|
||||
|
||||
.. _Guppy library: https://pypi.python.org/pypi/guppy
|
||||
|
||||
If you use ``pip``, you can install Guppy with the following command::
|
||||
|
||||
pip install guppy
|
||||
|
||||
The telnet console also comes with a built-in shortcut (``hpy``) for accessing
|
||||
Guppy heap objects. Here's an example to view all Python objects available in
|
||||
the heap using Guppy:
|
||||
|
||||
>>> x = hpy.heap()
|
||||
>>> x.bytype
|
||||
Partition of a set of 297033 objects. Total size = 52587824 bytes.
|
||||
Index Count % Size % Cumulative % Type
|
||||
0 22307 8 16423880 31 16423880 31 dict
|
||||
1 122285 41 12441544 24 28865424 55 str
|
||||
2 68346 23 5966696 11 34832120 66 tuple
|
||||
3 227 0 5836528 11 40668648 77 unicode
|
||||
4 2461 1 2222272 4 42890920 82 type
|
||||
5 16870 6 2024400 4 44915320 85 function
|
||||
6 13949 5 1673880 3 46589200 89 types.CodeType
|
||||
7 13422 5 1653104 3 48242304 92 list
|
||||
8 3735 1 1173680 2 49415984 94 _sre.SRE_Pattern
|
||||
9 1209 0 456936 1 49872920 95 scrapy.http.headers.Headers
|
||||
<1676 more rows. Type e.g. '_.more' to view.>
|
||||
|
||||
You can see that most space is used by dicts. Then, if you want to see from
|
||||
which attribute those dicts are referenced, you could do:
|
||||
|
||||
>>> x.bytype[0].byvia
|
||||
Partition of a set of 22307 objects. Total size = 16423880 bytes.
|
||||
Index Count % Size % Cumulative % Referred Via:
|
||||
0 10982 49 9416336 57 9416336 57 '.__dict__'
|
||||
1 1820 8 2681504 16 12097840 74 '.__dict__', '.func_globals'
|
||||
2 3097 14 1122904 7 13220744 80
|
||||
3 990 4 277200 2 13497944 82 "['cookies']"
|
||||
4 987 4 276360 2 13774304 84 "['cache']"
|
||||
5 985 4 275800 2 14050104 86 "['meta']"
|
||||
6 897 4 251160 2 14301264 87 '[2]'
|
||||
7 1 0 196888 1 14498152 88 "['moduleDict']", "['modules']"
|
||||
8 672 3 188160 1 14686312 89 "['cb_kwargs']"
|
||||
9 27 0 155016 1 14841328 90 '[1]'
|
||||
<333 more rows. Type e.g. '_.more' to view.>
|
||||
|
||||
As you can see, the Guppy module is very powerful but also requires some deep
|
||||
knowledge about Python internals. For more info about Guppy, refer to the
|
||||
`Guppy documentation`_.
|
||||
|
||||
.. _Guppy documentation: http://guppy-pe.sourceforge.net/
|
||||
|
||||
.. _topics-leaks-muppy:
|
||||
|
||||
Debugging memory leaks with muppy
|
||||
=================================
|
||||
You can use muppy from `Pympler`_.
|
||||
|
||||
.. _Pympler: https://pypi.org/project/Pympler/
|
||||
|
|
@ -311,9 +257,9 @@ though neither Scrapy nor your project are leaking memory. This is due to a
|
|||
(not so well) known problem of Python, which may not return released memory to
|
||||
the operating system in some cases. For more information on this issue see:
|
||||
|
||||
* `Python Memory Management <http://www.evanjones.ca/python-memory.html>`_
|
||||
* `Python Memory Management Part 2 <http://www.evanjones.ca/python-memory-part2.html>`_
|
||||
* `Python Memory Management Part 3 <http://www.evanjones.ca/python-memory-part3.html>`_
|
||||
* `Python Memory Management <https://www.evanjones.ca/python-memory.html>`_
|
||||
* `Python Memory Management Part 2 <https://www.evanjones.ca/python-memory-part2.html>`_
|
||||
* `Python Memory Management Part 3 <https://www.evanjones.ca/python-memory-part3.html>`_
|
||||
|
||||
The improvements proposed by Evan Jones, which are detailed in `this paper`_,
|
||||
got merged in Python 2.5, but this only reduces the problem, it doesn't fix it
|
||||
|
|
@ -327,7 +273,7 @@ completely. To quote the paper:
|
|||
to move to a compacting garbage collector, which is able to move objects in
|
||||
memory. This would require significant changes to the Python interpreter.*
|
||||
|
||||
.. _this paper: http://www.evanjones.ca/memoryallocator/
|
||||
.. _this paper: https://www.evanjones.ca/memoryallocator/
|
||||
|
||||
To keep memory consumption reasonable you can split the job into several
|
||||
smaller jobs or enable :ref:`persistent job queue <topics-jobs>`
|
||||
|
|
|
|||
|
|
@ -49,7 +49,7 @@ LxmlLinkExtractor
|
|||
:type allow: a regular expression (or list of)
|
||||
|
||||
:param deny: a single regular expression (or list of regular expressions)
|
||||
that the (absolute) urls must match in order to be excluded (ie. not
|
||||
that the (absolute) urls must match in order to be excluded (i.e. not
|
||||
extracted). It has precedence over the ``allow`` parameter. If not
|
||||
given (or empty) it won't exclude any links.
|
||||
:type deny: a regular expression (or list of)
|
||||
|
|
@ -64,9 +64,13 @@ LxmlLinkExtractor
|
|||
|
||||
:param deny_extensions: a single value or list of strings containing
|
||||
extensions that should be ignored when extracting links.
|
||||
If not given, it will default to the
|
||||
``IGNORED_EXTENSIONS`` list defined in the
|
||||
`scrapy.linkextractors`_ package.
|
||||
If not given, it will default to
|
||||
:data:`scrapy.linkextractors.IGNORED_EXTENSIONS`.
|
||||
|
||||
.. versionchanged:: 2.0
|
||||
:data:`~scrapy.linkextractors.IGNORED_EXTENSIONS` now includes
|
||||
``7z``, ``7zip``, ``apk``, ``bz2``, ``cdr``, ``dmg``, ``ico``,
|
||||
``iso``, ``tar``, ``tar.gz``, ``webm``, and ``xz``.
|
||||
:type deny_extensions: list
|
||||
|
||||
:param restrict_xpaths: is an XPath (or list of XPath's) which defines
|
||||
|
|
|
|||
|
|
@ -136,6 +136,9 @@ with the data to be parsed, and return a parsed value. So you can use any
|
|||
function as input or output processor. The only requirement is that they must
|
||||
accept one (and only one) positional argument, which will be an iterable.
|
||||
|
||||
.. versionchanged:: 2.0
|
||||
Processors no longer need to be methods.
|
||||
|
||||
.. note:: Both input and output processors must receive an iterable as their
|
||||
first argument. The output of those functions can be anything. The result of
|
||||
input processors will be appended to an internal list (in the Loader)
|
||||
|
|
|
|||
|
|
@ -116,12 +116,6 @@ For the Images Pipeline, set the :setting:`IMAGES_STORE` setting::
|
|||
Supported Storage
|
||||
=================
|
||||
|
||||
File system is currently the only officially supported storage, but there are
|
||||
also support for storing files in `Amazon S3`_ and `Google Cloud Storage`_.
|
||||
|
||||
.. _Amazon S3: https://aws.amazon.com/s3/
|
||||
.. _Google Cloud Storage: https://cloud.google.com/storage/
|
||||
|
||||
File system storage
|
||||
-------------------
|
||||
|
||||
|
|
@ -147,9 +141,13 @@ Where:
|
|||
* ``full`` is a sub-directory to separate full images from thumbnails (if
|
||||
used). For more info see :ref:`topics-images-thumbnails`.
|
||||
|
||||
.. _media-pipeline-ftp:
|
||||
|
||||
FTP server storage
|
||||
------------------
|
||||
|
||||
.. versionadded:: 2.0
|
||||
|
||||
:setting:`FILES_STORE` and :setting:`IMAGES_STORE` can point to an FTP server.
|
||||
Scrapy will automatically upload the files to the server.
|
||||
|
||||
|
|
@ -573,6 +571,8 @@ See here the methods that you can override in your custom Images Pipeline:
|
|||
By default, the :meth:`item_completed` method returns the item.
|
||||
|
||||
|
||||
.. _media-pipeline-example:
|
||||
|
||||
Custom Images pipeline example
|
||||
==============================
|
||||
|
||||
|
|
|
|||
|
|
@ -31,6 +31,8 @@ Request objects
|
|||
a :class:`Response`.
|
||||
|
||||
:param url: the URL of this request
|
||||
|
||||
If the URL is invalid, a :exc:`ValueError` exception is raised.
|
||||
:type url: string
|
||||
|
||||
:param callback: the function that will be called with the response of this
|
||||
|
|
@ -125,6 +127,10 @@ Request objects
|
|||
:exc:`~twisted.python.failure.Failure` as first parameter.
|
||||
For more information,
|
||||
see :ref:`topics-request-response-ref-errbacks` below.
|
||||
|
||||
.. versionchanged:: 2.0
|
||||
The *callback* parameter is no longer required when the *errback*
|
||||
parameter is specified.
|
||||
:type errback: callable
|
||||
|
||||
:param flags: Flags sent to the request, can be used for logging or similar purposes.
|
||||
|
|
@ -396,7 +402,7 @@ The FormRequest class extends the base :class:`Request` with functionality for
|
|||
dealing with HTML forms. It uses `lxml.html forms`_ to pre-populate form
|
||||
fields with form data from :class:`Response` objects.
|
||||
|
||||
.. _lxml.html forms: http://lxml.de/lxmlhtml.html#forms
|
||||
.. _lxml.html forms: https://lxml.de/lxmlhtml.html#forms
|
||||
|
||||
.. class:: FormRequest(url, [formdata, ...])
|
||||
|
||||
|
|
@ -609,7 +615,10 @@ Response objects
|
|||
|
||||
:param request: the initial value of the :attr:`Response.request` attribute.
|
||||
This represents the :class:`Request` that generated this response.
|
||||
:type request: :class:`Request` object
|
||||
:type request: scrapy.http.Request
|
||||
|
||||
:param certificate: an object representing the server's SSL certificate.
|
||||
:type certificate: twisted.internet.ssl.Certificate
|
||||
|
||||
.. attribute:: Response.url
|
||||
|
||||
|
|
@ -664,7 +673,7 @@ Response objects
|
|||
.. attribute:: Response.meta
|
||||
|
||||
A shortcut to the :attr:`Request.meta` attribute of the
|
||||
:attr:`Response.request` object (ie. ``self.request.meta``).
|
||||
:attr:`Response.request` object (i.e. ``self.request.meta``).
|
||||
|
||||
Unlike the :attr:`Response.request` attribute, the :attr:`Response.meta`
|
||||
attribute is propagated along redirects and retries, so you will get
|
||||
|
|
@ -672,6 +681,20 @@ Response objects
|
|||
|
||||
.. seealso:: :attr:`Request.meta` attribute
|
||||
|
||||
.. attribute:: Response.cb_kwargs
|
||||
|
||||
.. versionadded:: 2.0
|
||||
|
||||
A shortcut to the :attr:`Request.cb_kwargs` attribute of the
|
||||
:attr:`Response.request` object (i.e. ``self.request.cb_kwargs``).
|
||||
|
||||
Unlike the :attr:`Response.request` attribute, the
|
||||
:attr:`Response.cb_kwargs` attribute is propagated along redirects and
|
||||
retries, so you will get the original :attr:`Request.cb_kwargs` sent
|
||||
from your spider.
|
||||
|
||||
.. seealso:: :attr:`Request.cb_kwargs` attribute
|
||||
|
||||
.. attribute:: Response.flags
|
||||
|
||||
A list that contains flags for this response. Flags are labels used for
|
||||
|
|
@ -679,6 +702,13 @@ Response objects
|
|||
they're shown on the string representation of the Response (`__str__`
|
||||
method) which is used by the engine for logging.
|
||||
|
||||
.. attribute:: Response.certificate
|
||||
|
||||
A :class:`twisted.internet.ssl.Certificate` object representing
|
||||
the server's SSL certificate.
|
||||
|
||||
Only populated for ``https`` responses, ``None`` otherwise.
|
||||
|
||||
.. method:: Response.copy()
|
||||
|
||||
Returns a new Response which is a copy of this Response.
|
||||
|
|
@ -760,7 +790,7 @@ TextResponse objects
|
|||
1. the encoding passed in the ``__init__`` method ``encoding`` argument
|
||||
|
||||
2. the encoding declared in the Content-Type HTTP header. If this
|
||||
encoding is not valid (ie. unknown), it is ignored and the next
|
||||
encoding is not valid (i.e. unknown), it is ignored and the next
|
||||
resolution mechanism is tried.
|
||||
|
||||
3. the encoding declared in the response body. The TextResponse class
|
||||
|
|
|
|||
|
|
@ -35,12 +35,11 @@ defines selectors to associate those styles with specific HTML elements.
|
|||
in speed and parsing accuracy to lxml.
|
||||
|
||||
.. _BeautifulSoup: https://www.crummy.com/software/BeautifulSoup/
|
||||
.. _lxml: http://lxml.de/
|
||||
.. _lxml: https://lxml.de/
|
||||
.. _ElementTree: https://docs.python.org/2/library/xml.etree.elementtree.html
|
||||
.. _cssselect: https://pypi.python.org/pypi/cssselect/
|
||||
.. _XPath: https://www.w3.org/TR/xpath
|
||||
.. _XPath: https://www.w3.org/TR/xpath/all/
|
||||
.. _CSS: https://www.w3.org/TR/selectors
|
||||
.. _parsel: https://parsel.readthedocs.io/
|
||||
.. _parsel: https://parsel.readthedocs.io/en/latest/
|
||||
|
||||
Using selectors
|
||||
===============
|
||||
|
|
@ -255,7 +254,7 @@ that Scrapy (parsel) implements a couple of **non-standard pseudo-elements**:
|
|||
They will most probably not work with other libraries like
|
||||
`lxml`_ or `PyQuery`_.
|
||||
|
||||
.. _PyQuery: https://pypi.python.org/pypi/pyquery
|
||||
.. _PyQuery: https://pypi.org/project/pyquery/
|
||||
|
||||
Examples:
|
||||
|
||||
|
|
@ -309,7 +308,7 @@ Examples:
|
|||
make much sense: text nodes do not have attributes, and attribute values
|
||||
are string values already and do not have children nodes.
|
||||
|
||||
.. _CSS Selectors: https://www.w3.org/TR/css3-selectors/#selectors
|
||||
.. _CSS Selectors: https://www.w3.org/TR/selectors-3/#selectors
|
||||
|
||||
.. _topics-selectors-nesting-selectors:
|
||||
|
||||
|
|
@ -504,7 +503,7 @@ Another common case would be to extract all direct ``<p>`` children:
|
|||
For more details about relative XPaths see the `Location Paths`_ section in the
|
||||
XPath specification.
|
||||
|
||||
.. _Location Paths: https://www.w3.org/TR/xpath#location-paths
|
||||
.. _Location Paths: https://www.w3.org/TR/xpath/all/#location-paths
|
||||
|
||||
When querying by class, consider using CSS
|
||||
------------------------------------------
|
||||
|
|
@ -612,7 +611,7 @@ But using the ``.`` to mean the node, works:
|
|||
>>> sel.xpath("//a[contains(., 'Next Page')]").getall()
|
||||
['<a href="#">Click here to go to the <strong>Next Page</strong></a>']
|
||||
|
||||
.. _`XPath string function`: https://www.w3.org/TR/xpath/#section-String-Functions
|
||||
.. _`XPath string function`: https://www.w3.org/TR/xpath/all/#section-String-Functions
|
||||
|
||||
.. _topics-selectors-xpath-variables:
|
||||
|
||||
|
|
@ -764,7 +763,7 @@ Set operations
|
|||
These can be handy for excluding parts of a document tree before
|
||||
extracting text elements for example.
|
||||
|
||||
Example extracting microdata (sample content taken from http://schema.org/Product)
|
||||
Example extracting microdata (sample content taken from https://schema.org/Product)
|
||||
with groups of itemscopes and corresponding itemprops::
|
||||
|
||||
>>> doc = u"""
|
||||
|
|
@ -986,7 +985,7 @@ a :class:`~scrapy.http.HtmlResponse` object like this::
|
|||
sel = Selector(html_response)
|
||||
|
||||
1. Select all ``<h1>`` elements from an HTML response body, returning a list of
|
||||
:class:`Selector` objects (ie. a :class:`SelectorList` object)::
|
||||
:class:`Selector` objects (i.e. a :class:`SelectorList` object)::
|
||||
|
||||
sel.xpath("//h1")
|
||||
|
||||
|
|
@ -1013,7 +1012,7 @@ instantiated with an :class:`~scrapy.http.XmlResponse` object::
|
|||
sel = Selector(xml_response)
|
||||
|
||||
1. Select all ``<product>`` elements from an XML response body, returning a list
|
||||
of :class:`Selector` objects (ie. a :class:`SelectorList` object)::
|
||||
of :class:`Selector` objects (i.e. a :class:`SelectorList` object)::
|
||||
|
||||
sel.xpath("//product")
|
||||
|
||||
|
|
|
|||
|
|
@ -124,7 +124,7 @@ Settings can be accessed through the :attr:`scrapy.crawler.Crawler.settings`
|
|||
attribute of the Crawler that is passed to ``from_crawler`` method in
|
||||
extensions, middlewares and item pipelines::
|
||||
|
||||
class MyExtension(object):
|
||||
class MyExtension:
|
||||
def __init__(self, log_is_enabled=False):
|
||||
if log_is_enabled:
|
||||
print("log is enabled!")
|
||||
|
|
@ -248,7 +248,7 @@ CONCURRENT_REQUESTS
|
|||
|
||||
Default: ``16``
|
||||
|
||||
The maximum number of concurrent (ie. simultaneous) requests that will be
|
||||
The maximum number of concurrent (i.e. simultaneous) requests that will be
|
||||
performed by the Scrapy downloader.
|
||||
|
||||
.. setting:: CONCURRENT_REQUESTS_PER_DOMAIN
|
||||
|
|
@ -258,7 +258,7 @@ CONCURRENT_REQUESTS_PER_DOMAIN
|
|||
|
||||
Default: ``8``
|
||||
|
||||
The maximum number of concurrent (ie. simultaneous) requests that will be
|
||||
The maximum number of concurrent (i.e. simultaneous) requests that will be
|
||||
performed to any single domain.
|
||||
|
||||
See also: :ref:`topics-autothrottle` and its
|
||||
|
|
@ -272,7 +272,7 @@ CONCURRENT_REQUESTS_PER_IP
|
|||
|
||||
Default: ``0``
|
||||
|
||||
The maximum number of concurrent (ie. simultaneous) requests that will be
|
||||
The maximum number of concurrent (i.e. simultaneous) requests that will be
|
||||
performed to any single IP. If non-zero, the
|
||||
:setting:`CONCURRENT_REQUESTS_PER_DOMAIN` setting is ignored, and this one is
|
||||
used instead. In other words, concurrency limits will be applied per IP, not
|
||||
|
|
@ -381,6 +381,8 @@ DNS in-memory cache size.
|
|||
DNS_RESOLVER
|
||||
------------
|
||||
|
||||
.. versionadded:: 2.0
|
||||
|
||||
Default: ``'scrapy.resolver.CachingThreadedResolver'``
|
||||
|
||||
The class to be used to resolve DNS names. The default ``scrapy.resolver.CachingThreadedResolver``
|
||||
|
|
@ -1275,6 +1277,9 @@ does not work together with :setting:`CONCURRENT_REQUESTS_PER_IP`.
|
|||
|
||||
SCRAPER_SLOT_MAX_ACTIVE_SIZE
|
||||
----------------------------
|
||||
|
||||
.. versionadded:: 2.0
|
||||
|
||||
Default: ``5_000_000``
|
||||
|
||||
Soft limit (in bytes) for response data being processed.
|
||||
|
|
@ -1464,24 +1469,95 @@ in the ``project`` subdirectory.
|
|||
TWISTED_REACTOR
|
||||
---------------
|
||||
|
||||
.. versionadded:: 2.0
|
||||
|
||||
Default: ``None``
|
||||
|
||||
Import path of a given Twisted reactor, for instance:
|
||||
:class:`twisted.internet.asyncioreactor.AsyncioSelectorReactor`.
|
||||
Import path of a given :mod:`~twisted.internet.reactor`.
|
||||
|
||||
Scrapy will install this reactor if no other is installed yet, such as when
|
||||
the ``scrapy`` CLI program is invoked or when using the
|
||||
:class:`~scrapy.crawler.CrawlerProcess` class. If you are using the
|
||||
:class:`~scrapy.crawler.CrawlerRunner` class, you need to install the correct
|
||||
reactor manually. An exception will be raised if the installation fails.
|
||||
Scrapy will install this reactor if no other reactor is installed yet, such as
|
||||
when the ``scrapy`` CLI program is invoked or when using the
|
||||
:class:`~scrapy.crawler.CrawlerProcess` class.
|
||||
|
||||
The default value for this option is currently ``None``, which means that Scrapy
|
||||
will not attempt to install any specific reactor, and the default one defined by
|
||||
Twisted for the current platform will be used. This is to maintain backward
|
||||
compatibility and avoid possible problems caused by using a non-default reactor.
|
||||
If you are using the :class:`~scrapy.crawler.CrawlerRunner` class, you also
|
||||
need to install the correct reactor manually. You can do that using
|
||||
:func:`~scrapy.utils.reactor.install_reactor`:
|
||||
|
||||
For additional information, please see
|
||||
:doc:`core/howto/choosing-reactor`.
|
||||
.. autofunction:: scrapy.utils.reactor.install_reactor
|
||||
|
||||
If a reactor is already installed,
|
||||
:func:`~scrapy.utils.reactor.install_reactor` has no effect.
|
||||
|
||||
:meth:`CrawlerRunner.__init__ <scrapy.crawler.CrawlerRunner.__init__>` raises
|
||||
:exc:`Exception` if the installed reactor does not match the
|
||||
:setting:`TWISTED_REACTOR` setting; therfore, having top-level
|
||||
:mod:`~twisted.internet.reactor` imports in project files and imported
|
||||
third-party libraries will make Scrapy raise :exc:`Exception` when
|
||||
it checks which reactor is installed.
|
||||
|
||||
In order to use the reactor installed by Scrapy::
|
||||
|
||||
import scrapy
|
||||
from twisted.internet import reactor
|
||||
|
||||
|
||||
class QuotesSpider(scrapy.Spider):
|
||||
name = 'quotes'
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
self.timeout = int(kwargs.pop('timeout', '60'))
|
||||
super(QuotesSpider, self).__init__(*args, **kwargs)
|
||||
|
||||
def start_requests(self):
|
||||
reactor.callLater(self.timeout, self.stop)
|
||||
|
||||
urls = ['http://quotes.toscrape.com/page/1']
|
||||
for url in urls:
|
||||
yield scrapy.Request(url=url, callback=self.parse)
|
||||
|
||||
def parse(self, response):
|
||||
for quote in response.css('div.quote'):
|
||||
yield {'text': quote.css('span.text::text').get()}
|
||||
|
||||
def stop(self):
|
||||
self.crawler.engine.close_spider(self, 'timeout')
|
||||
|
||||
|
||||
which raises :exc:`Exception`, becomes::
|
||||
|
||||
import scrapy
|
||||
|
||||
|
||||
class QuotesSpider(scrapy.Spider):
|
||||
name = 'quotes'
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
self.timeout = int(kwargs.pop('timeout', '60'))
|
||||
super(QuotesSpider, self).__init__(*args, **kwargs)
|
||||
|
||||
def start_requests(self):
|
||||
from twisted.internet import reactor
|
||||
reactor.callLater(self.timeout, self.stop)
|
||||
|
||||
urls = ['http://quotes.toscrape.com/page/1']
|
||||
for url in urls:
|
||||
yield scrapy.Request(url=url, callback=self.parse)
|
||||
|
||||
def parse(self, response):
|
||||
for quote in response.css('div.quote'):
|
||||
yield {'text': quote.css('span.text::text').get()}
|
||||
|
||||
def stop(self):
|
||||
self.crawler.engine.close_spider(self, 'timeout')
|
||||
|
||||
|
||||
The default value of the :setting:`TWISTED_REACTOR` setting is ``None``, which
|
||||
means that Scrapy will not attempt to install any specific reactor, and the
|
||||
default reactor defined by Twisted for the current platform will be used. This
|
||||
is to maintain backward compatibility and avoid possible problems caused by
|
||||
using a non-default reactor.
|
||||
|
||||
For additional information, see :doc:`core/howto/choosing-reactor`.
|
||||
|
||||
|
||||
.. setting:: URLLENGTH_LIMIT
|
||||
|
|
|
|||
|
|
@ -41,7 +41,7 @@ variable; or by defining it in your :ref:`scrapy.cfg <topics-config-settings>`::
|
|||
|
||||
.. _IPython: https://ipython.org/
|
||||
.. _IPython installation guide: https://ipython.org/install.html
|
||||
.. _bpython: https://www.bpython-interpreter.org/
|
||||
.. _bpython: https://bpython-interpreter.org/
|
||||
|
||||
Launch the shell
|
||||
================
|
||||
|
|
@ -142,7 +142,7 @@ Example of shell session
|
|||
========================
|
||||
|
||||
Here's an example of a typical shell session where we start by scraping the
|
||||
https://scrapy.org page, and then proceed to scrape the https://reddit.com
|
||||
https://scrapy.org page, and then proceed to scrape the https://old.reddit.com/
|
||||
page. Finally, we modify the (Reddit) request method to POST and re-fetch it
|
||||
getting an error. We end the session by typing Ctrl-D (in Unix systems) or
|
||||
Ctrl-Z in Windows.
|
||||
|
|
@ -182,7 +182,7 @@ After that, we can start playing with the objects:
|
|||
>>> response.xpath('//title/text()').get()
|
||||
'Scrapy | A Fast and Powerful Scraping and Web Crawling Framework'
|
||||
|
||||
>>> fetch("https://reddit.com")
|
||||
>>> fetch("https://old.reddit.com/")
|
||||
|
||||
>>> response.xpath('//title/text()').get()
|
||||
'reddit: the front page of the internet'
|
||||
|
|
|
|||
|
|
@ -46,6 +46,7 @@ Here is a simple example showing how you can catch signals and perform some acti
|
|||
def parse(self, response):
|
||||
pass
|
||||
|
||||
.. _signal-deferred:
|
||||
|
||||
Deferred signal handlers
|
||||
========================
|
||||
|
|
@ -141,7 +142,7 @@ item_error
|
|||
.. signal:: item_error
|
||||
.. function:: item_error(item, response, spider, failure)
|
||||
|
||||
Sent when a :ref:`topics-item-pipeline` generates an error (ie. raises
|
||||
Sent when a :ref:`topics-item-pipeline` generates an error (i.e. raises
|
||||
an exception), except :exc:`~scrapy.exceptions.DropItem` exception.
|
||||
|
||||
This signal supports returning deferreds from their handlers.
|
||||
|
|
@ -232,7 +233,7 @@ spider_error
|
|||
.. signal:: spider_error
|
||||
.. function:: spider_error(failure, response, spider)
|
||||
|
||||
Sent when a spider callback generates an error (ie. raises an exception).
|
||||
Sent when a spider callback generates an error (i.e. raises an exception).
|
||||
|
||||
This signal does not support returning deferreds from their handlers.
|
||||
|
||||
|
|
@ -301,6 +302,8 @@ request_left_downloader
|
|||
.. signal:: request_left_downloader
|
||||
.. function:: request_left_downloader(request, spider)
|
||||
|
||||
.. versionadded:: 2.0
|
||||
|
||||
Sent when a :class:`~scrapy.http.Request` leaves the downloader, even in case of
|
||||
failure.
|
||||
|
||||
|
|
|
|||
|
|
@ -299,8 +299,8 @@ The spider will not do any parsing on its own.
|
|||
If you were to set the ``start_urls`` attribute from the command line,
|
||||
you would have to parse it on your own into a list
|
||||
using something like
|
||||
`ast.literal_eval <https://docs.python.org/library/ast.html#ast.literal_eval>`_
|
||||
or `json.loads <https://docs.python.org/library/json.html#json.loads>`_
|
||||
`ast.literal_eval <https://docs.python.org/3/library/ast.html#ast.literal_eval>`_
|
||||
or `json.loads <https://docs.python.org/3/library/json.html#json.loads>`_
|
||||
and then set it as an attribute.
|
||||
Otherwise, you would cause iteration over a ``start_urls`` string
|
||||
(a very common python pitfall)
|
||||
|
|
@ -420,6 +420,9 @@ Crawling rules
|
|||
It receives a :class:`Twisted Failure <twisted.python.failure.Failure>`
|
||||
instance as first parameter.
|
||||
|
||||
.. versionadded:: 2.0
|
||||
The *errback* parameter.
|
||||
|
||||
CrawlSpider example
|
||||
~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
|
|
@ -811,6 +814,6 @@ Combine SitemapSpider with other sources of urls::
|
|||
|
||||
.. _Sitemaps: https://www.sitemaps.org/index.html
|
||||
.. _Sitemap index files: https://www.sitemaps.org/protocol.html#index
|
||||
.. _robots.txt: http://www.robotstxt.org/
|
||||
.. _robots.txt: https://www.robotstxt.org/
|
||||
.. _TLD: https://en.wikipedia.org/wiki/Top-level_domain
|
||||
.. _Scrapyd documentation: https://scrapyd.readthedocs.io/en/latest/
|
||||
|
|
|
|||
|
|
@ -32,7 +32,7 @@ Common Stats Collector uses
|
|||
Access the stats collector through the :attr:`~scrapy.crawler.Crawler.stats`
|
||||
attribute. Here is an example of an extension that access stats::
|
||||
|
||||
class ExtensionThatAccessStats(object):
|
||||
class ExtensionThatAccessStats:
|
||||
|
||||
def __init__(self, stats):
|
||||
self.stats = stats
|
||||
|
|
|
|||
115
pytest.ini
115
pytest.ini
|
|
@ -37,7 +37,7 @@ flake8-ignore =
|
|||
scrapy/commands/edit.py E501
|
||||
scrapy/commands/fetch.py E401 E501 E128 E731
|
||||
scrapy/commands/genspider.py E128 E501 E502
|
||||
scrapy/commands/parse.py E128 E501 E731 E226
|
||||
scrapy/commands/parse.py E128 E501 E731
|
||||
scrapy/commands/runspider.py E501
|
||||
scrapy/commands/settings.py E128
|
||||
scrapy/commands/shell.py E128 E501 E502
|
||||
|
|
@ -47,22 +47,22 @@ flake8-ignore =
|
|||
scrapy/contracts/__init__.py E501 W504
|
||||
scrapy/contracts/default.py E128
|
||||
# scrapy/core
|
||||
scrapy/core/engine.py E501 E128 E127 E306 E502
|
||||
scrapy/core/engine.py E501 E128 E127 E502
|
||||
scrapy/core/scheduler.py E501
|
||||
scrapy/core/scraper.py E501 E306 E128 W504
|
||||
scrapy/core/spidermw.py E501 E731 E126 E226
|
||||
scrapy/core/scraper.py E501 E128 W504
|
||||
scrapy/core/spidermw.py E501 E731 E126
|
||||
scrapy/core/downloader/__init__.py E501
|
||||
scrapy/core/downloader/contextfactory.py E501 E128 E126
|
||||
scrapy/core/downloader/middleware.py E501 E502
|
||||
scrapy/core/downloader/tls.py E501 E305 E241
|
||||
scrapy/core/downloader/webclient.py E731 E501 E128 E126 E226
|
||||
scrapy/core/downloader/tls.py E501 E241
|
||||
scrapy/core/downloader/webclient.py E731 E501 E128 E126
|
||||
scrapy/core/downloader/handlers/__init__.py E501
|
||||
scrapy/core/downloader/handlers/ftp.py E501 E305 E128 E127
|
||||
scrapy/core/downloader/handlers/ftp.py E501 E128 E127
|
||||
scrapy/core/downloader/handlers/http10.py E501
|
||||
scrapy/core/downloader/handlers/http11.py E501
|
||||
scrapy/core/downloader/handlers/s3.py E501 E128 E126
|
||||
# scrapy/downloadermiddlewares
|
||||
scrapy/downloadermiddlewares/ajaxcrawl.py E501 E226
|
||||
scrapy/downloadermiddlewares/ajaxcrawl.py E501
|
||||
scrapy/downloadermiddlewares/decompression.py E501
|
||||
scrapy/downloadermiddlewares/defaultheaders.py E501
|
||||
scrapy/downloadermiddlewares/httpcache.py E501 E126
|
||||
|
|
@ -76,7 +76,7 @@ flake8-ignore =
|
|||
scrapy/extensions/closespider.py E501 E128 E123
|
||||
scrapy/extensions/corestats.py E501
|
||||
scrapy/extensions/feedexport.py E128 E501
|
||||
scrapy/extensions/httpcache.py E128 E501 E303
|
||||
scrapy/extensions/httpcache.py E128 E501
|
||||
scrapy/extensions/memdebug.py E501
|
||||
scrapy/extensions/spiderstate.py E501
|
||||
scrapy/extensions/telnet.py E501 W504
|
||||
|
|
@ -91,7 +91,7 @@ flake8-ignore =
|
|||
scrapy/http/response/text.py E501 E128 E124
|
||||
# scrapy/linkextractors
|
||||
scrapy/linkextractors/__init__.py E731 E501 E402 W504
|
||||
scrapy/linkextractors/lxmlhtml.py E501 E731 E226
|
||||
scrapy/linkextractors/lxmlhtml.py E501 E731
|
||||
# scrapy/loader
|
||||
scrapy/loader/__init__.py E501 E128
|
||||
scrapy/loader/processors.py E501
|
||||
|
|
@ -105,7 +105,7 @@ flake8-ignore =
|
|||
scrapy/selector/unified.py E501 E111
|
||||
# scrapy/settings
|
||||
scrapy/settings/__init__.py E501
|
||||
scrapy/settings/default_settings.py E501 E114 E116 E226
|
||||
scrapy/settings/default_settings.py E501 E114 E116
|
||||
scrapy/settings/deprecated.py E501
|
||||
# scrapy/spidermiddlewares
|
||||
scrapy/spidermiddlewares/httperror.py E501
|
||||
|
|
@ -121,28 +121,27 @@ flake8-ignore =
|
|||
scrapy/utils/asyncio.py E501
|
||||
scrapy/utils/benchserver.py E501
|
||||
scrapy/utils/conf.py E402 E501
|
||||
scrapy/utils/console.py E306 E305
|
||||
scrapy/utils/datatypes.py E501 E226
|
||||
scrapy/utils/datatypes.py E501
|
||||
scrapy/utils/decorators.py E501
|
||||
scrapy/utils/defer.py E501 E128
|
||||
scrapy/utils/deprecate.py E128 E501 E127 E502
|
||||
scrapy/utils/gz.py E305 E501 W504
|
||||
scrapy/utils/http.py F403 E226
|
||||
scrapy/utils/gz.py E501 W504
|
||||
scrapy/utils/http.py F403
|
||||
scrapy/utils/httpobj.py E501
|
||||
scrapy/utils/iterators.py E501 E701
|
||||
scrapy/utils/iterators.py E501
|
||||
scrapy/utils/log.py E128 E501
|
||||
scrapy/utils/markup.py F403
|
||||
scrapy/utils/misc.py E501 E226
|
||||
scrapy/utils/misc.py E501
|
||||
scrapy/utils/multipart.py F403
|
||||
scrapy/utils/project.py E501
|
||||
scrapy/utils/python.py E501
|
||||
scrapy/utils/reactor.py E226 E501
|
||||
scrapy/utils/reactor.py E501
|
||||
scrapy/utils/reqser.py E501
|
||||
scrapy/utils/request.py E127 E501
|
||||
scrapy/utils/response.py E501 E128
|
||||
scrapy/utils/signal.py E501 E128
|
||||
scrapy/utils/sitemap.py E501
|
||||
scrapy/utils/spider.py E271 E501
|
||||
scrapy/utils/spider.py E501
|
||||
scrapy/utils/ssl.py E501
|
||||
scrapy/utils/test.py E501
|
||||
scrapy/utils/url.py E501 F403 E128 F405
|
||||
|
|
@ -152,7 +151,7 @@ flake8-ignore =
|
|||
scrapy/crawler.py E501
|
||||
scrapy/dupefilters.py E501 E202
|
||||
scrapy/exceptions.py E501
|
||||
scrapy/exporters.py E501 E226
|
||||
scrapy/exporters.py E501
|
||||
scrapy/interfaces.py E501
|
||||
scrapy/item.py E501 E128
|
||||
scrapy/link.py E501
|
||||
|
|
@ -161,93 +160,91 @@ flake8-ignore =
|
|||
scrapy/middleware.py E128 E501
|
||||
scrapy/pqueues.py E501
|
||||
scrapy/resolver.py E501
|
||||
scrapy/responsetypes.py E128 E501 E305
|
||||
scrapy/responsetypes.py E128 E501
|
||||
scrapy/robotstxt.py E501
|
||||
scrapy/shell.py E501
|
||||
scrapy/signalmanager.py E501
|
||||
scrapy/spiderloader.py E225 F841 E501 E126
|
||||
scrapy/spiderloader.py F841 E501 E126
|
||||
scrapy/squeues.py E128
|
||||
scrapy/statscollectors.py E501
|
||||
# tests
|
||||
tests/__init__.py E402 E501
|
||||
tests/mockserver.py E401 E501 E126 E123
|
||||
tests/pipelines.py F841 E226
|
||||
tests/pipelines.py F841
|
||||
tests/spiders.py E501 E127
|
||||
tests/test_closespider.py E501 E127
|
||||
tests/test_command_fetch.py E501
|
||||
tests/test_command_parse.py E501 E128 E303 E226
|
||||
tests/test_command_parse.py E501 E128
|
||||
tests/test_command_shell.py E501 E128
|
||||
tests/test_commands.py E128 E501
|
||||
tests/test_contracts.py E501 E128
|
||||
tests/test_crawl.py E501 E741 E265
|
||||
tests/test_crawler.py F841 E306 E501
|
||||
tests/test_dependencies.py F841 E501 E305
|
||||
tests/test_downloader_handlers.py E124 E127 E128 E225 E265 E501 E701 E126 E226 E123
|
||||
tests/test_crawler.py F841 E501
|
||||
tests/test_dependencies.py F841 E501
|
||||
tests/test_downloader_handlers.py E124 E127 E128 E265 E501 E126 E123
|
||||
tests/test_downloadermiddleware.py E501
|
||||
tests/test_downloadermiddleware_ajaxcrawlable.py E501
|
||||
tests/test_downloadermiddleware_cookies.py E731 E741 E501 E128 E303 E265 E126
|
||||
tests/test_downloadermiddleware_cookies.py E731 E741 E501 E128 E265 E126
|
||||
tests/test_downloadermiddleware_decompression.py E127
|
||||
tests/test_downloadermiddleware_defaultheaders.py E501
|
||||
tests/test_downloadermiddleware_downloadtimeout.py E501
|
||||
tests/test_downloadermiddleware_httpcache.py E501 E305
|
||||
tests/test_downloadermiddleware_httpcompression.py E501 E251 E126 E123
|
||||
tests/test_downloadermiddleware_httpcache.py E501
|
||||
tests/test_downloadermiddleware_httpcompression.py E501 E126 E123
|
||||
tests/test_downloadermiddleware_httpproxy.py E501 E128
|
||||
tests/test_downloadermiddleware_redirect.py E501 E303 E128 E306 E127 E305
|
||||
tests/test_downloadermiddleware_retry.py E501 E128 E251 E303 E126
|
||||
tests/test_downloadermiddleware_redirect.py E501 E128 E127
|
||||
tests/test_downloadermiddleware_retry.py E501 E128 E126
|
||||
tests/test_downloadermiddleware_robotstxt.py E501
|
||||
tests/test_downloadermiddleware_stats.py E501
|
||||
tests/test_dupefilters.py E221 E501 E741 E128 E124
|
||||
tests/test_dupefilters.py E501 E741 E128 E124
|
||||
tests/test_engine.py E401 E501 E128
|
||||
tests/test_exporters.py E501 E731 E306 E128 E124
|
||||
tests/test_exporters.py E501 E731 E128 E124
|
||||
tests/test_extension_telnet.py F841
|
||||
tests/test_feedexport.py E501 F841 E241
|
||||
tests/test_http_cookies.py E501
|
||||
tests/test_http_headers.py E501
|
||||
tests/test_http_request.py E402 E501 E127 E128 E128 E126 E123
|
||||
tests/test_http_response.py E501 E301 E128 E265
|
||||
tests/test_item.py E701 E128 F841 E306
|
||||
tests/test_http_response.py E501 E128 E265
|
||||
tests/test_item.py E128 F841
|
||||
tests/test_link.py E501
|
||||
tests/test_linkextractors.py E501 E128 E124
|
||||
tests/test_loader.py E501 E731 E303 E741 E128 E117 E241
|
||||
tests/test_loader.py E501 E731 E741 E128 E117 E241
|
||||
tests/test_logformatter.py E128 E501 E122
|
||||
tests/test_mail.py E128 E501 E305
|
||||
tests/test_mail.py E128 E501
|
||||
tests/test_middleware.py E501 E128
|
||||
tests/test_pipeline_crawl.py E131 E501 E128 E126
|
||||
tests/test_pipeline_files.py E501 E303 E272 E226
|
||||
tests/test_pipeline_images.py F841 E501 E303
|
||||
tests/test_pipeline_media.py E501 E741 E731 E128 E306 E502
|
||||
tests/test_pipeline_crawl.py E501 E128 E126
|
||||
tests/test_pipeline_files.py E501
|
||||
tests/test_pipeline_images.py F841 E501
|
||||
tests/test_pipeline_media.py E501 E741 E731 E128 E502
|
||||
tests/test_proxy_connect.py E501 E741
|
||||
tests/test_request_cb_kwargs.py E501
|
||||
tests/test_responsetypes.py E501 E305
|
||||
tests/test_responsetypes.py E501
|
||||
tests/test_robotstxt_interface.py E501 E501
|
||||
tests/test_scheduler.py E501 E126 E123
|
||||
tests/test_selector.py E501 E127
|
||||
tests/test_spider.py E501
|
||||
tests/test_spidermiddleware.py E501 E226
|
||||
tests/test_spidermiddleware.py E501
|
||||
tests/test_spidermiddleware_httperror.py E128 E501 E127 E121
|
||||
tests/test_spidermiddleware_offsite.py E501 E128 E111
|
||||
tests/test_spidermiddleware_output_chain.py E501 E226
|
||||
tests/test_spidermiddleware_output_chain.py E501
|
||||
tests/test_spidermiddleware_referer.py E501 F841 E125 E201 E124 E501 E241 E121
|
||||
tests/test_squeues.py E501 E701 E741
|
||||
tests/test_squeues.py E501 E741
|
||||
tests/test_utils_asyncio.py E501
|
||||
tests/test_utils_conf.py E501 E303 E128
|
||||
tests/test_utils_conf.py E501 E128
|
||||
tests/test_utils_curl.py E501
|
||||
tests/test_utils_datatypes.py E402 E501 E305
|
||||
tests/test_utils_defer.py E306 E501 F841 E226
|
||||
tests/test_utils_deprecate.py F841 E306 E501
|
||||
tests/test_utils_datatypes.py E402 E501
|
||||
tests/test_utils_defer.py E501 F841
|
||||
tests/test_utils_deprecate.py F841 E501
|
||||
tests/test_utils_http.py E501 E128 W504
|
||||
tests/test_utils_iterators.py E501 E128 E129 E303 E241
|
||||
tests/test_utils_log.py E741 E226
|
||||
tests/test_utils_python.py E501 E303 E731 E701 E305
|
||||
tests/test_utils_iterators.py E501 E128 E129 E241
|
||||
tests/test_utils_log.py E741
|
||||
tests/test_utils_python.py E501 E731
|
||||
tests/test_utils_reqser.py E501 E128
|
||||
tests/test_utils_request.py E501 E128 E305
|
||||
tests/test_utils_request.py E501 E128
|
||||
tests/test_utils_response.py E501
|
||||
tests/test_utils_signal.py E741 F841 E731 E226
|
||||
tests/test_utils_signal.py E741 F841 E731
|
||||
tests/test_utils_sitemap.py E128 E501 E124
|
||||
tests/test_utils_spider.py E305
|
||||
tests/test_utils_template.py E305
|
||||
tests/test_utils_url.py E501 E127 E305 E211 E125 E501 E226 E241 E126 E123
|
||||
tests/test_webclient.py E501 E128 E122 E303 E402 E306 E226 E241 E123 E126
|
||||
tests/test_utils_url.py E501 E127 E125 E501 E241 E126 E123
|
||||
tests/test_webclient.py E501 E128 E122 E402 E241 E123 E126
|
||||
tests/test_cmdline/__init__.py E501
|
||||
tests/test_settings/__init__.py E501 E128
|
||||
tests/test_spiderloader/__init__.py E128 E501
|
||||
|
|
|
|||
|
|
@ -1 +1 @@
|
|||
1.8.0
|
||||
2.0.0
|
||||
|
|
|
|||
|
|
@ -12,7 +12,6 @@ from scrapy.exceptions import UsageError
|
|||
from scrapy.utils.misc import walk_modules
|
||||
from scrapy.utils.project import inside_project, get_project_settings
|
||||
from scrapy.utils.python import garbage_collect
|
||||
from scrapy.settings.deprecated import check_deprecated_settings
|
||||
|
||||
|
||||
def _iter_command_classes(module_name):
|
||||
|
|
@ -118,7 +117,6 @@ def execute(argv=None, settings=None):
|
|||
pass
|
||||
else:
|
||||
settings['EDITOR'] = editor
|
||||
check_deprecated_settings(settings)
|
||||
|
||||
inproject = inside_project()
|
||||
cmds = _get_commands_dict(settings, inproject)
|
||||
|
|
|
|||
|
|
@ -9,7 +9,7 @@ from scrapy.utils.conf import arglist_to_dict
|
|||
from scrapy.exceptions import UsageError
|
||||
|
||||
|
||||
class ScrapyCommand(object):
|
||||
class ScrapyCommand:
|
||||
|
||||
requires_project = False
|
||||
crawler_process = None
|
||||
|
|
|
|||
|
|
@ -25,7 +25,7 @@ class Command(ScrapyCommand):
|
|||
self.crawler_process.start()
|
||||
|
||||
|
||||
class _BenchServer(object):
|
||||
class _BenchServer:
|
||||
|
||||
def __enter__(self):
|
||||
from scrapy.utils.test import get_testenv
|
||||
|
|
|
|||
|
|
@ -1,7 +1,5 @@
|
|||
import os
|
||||
from scrapy.commands import ScrapyCommand
|
||||
from scrapy.utils.conf import arglist_to_dict
|
||||
from scrapy.utils.python import without_none_values
|
||||
from scrapy.utils.conf import arglist_to_dict, feed_process_params_from_cli
|
||||
from scrapy.exceptions import UsageError
|
||||
|
||||
|
||||
|
|
@ -19,7 +17,7 @@ class Command(ScrapyCommand):
|
|||
ScrapyCommand.add_options(self, parser)
|
||||
parser.add_option("-a", dest="spargs", action="append", default=[], metavar="NAME=VALUE",
|
||||
help="set spider argument (may be repeated)")
|
||||
parser.add_option("-o", "--output", metavar="FILE",
|
||||
parser.add_option("-o", "--output", metavar="FILE", action="append",
|
||||
help="dump scraped items into FILE (use - for stdout)")
|
||||
parser.add_option("-t", "--output-format", metavar="FORMAT",
|
||||
help="format to use for dumping items with -o")
|
||||
|
|
@ -31,21 +29,8 @@ class Command(ScrapyCommand):
|
|||
except ValueError:
|
||||
raise UsageError("Invalid -a value, use -a NAME=VALUE", print_help=False)
|
||||
if opts.output:
|
||||
if opts.output == '-':
|
||||
self.settings.set('FEED_URI', 'stdout:', priority='cmdline')
|
||||
else:
|
||||
self.settings.set('FEED_URI', opts.output, priority='cmdline')
|
||||
feed_exporters = without_none_values(
|
||||
self.settings.getwithbase('FEED_EXPORTERS'))
|
||||
valid_output_formats = feed_exporters.keys()
|
||||
if not opts.output_format:
|
||||
opts.output_format = os.path.splitext(opts.output)[1].replace(".", "")
|
||||
if opts.output_format not in valid_output_formats:
|
||||
raise UsageError("Unrecognized output format '%s', set one"
|
||||
" using the '-t' switch or as a file extension"
|
||||
" from the supported list %s" % (opts.output_format,
|
||||
tuple(valid_output_formats)))
|
||||
self.settings.set('FEED_FORMAT', opts.output_format, priority='cmdline')
|
||||
feeds = feed_process_params_from_cli(self.settings, opts.output, opts.output_format)
|
||||
self.settings.set('FEEDS', feeds, priority='cmdline')
|
||||
|
||||
def run(self, args, opts):
|
||||
if len(args) < 1:
|
||||
|
|
@ -54,8 +39,13 @@ class Command(ScrapyCommand):
|
|||
raise UsageError("running 'scrapy crawl' with more than one spider is no longer supported")
|
||||
spname = args[0]
|
||||
|
||||
self.crawler_process.crawl(spname, **opts.spargs)
|
||||
self.crawler_process.start()
|
||||
crawl_defer = self.crawler_process.crawl(spname, **opts.spargs)
|
||||
|
||||
if self.crawler_process.bootstrap_failed:
|
||||
if getattr(crawl_defer, 'result', None) is not None and issubclass(crawl_defer.result.type, Exception):
|
||||
self.exitcode = 1
|
||||
else:
|
||||
self.crawler_process.start()
|
||||
|
||||
if self.crawler_process.bootstrap_failed or \
|
||||
(hasattr(self.crawler_process, 'has_exception') and self.crawler_process.has_exception):
|
||||
self.exitcode = 1
|
||||
|
|
|
|||
|
|
@ -80,7 +80,7 @@ class Command(ScrapyCommand):
|
|||
else:
|
||||
items = self.items.get(lvl, [])
|
||||
|
||||
print("# Scraped Items ", "-"*60)
|
||||
print("# Scraped Items ", "-" * 60)
|
||||
display.pprint([dict(x) for x in items], colorize=colour)
|
||||
|
||||
def print_requests(self, lvl=None, colour=True):
|
||||
|
|
@ -92,14 +92,14 @@ class Command(ScrapyCommand):
|
|||
else:
|
||||
requests = self.requests.get(lvl, [])
|
||||
|
||||
print("# Requests ", "-"*65)
|
||||
print("# Requests ", "-" * 65)
|
||||
display.pprint(requests, colorize=colour)
|
||||
|
||||
def print_results(self, opts):
|
||||
colour = not opts.nocolour
|
||||
|
||||
if opts.verbose:
|
||||
for level in range(1, self.max_level+1):
|
||||
for level in range(1, self.max_level + 1):
|
||||
print('\n>>> DEPTH LEVEL: %s <<<' % level)
|
||||
if not opts.noitems:
|
||||
self.print_items(level, colour)
|
||||
|
|
|
|||
|
|
@ -5,8 +5,7 @@ from importlib import import_module
|
|||
from scrapy.utils.spider import iter_spider_classes
|
||||
from scrapy.commands import ScrapyCommand
|
||||
from scrapy.exceptions import UsageError
|
||||
from scrapy.utils.conf import arglist_to_dict
|
||||
from scrapy.utils.python import without_none_values
|
||||
from scrapy.utils.conf import arglist_to_dict, feed_process_params_from_cli
|
||||
|
||||
|
||||
def _import_file(filepath):
|
||||
|
|
@ -43,7 +42,7 @@ class Command(ScrapyCommand):
|
|||
ScrapyCommand.add_options(self, parser)
|
||||
parser.add_option("-a", dest="spargs", action="append", default=[], metavar="NAME=VALUE",
|
||||
help="set spider argument (may be repeated)")
|
||||
parser.add_option("-o", "--output", metavar="FILE",
|
||||
parser.add_option("-o", "--output", metavar="FILE", action="append",
|
||||
help="dump scraped items into FILE (use - for stdout)")
|
||||
parser.add_option("-t", "--output-format", metavar="FORMAT",
|
||||
help="format to use for dumping items with -o")
|
||||
|
|
@ -55,19 +54,8 @@ class Command(ScrapyCommand):
|
|||
except ValueError:
|
||||
raise UsageError("Invalid -a value, use -a NAME=VALUE", print_help=False)
|
||||
if opts.output:
|
||||
if opts.output == '-':
|
||||
self.settings.set('FEED_URI', 'stdout:', priority='cmdline')
|
||||
else:
|
||||
self.settings.set('FEED_URI', opts.output, priority='cmdline')
|
||||
feed_exporters = without_none_values(self.settings.getwithbase('FEED_EXPORTERS'))
|
||||
if not opts.output_format:
|
||||
opts.output_format = os.path.splitext(opts.output)[1].replace(".", "")
|
||||
if opts.output_format not in feed_exporters:
|
||||
raise UsageError("Unrecognized output format '%s', set one"
|
||||
" using the '-t' switch or as a file extension"
|
||||
" from the supported list %s" % (opts.output_format,
|
||||
tuple(feed_exporters)))
|
||||
self.settings.set('FEED_FORMAT', opts.output_format, priority='cmdline')
|
||||
feeds = feed_process_params_from_cli(self.settings, opts.output, opts.output_format)
|
||||
self.settings.set('FEEDS', feeds, priority='cmdline')
|
||||
|
||||
def run(self, args, opts):
|
||||
if len(args) != 1:
|
||||
|
|
|
|||
|
|
@ -9,7 +9,7 @@ from scrapy.utils.spider import iterate_spider_output
|
|||
from scrapy.utils.python import get_spec
|
||||
|
||||
|
||||
class ContractsManager(object):
|
||||
class ContractsManager:
|
||||
contracts = {}
|
||||
|
||||
def __init__(self, contracts):
|
||||
|
|
@ -107,7 +107,7 @@ class ContractsManager(object):
|
|||
request.errback = eb_wrapper
|
||||
|
||||
|
||||
class Contract(object):
|
||||
class Contract:
|
||||
""" Abstract class for contracts """
|
||||
request_cls = None
|
||||
|
||||
|
|
|
|||
|
|
@ -3,7 +3,7 @@ from time import time
|
|||
from datetime import datetime
|
||||
from collections import deque
|
||||
|
||||
from twisted.internet import reactor, defer, task
|
||||
from twisted.internet import defer, task
|
||||
|
||||
from scrapy.utils.defer import mustbe_deferred
|
||||
from scrapy.utils.httpobj import urlparse_cached
|
||||
|
|
@ -13,7 +13,7 @@ from scrapy.core.downloader.middleware import DownloaderMiddlewareManager
|
|||
from scrapy.core.downloader.handlers import DownloadHandlers
|
||||
|
||||
|
||||
class Slot(object):
|
||||
class Slot:
|
||||
"""Downloader slot"""
|
||||
|
||||
def __init__(self, concurrency, delay, randomize_delay):
|
||||
|
|
@ -66,7 +66,7 @@ def _get_concurrency_delay(concurrency, spider, settings):
|
|||
return concurrency, delay
|
||||
|
||||
|
||||
class Downloader(object):
|
||||
class Downloader:
|
||||
|
||||
DOWNLOAD_SLOT = 'download_slot'
|
||||
|
||||
|
|
@ -133,6 +133,7 @@ class Downloader(object):
|
|||
return deferred
|
||||
|
||||
def _process_queue(self, spider, slot):
|
||||
from twisted.internet import reactor
|
||||
if slot.latercall and slot.latercall.active():
|
||||
return
|
||||
|
||||
|
|
|
|||
|
|
@ -32,7 +32,6 @@ import re
|
|||
from io import BytesIO
|
||||
from urllib.parse import unquote
|
||||
|
||||
from twisted.internet import reactor
|
||||
from twisted.internet.protocol import ClientCreator, Protocol
|
||||
from twisted.protocols.ftp import CommandFailed, FTPClient
|
||||
|
||||
|
|
@ -81,6 +80,7 @@ class FTPDownloadHandler:
|
|||
return cls(crawler.settings)
|
||||
|
||||
def download_request(self, request, spider):
|
||||
from twisted.internet import reactor
|
||||
parsed_url = urlparse_cached(request)
|
||||
user = request.meta.get("ftp_user", self.default_user)
|
||||
password = request.meta.get("ftp_password", self.default_password)
|
||||
|
|
|
|||
|
|
@ -1,7 +1,5 @@
|
|||
"""Download handlers for http and https schemes
|
||||
"""
|
||||
from twisted.internet import reactor
|
||||
|
||||
from scrapy.utils.misc import create_instance, load_object
|
||||
from scrapy.utils.python import to_unicode
|
||||
|
||||
|
|
@ -26,6 +24,7 @@ class HTTP10DownloadHandler:
|
|||
return factory.deferred
|
||||
|
||||
def _connect(self, factory):
|
||||
from twisted.internet import reactor
|
||||
host, port = to_unicode(factory.host), factory.port
|
||||
if factory.scheme == b'https':
|
||||
client_context_factory = create_instance(
|
||||
|
|
|
|||
|
|
@ -3,11 +3,12 @@
|
|||
import logging
|
||||
import re
|
||||
import warnings
|
||||
from contextlib import suppress
|
||||
from io import BytesIO
|
||||
from time import time
|
||||
from urllib.parse import urldefrag
|
||||
|
||||
from twisted.internet import defer, protocol, reactor
|
||||
from twisted.internet import defer, protocol, ssl
|
||||
from twisted.internet.endpoints import TCP4ClientEndpoint
|
||||
from twisted.internet.error import TimeoutError
|
||||
from twisted.web.client import Agent, HTTPConnectionPool, ResponseDone, ResponseFailed, URI
|
||||
|
|
@ -32,6 +33,7 @@ class HTTP11DownloadHandler:
|
|||
lazy = False
|
||||
|
||||
def __init__(self, settings, crawler=None):
|
||||
from twisted.internet import reactor
|
||||
self._pool = HTTPConnectionPool(reactor, persistent=True)
|
||||
self._pool.maxPersistentPerHost = settings.getint('CONCURRENT_REQUESTS_PER_DOMAIN')
|
||||
self._pool._factory.noisy = False
|
||||
|
|
@ -80,6 +82,7 @@ class HTTP11DownloadHandler:
|
|||
return agent.download_request(request)
|
||||
|
||||
def close(self):
|
||||
from twisted.internet import reactor
|
||||
d = self._pool.closeCachedConnections()
|
||||
# closeCachedConnections will hang on network or server issues, so
|
||||
# we'll manually timeout the deferred.
|
||||
|
|
@ -265,7 +268,7 @@ class ScrapyProxyAgent(Agent):
|
|||
)
|
||||
|
||||
|
||||
class ScrapyAgent(object):
|
||||
class ScrapyAgent:
|
||||
|
||||
_Agent = Agent
|
||||
_ProxyAgent = ScrapyProxyAgent
|
||||
|
|
@ -283,6 +286,7 @@ class ScrapyAgent(object):
|
|||
self._txresponse = None
|
||||
|
||||
def _get_agent(self, request, timeout):
|
||||
from twisted.internet import reactor
|
||||
bindaddress = request.meta.get('bindaddress') or self._bindAddress
|
||||
proxy = request.meta.get('proxy')
|
||||
if proxy:
|
||||
|
|
@ -325,6 +329,7 @@ class ScrapyAgent(object):
|
|||
)
|
||||
|
||||
def download_request(self, request):
|
||||
from twisted.internet import reactor
|
||||
timeout = request.meta.get('download_timeout') or self._connectTimeout
|
||||
agent = self._get_agent(request, timeout)
|
||||
|
||||
|
|
@ -382,7 +387,7 @@ class ScrapyAgent(object):
|
|||
def _cb_bodyready(self, txresponse, request):
|
||||
# deliverBody hangs for responses without body
|
||||
if txresponse.length == 0:
|
||||
return txresponse, b'', None
|
||||
return txresponse, b'', None, None
|
||||
|
||||
maxsize = request.meta.get('download_maxsize', self._maxsize)
|
||||
warnsize = request.meta.get('download_warnsize', self._warnsize)
|
||||
|
|
@ -418,15 +423,16 @@ class ScrapyAgent(object):
|
|||
return d
|
||||
|
||||
def _cb_bodydone(self, result, request, url):
|
||||
txresponse, body, flags = result
|
||||
txresponse, body, flags, certificate = result
|
||||
status = int(txresponse.code)
|
||||
headers = Headers(txresponse.headers.getAllRawHeaders())
|
||||
respcls = responsetypes.from_args(headers=headers, url=url, body=body)
|
||||
return respcls(url=url, status=status, headers=headers, body=body, flags=flags)
|
||||
return respcls(url=url, status=status, headers=headers, body=body,
|
||||
flags=flags, certificate=certificate)
|
||||
|
||||
|
||||
@implementer(IBodyProducer)
|
||||
class _RequestBodyProducer(object):
|
||||
class _RequestBodyProducer:
|
||||
|
||||
def __init__(self, body):
|
||||
self.body = body
|
||||
|
|
@ -456,6 +462,12 @@ class _ResponseReader(protocol.Protocol):
|
|||
self._fail_on_dataloss_warned = False
|
||||
self._reached_warnsize = False
|
||||
self._bytes_received = 0
|
||||
self._certificate = None
|
||||
|
||||
def connectionMade(self):
|
||||
if self._certificate is None:
|
||||
with suppress(AttributeError):
|
||||
self._certificate = ssl.Certificate(self.transport._producer.getPeerCertificate())
|
||||
|
||||
def dataReceived(self, bodyBytes):
|
||||
# This maybe called several times after cancel was called with buffered data.
|
||||
|
|
@ -488,16 +500,16 @@ class _ResponseReader(protocol.Protocol):
|
|||
|
||||
body = self._bodybuf.getvalue()
|
||||
if reason.check(ResponseDone):
|
||||
self._finished.callback((self._txresponse, body, None))
|
||||
self._finished.callback((self._txresponse, body, None, self._certificate))
|
||||
return
|
||||
|
||||
if reason.check(PotentialDataLoss):
|
||||
self._finished.callback((self._txresponse, body, ['partial']))
|
||||
self._finished.callback((self._txresponse, body, ['partial'], self._certificate))
|
||||
return
|
||||
|
||||
if reason.check(ResponseFailed) and any(r.check(_DataLoss) for r in reason.value.reasons):
|
||||
if not self._fail_on_dataloss:
|
||||
self._finished.callback((self._txresponse, body, ['dataloss']))
|
||||
self._finished.callback((self._txresponse, body, ['dataloss'], self._certificate))
|
||||
return
|
||||
|
||||
elif not self._fail_on_dataloss_warned:
|
||||
|
|
|
|||
|
|
@ -89,4 +89,5 @@ class ScrapyClientTLSOptions(ClientTLSOptions):
|
|||
'from host "{}" (exception: {})'.format(
|
||||
self._hostnameASCII, repr(e)))
|
||||
|
||||
|
||||
DEFAULT_CIPHERS = AcceptableCiphers.fromOpenSSLCipherString('DEFAULT')
|
||||
|
|
|
|||
|
|
@ -140,7 +140,7 @@ class ScrapyHTTPClientFactory(HTTPClientFactory):
|
|||
self.headers['Content-Length'] = 0
|
||||
|
||||
def _build_response(self, body, request):
|
||||
request.meta['download_latency'] = self.headers_time-self.start_time
|
||||
request.meta['download_latency'] = self.headers_time - self.start_time
|
||||
status = int(self.status)
|
||||
headers = Headers(self.response_headers)
|
||||
respcls = responsetypes.from_args(headers=headers, url=self._url)
|
||||
|
|
|
|||
|
|
@ -21,7 +21,7 @@ from scrapy.utils.log import logformatter_adapter, failure_to_exc_info
|
|||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class Slot(object):
|
||||
class Slot:
|
||||
|
||||
def __init__(self, start_requests, close_if_idle, nextcall, scheduler):
|
||||
self.closing = False
|
||||
|
|
@ -53,7 +53,7 @@ class Slot(object):
|
|||
self.closing.callback(None)
|
||||
|
||||
|
||||
class ExecutionEngine(object):
|
||||
class ExecutionEngine:
|
||||
|
||||
def __init__(self, crawler, spider_closed_callback):
|
||||
self.crawler = crawler
|
||||
|
|
@ -230,6 +230,7 @@ class ExecutionEngine(object):
|
|||
def _download(self, request, spider):
|
||||
slot = self.slot
|
||||
slot.add_request(request)
|
||||
|
||||
def _on_success(response):
|
||||
assert isinstance(response, (Response, Request))
|
||||
if isinstance(response, Response):
|
||||
|
|
|
|||
|
|
@ -14,7 +14,7 @@ from scrapy.utils.deprecate import ScrapyDeprecationWarning
|
|||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class Scheduler(object):
|
||||
class Scheduler:
|
||||
"""
|
||||
Scrapy Scheduler. It allows to enqueue requests and then get
|
||||
a next request to download. Scheduler is also handling duplication
|
||||
|
|
@ -66,8 +66,7 @@ class Scheduler(object):
|
|||
|
||||
dqclass = load_object(settings['SCHEDULER_DISK_QUEUE'])
|
||||
mqclass = load_object(settings['SCHEDULER_MEMORY_QUEUE'])
|
||||
logunser = settings.getbool('LOG_UNSERIALIZABLE_REQUESTS',
|
||||
settings.getbool('SCHEDULER_DEBUG'))
|
||||
logunser = settings.getbool('SCHEDULER_DEBUG')
|
||||
return cls(dupefilter, jobdir=job_dir(settings), logunser=logunser,
|
||||
stats=crawler.stats, pqclass=pqclass, dqclass=dqclass,
|
||||
mqclass=mqclass, crawler=crawler)
|
||||
|
|
@ -119,7 +118,7 @@ class Scheduler(object):
|
|||
if self.dqs is None:
|
||||
return
|
||||
try:
|
||||
self.dqs.push(request, -request.priority)
|
||||
self.dqs.push(request)
|
||||
except ValueError as e: # non serializable request
|
||||
if self.logunser:
|
||||
msg = ("Unable to serialize request: %(request)s - reason:"
|
||||
|
|
@ -135,35 +134,29 @@ class Scheduler(object):
|
|||
return True
|
||||
|
||||
def _mqpush(self, request):
|
||||
self.mqs.push(request, -request.priority)
|
||||
self.mqs.push(request)
|
||||
|
||||
def _dqpop(self):
|
||||
if self.dqs:
|
||||
return self.dqs.pop()
|
||||
|
||||
def _newmq(self, priority):
|
||||
""" Factory for creating memory queues. """
|
||||
return self.mqclass()
|
||||
|
||||
def _newdq(self, priority):
|
||||
""" Factory for creating disk queues. """
|
||||
path = join(self.dqdir, 'p%s' % (priority, ))
|
||||
return self.dqclass(path)
|
||||
|
||||
def _mq(self):
|
||||
""" Create a new priority queue instance, with in-memory storage """
|
||||
return create_instance(self.pqclass, None, self.crawler, self._newmq,
|
||||
serialize=False)
|
||||
return create_instance(self.pqclass,
|
||||
settings=None,
|
||||
crawler=self.crawler,
|
||||
downstream_queue_cls=self.mqclass,
|
||||
key='')
|
||||
|
||||
def _dq(self):
|
||||
""" Create a new priority queue instance, with disk storage """
|
||||
state = self._read_dqs_state(self.dqdir)
|
||||
q = create_instance(self.pqclass,
|
||||
None,
|
||||
self.crawler,
|
||||
self._newdq,
|
||||
state,
|
||||
serialize=True)
|
||||
settings=None,
|
||||
crawler=self.crawler,
|
||||
downstream_queue_cls=self.dqclass,
|
||||
key=self.dqdir,
|
||||
startprios=state)
|
||||
if q:
|
||||
logger.info("Resuming crawl (%(queuesize)d requests scheduled)",
|
||||
{'queuesize': len(q)}, extra={'spider': self.spider})
|
||||
|
|
|
|||
|
|
@ -16,13 +16,12 @@ from scrapy import signals
|
|||
from scrapy.http import Request, Response
|
||||
from scrapy.item import BaseItem
|
||||
from scrapy.core.spidermw import SpiderMiddlewareManager
|
||||
from scrapy.utils.request import referer_str
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class Slot(object):
|
||||
class Slot:
|
||||
"""Scraper slot (one per running spider)"""
|
||||
|
||||
MIN_RESPONSE_SIZE = 1024
|
||||
|
|
@ -63,7 +62,7 @@ class Slot(object):
|
|||
return self.active_size > self.max_active_size
|
||||
|
||||
|
||||
class Scraper(object):
|
||||
class Scraper:
|
||||
|
||||
def __init__(self, crawler):
|
||||
self.slot = None
|
||||
|
|
@ -158,9 +157,9 @@ class Scraper(object):
|
|||
if isinstance(exc, CloseSpider):
|
||||
self.crawler.engine.close_spider(spider, exc.reason or 'cancelled')
|
||||
return
|
||||
logger.error(
|
||||
"Spider error processing %(request)s (referer: %(referer)s)",
|
||||
{'request': request, 'referer': referer_str(request)},
|
||||
logkws = self.logformatter.spider_error(_failure, request, response, spider)
|
||||
logger.log(
|
||||
*logformatter_adapter(logkws),
|
||||
exc_info=failure_to_exc_info(_failure),
|
||||
extra={'spider': spider}
|
||||
)
|
||||
|
|
@ -208,16 +207,21 @@ class Scraper(object):
|
|||
"""
|
||||
if isinstance(download_failure, Failure) and not download_failure.check(IgnoreRequest):
|
||||
if download_failure.frames:
|
||||
logger.error('Error downloading %(request)s',
|
||||
{'request': request},
|
||||
exc_info=failure_to_exc_info(download_failure),
|
||||
extra={'spider': spider})
|
||||
logkws = self.logformatter.download_error(download_failure, request, spider)
|
||||
logger.log(
|
||||
*logformatter_adapter(logkws),
|
||||
extra={'spider': spider},
|
||||
exc_info=failure_to_exc_info(download_failure),
|
||||
)
|
||||
else:
|
||||
errmsg = download_failure.getErrorMessage()
|
||||
if errmsg:
|
||||
logger.error('Error downloading %(request)s: %(errmsg)s',
|
||||
{'request': request, 'errmsg': errmsg},
|
||||
extra={'spider': spider})
|
||||
logkws = self.logformatter.download_error(
|
||||
download_failure, request, spider, errmsg)
|
||||
logger.log(
|
||||
*logformatter_adapter(logkws),
|
||||
extra={'spider': spider},
|
||||
)
|
||||
|
||||
if spider_failure is not download_failure:
|
||||
return spider_failure
|
||||
|
|
@ -236,7 +240,7 @@ class Scraper(object):
|
|||
signal=signals.item_dropped, item=item, response=response,
|
||||
spider=spider, exception=output.value)
|
||||
else:
|
||||
logkws = self.logformatter.error(item, ex, response, spider)
|
||||
logkws = self.logformatter.item_error(item, ex, response, spider)
|
||||
logger.log(*logformatter_adapter(logkws), extra={'spider': spider},
|
||||
exc_info=failure_to_exc_info(output))
|
||||
return self.signals.send_catch_log_deferred(
|
||||
|
|
|
|||
|
|
@ -82,7 +82,7 @@ class SpiderMiddlewareManager(MiddlewareManager):
|
|||
if _isiterable(result):
|
||||
# stop exception handling by handing control over to the
|
||||
# process_spider_output chain if an iterable has been returned
|
||||
return process_spider_output(result, method_index+1)
|
||||
return process_spider_output(result, method_index + 1)
|
||||
elif result is None:
|
||||
continue
|
||||
else:
|
||||
|
|
@ -103,12 +103,12 @@ class SpiderMiddlewareManager(MiddlewareManager):
|
|||
# might fail directly if the output value is not a generator
|
||||
result = method(response=response, result=result, spider=spider)
|
||||
except Exception as ex:
|
||||
exception_result = process_spider_exception(Failure(ex), method_index+1)
|
||||
exception_result = process_spider_exception(Failure(ex), method_index + 1)
|
||||
if isinstance(exception_result, Failure):
|
||||
raise
|
||||
return exception_result
|
||||
if _isiterable(result):
|
||||
result = _evaluate_iterable(result, method_index+1, recovered)
|
||||
result = _evaluate_iterable(result, method_index + 1, recovered)
|
||||
else:
|
||||
msg = "Middleware {} must return an iterable, got {}"
|
||||
raise _InvalidOutput(msg.format(_fname(method), type(result)))
|
||||
|
|
|
|||
|
|
@ -68,17 +68,6 @@ class Crawler:
|
|||
self.spider = None
|
||||
self.engine = None
|
||||
|
||||
@property
|
||||
def spiders(self):
|
||||
if not hasattr(self, '_spiders'):
|
||||
warnings.warn("Crawler.spiders is deprecated, use "
|
||||
"CrawlerRunner.spider_loader or instantiate "
|
||||
"scrapy.spiderloader.SpiderLoader with your "
|
||||
"settings.",
|
||||
category=ScrapyDeprecationWarning, stacklevel=2)
|
||||
self._spiders = _get_spider_loader(self.settings.frozencopy())
|
||||
return self._spiders
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def crawl(self, *args, **kwargs):
|
||||
assert not self.crawling, "Crawling already taking place"
|
||||
|
|
@ -130,11 +119,27 @@ class CrawlerRunner:
|
|||
":meth:`crawl` and managed by this class."
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _get_spider_loader(settings):
|
||||
""" Get SpiderLoader instance from settings """
|
||||
cls_path = settings.get('SPIDER_LOADER_CLASS')
|
||||
loader_cls = load_object(cls_path)
|
||||
try:
|
||||
verifyClass(ISpiderLoader, loader_cls)
|
||||
except DoesNotImplement:
|
||||
warnings.warn(
|
||||
'SPIDER_LOADER_CLASS (previously named SPIDER_MANAGER_CLASS) does '
|
||||
'not fully implement scrapy.interfaces.ISpiderLoader interface. '
|
||||
'Please add all missing methods to avoid unexpected runtime errors.',
|
||||
category=ScrapyDeprecationWarning, stacklevel=2
|
||||
)
|
||||
return loader_cls.from_settings(settings.frozencopy())
|
||||
|
||||
def __init__(self, settings=None):
|
||||
if isinstance(settings, dict) or settings is None:
|
||||
settings = Settings(settings)
|
||||
self.settings = settings
|
||||
self.spider_loader = _get_spider_loader(settings)
|
||||
self.spider_loader = self._get_spider_loader(settings)
|
||||
self._crawlers = set()
|
||||
self._active = set()
|
||||
self.bootstrap_failed = False
|
||||
|
|
@ -327,19 +332,3 @@ class CrawlerProcess(CrawlerRunner):
|
|||
if self.settings.get("TWISTED_REACTOR"):
|
||||
install_reactor(self.settings["TWISTED_REACTOR"])
|
||||
super()._handle_twisted_reactor()
|
||||
|
||||
|
||||
def _get_spider_loader(settings):
|
||||
""" Get SpiderLoader instance from settings """
|
||||
cls_path = settings.get('SPIDER_LOADER_CLASS')
|
||||
loader_cls = load_object(cls_path)
|
||||
try:
|
||||
verifyClass(ISpiderLoader, loader_cls)
|
||||
except DoesNotImplement:
|
||||
warnings.warn(
|
||||
'SPIDER_LOADER_CLASS (previously named SPIDER_MANAGER_CLASS) does '
|
||||
'not fully implement scrapy.interfaces.ISpiderLoader interface. '
|
||||
'Please add all missing methods to avoid unexpected runtime errors.',
|
||||
category=ScrapyDeprecationWarning, stacklevel=2
|
||||
)
|
||||
return loader_cls.from_settings(settings.frozencopy())
|
||||
|
|
|
|||
|
|
@ -11,7 +11,7 @@ from scrapy.http import HtmlResponse
|
|||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class AjaxCrawlMiddleware(object):
|
||||
class AjaxCrawlMiddleware:
|
||||
"""
|
||||
Handle 'AJAX crawlable' pages marked as crawlable via meta tag.
|
||||
For more info see https://developers.google.com/webmasters/ajax-crawling/docs/getting-started.
|
||||
|
|
@ -47,7 +47,7 @@ class AjaxCrawlMiddleware(object):
|
|||
return response
|
||||
|
||||
# scrapy already handles #! links properly
|
||||
ajax_crawl_request = request.replace(url=request.url+'#!')
|
||||
ajax_crawl_request = request.replace(url=request.url + '#!')
|
||||
logger.debug("Downloading AJAX crawlable %(ajax_crawl_request)s instead of %(request)s",
|
||||
{'ajax_crawl_request': ajax_crawl_request, 'request': request},
|
||||
extra={'spider': spider})
|
||||
|
|
|
|||
|
|
@ -1,21 +0,0 @@
|
|||
import warnings
|
||||
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
from scrapy.utils.http import decode_chunked_transfer
|
||||
|
||||
|
||||
warnings.warn("Module `scrapy.downloadermiddlewares.chunked` is deprecated, "
|
||||
"chunked transfers are supported by default.",
|
||||
ScrapyDeprecationWarning, stacklevel=2)
|
||||
|
||||
|
||||
class ChunkedTransferMiddleware(object):
|
||||
"""This middleware adds support for chunked transfer encoding, as
|
||||
documented in: https://en.wikipedia.org/wiki/Chunked_transfer_encoding
|
||||
"""
|
||||
|
||||
def process_response(self, request, response, spider):
|
||||
if response.headers.get('Transfer-Encoding') == 'chunked':
|
||||
body = decode_chunked_transfer(response.body)
|
||||
return response.replace(body=body)
|
||||
return response
|
||||
|
|
@ -10,7 +10,7 @@ from scrapy.utils.python import to_unicode
|
|||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class CookiesMiddleware(object):
|
||||
class CookiesMiddleware:
|
||||
"""This middleware enables working with sites that need cookies"""
|
||||
|
||||
def __init__(self, debug=False):
|
||||
|
|
|
|||
|
|
@ -16,7 +16,7 @@ from scrapy.responsetypes import responsetypes
|
|||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class DecompressionMiddleware(object):
|
||||
class DecompressionMiddleware:
|
||||
""" This middleware tries to recognise and extract the possibly compressed
|
||||
responses that may arrive. """
|
||||
|
||||
|
|
|
|||
|
|
@ -7,7 +7,7 @@ See documentation in docs/topics/downloader-middleware.rst
|
|||
from scrapy.utils.python import without_none_values
|
||||
|
||||
|
||||
class DefaultHeadersMiddleware(object):
|
||||
class DefaultHeadersMiddleware:
|
||||
|
||||
def __init__(self, headers):
|
||||
self._headers = headers
|
||||
|
|
|
|||
|
|
@ -7,7 +7,7 @@ See documentation in docs/topics/downloader-middleware.rst
|
|||
from scrapy import signals
|
||||
|
||||
|
||||
class DownloadTimeoutMiddleware(object):
|
||||
class DownloadTimeoutMiddleware:
|
||||
|
||||
def __init__(self, timeout=180):
|
||||
self._timeout = timeout
|
||||
|
|
|
|||
|
|
@ -9,7 +9,7 @@ from w3lib.http import basic_auth_header
|
|||
from scrapy import signals
|
||||
|
||||
|
||||
class HttpAuthMiddleware(object):
|
||||
class HttpAuthMiddleware:
|
||||
"""Set Basic HTTP Authorization header
|
||||
(http_user and http_pass spider class attributes)"""
|
||||
|
||||
|
|
|
|||
|
|
@ -17,7 +17,7 @@ from scrapy.exceptions import IgnoreRequest, NotConfigured
|
|||
from scrapy.utils.misc import load_object
|
||||
|
||||
|
||||
class HttpCacheMiddleware(object):
|
||||
class HttpCacheMiddleware:
|
||||
|
||||
DOWNLOAD_EXCEPTIONS = (defer.TimeoutError, TimeoutError, DNSLookupError,
|
||||
ConnectionRefusedError, ConnectionDone, ConnectError,
|
||||
|
|
|
|||
|
|
@ -15,7 +15,7 @@ except ImportError:
|
|||
pass
|
||||
|
||||
|
||||
class HttpCompressionMiddleware(object):
|
||||
class HttpCompressionMiddleware:
|
||||
"""This middleware allows compressed (gzip, deflate) traffic to be
|
||||
sent/received from web sites"""
|
||||
@classmethod
|
||||
|
|
|
|||
|
|
@ -7,7 +7,7 @@ from scrapy.utils.httpobj import urlparse_cached
|
|||
from scrapy.utils.python import to_bytes
|
||||
|
||||
|
||||
class HttpProxyMiddleware(object):
|
||||
class HttpProxyMiddleware:
|
||||
|
||||
def __init__(self, auth_encoding='latin-1'):
|
||||
self.auth_encoding = auth_encoding
|
||||
|
|
|
|||
|
|
@ -11,7 +11,7 @@ from scrapy.exceptions import IgnoreRequest, NotConfigured
|
|||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class BaseRedirectMiddleware(object):
|
||||
class BaseRedirectMiddleware:
|
||||
|
||||
enabled_setting = 'REDIRECT_ENABLED'
|
||||
|
||||
|
|
@ -93,8 +93,7 @@ class MetaRefreshMiddleware(BaseRedirectMiddleware):
|
|||
def __init__(self, settings):
|
||||
super(MetaRefreshMiddleware, self).__init__(settings)
|
||||
self._ignore_tags = settings.getlist('METAREFRESH_IGNORE_TAGS')
|
||||
self._maxdelay = settings.getint('REDIRECT_MAX_METAREFRESH_DELAY',
|
||||
settings.getint('METAREFRESH_MAXDELAY'))
|
||||
self._maxdelay = settings.getint('METAREFRESH_MAXDELAY')
|
||||
|
||||
def process_response(self, request, response, spider):
|
||||
if request.meta.get('dont_redirect', False) or request.method == 'HEAD' or \
|
||||
|
|
|
|||
|
|
@ -25,7 +25,7 @@ from scrapy.utils.python import global_object_name
|
|||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class RetryMiddleware(object):
|
||||
class RetryMiddleware:
|
||||
|
||||
# IOError is raised by the HttpCompression middleware when trying to
|
||||
# decompress an empty response
|
||||
|
|
|
|||
|
|
@ -16,7 +16,7 @@ from scrapy.utils.misc import load_object
|
|||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class RobotsTxtMiddleware(object):
|
||||
class RobotsTxtMiddleware:
|
||||
DOWNLOAD_PRIORITY = 1000
|
||||
|
||||
def __init__(self, crawler):
|
||||
|
|
|
|||
|
|
@ -4,7 +4,7 @@ from scrapy.utils.response import response_httprepr
|
|||
from scrapy.utils.python import global_object_name
|
||||
|
||||
|
||||
class DownloaderStats(object):
|
||||
class DownloaderStats:
|
||||
|
||||
def __init__(self, stats):
|
||||
self.stats = stats
|
||||
|
|
|
|||
|
|
@ -3,7 +3,7 @@
|
|||
from scrapy import signals
|
||||
|
||||
|
||||
class UserAgentMiddleware(object):
|
||||
class UserAgentMiddleware:
|
||||
"""This middleware allows spiders to override the user_agent"""
|
||||
|
||||
def __init__(self, user_agent='Scrapy'):
|
||||
|
|
|
|||
|
|
@ -5,7 +5,7 @@ from scrapy.utils.job import job_dir
|
|||
from scrapy.utils.request import referer_str, request_fingerprint
|
||||
|
||||
|
||||
class BaseDupeFilter(object):
|
||||
class BaseDupeFilter:
|
||||
|
||||
@classmethod
|
||||
def from_settings(cls, settings):
|
||||
|
|
|
|||
|
|
@ -21,9 +21,9 @@ __all__ = ['BaseItemExporter', 'PprintItemExporter', 'PickleItemExporter',
|
|||
'JsonItemExporter', 'MarshalItemExporter']
|
||||
|
||||
|
||||
class BaseItemExporter(object):
|
||||
class BaseItemExporter:
|
||||
|
||||
def __init__(self, dont_fail=False, **kwargs):
|
||||
def __init__(self, *, dont_fail=False, **kwargs):
|
||||
self._kwargs = kwargs
|
||||
self._configure(kwargs, dont_fail=dont_fail)
|
||||
|
||||
|
|
@ -173,12 +173,12 @@ class XmlItemExporter(BaseItemExporter):
|
|||
if hasattr(serialized_value, 'items'):
|
||||
self._beautify_newline()
|
||||
for subname, value in serialized_value.items():
|
||||
self._export_xml_field(subname, value, depth=depth+1)
|
||||
self._export_xml_field(subname, value, depth=depth + 1)
|
||||
self._beautify_indent(depth=depth)
|
||||
elif is_listlike(serialized_value):
|
||||
self._beautify_newline()
|
||||
for value in serialized_value:
|
||||
self._export_xml_field('value', value, depth=depth+1)
|
||||
self._export_xml_field('value', value, depth=depth + 1)
|
||||
self._beautify_indent(depth=depth)
|
||||
elif isinstance(serialized_value, str):
|
||||
self.xg.characters(serialized_value)
|
||||
|
|
|
|||
|
|
@ -6,13 +6,11 @@ See documentation in docs/topics/extensions.rst
|
|||
|
||||
from collections import defaultdict
|
||||
|
||||
from twisted.internet import reactor
|
||||
|
||||
from scrapy import signals
|
||||
from scrapy.exceptions import NotConfigured
|
||||
|
||||
|
||||
class CloseSpider(object):
|
||||
class CloseSpider:
|
||||
|
||||
def __init__(self, crawler):
|
||||
self.crawler = crawler
|
||||
|
|
@ -54,6 +52,7 @@ class CloseSpider(object):
|
|||
self.crawler.engine.close_spider(spider, 'closespider_pagecount')
|
||||
|
||||
def spider_opened(self, spider):
|
||||
from twisted.internet import reactor
|
||||
self.task = reactor.callLater(self.close_on['timeout'],
|
||||
self.crawler.engine.close_spider, spider,
|
||||
reason='closespider_timeout')
|
||||
|
|
|
|||
|
|
@ -6,7 +6,7 @@ from datetime import datetime
|
|||
from scrapy import signals
|
||||
|
||||
|
||||
class CoreStats(object):
|
||||
class CoreStats:
|
||||
|
||||
def __init__(self, stats):
|
||||
self.stats = stats
|
||||
|
|
|
|||
|
|
@ -17,7 +17,7 @@ from scrapy.utils.trackref import format_live_refs
|
|||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class StackTraceDump(object):
|
||||
class StackTraceDump:
|
||||
|
||||
def __init__(self, crawler=None):
|
||||
self.crawler = crawler
|
||||
|
|
@ -52,7 +52,7 @@ class StackTraceDump(object):
|
|||
return dumps
|
||||
|
||||
|
||||
class Debugger(object):
|
||||
class Debugger:
|
||||
def __init__(self):
|
||||
try:
|
||||
signal.signal(signal.SIGUSR2, self._enter_debugger)
|
||||
|
|
|
|||
|
|
@ -4,24 +4,27 @@ Feed Exports extension
|
|||
See documentation in docs/topics/feed-exports.rst
|
||||
"""
|
||||
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
import logging
|
||||
from tempfile import NamedTemporaryFile
|
||||
import warnings
|
||||
from datetime import datetime
|
||||
from urllib.parse import urlparse, unquote
|
||||
from tempfile import NamedTemporaryFile
|
||||
from urllib.parse import unquote, urlparse
|
||||
|
||||
from zope.interface import Interface, implementer
|
||||
from twisted.internet import defer, threads
|
||||
from w3lib.url import file_uri_to_path
|
||||
from zope.interface import implementer, Interface
|
||||
|
||||
from scrapy import signals
|
||||
from scrapy.utils.ftp import ftp_store_file
|
||||
from scrapy.exceptions import NotConfigured
|
||||
from scrapy.utils.misc import create_instance, load_object
|
||||
from scrapy.utils.log import failure_to_exc_info
|
||||
from scrapy.utils.python import without_none_values
|
||||
from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning
|
||||
from scrapy.utils.boto import is_botocore
|
||||
from scrapy.utils.conf import feed_complete_default_values_from_settings
|
||||
from scrapy.utils.ftp import ftp_store_file
|
||||
from scrapy.utils.log import failure_to_exc_info
|
||||
from scrapy.utils.misc import create_instance, load_object
|
||||
from scrapy.utils.python import without_none_values
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
|
@ -41,7 +44,7 @@ class IFeedStorage(Interface):
|
|||
|
||||
|
||||
@implementer(IFeedStorage)
|
||||
class BlockingFeedStorage(object):
|
||||
class BlockingFeedStorage:
|
||||
|
||||
def open(self, spider):
|
||||
path = spider.crawler.settings['FEED_TEMPDIR']
|
||||
|
|
@ -58,7 +61,7 @@ class BlockingFeedStorage(object):
|
|||
|
||||
|
||||
@implementer(IFeedStorage)
|
||||
class StdoutFeedStorage(object):
|
||||
class StdoutFeedStorage:
|
||||
|
||||
def __init__(self, uri, _stdout=None):
|
||||
if not _stdout:
|
||||
|
|
@ -73,7 +76,7 @@ class StdoutFeedStorage(object):
|
|||
|
||||
|
||||
@implementer(IFeedStorage)
|
||||
class FileFeedStorage(object):
|
||||
class FileFeedStorage:
|
||||
|
||||
def __init__(self, uri):
|
||||
self.path = file_uri_to_path(uri)
|
||||
|
|
@ -98,8 +101,6 @@ class S3FeedStorage(BlockingFeedStorage):
|
|||
from scrapy.utils.project import get_project_settings
|
||||
settings = get_project_settings()
|
||||
if 'AWS_ACCESS_KEY_ID' in settings or 'AWS_SECRET_ACCESS_KEY' in settings:
|
||||
import warnings
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
warnings.warn(
|
||||
"Initialising `scrapy.extensions.feedexport.S3FeedStorage` "
|
||||
"without AWS keys is deprecated. Please supply credentials or "
|
||||
|
|
@ -204,88 +205,117 @@ class FTPFeedStorage(BlockingFeedStorage):
|
|||
)
|
||||
|
||||
|
||||
class SpiderSlot(object):
|
||||
def __init__(self, file, exporter, storage, uri):
|
||||
class _FeedSlot:
|
||||
def __init__(self, file, exporter, storage, uri, format, store_empty):
|
||||
self.file = file
|
||||
self.exporter = exporter
|
||||
self.storage = storage
|
||||
# feed params
|
||||
self.uri = uri
|
||||
self.format = format
|
||||
self.store_empty = store_empty
|
||||
# flags
|
||||
self.itemcount = 0
|
||||
|
||||
|
||||
class FeedExporter(object):
|
||||
|
||||
def __init__(self, settings):
|
||||
self.settings = settings
|
||||
if not settings['FEED_URI']:
|
||||
raise NotConfigured
|
||||
self.urifmt = str(settings['FEED_URI'])
|
||||
self.format = settings['FEED_FORMAT'].lower()
|
||||
self.export_encoding = settings['FEED_EXPORT_ENCODING']
|
||||
self.storages = self._load_components('FEED_STORAGES')
|
||||
self.exporters = self._load_components('FEED_EXPORTERS')
|
||||
if not self._storage_supported(self.urifmt):
|
||||
raise NotConfigured
|
||||
if not self._exporter_supported(self.format):
|
||||
raise NotConfigured
|
||||
self.store_empty = settings.getbool('FEED_STORE_EMPTY')
|
||||
self._exporting = False
|
||||
self.export_fields = settings.getlist('FEED_EXPORT_FIELDS') or None
|
||||
self.indent = None
|
||||
if settings.get('FEED_EXPORT_INDENT') is not None:
|
||||
self.indent = settings.getint('FEED_EXPORT_INDENT')
|
||||
uripar = settings['FEED_URI_PARAMS']
|
||||
self._uripar = load_object(uripar) if uripar else lambda x, y: None
|
||||
|
||||
def start_exporting(self):
|
||||
if not self._exporting:
|
||||
self.exporter.start_exporting()
|
||||
self._exporting = True
|
||||
|
||||
def finish_exporting(self):
|
||||
if self._exporting:
|
||||
self.exporter.finish_exporting()
|
||||
self._exporting = False
|
||||
|
||||
|
||||
class FeedExporter:
|
||||
|
||||
@classmethod
|
||||
def from_crawler(cls, crawler):
|
||||
o = cls(crawler.settings)
|
||||
o.crawler = crawler
|
||||
crawler.signals.connect(o.open_spider, signals.spider_opened)
|
||||
crawler.signals.connect(o.close_spider, signals.spider_closed)
|
||||
crawler.signals.connect(o.item_scraped, signals.item_scraped)
|
||||
return o
|
||||
exporter = cls(crawler)
|
||||
crawler.signals.connect(exporter.open_spider, signals.spider_opened)
|
||||
crawler.signals.connect(exporter.close_spider, signals.spider_closed)
|
||||
crawler.signals.connect(exporter.item_scraped, signals.item_scraped)
|
||||
return exporter
|
||||
|
||||
def __init__(self, crawler):
|
||||
self.crawler = crawler
|
||||
self.settings = crawler.settings
|
||||
self.feeds = {}
|
||||
self.slots = []
|
||||
|
||||
if not self.settings['FEEDS'] and not self.settings['FEED_URI']:
|
||||
raise NotConfigured
|
||||
|
||||
# Begin: Backward compatibility for FEED_URI and FEED_FORMAT settings
|
||||
if self.settings['FEED_URI']:
|
||||
warnings.warn(
|
||||
'The `FEED_URI` and `FEED_FORMAT` settings have been deprecated in favor of '
|
||||
'the `FEEDS` setting. Please see the `FEEDS` setting docs for more details',
|
||||
category=ScrapyDeprecationWarning, stacklevel=2,
|
||||
)
|
||||
uri = str(self.settings['FEED_URI']) # handle pathlib.Path objects
|
||||
feed = {'format': self.settings.get('FEED_FORMAT', 'jsonlines')}
|
||||
self.feeds[uri] = feed_complete_default_values_from_settings(feed, self.settings)
|
||||
# End: Backward compatibility for FEED_URI and FEED_FORMAT settings
|
||||
|
||||
# 'FEEDS' setting takes precedence over 'FEED_URI'
|
||||
for uri, feed in self.settings.getdict('FEEDS').items():
|
||||
uri = str(uri) # handle pathlib.Path objects
|
||||
self.feeds[uri] = feed_complete_default_values_from_settings(feed, self.settings)
|
||||
|
||||
self.storages = self._load_components('FEED_STORAGES')
|
||||
self.exporters = self._load_components('FEED_EXPORTERS')
|
||||
for uri, feed in self.feeds.items():
|
||||
if not self._storage_supported(uri):
|
||||
raise NotConfigured
|
||||
if not self._exporter_supported(feed['format']):
|
||||
raise NotConfigured
|
||||
|
||||
def open_spider(self, spider):
|
||||
uri = self.urifmt % self._get_uri_params(spider)
|
||||
storage = self._get_storage(uri)
|
||||
file = storage.open(spider)
|
||||
exporter = self._get_exporter(file, fields_to_export=self.export_fields,
|
||||
encoding=self.export_encoding, indent=self.indent)
|
||||
if self.store_empty:
|
||||
exporter.start_exporting()
|
||||
self._exporting = True
|
||||
self.slot = SpiderSlot(file, exporter, storage, uri)
|
||||
for uri, feed in self.feeds.items():
|
||||
uri = uri % self._get_uri_params(spider, feed['uri_params'])
|
||||
storage = self._get_storage(uri)
|
||||
file = storage.open(spider)
|
||||
exporter = self._get_exporter(
|
||||
file=file,
|
||||
format=feed['format'],
|
||||
fields_to_export=feed['fields'],
|
||||
encoding=feed['encoding'],
|
||||
indent=feed['indent'],
|
||||
)
|
||||
slot = _FeedSlot(file, exporter, storage, uri, feed['format'], feed['store_empty'])
|
||||
self.slots.append(slot)
|
||||
if slot.store_empty:
|
||||
slot.start_exporting()
|
||||
|
||||
def close_spider(self, spider):
|
||||
slot = self.slot
|
||||
if not slot.itemcount and not self.store_empty:
|
||||
# We need to call slot.storage.store nonetheless to get the file
|
||||
# properly closed.
|
||||
return defer.maybeDeferred(slot.storage.store, slot.file)
|
||||
if self._exporting:
|
||||
slot.exporter.finish_exporting()
|
||||
self._exporting = False
|
||||
logfmt = "%s %%(format)s feed (%%(itemcount)d items) in: %%(uri)s"
|
||||
log_args = {'format': self.format,
|
||||
'itemcount': slot.itemcount,
|
||||
'uri': slot.uri}
|
||||
d = defer.maybeDeferred(slot.storage.store, slot.file)
|
||||
d.addCallback(lambda _: logger.info(logfmt % "Stored", log_args,
|
||||
extra={'spider': spider}))
|
||||
d.addErrback(lambda f: logger.error(logfmt % "Error storing", log_args,
|
||||
exc_info=failure_to_exc_info(f),
|
||||
extra={'spider': spider}))
|
||||
return d
|
||||
deferred_list = []
|
||||
for slot in self.slots:
|
||||
if not slot.itemcount and not slot.store_empty:
|
||||
# We need to call slot.storage.store nonetheless to get the file
|
||||
# properly closed.
|
||||
return defer.maybeDeferred(slot.storage.store, slot.file)
|
||||
slot.finish_exporting()
|
||||
logfmt = "%s %%(format)s feed (%%(itemcount)d items) in: %%(uri)s"
|
||||
log_args = {'format': slot.format,
|
||||
'itemcount': slot.itemcount,
|
||||
'uri': slot.uri}
|
||||
d = defer.maybeDeferred(slot.storage.store, slot.file)
|
||||
d.addCallback(lambda _: logger.info(logfmt % "Stored", log_args,
|
||||
extra={'spider': spider}))
|
||||
d.addErrback(lambda f: logger.error(logfmt % "Error storing", log_args,
|
||||
exc_info=failure_to_exc_info(f),
|
||||
extra={'spider': spider}))
|
||||
deferred_list.append(d)
|
||||
return defer.DeferredList(deferred_list) if deferred_list else None
|
||||
|
||||
def item_scraped(self, item, spider):
|
||||
slot = self.slot
|
||||
if not self._exporting:
|
||||
slot.exporter.start_exporting()
|
||||
self._exporting = True
|
||||
slot.exporter.export_item(item)
|
||||
slot.itemcount += 1
|
||||
return item
|
||||
for slot in self.slots:
|
||||
slot.start_exporting()
|
||||
slot.exporter.export_item(item)
|
||||
slot.itemcount += 1
|
||||
|
||||
def _load_components(self, setting_prefix):
|
||||
conf = without_none_values(self.settings.getwithbase(setting_prefix))
|
||||
|
|
@ -321,17 +351,18 @@ class FeedExporter(object):
|
|||
objcls, self.settings, getattr(self, 'crawler', None),
|
||||
*args, **kwargs)
|
||||
|
||||
def _get_exporter(self, *args, **kwargs):
|
||||
return self._get_instance(self.exporters[self.format], *args, **kwargs)
|
||||
def _get_exporter(self, file, format, *args, **kwargs):
|
||||
return self._get_instance(self.exporters[format], file, *args, **kwargs)
|
||||
|
||||
def _get_storage(self, uri):
|
||||
return self._get_instance(self.storages[urlparse(uri).scheme], uri)
|
||||
|
||||
def _get_uri_params(self, spider):
|
||||
def _get_uri_params(self, spider, uri_params):
|
||||
params = {}
|
||||
for k in dir(spider):
|
||||
params[k] = getattr(spider, k)
|
||||
ts = datetime.utcnow().replace(microsecond=0).isoformat().replace(':', '-')
|
||||
params['time'] = ts
|
||||
self._uripar(params, spider)
|
||||
uripar_function = load_object(uri_params) if uri_params else lambda x, y: None
|
||||
uripar_function(params, spider)
|
||||
return params
|
||||
|
|
|
|||
|
|
@ -20,7 +20,7 @@ from scrapy.utils.request import request_fingerprint
|
|||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class DummyPolicy(object):
|
||||
class DummyPolicy:
|
||||
|
||||
def __init__(self, settings):
|
||||
self.ignore_schemes = settings.getlist('HTTPCACHE_IGNORE_SCHEMES')
|
||||
|
|
@ -39,7 +39,7 @@ class DummyPolicy(object):
|
|||
return True
|
||||
|
||||
|
||||
class RFC2616Policy(object):
|
||||
class RFC2616Policy:
|
||||
|
||||
MAXAGE = 3600 * 24 * 365 # one year
|
||||
|
||||
|
|
@ -213,7 +213,7 @@ class RFC2616Policy(object):
|
|||
return currentage
|
||||
|
||||
|
||||
class DbmCacheStorage(object):
|
||||
class DbmCacheStorage:
|
||||
|
||||
def __init__(self, settings):
|
||||
self.cachedir = data_path(settings['HTTPCACHE_DIR'], createdir=True)
|
||||
|
|
@ -270,7 +270,7 @@ class DbmCacheStorage(object):
|
|||
return request_fingerprint(request)
|
||||
|
||||
|
||||
class FilesystemCacheStorage(object):
|
||||
class FilesystemCacheStorage:
|
||||
|
||||
def __init__(self, settings):
|
||||
self.cachedir = data_path(settings['HTTPCACHE_DIR'])
|
||||
|
|
|
|||
|
|
@ -8,7 +8,7 @@ from scrapy import signals
|
|||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class LogStats(object):
|
||||
class LogStats:
|
||||
"""Log basic scraping stats periodically"""
|
||||
|
||||
def __init__(self, stats, interval=60.0):
|
||||
|
|
|
|||
|
|
@ -11,7 +11,7 @@ from scrapy.exceptions import NotConfigured
|
|||
from scrapy.utils.trackref import live_refs
|
||||
|
||||
|
||||
class MemoryDebugger(object):
|
||||
class MemoryDebugger:
|
||||
|
||||
def __init__(self, stats):
|
||||
self.stats = stats
|
||||
|
|
|
|||
|
|
@ -19,7 +19,7 @@ from scrapy.utils.engine import get_engine_status
|
|||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class MemoryUsage(object):
|
||||
class MemoryUsage:
|
||||
|
||||
def __init__(self, crawler):
|
||||
if not crawler.settings.getbool('MEMUSAGE_ENABLED'):
|
||||
|
|
@ -47,7 +47,7 @@ class MemoryUsage(object):
|
|||
def get_virtual_size(self):
|
||||
size = self.resource.getrusage(self.resource.RUSAGE_SELF).ru_maxrss
|
||||
if sys.platform != 'darwin':
|
||||
# on Mac OS X ru_maxrss is in bytes, on Linux it is in KB
|
||||
# on macOS ru_maxrss is in bytes, on Linux it is in KB
|
||||
size *= 1024
|
||||
return size
|
||||
|
||||
|
|
|
|||
|
|
@ -6,7 +6,7 @@ from scrapy.exceptions import NotConfigured
|
|||
from scrapy.utils.job import job_dir
|
||||
|
||||
|
||||
class SpiderState(object):
|
||||
class SpiderState:
|
||||
"""Store and load spider state during a scraping job"""
|
||||
|
||||
def __init__(self, jobdir=None):
|
||||
|
|
|
|||
|
|
@ -8,7 +8,7 @@ from scrapy import signals
|
|||
from scrapy.mail import MailSender
|
||||
from scrapy.exceptions import NotConfigured
|
||||
|
||||
class StatsMailer(object):
|
||||
class StatsMailer:
|
||||
|
||||
def __init__(self, stats, recipients, mail):
|
||||
self.stats = stats
|
||||
|
|
|
|||
|
|
@ -26,11 +26,6 @@ from scrapy.utils.engine import print_engine_status
|
|||
from scrapy.utils.reactor import listen_tcp
|
||||
from scrapy.utils.decorators import defers
|
||||
|
||||
try:
|
||||
import guppy
|
||||
hpy = guppy.hpy()
|
||||
except ImportError:
|
||||
hpy = None
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
|
@ -110,7 +105,6 @@ class TelnetConsole(protocol.ServerFactory):
|
|||
'est': lambda: print_engine_status(self.crawler.engine),
|
||||
'p': pprint.pprint,
|
||||
'prefs': print_live_refs,
|
||||
'hpy': hpy,
|
||||
'help': "This is Scrapy telnet console. For more info see: "
|
||||
"https://docs.scrapy.org/en/latest/topics/telnetconsole.html",
|
||||
}
|
||||
|
|
|
|||
|
|
@ -6,7 +6,7 @@ from scrapy import signals
|
|||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class AutoThrottle(object):
|
||||
class AutoThrottle:
|
||||
|
||||
def __init__(self, crawler):
|
||||
self.crawler = crawler
|
||||
|
|
|
|||
|
|
@ -5,7 +5,7 @@ from scrapy.utils.httpobj import urlparse_cached
|
|||
from scrapy.utils.python import to_unicode
|
||||
|
||||
|
||||
class CookieJar(object):
|
||||
class CookieJar:
|
||||
def __init__(self, policy=None, check_expired_frequency=10000):
|
||||
self.policy = policy or DefaultCookiePolicy()
|
||||
self.jar = _CookieJar(self.policy)
|
||||
|
|
@ -100,7 +100,7 @@ def potential_domain_matches(domain):
|
|||
return matches + ['.' + d for d in matches]
|
||||
|
||||
|
||||
class _DummyLock(object):
|
||||
class _DummyLock:
|
||||
def acquire(self):
|
||||
pass
|
||||
|
||||
|
|
@ -108,7 +108,7 @@ class _DummyLock(object):
|
|||
pass
|
||||
|
||||
|
||||
class WrappedRequest(object):
|
||||
class WrappedRequest:
|
||||
"""Wraps a scrapy Request class with methods defined by urllib2.Request class to interact with CookieJar class
|
||||
|
||||
see http://docs.python.org/library/urllib2.html#urllib2.Request
|
||||
|
|
@ -178,7 +178,7 @@ class WrappedRequest(object):
|
|||
self.request.headers.appendlist(name, value)
|
||||
|
||||
|
||||
class WrappedResponse(object):
|
||||
class WrappedResponse:
|
||||
|
||||
def __init__(self, response):
|
||||
self.response = response
|
||||
|
|
|
|||
|
|
@ -17,13 +17,24 @@ from scrapy.utils.trackref import object_ref
|
|||
|
||||
class Response(object_ref):
|
||||
|
||||
def __init__(self, url, status=200, headers=None, body=b'', flags=None, request=None):
|
||||
def __init__(self, url, status=200, headers=None, body=b'', flags=None, request=None, certificate=None):
|
||||
self.headers = Headers(headers or {})
|
||||
self.status = int(status)
|
||||
self._set_body(body)
|
||||
self._set_url(url)
|
||||
self.request = request
|
||||
self.flags = [] if flags is None else list(flags)
|
||||
self.certificate = certificate
|
||||
|
||||
@property
|
||||
def cb_kwargs(self):
|
||||
try:
|
||||
return self.request.cb_kwargs
|
||||
except AttributeError:
|
||||
raise AttributeError(
|
||||
"Response.cb_kwargs not available, this response "
|
||||
"is not tied to any request"
|
||||
)
|
||||
|
||||
@property
|
||||
def meta(self):
|
||||
|
|
@ -76,7 +87,7 @@ class Response(object_ref):
|
|||
"""Create a new Response with the same attributes except for those
|
||||
given new values.
|
||||
"""
|
||||
for x in ['url', 'status', 'headers', 'body', 'request', 'flags']:
|
||||
for x in ['url', 'status', 'headers', 'body', 'request', 'flags', 'certificate']:
|
||||
kwargs.setdefault(x, getattr(self, x))
|
||||
cls = kwargs.pop('cls', self.__class__)
|
||||
return cls(*args, **kwargs)
|
||||
|
|
@ -107,7 +118,7 @@ class Response(object_ref):
|
|||
|
||||
def follow(self, url, callback=None, method='GET', headers=None, body=None,
|
||||
cookies=None, meta=None, encoding='utf-8', priority=0,
|
||||
dont_filter=False, errback=None, cb_kwargs=None):
|
||||
dont_filter=False, errback=None, cb_kwargs=None, flags=None):
|
||||
# type: (...) -> Request
|
||||
"""
|
||||
Return a :class:`~.Request` instance to follow a link ``url``.
|
||||
|
|
@ -118,12 +129,16 @@ class Response(object_ref):
|
|||
:class:`~.TextResponse` provides a :meth:`~.TextResponse.follow`
|
||||
method which supports selectors in addition to absolute/relative URLs
|
||||
and Link objects.
|
||||
|
||||
.. versionadded:: 2.0
|
||||
The *flags* parameter.
|
||||
"""
|
||||
if isinstance(url, Link):
|
||||
url = url.url
|
||||
elif url is None:
|
||||
raise ValueError("url can't be None")
|
||||
url = self.urljoin(url)
|
||||
|
||||
return Request(
|
||||
url=url,
|
||||
callback=callback,
|
||||
|
|
@ -137,13 +152,16 @@ class Response(object_ref):
|
|||
dont_filter=dont_filter,
|
||||
errback=errback,
|
||||
cb_kwargs=cb_kwargs,
|
||||
flags=flags,
|
||||
)
|
||||
|
||||
def follow_all(self, urls, callback=None, method='GET', headers=None, body=None,
|
||||
cookies=None, meta=None, encoding='utf-8', priority=0,
|
||||
dont_filter=False, errback=None, cb_kwargs=None):
|
||||
dont_filter=False, errback=None, cb_kwargs=None, flags=None):
|
||||
# type: (...) -> Generator[Request, None, None]
|
||||
"""
|
||||
.. versionadded:: 2.0
|
||||
|
||||
Return an iterable of :class:`~.Request` instances to follow all links
|
||||
in ``urls``. It accepts the same arguments as ``Request.__init__`` method,
|
||||
but elements of ``urls`` can be relative URLs or :class:`~scrapy.link.Link` objects,
|
||||
|
|
@ -169,6 +187,7 @@ class Response(object_ref):
|
|||
dont_filter=dont_filter,
|
||||
errback=errback,
|
||||
cb_kwargs=cb_kwargs,
|
||||
flags=flags,
|
||||
)
|
||||
for url in urls
|
||||
)
|
||||
|
|
|
|||
|
|
@ -121,7 +121,7 @@ class TextResponse(Response):
|
|||
|
||||
def follow(self, url, callback=None, method='GET', headers=None, body=None,
|
||||
cookies=None, meta=None, encoding=None, priority=0,
|
||||
dont_filter=False, errback=None, cb_kwargs=None):
|
||||
dont_filter=False, errback=None, cb_kwargs=None, flags=None):
|
||||
# type: (...) -> Request
|
||||
"""
|
||||
Return a :class:`~.Request` instance to follow a link ``url``.
|
||||
|
|
@ -157,11 +157,12 @@ class TextResponse(Response):
|
|||
dont_filter=dont_filter,
|
||||
errback=errback,
|
||||
cb_kwargs=cb_kwargs,
|
||||
flags=flags,
|
||||
)
|
||||
|
||||
def follow_all(self, urls=None, callback=None, method='GET', headers=None, body=None,
|
||||
cookies=None, meta=None, encoding=None, priority=0,
|
||||
dont_filter=False, errback=None, cb_kwargs=None,
|
||||
dont_filter=False, errback=None, cb_kwargs=None, flags=None,
|
||||
css=None, xpath=None):
|
||||
# type: (...) -> Generator[Request, None, None]
|
||||
"""
|
||||
|
|
@ -187,9 +188,11 @@ class TextResponse(Response):
|
|||
selectors from which links cannot be obtained (for instance, anchor tags without an
|
||||
``href`` attribute)
|
||||
"""
|
||||
arg_count = len(list(filter(None, (urls, css, xpath))))
|
||||
if arg_count != 1:
|
||||
raise ValueError('Please supply exactly one of the following arguments: urls, css, xpath')
|
||||
arguments = [x for x in (urls, css, xpath) if x is not None]
|
||||
if len(arguments) != 1:
|
||||
raise ValueError(
|
||||
"Please supply exactly one of the following arguments: urls, css, xpath"
|
||||
)
|
||||
if not urls:
|
||||
if css:
|
||||
urls = self.css(css)
|
||||
|
|
@ -214,6 +217,7 @@ class TextResponse(Response):
|
|||
dont_filter=dont_filter,
|
||||
errback=errback,
|
||||
cb_kwargs=cb_kwargs,
|
||||
flags=flags,
|
||||
)
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -6,7 +6,7 @@ its documentation in: docs/topics/link-extractors.rst
|
|||
"""
|
||||
|
||||
|
||||
class Link(object):
|
||||
class Link:
|
||||
"""Link objects represent an extracted link by the LinkExtractor."""
|
||||
|
||||
__slots__ = ['url', 'text', 'fragment', 'nofollow']
|
||||
|
|
|
|||
|
|
@ -49,7 +49,7 @@ _matches = lambda url, regexs: any(r.search(url) for r in regexs)
|
|||
_is_valid_url = lambda url: url.split('://', 1)[0] in {'http', 'https', 'file', 'ftp'}
|
||||
|
||||
|
||||
class FilteringLinkExtractor(object):
|
||||
class FilteringLinkExtractor:
|
||||
|
||||
_csstranslator = HTMLTranslator()
|
||||
|
||||
|
|
|
|||
|
|
@ -5,11 +5,11 @@ from urllib.parse import urljoin
|
|||
|
||||
import lxml.etree as etree
|
||||
from w3lib.html import strip_html5_whitespace
|
||||
from w3lib.url import canonicalize_url
|
||||
from w3lib.url import canonicalize_url, safe_url_string
|
||||
|
||||
from scrapy.link import Link
|
||||
from scrapy.utils.misc import arg_to_iter, rel_has_nofollow
|
||||
from scrapy.utils.python import unique as unique_list, to_unicode
|
||||
from scrapy.utils.python import unique as unique_list
|
||||
from scrapy.utils.response import get_base_url
|
||||
from scrapy.linkextractors import FilteringLinkExtractor
|
||||
|
||||
|
|
@ -22,12 +22,12 @@ _collect_string_content = etree.XPath("string()")
|
|||
|
||||
def _nons(tag):
|
||||
if isinstance(tag, str):
|
||||
if tag[0] == '{' and tag[1:len(XHTML_NAMESPACE)+1] == XHTML_NAMESPACE:
|
||||
if tag[0] == '{' and tag[1:len(XHTML_NAMESPACE) + 1] == XHTML_NAMESPACE:
|
||||
return tag.split('}')[-1]
|
||||
return tag
|
||||
|
||||
|
||||
class LxmlParserLinkExtractor(object):
|
||||
class LxmlParserLinkExtractor:
|
||||
def __init__(self, tag="a", attr="href", process=None, unique=False,
|
||||
strip=True, canonicalized=False):
|
||||
self.scan_tag = tag if callable(tag) else lambda t: t == tag
|
||||
|
|
@ -66,7 +66,7 @@ class LxmlParserLinkExtractor(object):
|
|||
url = self.process_attr(attr_val)
|
||||
if url is None:
|
||||
continue
|
||||
url = to_unicode(url, encoding=response_encoding)
|
||||
url = safe_url_string(url, encoding=response_encoding)
|
||||
# to fix relative links after process_value
|
||||
url = urljoin(response_url, url)
|
||||
link = Link(url, _collect_string_content(el) or u'',
|
||||
|
|
|
|||
|
|
@ -25,7 +25,7 @@ def unbound_method(method):
|
|||
return method
|
||||
|
||||
|
||||
class ItemLoader(object):
|
||||
class ItemLoader:
|
||||
|
||||
default_item_class = Item
|
||||
default_input_processor = Identity()
|
||||
|
|
|
|||
Some files were not shown because too many files have changed in this diff Show More
Loading…
Reference in New Issue