diff --git a/.bumpversion.cfg b/.bumpversion.cfg index c9f1abea5..f347a0cd0 100644 --- a/.bumpversion.cfg +++ b/.bumpversion.cfg @@ -1,8 +1,7 @@ [bumpversion] -current_version = 1.8.0 +current_version = 2.0.0 commit = True tag = True tag_name = {new_version} [bumpversion:file:scrapy/VERSION] - diff --git a/.readthedocs.yml b/.readthedocs.yml index 563add75f..17eba34f3 100644 --- a/.readthedocs.yml +++ b/.readthedocs.yml @@ -1,9 +1,11 @@ version: 2 sphinx: configuration: docs/conf.py + fail_on_warning: true python: # For available versions, see: # https://docs.readthedocs.io/en/stable/config-file/v2.html#build-image version: 3.7 # Keep in sync with .travis.yml install: - requirements: docs/requirements.txt + - path: . diff --git a/Makefile.buildbot b/Makefile.buildbot deleted file mode 100644 index 775538259..000000000 --- a/Makefile.buildbot +++ /dev/null @@ -1,24 +0,0 @@ -TRIAL := $(shell which trial) -BRANCH := $(shell git rev-parse --abbrev-ref HEAD) -export PYTHONPATH=$(PWD) - -test: - coverage run --branch $(TRIAL) --reporter=text tests - rm -rf htmlcov && coverage html - -s3cmd sync -P htmlcov/ s3://static.scrapy.org/coverage-scrapy-$(BRANCH)/ - -build: - git describe --tags --match '[0-9]*' |sed 's/-/.post/;s/-g/+g/' >scrapy/VERSION - debchange -m -D unstable --force-distribution -v \ - $$(python setup.py --version |sed -r 's/([0-9]+.[0-9]+.[0-9]+)(a|b|rc|dev)([0-9]*)/\1~\2\3/')-$$(date +%s) \ - "Automatic build" - debuild -us -uc -b - -clean: - git checkout debian scrapy/VERSION - git clean -dfq - -pypi: - umask 0022 && chmod -R a+rX . && python setup.py sdist upload - -.PHONY: clean test build diff --git a/README.rst b/README.rst index 7fefaeec9..ce5973bcd 100644 --- a/README.rst +++ b/README.rst @@ -41,7 +41,7 @@ Requirements ============ * Python 3.5+ -* Works on Linux, Windows, Mac OSX, BSD +* Works on Linux, Windows, macOS, BSD Install ======= diff --git a/conftest.py b/conftest.py index c0de09909..be5fbabf4 100644 --- a/conftest.py +++ b/conftest.py @@ -11,7 +11,9 @@ collect_ignore = [ # not a test, but looks like a test "scrapy/utils/testsite.py", # contains scripts to be run by tests/test_crawler.py::CrawlerProcessSubprocess - *_py_files("tests/CrawlerProcess") + *_py_files("tests/CrawlerProcess"), + # Py36-only parts of respective tests + *_py_files("tests/py36"), ] for line in open('tests/ignores.txt'): diff --git a/debian/changelog b/debian/changelog deleted file mode 100644 index dde97f9e3..000000000 --- a/debian/changelog +++ /dev/null @@ -1,5 +0,0 @@ -scrapy (0.11) unstable; urgency=low - - * Initial release. - - -- Scrapinghub Team Thu, 10 Jun 2010 17:24:02 -0300 diff --git a/debian/compat b/debian/compat deleted file mode 100644 index 7f8f011eb..000000000 --- a/debian/compat +++ /dev/null @@ -1 +0,0 @@ -7 diff --git a/debian/control b/debian/control deleted file mode 100644 index 2cc8eedf4..000000000 --- a/debian/control +++ /dev/null @@ -1,20 +0,0 @@ -Source: scrapy -Section: python -Priority: optional -Maintainer: Scrapinghub Team -Build-Depends: debhelper (>= 7.0.50), python (>=2.7), python-twisted, python-w3lib, python-lxml, python-six (>=1.5.2) -Standards-Version: 3.8.4 -Homepage: https://scrapy.org/ - -Package: scrapy -Architecture: all -Depends: ${python:Depends}, python-lxml, python-twisted, python-openssl, - python-w3lib (>= 1.8.0), python-queuelib, python-cssselect (>= 0.9), python-six (>=1.5.2) -Recommends: python-setuptools -Conflicts: python-scrapy, scrapy-0.25 -Provides: python-scrapy, scrapy-0.25 -Description: Python web crawling and web scraping framework - Scrapy is a fast high-level web crawling and web scraping framework, - used to crawl websites and extract structured data from their pages. - It can be used for a wide range of purposes, from data mining to - monitoring and automated testing. diff --git a/debian/copyright b/debian/copyright deleted file mode 100644 index c1bf47565..000000000 --- a/debian/copyright +++ /dev/null @@ -1,40 +0,0 @@ -This package was debianized by the Scrapinghub team . - -It was downloaded from https://scrapy.org - -Upstream Author: Scrapy Developers - -Copyright: 2007-2013 Scrapy Developers - -License: bsd - -Copyright (c) Scrapy developers. -All rights reserved. - -Redistribution and use in source and binary forms, with or without modification, -are permitted provided that the following conditions are met: - - 1. Redistributions of source code must retain the above copyright notice, - this list of conditions and the following disclaimer. - - 2. Redistributions in binary form must reproduce the above copyright - notice, this list of conditions and the following disclaimer in the - documentation and/or other materials provided with the distribution. - - 3. Neither the name of Scrapy nor the names of its contributors may be used - to endorse or promote products derived from this software without - specific prior written permission. - -THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND -ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED -WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE -DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR -ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES -(INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; -LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON -ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT -(INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS -SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - -The Debian packaging is (C) 2010-2013, Scrapinghub and -is licensed under the BSD, see `/usr/share/common-licenses/BSD'. diff --git a/debian/pyversions b/debian/pyversions deleted file mode 100644 index 1effb0034..000000000 --- a/debian/pyversions +++ /dev/null @@ -1 +0,0 @@ -2.7 diff --git a/debian/rules b/debian/rules deleted file mode 100755 index b8796e6e3..000000000 --- a/debian/rules +++ /dev/null @@ -1,5 +0,0 @@ -#!/usr/bin/make -f -# -*- makefile -*- - -%: - dh $@ diff --git a/debian/scrapy.docs b/debian/scrapy.docs deleted file mode 100644 index c19ffba4d..000000000 --- a/debian/scrapy.docs +++ /dev/null @@ -1,2 +0,0 @@ -README.rst -AUTHORS diff --git a/debian/scrapy.install b/debian/scrapy.install deleted file mode 100644 index c288ebed3..000000000 --- a/debian/scrapy.install +++ /dev/null @@ -1,2 +0,0 @@ -extras/scrapy_bash_completion etc/bash_completion.d/ -extras/scrapy_zsh_completion /usr/share/zsh/vendor-completions/_scrapy diff --git a/debian/scrapy.lintian-overrides b/debian/scrapy.lintian-overrides deleted file mode 100644 index b5de7f67d..000000000 --- a/debian/scrapy.lintian-overrides +++ /dev/null @@ -1 +0,0 @@ -new-package-should-close-itp-bug diff --git a/debian/scrapy.manpages b/debian/scrapy.manpages deleted file mode 100644 index 4818e9c92..000000000 --- a/debian/scrapy.manpages +++ /dev/null @@ -1 +0,0 @@ -extras/scrapy.1 diff --git a/docs/conf.py b/docs/conf.py index c3418cfb3..6e2399f66 100644 --- a/docs/conf.py +++ b/docs/conf.py @@ -281,6 +281,7 @@ coverage_ignore_pyobjects = [ intersphinx_mapping = { 'coverage': ('https://coverage.readthedocs.io/en/stable', None), + 'cssselect': ('https://cssselect.readthedocs.io/en/latest', None), 'pytest': ('https://docs.pytest.org/en/latest', None), 'python': ('https://docs.python.org/3', None), 'sphinx': ('https://www.sphinx-doc.org/en/master', None), diff --git a/docs/contributing.rst b/docs/contributing.rst index f40a6bba2..aed5ab92e 100644 --- a/docs/contributing.rst +++ b/docs/contributing.rst @@ -143,7 +143,7 @@ by running ``git fetch upstream pull/$PR_NUMBER/head:$BRANCH_NAME_TO_CREATE`` (replace 'upstream' with a remote name for scrapy repository, ``$PR_NUMBER`` with an ID of the pull request, and ``$BRANCH_NAME_TO_CREATE`` with a name of the branch you want to create locally). -See also: https://help.github.com/articles/checking-out-pull-requests-locally/#modifying-an-inactive-pull-request-locally. +See also: https://help.github.com/en/github/collaborating-with-issues-and-pull-requests/checking-out-pull-requests-locally#modifying-an-inactive-pull-request-locally. When writing GitHub pull requests, try to keep titles short but descriptive. E.g. For bug #411: "Scrapy hangs if an exception raises in start_requests" @@ -168,7 +168,7 @@ Scrapy: * Don't put your name in the code you contribute; git provides enough metadata to identify author of the code. - See https://help.github.com/articles/setting-your-username-in-git/ for + See https://help.github.com/en/github/using-git/setting-your-username-in-git for setup instructions. .. _documentation-policies: @@ -266,5 +266,5 @@ And their unit-tests are in:: .. _tests/: https://github.com/scrapy/scrapy/tree/master/tests .. _open issues: https://github.com/scrapy/scrapy/issues .. _PEP 257: https://www.python.org/dev/peps/pep-0257/ -.. _pull request: https://help.github.com/en/articles/creating-a-pull-request +.. _pull request: https://help.github.com/en/github/collaborating-with-issues-and-pull-requests/creating-a-pull-request .. _pytest-xdist: https://github.com/pytest-dev/pytest-xdist diff --git a/docs/faq.rst b/docs/faq.rst index f72e4cf01..75a0f4864 100644 --- a/docs/faq.rst +++ b/docs/faq.rst @@ -22,8 +22,8 @@ In other words, comparing `BeautifulSoup`_ (or `lxml`_) to Scrapy is like comparing `jinja2`_ to `Django`_. .. _BeautifulSoup: https://www.crummy.com/software/BeautifulSoup/ -.. _lxml: http://lxml.de/ -.. _jinja2: http://jinja.pocoo.org/ +.. _lxml: https://lxml.de/ +.. _jinja2: https://palletsprojects.com/p/jinja/ .. _Django: https://www.djangoproject.com/ Can I use Scrapy with BeautifulSoup? @@ -269,7 +269,7 @@ The ``__VIEWSTATE`` parameter is used in sites built with ASP.NET/VB.NET. For more info on how it works see `this page`_. Also, here's an `example spider`_ which scrapes one of these sites. -.. _this page: http://search.cpan.org/~ecarroll/HTML-TreeBuilderX-ASP_NET-0.09/lib/HTML/TreeBuilderX/ASP_NET.pm +.. _this page: https://metacpan.org/pod/release/ECARROLL/HTML-TreeBuilderX-ASP_NET-0.09/lib/HTML/TreeBuilderX/ASP_NET.pm .. _example spider: https://github.com/AmbientLighter/rpn-fas/blob/master/fas/spiders/rnp.py What's the best way to parse big XML/CSV data feeds? diff --git a/docs/index.rst b/docs/index.rst index a4343b7e0..11aa5c9be 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -165,6 +165,8 @@ Solving specific problems topics/autothrottle topics/benchmarking topics/jobs + topics/coroutines + topics/asyncio :doc:`faq` Get answers to most frequently asked questions. @@ -205,6 +207,12 @@ Solving specific problems :doc:`topics/jobs` Learn how to pause and resume crawls for large spiders. +:doc:`topics/coroutines` + Use the :ref:`coroutine syntax `. + +:doc:`topics/asyncio` + Use :mod:`asyncio` and :mod:`asyncio`-powered libraries. + .. _extending-scrapy: Extending Scrapy diff --git a/docs/intro/install.rst b/docs/intro/install.rst index 178be723c..6356e0eea 100644 --- a/docs/intro/install.rst +++ b/docs/intro/install.rst @@ -7,12 +7,12 @@ Installation guide Installing Scrapy ================= -Scrapy runs on Python 3.5 or above -under CPython (default Python implementation) and PyPy (starting with PyPy 5.9). +Scrapy runs on Python 3.5 or above under CPython (default Python +implementation) and PyPy (starting with PyPy 5.9). If you're using `Anaconda`_ or `Miniconda`_, you can install the package from the `conda-forge`_ channel, which has up-to-date packages for Linux, Windows -and OS X. +and macOS. To install Scrapy using ``conda``, run:: @@ -65,7 +65,7 @@ please refer to their respective installation instructions: * `lxml installation`_ * `cryptography installation`_ -.. _lxml installation: http://lxml.de/installation.html +.. _lxml installation: https://lxml.de/installation.html .. _cryptography installation: https://cryptography.io/en/latest/installation/ @@ -81,35 +81,18 @@ Python packages can be installed either globally (a.k.a system wide), or in user-space. We do not recommend installing Scrapy system wide. Instead, we recommend that you install Scrapy within a so-called -"virtual environment" (`virtualenv`_). -Virtualenvs allow you to not conflict with already-installed Python +"virtual environment" (:mod:`venv`). +Virtual environments allow you to not conflict with already-installed Python system packages (which could break some of your system tools and scripts), and still install packages normally with ``pip`` (without ``sudo`` and the likes). -To get started with virtual environments, see `virtualenv installation instructions`_. -To install it globally (having it globally installed actually helps here), -it should be a matter of running:: +See :ref:`tut-venv` on how to create your virtual environment. - $ [sudo] pip install virtualenv - -Check this `user guide`_ on how to create your virtualenv. - -.. note:: - If you use Linux or OS X, `virtualenvwrapper`_ is a handy tool to create virtualenvs. - -Once you have created a virtualenv, you can install Scrapy inside it with ``pip``, +Once you have created a virtual environment, you can install Scrapy inside it with ``pip``, just like any other Python package. (See :ref:`platform-specific guides ` below for non-Python dependencies that you may need to install beforehand). -Python virtualenvs can be created to use Python 2 by default, or Python 3 by default. As Scrapy -only supports Python 3, make sure you created a Python 3 virtualenv. - -.. _virtualenv: https://virtualenv.pypa.io -.. _virtualenv installation instructions: https://virtualenv.pypa.io/en/stable/installation/ -.. _virtualenvwrapper: https://virtualenvwrapper.readthedocs.io/en/latest/install.html -.. _user guide: https://virtualenv.pypa.io/en/stable/userguide/ - .. _intro-install-platform-notes: @@ -165,11 +148,11 @@ you can install Scrapy with ``pip`` after that:: .. _intro-install-macos: -Mac OS X --------- +macOS +----- Building Scrapy's dependencies requires the presence of a C compiler and -development headers. On OS X this is typically provided by Apple’s Xcode +development headers. On macOS this is typically provided by Apple’s Xcode development tools. To install the Xcode command line tools open a terminal window and run:: @@ -205,15 +188,12 @@ solutions: brew update; brew upgrade python -* *(Optional)* Install Scrapy inside an isolated python environment. +* *(Optional)* :ref:`Install Scrapy inside a Python virtual environment + `. - This method is a workaround for the above OS X issue, but it's an overall + This method is a workaround for the above macOS issue, but it's an overall good practice for managing dependencies and can complement the first method. - `virtualenv`_ is a tool you can use to create virtual environments in python. - We recommended reading a tutorial like - http://docs.python-guide.org/en/latest/dev/virtualenvs/ to get started. - After any of these workarounds you should be able to install Scrapy:: pip install Scrapy @@ -227,7 +207,7 @@ For PyPy3, only Linux installation was tested. Most Scrapy dependencides now have binary wheels for CPython, but not for PyPy. This means that these dependecies will be built during installation. -On OS X, you are likely to face an issue with building Cryptography dependency, +On macOS, you are likely to face an issue with building Cryptography dependency, solution to this problem is described `here `_, that is to ``brew install openssl`` and then export the flags that this command @@ -273,11 +253,11 @@ For details, see `Issue #2473 `_. .. _Python: https://www.python.org/ .. _pip: https://pip.pypa.io/en/latest/installing/ .. _lxml: https://lxml.de/index.html -.. _parsel: https://pypi.python.org/pypi/parsel -.. _w3lib: https://pypi.python.org/pypi/w3lib -.. _twisted: https://twistedmatrix.com/ -.. _cryptography: https://cryptography.io/ -.. _pyOpenSSL: https://pypi.python.org/pypi/pyOpenSSL +.. _parsel: https://pypi.org/project/parsel/ +.. _w3lib: https://pypi.org/project/w3lib/ +.. _twisted: https://twistedmatrix.com/trac/ +.. _cryptography: https://cryptography.io/en/latest/ +.. _pyOpenSSL: https://pypi.org/project/pyOpenSSL/ .. _setuptools: https://pypi.python.org/pypi/setuptools .. _AUR Scrapy package: https://aur.archlinux.org/packages/scrapy/ .. _homebrew: https://brew.sh/ diff --git a/docs/intro/tutorial.rst b/docs/intro/tutorial.rst index ee10048b5..1768badbb 100644 --- a/docs/intro/tutorial.rst +++ b/docs/intro/tutorial.rst @@ -212,7 +212,7 @@ using the :ref:`Scrapy shell `. Run:: .. note:: Remember to always enclose urls in quotes when running Scrapy shell from - command-line, otherwise urls containing arguments (ie. ``&`` character) + command-line, otherwise urls containing arguments (i.e. ``&`` character) will not work. On Windows, use double quotes instead:: @@ -306,7 +306,7 @@ with a selector (see :ref:`topics-developer-tools`). visually selected elements, which works in many browsers. .. _regular expressions: https://docs.python.org/3/library/re.html -.. _Selector Gadget: http://selectorgadget.com/ +.. _Selector Gadget: https://selectorgadget.com/ XPath: a brief intro @@ -337,7 +337,7 @@ recommend `this tutorial to learn XPath through examples `_, and `this tutorial to learn "how to think in XPath" `_. -.. _XPath: https://www.w3.org/TR/xpath +.. _XPath: https://www.w3.org/TR/xpath/all/ .. _CSS: https://www.w3.org/TR/selectors Extracting quotes and authors diff --git a/docs/news.rst b/docs/news.rst index e4b985c77..dd5e00223 100644 --- a/docs/news.rst +++ b/docs/news.rst @@ -3,8 +3,452 @@ Release notes ============= -.. note:: Scrapy 1.x will be the last series supporting Python 2. Scrapy 2.0, - planned for Q4 2019 or Q1 2020, will support **Python 3 only**. +.. _release-2.0.0: + +Scrapy 2.0.0 (2020-03-03) +------------------------- + +Highlights: + +* Python 2 support has been removed +* :doc:`Partial ` :ref:`coroutine syntax ` support + and :doc:`experimental ` :mod:`asyncio` support +* New :meth:`Response.follow_all ` method +* :ref:`FTP support ` for media pipelines +* New :attr:`Response.certificate ` + attribute +* IPv6 support through :setting:`DNS_RESOLVER` + +Backward-incompatible changes +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +* Python 2 support has been removed, following `Python 2 end-of-life on + January 1, 2020`_ (:issue:`4091`, :issue:`4114`, :issue:`4115`, + :issue:`4121`, :issue:`4138`, :issue:`4231`, :issue:`4242`, :issue:`4304`, + :issue:`4309`, :issue:`4373`) + +* Retry gaveups (see :setting:`RETRY_TIMES`) are now logged as errors instead + of as debug information (:issue:`3171`, :issue:`3566`) + +* File extensions that + :class:`LinkExtractor ` + ignores by default now also include ``7z``, ``7zip``, ``apk``, ``bz2``, + ``cdr``, ``dmg``, ``ico``, ``iso``, ``tar``, ``tar.gz``, ``webm``, and + ``xz`` (:issue:`1837`, :issue:`2067`, :issue:`4066`) + +* The :setting:`METAREFRESH_IGNORE_TAGS` setting is now an empty list by + default, following web browser behavior (:issue:`3844`, :issue:`4311`) + +* The + :class:`~scrapy.downloadermiddlewares.httpcompression.HttpCompressionMiddleware` + now includes spaces after commas in the value of the ``Accept-Encoding`` + header that it sets, following web browser behavior (:issue:`4293`) + +* The ``__init__`` method of custom download handlers (see + :setting:`DOWNLOAD_HANDLERS`) or subclasses of the following downloader + handlers no longer receives a ``settings`` parameter: + + * :class:`scrapy.core.downloader.handlers.datauri.DataURIDownloadHandler` + + * :class:`scrapy.core.downloader.handlers.file.FileDownloadHandler` + + Use the ``from_settings`` or ``from_crawler`` class methods to expose such + a parameter to your custom download handlers. + + (:issue:`4126`) + +* We have refactored the :class:`scrapy.core.scheduler.Scheduler` class and + related queue classes (see :setting:`SCHEDULER_PRIORITY_QUEUE`, + :setting:`SCHEDULER_DISK_QUEUE` and :setting:`SCHEDULER_MEMORY_QUEUE`) to + make it easier to implement custom scheduler queue classes. See + :ref:`2-0-0-scheduler-queue-changes` below for details. + +* Overridden settings are now logged in a different format. This is more in + line with similar information logged at startup (:issue:`4199`) + +.. _Python 2 end-of-life on January 1, 2020: https://www.python.org/doc/sunset-python-2/ + + +Deprecation removals +~~~~~~~~~~~~~~~~~~~~ + +* The :ref:`Scrapy shell ` no longer provides a `sel` proxy + object, use :meth:`response.selector ` + instead (:issue:`4347`) + +* LevelDB support has been removed (:issue:`4112`) + +* The following functions have been removed from :mod:`scrapy.utils.python`: + ``isbinarytext``, ``is_writable``, ``setattr_default``, ``stringify_dict`` + (:issue:`4362`) + + +Deprecations +~~~~~~~~~~~~ + +* Using environment variables prefixed with ``SCRAPY_`` to override settings + is deprecated (:issue:`4300`, :issue:`4374`, :issue:`4375`) + +* :class:`scrapy.linkextractors.FilteringLinkExtractor` is deprecated, use + :class:`scrapy.linkextractors.LinkExtractor + ` instead (:issue:`4045`) + +* The ``noconnect`` query string argument of proxy URLs is deprecated and + should be removed from proxy URLs (:issue:`4198`) + +* The :meth:`next ` method of + :class:`scrapy.utils.python.MutableChain` is deprecated, use the global + :func:`next` function or :meth:`MutableChain.__next__ + ` instead (:issue:`4153`) + + +New features +~~~~~~~~~~~~ + +* Added :doc:`partial support ` for Python’s + :ref:`coroutine syntax ` and :doc:`experimental support + ` for :mod:`asyncio` and :mod:`asyncio`-powered libraries + (:issue:`4010`, :issue:`4259`, :issue:`4269`, :issue:`4270`, :issue:`4271`, + :issue:`4316`, :issue:`4318`) + +* The new :meth:`Response.follow_all ` + method offers the same functionality as + :meth:`Response.follow ` but supports an + iterable of URLs as input and returns an iterable of requests + (:issue:`2582`, :issue:`4057`, :issue:`4286`) + +* :ref:`Media pipelines ` now support :ref:`FTP + storage ` (:issue:`3928`, :issue:`3961`) + +* The new :attr:`Response.certificate ` + attribute exposes the SSL certificate of the server as a + :class:`twisted.internet.ssl.Certificate` object for HTTPS responses + (:issue:`2726`, :issue:`4054`) + +* A new :setting:`DNS_RESOLVER` setting allows enabling IPv6 support + (:issue:`1031`, :issue:`4227`) + +* A new :setting:`SCRAPER_SLOT_MAX_ACTIVE_SIZE` setting allows configuring + the existing soft limit that pauses request downloads when the total + response data being processed is too high (:issue:`1410`, :issue:`3551`) + +* A new :setting:`TWISTED_REACTOR` setting allows customizing the + :mod:`~twisted.internet.reactor` that Scrapy uses, allowing to + :doc:`enable asyncio support ` or deal with a + :ref:`common macOS issue ` (:issue:`2905`, + :issue:`4294`) + +* Scheduler disk and memory queues may now use the class methods + ``from_crawler`` or ``from_settings`` (:issue:`3884`) + +* The new :attr:`Response.cb_kwargs ` + attribute serves as a shortcut for :attr:`Response.request.cb_kwargs + ` (:issue:`4331`) + +* :meth:`Response.follow ` now supports a + ``flags`` parameter, for consistency with :class:`~scrapy.http.Request` + (:issue:`4277`, :issue:`4279`) + +* :ref:`Item loader processors ` can now be + regular functions, they no longer need to be methods (:issue:`3899`) + +* :class:`~scrapy.spiders.Rule` now accepts an ``errback`` parameter + (:issue:`4000`) + +* :class:`~scrapy.http.Request` no longer requires a ``callback`` parameter + when an ``errback`` parameter is specified (:issue:`3586`, :issue:`4008`) + +* :class:`~scrapy.logformatter.LogFormatter` now supports some additional + methods: + + * :class:`~scrapy.logformatter.LogFormatter.download_error` for + download errors + + * :class:`~scrapy.logformatter.LogFormatter.item_error` for exceptions + raised during item processing by :ref:`item pipelines + ` + + * :class:`~scrapy.logformatter.LogFormatter.spider_error` for exceptions + raised from :ref:`spider callbacks ` + + (:issue:`374`, :issue:`3986`, :issue:`3989`, :issue:`4176`, :issue:`4188`) + +* The :setting:`FEED_URI` setting now supports :class:`pathlib.Path` values + (:issue:`3731`, :issue:`4074`) + +* A new :signal:`request_left_downloader` signal is sent when a request + leaves the downloader (:issue:`4303`) + +* Scrapy logs a warning when it detects a request callback or errback that + uses ``yield`` but also returns a value, since the returned value would be + lost (:issue:`3484`, :issue:`3869`) + +* :class:`~scrapy.spiders.Spider` objects now raise an :exc:`AttributeError` + exception if they do not have a :class:`~scrapy.spiders.Spider.start_urls` + attribute nor reimplement :class:`~scrapy.spiders.Spider.start_requests`, + but have a ``start_url`` attribute (:issue:`4133`, :issue:`4170`) + +* :class:`~scrapy.exporters.BaseItemExporter` subclasses may now use + ``super().__init__(**kwargs)`` instead of ``self._configure(kwargs)`` in + their ``__init__`` method, passing ``dont_fail=True`` to the parent + ``__init__`` method if needed, and accessing ``kwargs`` at ``self._kwargs`` + after calling their parent ``__init__`` method (:issue:`4193`, + :issue:`4370`) + +* A new ``keep_fragments`` parameter of + :func:`scrapy.utils.request.request_fingerprint` allows to generate + different fingerprints for requests with different fragments in their URL + (:issue:`4104`) + +* Download handlers (see :setting:`DOWNLOAD_HANDLERS`) may now use the + ``from_settings`` and ``from_crawler`` class methods that other Scrapy + components already supported (:issue:`4126`) + +* :class:`scrapy.utils.python.MutableChain.__iter__` now returns ``self``, + `allowing it to be used as a sequence `_ + (:issue:`4153`) + + +Bug fixes +~~~~~~~~~ + +* The :command:`crawl` command now also exits with exit code 1 when an + exception happens before the crawling starts (:issue:`4175`, :issue:`4207`) + +* :class:`LinkExtractor.extract_links + ` no longer + re-encodes the query string or URLs from non-UTF-8 responses in UTF-8 + (:issue:`998`, :issue:`1403`, :issue:`1949`, :issue:`4321`) + +* The first spider middleware (see :setting:`SPIDER_MIDDLEWARES`) now also + processes exceptions raised from callbacks that are generators + (:issue:`4260`, :issue:`4272`) + +* Redirects to URLs starting with 3 slashes (``///``) are now supported + (:issue:`4032`, :issue:`4042`) + +* :class:`~scrapy.http.Request` no longer accepts strings as ``url`` simply + because they have a colon (:issue:`2552`, :issue:`4094`) + +* The correct encoding is now used for attach names in + :class:`~scrapy.mail.MailSender` (:issue:`4229`, :issue:`4239`) + +* :class:`~scrapy.dupefilters.RFPDupeFilter`, the default + :setting:`DUPEFILTER_CLASS`, no longer writes an extra ``\r`` character on + each line in Windows, which made the size of the ``requests.seen`` file + unnecessarily large on that platform (:issue:`4283`) + +* Z shell auto-completion now looks for ``.html`` files, not ``.http`` files, + and covers the ``-h`` command-line switch (:issue:`4122`, :issue:`4291`) + +* Adding items to a :class:`scrapy.utils.datatypes.LocalCache` object + without a ``limit`` defined no longer raises a :exc:`TypeError` exception + (:issue:`4123`) + +* Fixed a typo in the message of the :exc:`ValueError` exception raised when + :func:`scrapy.utils.misc.create_instance` gets both ``settings`` and + ``crawler`` set to ``None`` (:issue:`4128`) + + +Documentation +~~~~~~~~~~~~~ + +* API documentation now links to an online, syntax-highlighted view of the + corresponding source code (:issue:`4148`) + +* Links to unexisting documentation pages now allow access to the sidebar + (:issue:`4152`, :issue:`4169`) + +* Cross-references within our documentation now display a tooltip when + hovered (:issue:`4173`, :issue:`4183`) + +* Improved the documentation about :meth:`LinkExtractor.extract_links + ` and + simplified :ref:`topics-link-extractors` (:issue:`4045`) + +* Clarified how :class:`ItemLoader.item ` + works (:issue:`3574`, :issue:`4099`) + +* Clarified that :func:`logging.basicConfig` should not be used when also + using :class:`~scrapy.crawler.CrawlerProcess` (:issue:`2149`, + :issue:`2352`, :issue:`3146`, :issue:`3960`) + +* Clarified the requirements for :class:`~scrapy.http.Request` objects + :ref:`when using persistence ` (:issue:`4124`, + :issue:`4139`) + +* Clarified how to install a :ref:`custom image pipeline + ` (:issue:`4034`, :issue:`4252`) + +* Fixed the signatures of the ``file_path`` method in :ref:`media pipeline + ` examples (:issue:`4290`) + +* Covered a backward-incompatible change in Scrapy 1.7.0 affecting custom + :class:`scrapy.core.scheduler.Scheduler` subclasses (:issue:`4274`) + +* Improved the ``README.rst`` and ``CODE_OF_CONDUCT.md`` files + (:issue:`4059`) + +* Documentation examples are now checked as part of our test suite and we + have fixed some of the issues detected (:issue:`4142`, :issue:`4146`, + :issue:`4171`, :issue:`4184`, :issue:`4190`) + +* Fixed logic issues, broken links and typos (:issue:`4247`, :issue:`4258`, + :issue:`4282`, :issue:`4288`, :issue:`4305`, :issue:`4308`, :issue:`4323`, + :issue:`4338`, :issue:`4359`, :issue:`4361`) + +* Improved consistency when referring to the ``__init__`` method of an object + (:issue:`4086`, :issue:`4088`) + +* Fixed an inconsistency between code and output in :ref:`intro-overview` + (:issue:`4213`) + +* Extended :mod:`~sphinx.ext.intersphinx` usage (:issue:`4147`, + :issue:`4172`, :issue:`4185`, :issue:`4194`, :issue:`4197`) + +* We now use a recent version of Python to build the documentation + (:issue:`4140`, :issue:`4249`) + +* Cleaned up documentation (:issue:`4143`, :issue:`4275`) + + +Quality assurance +~~~~~~~~~~~~~~~~~ + +* Re-enabled proxy ``CONNECT`` tests (:issue:`2545`, :issue:`4114`) + +* Added Bandit_ security checks to our test suite (:issue:`4162`, + :issue:`4181`) + +* Added Flake8_ style checks to our test suite and applied many of the + corresponding changes (:issue:`3944`, :issue:`3945`, :issue:`4137`, + :issue:`4157`, :issue:`4167`, :issue:`4174`, :issue:`4186`, :issue:`4195`, + :issue:`4238`, :issue:`4246`, :issue:`4355`, :issue:`4360`, :issue:`4365`) + +* Improved test coverage (:issue:`4097`, :issue:`4218`, :issue:`4236`) + +* Started reporting slowest tests, and improved the performance of some of + them (:issue:`4163`, :issue:`4164`) + +* Fixed broken tests and refactored some tests (:issue:`4014`, :issue:`4095`, + :issue:`4244`, :issue:`4268`, :issue:`4372`) + +* Modified the :doc:`tox ` configuration to allow running tests + with any Python version, run Bandit_ and Flake8_ tests by default, and + enforce a minimum tox version programmatically (:issue:`4179`) + +* Cleaned up code (:issue:`3937`, :issue:`4208`, :issue:`4209`, + :issue:`4210`, :issue:`4212`, :issue:`4369`, :issue:`4376`, :issue:`4378`) + +.. _Bandit: https://bandit.readthedocs.io/ +.. _Flake8: https://flake8.pycqa.org/en/latest/ + + +.. _2-0-0-scheduler-queue-changes: + +Changes to scheduler queue classes +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +The following changes may impact any custom queue classes of all types: + +* The ``push`` method no longer receives a second positional parameter + containing ``request.priority * -1``. If you need that value, get it + from the first positional parameter, ``request``, instead, or use + the new :meth:`~scrapy.core.scheduler.ScrapyPriorityQueue.priority` + method in :class:`scrapy.core.scheduler.ScrapyPriorityQueue` + subclasses. + +The following changes may impact custom priority queue classes: + +* In the ``__init__`` method or the ``from_crawler`` or ``from_settings`` + class methods: + + * The parameter that used to contain a factory function, + ``qfactory``, is now passed as a keyword parameter named + ``downstream_queue_cls``. + + * A new keyword parameter has been added: ``key``. It is a string + that is always an empty string for memory queues and indicates the + :setting:`JOB_DIR` value for disk queues. + + * The parameter for disk queues that contains data from the previous + crawl, ``startprios`` or ``slot_startprios``, is now passed as a + keyword parameter named ``startprios``. + + * The ``serialize`` parameter is no longer passed. The disk queue + class must take care of request serialization on its own before + writing to disk, using the + :func:`~scrapy.utils.reqser.request_to_dict` and + :func:`~scrapy.utils.reqser.request_from_dict` functions from the + :mod:`scrapy.utils.reqser` module. + +The following changes may impact custom disk and memory queue classes: + +* The signature of the ``__init__`` method is now + ``__init__(self, crawler, key)``. + +The following changes affect specifically the +:class:`~scrapy.core.scheduler.ScrapyPriorityQueue` and +:class:`~scrapy.core.scheduler.DownloaderAwarePriorityQueue` classes from +:mod:`scrapy.core.scheduler` and may affect subclasses: + +* In the ``__init__`` method, most of the changes described above apply. + + ``__init__`` may still receive all parameters as positional parameters, + however: + + * ``downstream_queue_cls``, which replaced ``qfactory``, must be + instantiated differently. + + ``qfactory`` was instantiated with a priority value (integer). + + Instances of ``downstream_queue_cls`` should be created using + the new + :meth:`ScrapyPriorityQueue.qfactory ` + or + :meth:`DownloaderAwarePriorityQueue.pqfactory ` + methods. + + * The new ``key`` parameter displaced the ``startprios`` + parameter 1 position to the right. + +* The following class attributes have been added: + + * :attr:`~scrapy.core.scheduler.ScrapyPriorityQueue.crawler` + + * :attr:`~scrapy.core.scheduler.ScrapyPriorityQueue.downstream_queue_cls` + (details above) + + * :attr:`~scrapy.core.scheduler.ScrapyPriorityQueue.key` (details above) + +* The ``serialize`` attribute has been removed (details above) + +The following changes affect specifically the +:class:`~scrapy.core.scheduler.ScrapyPriorityQueue` class and may affect +subclasses: + +* A new :meth:`~scrapy.core.scheduler.ScrapyPriorityQueue.priority` + method has been added which, given a request, returns + ``request.priority * -1``. + + It is used in :meth:`~scrapy.core.scheduler.ScrapyPriorityQueue.push` + to make up for the removal of its ``priority`` parameter. + +* The ``spider`` attribute has been removed. Use + :attr:`crawler.spider ` + instead. + +The following changes affect specifically the +:class:`~scrapy.core.scheduler.DownloaderAwarePriorityQueue` class and may +affect subclasses: + +* A new :attr:`~scrapy.core.scheduler.DownloaderAwarePriorityQueue.pqueues` + attribute offers a mapping of downloader slot names to the + corresponding instances of + :attr:`~scrapy.core.scheduler.DownloaderAwarePriorityQueue.downstream_queue_cls`. + +(:issue:`3884`) + .. _release-1.8.0: @@ -26,7 +470,7 @@ Backward-incompatible changes * Python 3.4 is no longer supported, and some of the minimum requirements of Scrapy have also changed: - * cssselect_ 0.9.1 + * :doc:`cssselect ` 0.9.1 * cryptography_ 2.0 * lxml_ 3.5.0 * pyOpenSSL_ 16.2.0 @@ -288,12 +732,12 @@ Backward-incompatible changes :class:`~scrapy.http.Request` objects instead of arbitrary Python data structures. -* An additional ``crawler`` parameter has been added to the ``__init__`` method - of the :class:`scrapy.core.scheduler.Scheduler` class. - Custom scheduler subclasses which don't accept arbitrary parameters in - their ``__init__`` method might break because of this change. +* An additional ``crawler`` parameter has been added to the ``__init__`` + method of the :class:`~scrapy.core.scheduler.Scheduler` class. Custom + scheduler subclasses which don't accept arbitrary parameters in their + ``__init__`` method might break because of this change. - For more information, refer to the documentation for the :setting:`SCHEDULER` setting. + For more information, see :setting:`SCHEDULER`. See also :ref:`1.7-deprecation-removals` below. @@ -1076,7 +1520,7 @@ Cleanups & Refactoring ~~~~~~~~~~~~~~~~~~~~~~ - Tests: remove temp files and folders (:issue:`2570`), - fixed ProjectUtilsTest on OS X (:issue:`2569`), + fixed ProjectUtilsTest on macOS (:issue:`2569`), use portable pypy for Linux on Travis CI (:issue:`2710`) - Separate building request from ``_requests_to_follow`` in CrawlSpider (:issue:`2562`) - Remove “Python 3 progress” badge (:issue:`2567`) @@ -1616,7 +2060,7 @@ Deprecations and Removals + ``scrapy.utils.datatypes.SiteNode`` - The previously bundled ``scrapy.xlib.pydispatch`` library was deprecated and - replaced by `pydispatcher `_. + replaced by `pydispatcher `_. Relocations @@ -1645,7 +2089,7 @@ Bugfixes - Makes ``_monkeypatches`` more robust (:issue:`1634`). - Fixed bug on ``XMLItemExporter`` with non-string fields in items (:issue:`1738`). -- Fixed startproject command in OS X (:issue:`1635`). +- Fixed startproject command in macOS (:issue:`1635`). - Fixed :class:`~scrapy.exporters.PythonItemExporter` and CSVExporter for non-string item types (:issue:`1737`). - Various logging related fixes (:issue:`1294`, :issue:`1419`, :issue:`1263`, @@ -1713,12 +2157,12 @@ Scrapy 1.0.4 (2015-12-30) - Typos corrections (:commit:`7067117`) - fix typos in downloader-middleware.rst and exceptions.rst, middlware -> middleware (:commit:`32f115c`) - Add note to Ubuntu install section about Debian compatibility (:commit:`23fda69`) -- Replace alternative OSX install workaround with virtualenv (:commit:`98b63ee`) +- Replace alternative macOS install workaround with virtualenv (:commit:`98b63ee`) - Reference Homebrew's homepage for installation instructions (:commit:`1925db1`) - Add oldest supported tox version to contributing docs (:commit:`5d10d6d`) - Note in install docs about pip being already included in python>=2.7.9 (:commit:`85c980e`) - Add non-python dependencies to Ubuntu install section in the docs (:commit:`fbd010d`) -- Add OS X installation section to docs (:commit:`d8f4cba`) +- Add macOS installation section to docs (:commit:`d8f4cba`) - DOC(ENH): specify path to rtd theme explicitly (:commit:`de73b1a`) - minor: scrapy.Spider docs grammar (:commit:`1ddcc7b`) - Make common practices sample code match the comments (:commit:`1b85bcf`) @@ -2450,7 +2894,7 @@ Other ~~~~~ - Dropped Python 2.6 support (:issue:`448`) -- Add `cssselect`_ python package as install dependency +- Add :doc:`cssselect ` python package as install dependency - Drop libxml2 and multi selector's backend support, `lxml`_ is required from now on. - Minimum Twisted version increased to 10.0.0, dropped Twisted 8.0 support. - Running test suite now requires ``mock`` python library (:issue:`390`) @@ -2571,7 +3015,7 @@ Scrapy 0.18.0 (released 2013-08-09) - MetaRefreshMiddldeware and RedirectMiddleware have different priorities to address #62 - added from_crawler method to spiders - added system tests with mock server -- more improvements to Mac OS compatibility (thanks Alex Cepoi) +- more improvements to macOS compatibility (thanks Alex Cepoi) - several more cleanups to singletons and multi-spider support (thanks Nicolas Ramirez) - support custom download slots - added --spider option to "shell" command. @@ -2647,7 +3091,7 @@ Scrapy 0.16.3 (released 2012-12-07) - Remove concurrency limitation when using download delays and still ensure inter-request delays are enforced (:commit:`487b9b5`) - add error details when image pipeline fails (:commit:`8232569`) -- improve mac os compatibility (:commit:`8dcf8aa`) +- improve macOS compatibility (:commit:`8dcf8aa`) - setup.py: use README.rst to populate long_description (:commit:`7b5310d`) - doc: removed obsolete references to ClientForm (:commit:`80f9bb6`) - correct docs for default storage backend (:commit:`2aa491b`) @@ -3047,17 +3491,16 @@ Scrapy 0.7 First release of Scrapy. -.. _AJAX crawleable urls: https://developers.google.com/webmasters/ajax-crawling/docs/getting-started?csw=1 +.. _AJAX crawleable urls: https://developers.google.com/search/docs/ajax-crawling/docs/getting-started?csw=1 .. _botocore: https://github.com/boto/botocore .. _chunked transfer encoding: https://en.wikipedia.org/wiki/Chunked_transfer_encoding .. _ClientForm: http://wwwsearch.sourceforge.net/old/ClientForm/ .. _Creating a pull request: https://help.github.com/en/articles/creating-a-pull-request .. _cryptography: https://cryptography.io/en/latest/ -.. _cssselect: https://github.com/scrapy/cssselect/ -.. _docstrings: https://docs.python.org/glossary.html#term-docstring -.. _KeyboardInterrupt: https://docs.python.org/library/exceptions.html#KeyboardInterrupt +.. _docstrings: https://docs.python.org/3/glossary.html#term-docstring +.. _KeyboardInterrupt: https://docs.python.org/3/library/exceptions.html#KeyboardInterrupt .. _LevelDB: https://github.com/google/leveldb -.. _lxml: http://lxml.de/ +.. _lxml: https://lxml.de/ .. _marshal: https://docs.python.org/2/library/marshal.html .. _parsel.csstranslator.GenericTranslator: https://parsel.readthedocs.io/en/latest/parsel.html#parsel.csstranslator.GenericTranslator .. _parsel.csstranslator.HTMLTranslator: https://parsel.readthedocs.io/en/latest/parsel.html#parsel.csstranslator.HTMLTranslator @@ -3068,11 +3511,11 @@ First release of Scrapy. .. _queuelib: https://github.com/scrapy/queuelib .. _registered with IANA: https://www.iana.org/assignments/media-types/media-types.xhtml .. _resource: https://docs.python.org/2/library/resource.html -.. _robots.txt: http://www.robotstxt.org/ +.. _robots.txt: https://www.robotstxt.org/ .. _scrapely: https://github.com/scrapy/scrapely .. _service_identity: https://service-identity.readthedocs.io/en/stable/ .. _six: https://six.readthedocs.io/ -.. _tox: https://pypi.python.org/pypi/tox +.. _tox: https://pypi.org/project/tox/ .. _Twisted: https://twistedmatrix.com/trac/ .. _Twisted - hello, asynchronous programming: http://jessenoller.com/blog/2009/02/11/twisted-hello-asynchronous-programming/ .. _w3lib: https://github.com/scrapy/w3lib diff --git a/docs/topics/asyncio.rst b/docs/topics/asyncio.rst new file mode 100644 index 000000000..038a459fd --- /dev/null +++ b/docs/topics/asyncio.rst @@ -0,0 +1,28 @@ +======= +asyncio +======= + +.. versionadded:: 2.0 + +Scrapy has partial support :mod:`asyncio`. After you :ref:`install the asyncio +reactor `, you may use :mod:`asyncio` and +:mod:`asyncio`-powered libraries in any :doc:`coroutine `. + +.. warning:: :mod:`asyncio` support in Scrapy is experimental. Future Scrapy + versions may introduce related changes without a deprecation + period or warning. + +.. _install-asyncio: + +Installing the asyncio reactor +============================== + +To enable :mod:`asyncio` support, set the :setting:`TWISTED_REACTOR` setting to +``'twisted.internet.asyncioreactor.AsyncioSelectorReactor'``. + +If you are using :class:`~scrapy.crawler.CrawlerRunner`, you also need to +install the :class:`~twisted.internet.asyncioreactor.AsyncioSelectorReactor` +reactor manually. You can do that using +:func:`~scrapy.utils.reactor.install_reactor`:: + + install_reactor('twisted.internet.asyncioreactor.AsyncioSelectorReactor') diff --git a/docs/topics/broad-crawls.rst b/docs/topics/broad-crawls.rst index 4922694ee..63b60312e 100644 --- a/docs/topics/broad-crawls.rst +++ b/docs/topics/broad-crawls.rst @@ -188,7 +188,7 @@ AjaxCrawlMiddleware helps to crawl them correctly. It is turned OFF by default because it has some performance overhead, and enabling it for focused crawls doesn't make much sense. -.. _ajax crawlable: https://developers.google.com/webmasters/ajax-crawling/docs/getting-started +.. _ajax crawlable: https://developers.google.com/search/docs/ajax-crawling/docs/getting-started .. _broad-crawls-bfo: diff --git a/docs/topics/coroutines.rst b/docs/topics/coroutines.rst new file mode 100644 index 000000000..487cf4c6c --- /dev/null +++ b/docs/topics/coroutines.rst @@ -0,0 +1,110 @@ +========== +Coroutines +========== + +.. versionadded:: 2.0 + +Scrapy has :ref:`partial support ` for the +:ref:`coroutine syntax `. + +.. warning:: :mod:`asyncio` support in Scrapy is experimental. Future Scrapy + versions may introduce related API and behavior changes without a + deprecation period or warning. + +.. _coroutine-support: + +Supported callables +=================== + +The following callables may be defined as coroutines using ``async def``, and +hence use coroutine syntax (e.g. ``await``, ``async for``, ``async with``): + +- :class:`~scrapy.http.Request` callbacks. + + The following are known caveats of the current implementation that we aim + to address in future versions of Scrapy: + + - The callback output is not processed until the whole callback finishes. + + As a side effect, if the callback raises an exception, none of its + output is processed. + + - Because `asynchronous generators were introduced in Python 3.6`_, you + can only use ``yield`` if you are using Python 3.6 or later. + + If you need to output multiple items or requests and you are using + Python 3.5, return an iterable (e.g. a list) instead. + +- The :meth:`process_item` method of + :ref:`item pipelines `. + +- The + :meth:`~scrapy.downloadermiddlewares.DownloaderMiddleware.process_request`, + :meth:`~scrapy.downloadermiddlewares.DownloaderMiddleware.process_response`, + and + :meth:`~scrapy.downloadermiddlewares.DownloaderMiddleware.process_exception` + methods of + :ref:`downloader middlewares `. + +- :ref:`Signal handlers that support deferreds `. + +.. _asynchronous generators were introduced in Python 3.6: https://www.python.org/dev/peps/pep-0525/ + +Usage +===== + +There are several use cases for coroutines in Scrapy. Code that would +return Deferreds when written for previous Scrapy versions, such as downloader +middlewares and signal handlers, can be rewritten to be shorter and cleaner:: + + class DbPipeline: + def _update_item(self, data, item): + item['field'] = data + return item + + def process_item(self, item, spider): + dfd = db.get_some_data(item['id']) + dfd.addCallback(self._update_item, item) + return dfd + +becomes:: + + class DbPipeline: + async def process_item(self, item, spider): + item['field'] = await db.get_some_data(item['id']) + return item + +Coroutines may be used to call asynchronous code. This includes other +coroutines, functions that return Deferreds and functions that return +`awaitable objects`_ such as :class:`~asyncio.Future`. This means you can use +many useful Python libraries providing such code:: + + class MySpider(Spider): + # ... + async def parse_with_deferred(self, response): + additional_response = await treq.get('https://additional.url') + additional_data = await treq.content(additional_response) + # ... use response and additional_data to yield items and requests + + async def parse_with_asyncio(self, response): + async with aiohttp.ClientSession() as session: + async with session.get('https://additional.url') as additional_response: + additional_data = await r.text() + # ... use response and additional_data to yield items and requests + +.. note:: Many libraries that use coroutines, such as `aio-libs`_, require the + :mod:`asyncio` loop and to use them you need to + :doc:`enable asyncio support in Scrapy`. + +Common use cases for asynchronous code include: + +* requesting data from websites, databases and other services (in callbacks, + pipelines and middlewares); +* storing data in databases (in pipelines and middlewares); +* delaying the spider initialization until some external event (in the + :signal:`spider_opened` handler); +* calling asynchronous Scrapy methods like ``ExecutionEngine.download`` (see + :ref:`the screenshot pipeline example`). + +.. _aio-libs: https://github.com/aio-libs +.. _awaitable objects: https://docs.python.org/3/glossary.html#term-awaitable diff --git a/docs/topics/downloader-middleware.rst b/docs/topics/downloader-middleware.rst index 3ec6e0c17..73648994d 100644 --- a/docs/topics/downloader-middleware.rst +++ b/docs/topics/downloader-middleware.rst @@ -259,8 +259,8 @@ COOKIES_DEBUG Default: ``False`` -If enabled, Scrapy will log all cookies sent in requests (ie. ``Cookie`` -header) and all cookies received in responses (ie. ``Set-Cookie`` header). +If enabled, Scrapy will log all cookies sent in requests (i.e. ``Cookie`` +header) and all cookies received in responses (i.e. ``Set-Cookie`` header). Here's an example of a log with :setting:`COOKIES_DEBUG` enabled:: @@ -709,7 +709,7 @@ HttpCompressionMiddleware provided `brotlipy`_ is installed. .. _brotli-compressed: https://www.ietf.org/rfc/rfc7932.txt -.. _brotlipy: https://pypi.python.org/pypi/brotlipy +.. _brotlipy: https://pypi.org/project/brotlipy/ HttpCompressionMiddleware Settings ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ @@ -872,6 +872,10 @@ Default: ``[]`` Meta tags within these tags are ignored. +.. versionchanged:: 2.0 + The default value of :setting:`METAREFRESH_IGNORE_TAGS` changed from + ``['script', 'noscript']`` to ``[]``. + .. setting:: METAREFRESH_MAXDELAY METAREFRESH_MAXDELAY @@ -1038,7 +1042,7 @@ Based on `RobotFileParser * is Python's built-in robots.txt_ parser * is compliant with `Martijn Koster's 1996 draft specification - `_ + `_ * lacks support for wildcard matching @@ -1061,7 +1065,7 @@ Based on `Reppy `_: `_ * is compliant with `Martijn Koster's 1996 draft specification - `_ + `_ * supports wildcard matching @@ -1086,7 +1090,7 @@ Based on `Robotexclusionrulesparser `_: * implemented in Python * is compliant with `Martijn Koster's 1996 draft specification - `_ + `_ * supports wildcard matching @@ -1115,7 +1119,7 @@ implementing the methods described below. .. autoclass:: RobotParser :members: -.. _robots.txt: http://www.robotstxt.org/ +.. _robots.txt: https://www.robotstxt.org/ DownloaderStats --------------- @@ -1155,7 +1159,7 @@ AjaxCrawlMiddleware Middleware that finds 'AJAX crawlable' page variants based on meta-fragment html tag. See - https://developers.google.com/webmasters/ajax-crawling/docs/getting-started + https://developers.google.com/search/docs/ajax-crawling/docs/getting-started for more info. .. note:: diff --git a/docs/topics/dynamic-content.rst b/docs/topics/dynamic-content.rst index 1c3607860..b98133676 100644 --- a/docs/topics/dynamic-content.rst +++ b/docs/topics/dynamic-content.rst @@ -241,12 +241,12 @@ along with `scrapy-selenium`_ for seamless integration. .. _headless browser: https://en.wikipedia.org/wiki/Headless_browser .. _JavaScript: https://en.wikipedia.org/wiki/JavaScript .. _js2xml: https://github.com/scrapinghub/js2xml -.. _json.loads: https://docs.python.org/library/json.html#json.loads +.. _json.loads: https://docs.python.org/3/library/json.html#json.loads .. _pytesseract: https://github.com/madmaze/pytesseract -.. _regular expression: https://docs.python.org/library/re.html +.. _regular expression: https://docs.python.org/3/library/re.html .. _scrapy-selenium: https://github.com/clemfromspace/scrapy-selenium .. _scrapy-splash: https://github.com/scrapy-plugins/scrapy-splash -.. _Selenium: https://www.seleniumhq.org/ +.. _Selenium: https://www.selenium.dev/ .. _Splash: https://github.com/scrapinghub/splash .. _tabula-py: https://github.com/chezou/tabula-py .. _wget: https://www.gnu.org/software/wget/ diff --git a/docs/topics/exporters.rst b/docs/topics/exporters.rst index b8d898022..d411e2eed 100644 --- a/docs/topics/exporters.rst +++ b/docs/topics/exporters.rst @@ -137,7 +137,7 @@ output examples, which assume you're exporting these two items:: BaseItemExporter ---------------- -.. class:: BaseItemExporter(fields_to_export=None, export_empty_fields=False, encoding='utf-8', indent=0) +.. class:: BaseItemExporter(fields_to_export=None, export_empty_fields=False, encoding='utf-8', indent=0, dont_fail=False) This is the (abstract) base class for all Item Exporters. It provides support for common features used by all (concrete) Item Exporters, such as @@ -148,6 +148,9 @@ BaseItemExporter populate their respective instance attributes: :attr:`fields_to_export`, :attr:`export_empty_fields`, :attr:`encoding`, :attr:`indent`. + .. versionadded:: 2.0 + The *dont_fail* parameter. + .. method:: export_item(item) Exports the given item. This method must be implemented in subclasses. diff --git a/docs/topics/extensions.rst b/docs/topics/extensions.rst index 0a7455ec9..dc057f6b6 100644 --- a/docs/topics/extensions.rst +++ b/docs/topics/extensions.rst @@ -63,7 +63,7 @@ but disabled unless the :setting:`HTTPCACHE_ENABLED` setting is set. Disabling an extension ====================== -In order to disable an extension that comes enabled by default (ie. those +In order to disable an extension that comes enabled by default (i.e. those included in the :setting:`EXTENSIONS_BASE` setting) you must set its order to ``None``. For example:: @@ -345,7 +345,7 @@ signal is received. The information dumped is the following: After the stack trace and engine status is dumped, the Scrapy process continues running normally. -This extension only works on POSIX-compliant platforms (ie. not Windows), +This extension only works on POSIX-compliant platforms (i.e. not Windows), because the `SIGQUIT`_ and `SIGUSR2`_ signals are not available on Windows. There are at least two ways to send Scrapy the `SIGQUIT`_ signal: @@ -370,7 +370,7 @@ running normally. For more info see `Debugging in Python`_. -This extension only works on POSIX-compliant platforms (ie. not Windows). +This extension only works on POSIX-compliant platforms (i.e. not Windows). .. _Python debugger: https://docs.python.org/2/library/pdb.html .. _Debugging in Python: https://pythonconquerstheuniverse.wordpress.com/2009/09/10/debugging-in-python/ diff --git a/docs/topics/feed-exports.rst b/docs/topics/feed-exports.rst index 7481b1a99..9e5968a29 100644 --- a/docs/topics/feed-exports.rst +++ b/docs/topics/feed-exports.rst @@ -12,7 +12,7 @@ generating an "export file" with the scraped data (commonly called "export feed") to be consumed by other systems. Scrapy provides this functionality out of the box with the Feed Exports, which -allows you to generate a feed with the scraped items, using multiple +allows you to generate feeds with the scraped items, using multiple serialization formats and storage backends. .. _topics-feed-format: @@ -36,7 +36,7 @@ But you can also extend the supported format through the JSON ---- - * :setting:`FEED_FORMAT`: ``json`` + * Value for the ``format`` key in the :setting:`FEEDS` setting: ``json`` * Exporter used: :class:`~scrapy.exporters.JsonItemExporter` * See :ref:`this warning ` if you're using JSON with large feeds. @@ -46,7 +46,7 @@ JSON JSON lines ---------- - * :setting:`FEED_FORMAT`: ``jsonlines`` + * Value for the ``format`` key in the :setting:`FEEDS` setting: ``jsonlines`` * Exporter used: :class:`~scrapy.exporters.JsonLinesItemExporter` .. _topics-feed-format-csv: @@ -54,7 +54,7 @@ JSON lines CSV --- - * :setting:`FEED_FORMAT`: ``csv`` + * Value for the ``format`` key in the :setting:`FEEDS` setting: ``csv`` * Exporter used: :class:`~scrapy.exporters.CsvItemExporter` * To specify columns to export and their order use :setting:`FEED_EXPORT_FIELDS`. Other feed exporters can also use this @@ -66,7 +66,7 @@ CSV XML --- - * :setting:`FEED_FORMAT`: ``xml`` + * Value for the ``format`` key in the :setting:`FEEDS` setting: ``xml`` * Exporter used: :class:`~scrapy.exporters.XmlItemExporter` .. _topics-feed-format-pickle: @@ -74,7 +74,7 @@ XML Pickle ------ - * :setting:`FEED_FORMAT`: ``pickle`` + * Value for the ``format`` key in the :setting:`FEEDS` setting: ``pickle`` * Exporter used: :class:`~scrapy.exporters.PickleItemExporter` .. _topics-feed-format-marshal: @@ -82,7 +82,7 @@ Pickle Marshal ------- - * :setting:`FEED_FORMAT`: ``marshal`` + * Value for the ``format`` key in the :setting:`FEEDS` setting: ``marshal`` * Exporter used: :class:`~scrapy.exporters.MarshalItemExporter` @@ -91,8 +91,8 @@ Marshal Storages ======== -When using the feed exports you define where to store the feed using a URI_ -(through the :setting:`FEED_URI` setting). The feed exports supports multiple +When using the feed exports you define where to store the feed using one or multiple URIs_ +(through the :setting:`FEEDS` setting). The feed exports supports multiple storage backend types which are defined by the URI scheme. The storages backends supported out of the box are: @@ -211,38 +211,66 @@ Settings These are the settings used for configuring the feed exports: - * :setting:`FEED_URI` (mandatory) - * :setting:`FEED_FORMAT` + * :setting:`FEEDS` (mandatory) + * :setting:`FEED_EXPORT_ENCODING` + * :setting:`FEED_STORE_EMPTY` + * :setting:`FEED_EXPORT_FIELDS` + * :setting:`FEED_EXPORT_INDENT` * :setting:`FEED_STORAGES` * :setting:`FEED_STORAGE_FTP_ACTIVE` * :setting:`FEED_STORAGE_S3_ACL` * :setting:`FEED_EXPORTERS` - * :setting:`FEED_STORE_EMPTY` - * :setting:`FEED_EXPORT_ENCODING` - * :setting:`FEED_EXPORT_FIELDS` - * :setting:`FEED_EXPORT_INDENT` .. currentmodule:: scrapy.extensions.feedexport -.. setting:: FEED_URI +.. setting:: FEEDS -FEED_URI --------- +FEEDS +----- -Default: ``None`` +.. versionadded:: 2.1 -The URI of the export feed. See :ref:`topics-feed-storage-backends` for -supported URI schemes. +Default: ``{}`` -This setting is required for enabling the feed exports. +A dictionary in which every key is a feed URI (or a :class:`pathlib.Path` +object) and each value is a nested dictionary containing configuration +parameters for the specific feed. +This setting is required for enabling the feed export feature. -.. setting:: FEED_FORMAT +See :ref:`topics-feed-storage-backends` for supported URI schemes. -FEED_FORMAT ------------ +For instance:: -The serialization format to be used for the feed. See -:ref:`topics-feed-format` for possible values. + { + 'items.json': { + 'format': 'json', + 'encoding': 'utf8', + 'store_empty': False, + 'fields': None, + 'indent': 4, + }, + '/home/user/documents/items.xml': { + 'format': 'xml', + 'fields': ['name', 'price'], + 'encoding': 'latin1', + 'indent': 8, + }, + pathlib.Path('items.csv'): { + 'format': 'csv', + 'fields': ['price', 'name'], + }, + } + +The following is a list of the accepted keys and the setting that is used +as a fallback value if that key is not provided for a specific feed definition. + +* ``format``: the serialization format to be used for the feed. + See :ref:`topics-feed-format` for possible values. + Mandatory, no fallback setting +* ``encoding``: falls back to :setting:`FEED_EXPORT_ENCODING` +* ``fields``: falls back to :setting:`FEED_EXPORT_FIELDS` +* ``indent``: falls back to :setting:`FEED_EXPORT_INDENT` +* ``store_empty``: falls back to :setting:`FEED_STORE_EMPTY` .. setting:: FEED_EXPORT_ENCODING @@ -301,7 +329,7 @@ FEED_STORE_EMPTY Default: ``False`` -Whether to export empty feeds (ie. feeds with no items). +Whether to export empty feeds (i.e. feeds with no items). .. setting:: FEED_STORAGES @@ -397,7 +425,7 @@ format in :setting:`FEED_EXPORTERS`. E.g., to disable the built-in CSV exporter 'csv': None, } -.. _URI: https://en.wikipedia.org/wiki/Uniform_Resource_Identifier +.. _URIs: https://en.wikipedia.org/wiki/Uniform_Resource_Identifier .. _Amazon S3: https://aws.amazon.com/s3/ .. _botocore: https://github.com/boto/botocore .. _Canned ACL: https://docs.aws.amazon.com/AmazonS3/latest/dev/acl-overview.html#canned-acl diff --git a/docs/topics/item-pipeline.rst b/docs/topics/item-pipeline.rst index cdc4953c2..98e2506e5 100644 --- a/docs/topics/item-pipeline.rst +++ b/docs/topics/item-pipeline.rst @@ -158,18 +158,20 @@ method and how to clean up the resources properly.:: self.db[self.collection_name].insert_one(dict(item)) return item -.. _MongoDB: https://www.mongodb.org/ -.. _pymongo: https://api.mongodb.org/python/current/ +.. _MongoDB: https://www.mongodb.com/ +.. _pymongo: https://api.mongodb.com/python/current/ +.. _ScreenshotPipeline: + Take screenshot of item ----------------------- This example demonstrates how to return a :class:`~twisted.internet.defer.Deferred` from the :meth:`process_item` method. It uses Splash_ to render screenshot of item url. Pipeline -makes request to locally running instance of Splash_. After request is downloaded -and Deferred callback fires, it saves item to a file and adds filename to an item. +makes request to locally running instance of Splash_. After request is downloaded, +it saves the screenshot to a file and adds filename to the item. :: @@ -184,15 +186,12 @@ and Deferred callback fires, it saves item to a file and adds filename to an ite SPLASH_URL = "http://localhost:8050/render.png?url={}" - def process_item(self, item, spider): + async def process_item(self, item, spider): encoded_item_url = quote(item["url"]) screenshot_url = self.SPLASH_URL.format(encoded_item_url) request = scrapy.Request(screenshot_url) - dfd = spider.crawler.engine.download(request, spider) - dfd.addBoth(self.return_item, item) - return dfd + response = await spider.crawler.engine.download(request, spider) - def return_item(self, response, item): if response.status != 200: # Error happened, return item. return item diff --git a/docs/topics/items.rst b/docs/topics/items.rst index 15313775b..44643cb67 100644 --- a/docs/topics/items.rst +++ b/docs/topics/items.rst @@ -166,7 +166,7 @@ If your item contains mutable_ values like lists or dictionaries, a shallow copy will keep references to the same mutable values across all different copies. -.. _mutable: https://docs.python.org/glossary.html#term-mutable +.. _mutable: https://docs.python.org/3/glossary.html#term-mutable For example, if you have an item with a list of tags, and you create a shallow copy of that item, both the original item and the copy have the same list of @@ -177,7 +177,7 @@ If that is not the desired behavior, use a deep copy instead. See the `documentation of the copy module`_ for more information. -.. _documentation of the copy module: https://docs.python.org/library/copy.html +.. _documentation of the copy module: https://docs.python.org/3/library/copy.html To create a shallow copy of an item, you can either call :meth:`~scrapy.item.Item.copy` on an existing item diff --git a/docs/topics/jobs.rst b/docs/topics/jobs.rst index 8816a028c..58601824a 100644 --- a/docs/topics/jobs.rst +++ b/docs/topics/jobs.rst @@ -22,7 +22,7 @@ Job directory To enable persistence support you just need to define a *job directory* through the ``JOBDIR`` setting. This directory will be for storing all required data to -keep the state of a single job (ie. a spider run). It's important to note that +keep the state of a single job (i.e. a spider run). It's important to note that this directory must not be shared by different spiders, or even different jobs/runs of the same spider, as it's meant to be used for storing the state of a *single* job. @@ -68,6 +68,9 @@ Cookies may expire. So, if you don't resume your spider quickly the requests scheduled may no longer work. This won't be an issue if you spider doesn't rely on cookies. + +.. _request-serialization: + Request serialization --------------------- diff --git a/docs/topics/leaks.rst b/docs/topics/leaks.rst index 9fee333ac..c0c83fc84 100644 --- a/docs/topics/leaks.rst +++ b/docs/topics/leaks.rst @@ -206,7 +206,7 @@ objects. If this is your case, and you can't find your leaks using ``trackref``, you still have another resource: the `Guppy library`_. If you're using Python3, see :ref:`topics-leaks-muppy`. -.. _Guppy library: https://pypi.python.org/pypi/guppy +.. _Guppy library: https://pypi.org/project/guppy/ If you use ``pip``, you can install Guppy with the following command:: @@ -311,9 +311,9 @@ though neither Scrapy nor your project are leaking memory. This is due to a (not so well) known problem of Python, which may not return released memory to the operating system in some cases. For more information on this issue see: -* `Python Memory Management `_ -* `Python Memory Management Part 2 `_ -* `Python Memory Management Part 3 `_ +* `Python Memory Management `_ +* `Python Memory Management Part 2 `_ +* `Python Memory Management Part 3 `_ The improvements proposed by Evan Jones, which are detailed in `this paper`_, got merged in Python 2.5, but this only reduces the problem, it doesn't fix it @@ -327,7 +327,7 @@ completely. To quote the paper: to move to a compacting garbage collector, which is able to move objects in memory. This would require significant changes to the Python interpreter.* -.. _this paper: http://www.evanjones.ca/memoryallocator/ +.. _this paper: https://www.evanjones.ca/memoryallocator/ To keep memory consumption reasonable you can split the job into several smaller jobs or enable :ref:`persistent job queue ` diff --git a/docs/topics/link-extractors.rst b/docs/topics/link-extractors.rst index 2119cb8f8..0162a331a 100644 --- a/docs/topics/link-extractors.rst +++ b/docs/topics/link-extractors.rst @@ -49,7 +49,7 @@ LxmlLinkExtractor :type allow: a regular expression (or list of) :param deny: a single regular expression (or list of regular expressions) - that the (absolute) urls must match in order to be excluded (ie. not + that the (absolute) urls must match in order to be excluded (i.e. not extracted). It has precedence over the ``allow`` parameter. If not given (or empty) it won't exclude any links. :type deny: a regular expression (or list of) @@ -64,9 +64,13 @@ LxmlLinkExtractor :param deny_extensions: a single value or list of strings containing extensions that should be ignored when extracting links. - If not given, it will default to the - ``IGNORED_EXTENSIONS`` list defined in the - `scrapy.linkextractors`_ package. + If not given, it will default to + :data:`scrapy.linkextractors.IGNORED_EXTENSIONS`. + + .. versionchanged:: 2.0 + :data:`~scrapy.linkextractors.IGNORED_EXTENSIONS` now includes + ``7z``, ``7zip``, ``apk``, ``bz2``, ``cdr``, ``dmg``, ``ico``, + ``iso``, ``tar``, ``tar.gz``, ``webm``, and ``xz``. :type deny_extensions: list :param restrict_xpaths: is an XPath (or list of XPath's) which defines diff --git a/docs/topics/loaders.rst b/docs/topics/loaders.rst index 9d5fccbbc..5f75ccbff 100644 --- a/docs/topics/loaders.rst +++ b/docs/topics/loaders.rst @@ -136,6 +136,9 @@ with the data to be parsed, and return a parsed value. So you can use any function as input or output processor. The only requirement is that they must accept one (and only one) positional argument, which will be an iterable. +.. versionchanged:: 2.0 + Processors no longer need to be methods. + .. note:: Both input and output processors must receive an iterable as their first argument. The output of those functions can be anything. The result of input processors will be appended to an internal list (in the Loader) diff --git a/docs/topics/media-pipeline.rst b/docs/topics/media-pipeline.rst index 67a0bfdba..cd84905c5 100644 --- a/docs/topics/media-pipeline.rst +++ b/docs/topics/media-pipeline.rst @@ -116,12 +116,6 @@ For the Images Pipeline, set the :setting:`IMAGES_STORE` setting:: Supported Storage ================= -File system is currently the only officially supported storage, but there are -also support for storing files in `Amazon S3`_ and `Google Cloud Storage`_. - -.. _Amazon S3: https://aws.amazon.com/s3/ -.. _Google Cloud Storage: https://cloud.google.com/storage/ - File system storage ------------------- @@ -147,9 +141,13 @@ Where: * ``full`` is a sub-directory to separate full images from thumbnails (if used). For more info see :ref:`topics-images-thumbnails`. +.. _media-pipeline-ftp: + FTP server storage ------------------ +.. versionadded:: 2.0 + :setting:`FILES_STORE` and :setting:`IMAGES_STORE` can point to an FTP server. Scrapy will automatically upload the files to the server. @@ -573,6 +571,8 @@ See here the methods that you can override in your custom Images Pipeline: By default, the :meth:`item_completed` method returns the item. +.. _media-pipeline-example: + Custom Images pipeline example ============================== diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index 8997a7f19..b2a60ff39 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -31,6 +31,8 @@ Request objects a :class:`Response`. :param url: the URL of this request + + If the URL is invalid, a :exc:`ValueError` exception is raised. :type url: string :param callback: the function that will be called with the response of this @@ -125,6 +127,10 @@ Request objects :exc:`~twisted.python.failure.Failure` as first parameter. For more information, see :ref:`topics-request-response-ref-errbacks` below. + + .. versionchanged:: 2.0 + The *callback* parameter is no longer required when the *errback* + parameter is specified. :type errback: callable :param flags: Flags sent to the request, can be used for logging or similar purposes. @@ -396,7 +402,7 @@ The FormRequest class extends the base :class:`Request` with functionality for dealing with HTML forms. It uses `lxml.html forms`_ to pre-populate form fields with form data from :class:`Response` objects. -.. _lxml.html forms: http://lxml.de/lxmlhtml.html#forms +.. _lxml.html forms: https://lxml.de/lxmlhtml.html#forms .. class:: FormRequest(url, [formdata, ...]) @@ -609,7 +615,10 @@ Response objects :param request: the initial value of the :attr:`Response.request` attribute. This represents the :class:`Request` that generated this response. - :type request: :class:`Request` object + :type request: scrapy.http.Request + + :param certificate: an object representing the server's SSL certificate. + :type certificate: twisted.internet.ssl.Certificate .. attribute:: Response.url @@ -664,7 +673,7 @@ Response objects .. attribute:: Response.meta A shortcut to the :attr:`Request.meta` attribute of the - :attr:`Response.request` object (ie. ``self.request.meta``). + :attr:`Response.request` object (i.e. ``self.request.meta``). Unlike the :attr:`Response.request` attribute, the :attr:`Response.meta` attribute is propagated along redirects and retries, so you will get @@ -672,6 +681,20 @@ Response objects .. seealso:: :attr:`Request.meta` attribute + .. attribute:: Response.cb_kwargs + + .. versionadded:: 2.0 + + A shortcut to the :attr:`Request.cb_kwargs` attribute of the + :attr:`Response.request` object (i.e. ``self.request.cb_kwargs``). + + Unlike the :attr:`Response.request` attribute, the + :attr:`Response.cb_kwargs` attribute is propagated along redirects and + retries, so you will get the original :attr:`Request.cb_kwargs` sent + from your spider. + + .. seealso:: :attr:`Request.cb_kwargs` attribute + .. attribute:: Response.flags A list that contains flags for this response. Flags are labels used for @@ -679,6 +702,13 @@ Response objects they're shown on the string representation of the Response (`__str__` method) which is used by the engine for logging. + .. attribute:: Response.certificate + + A :class:`twisted.internet.ssl.Certificate` object representing + the server's SSL certificate. + + Only populated for ``https`` responses, ``None`` otherwise. + .. method:: Response.copy() Returns a new Response which is a copy of this Response. @@ -760,7 +790,7 @@ TextResponse objects 1. the encoding passed in the ``__init__`` method ``encoding`` argument 2. the encoding declared in the Content-Type HTTP header. If this - encoding is not valid (ie. unknown), it is ignored and the next + encoding is not valid (i.e. unknown), it is ignored and the next resolution mechanism is tried. 3. the encoding declared in the response body. The TextResponse class diff --git a/docs/topics/selectors.rst b/docs/topics/selectors.rst index 8ec758b0e..1f7802c98 100644 --- a/docs/topics/selectors.rst +++ b/docs/topics/selectors.rst @@ -35,12 +35,11 @@ defines selectors to associate those styles with specific HTML elements. in speed and parsing accuracy to lxml. .. _BeautifulSoup: https://www.crummy.com/software/BeautifulSoup/ -.. _lxml: http://lxml.de/ +.. _lxml: https://lxml.de/ .. _ElementTree: https://docs.python.org/2/library/xml.etree.elementtree.html -.. _cssselect: https://pypi.python.org/pypi/cssselect/ -.. _XPath: https://www.w3.org/TR/xpath +.. _XPath: https://www.w3.org/TR/xpath/all/ .. _CSS: https://www.w3.org/TR/selectors -.. _parsel: https://parsel.readthedocs.io/ +.. _parsel: https://parsel.readthedocs.io/en/latest/ Using selectors =============== @@ -255,7 +254,7 @@ that Scrapy (parsel) implements a couple of **non-standard pseudo-elements**: They will most probably not work with other libraries like `lxml`_ or `PyQuery`_. -.. _PyQuery: https://pypi.python.org/pypi/pyquery +.. _PyQuery: https://pypi.org/project/pyquery/ Examples: @@ -309,7 +308,7 @@ Examples: make much sense: text nodes do not have attributes, and attribute values are string values already and do not have children nodes. -.. _CSS Selectors: https://www.w3.org/TR/css3-selectors/#selectors +.. _CSS Selectors: https://www.w3.org/TR/selectors-3/#selectors .. _topics-selectors-nesting-selectors: @@ -504,7 +503,7 @@ Another common case would be to extract all direct ``

`` children: For more details about relative XPaths see the `Location Paths`_ section in the XPath specification. -.. _Location Paths: https://www.w3.org/TR/xpath#location-paths +.. _Location Paths: https://www.w3.org/TR/xpath/all/#location-paths When querying by class, consider using CSS ------------------------------------------ @@ -612,7 +611,7 @@ But using the ``.`` to mean the node, works: >>> sel.xpath("//a[contains(., 'Next Page')]").getall() ['Click here to go to the Next Page'] -.. _`XPath string function`: https://www.w3.org/TR/xpath/#section-String-Functions +.. _`XPath string function`: https://www.w3.org/TR/xpath/all/#section-String-Functions .. _topics-selectors-xpath-variables: @@ -764,7 +763,7 @@ Set operations These can be handy for excluding parts of a document tree before extracting text elements for example. -Example extracting microdata (sample content taken from http://schema.org/Product) +Example extracting microdata (sample content taken from https://schema.org/Product) with groups of itemscopes and corresponding itemprops:: >>> doc = u""" @@ -986,7 +985,7 @@ a :class:`~scrapy.http.HtmlResponse` object like this:: sel = Selector(html_response) 1. Select all ``

`` elements from an HTML response body, returning a list of - :class:`Selector` objects (ie. a :class:`SelectorList` object):: + :class:`Selector` objects (i.e. a :class:`SelectorList` object):: sel.xpath("//h1") @@ -1013,7 +1012,7 @@ instantiated with an :class:`~scrapy.http.XmlResponse` object:: sel = Selector(xml_response) 1. Select all ```` elements from an XML response body, returning a list - of :class:`Selector` objects (ie. a :class:`SelectorList` object):: + of :class:`Selector` objects (i.e. a :class:`SelectorList` object):: sel.xpath("//product") diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index fa63a5807..c01202a10 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -248,7 +248,7 @@ CONCURRENT_REQUESTS Default: ``16`` -The maximum number of concurrent (ie. simultaneous) requests that will be +The maximum number of concurrent (i.e. simultaneous) requests that will be performed by the Scrapy downloader. .. setting:: CONCURRENT_REQUESTS_PER_DOMAIN @@ -258,7 +258,7 @@ CONCURRENT_REQUESTS_PER_DOMAIN Default: ``8`` -The maximum number of concurrent (ie. simultaneous) requests that will be +The maximum number of concurrent (i.e. simultaneous) requests that will be performed to any single domain. See also: :ref:`topics-autothrottle` and its @@ -272,7 +272,7 @@ CONCURRENT_REQUESTS_PER_IP Default: ``0`` -The maximum number of concurrent (ie. simultaneous) requests that will be +The maximum number of concurrent (i.e. simultaneous) requests that will be performed to any single IP. If non-zero, the :setting:`CONCURRENT_REQUESTS_PER_DOMAIN` setting is ignored, and this one is used instead. In other words, concurrency limits will be applied per IP, not @@ -381,6 +381,8 @@ DNS in-memory cache size. DNS_RESOLVER ------------ +.. versionadded:: 2.0 + Default: ``'scrapy.resolver.CachingThreadedResolver'`` The class to be used to resolve DNS names. The default ``scrapy.resolver.CachingThreadedResolver`` @@ -1258,6 +1260,9 @@ does not work together with :setting:`CONCURRENT_REQUESTS_PER_IP`. SCRAPER_SLOT_MAX_ACTIVE_SIZE ---------------------------- + +.. versionadded:: 2.0 + Default: ``5_000_000`` Soft limit (in bytes) for response data being processed. @@ -1447,24 +1452,95 @@ in the ``project`` subdirectory. TWISTED_REACTOR --------------- +.. versionadded:: 2.0 + Default: ``None`` -Import path of a given Twisted reactor, for instance: -:class:`twisted.internet.asyncioreactor.AsyncioSelectorReactor`. +Import path of a given :mod:`~twisted.internet.reactor`. -Scrapy will install this reactor if no other is installed yet, such as when -the ``scrapy`` CLI program is invoked or when using the -:class:`~scrapy.crawler.CrawlerProcess` class. If you are using the -:class:`~scrapy.crawler.CrawlerRunner` class, you need to install the correct -reactor manually. An exception will be raised if the installation fails. +Scrapy will install this reactor if no other reactor is installed yet, such as +when the ``scrapy`` CLI program is invoked or when using the +:class:`~scrapy.crawler.CrawlerProcess` class. -The default value for this option is currently ``None``, which means that Scrapy -will not attempt to install any specific reactor, and the default one defined by -Twisted for the current platform will be used. This is to maintain backward -compatibility and avoid possible problems caused by using a non-default reactor. +If you are using the :class:`~scrapy.crawler.CrawlerRunner` class, you also +need to install the correct reactor manually. You can do that using +:func:`~scrapy.utils.reactor.install_reactor`: -For additional information, please see -:doc:`core/howto/choosing-reactor`. +.. autofunction:: scrapy.utils.reactor.install_reactor + +If a reactor is already installed, +:func:`~scrapy.utils.reactor.install_reactor` has no effect. + +:meth:`CrawlerRunner.__init__ ` raises +:exc:`Exception` if the installed reactor does not match the +:setting:`TWISTED_REACTOR` setting; therfore, having top-level +:mod:`~twisted.internet.reactor` imports in project files and imported +third-party libraries will make Scrapy raise :exc:`Exception` when +it checks which reactor is installed. + +In order to use the reactor installed by Scrapy:: + + import scrapy + from twisted.internet import reactor + + + class QuotesSpider(scrapy.Spider): + name = 'quotes' + + def __init__(self, *args, **kwargs): + self.timeout = int(kwargs.pop('timeout', '60')) + super(QuotesSpider, self).__init__(*args, **kwargs) + + def start_requests(self): + reactor.callLater(self.timeout, self.stop) + + urls = ['http://quotes.toscrape.com/page/1'] + for url in urls: + yield scrapy.Request(url=url, callback=self.parse) + + def parse(self, response): + for quote in response.css('div.quote'): + yield {'text': quote.css('span.text::text').get()} + + def stop(self): + self.crawler.engine.close_spider(self, 'timeout') + + +which raises :exc:`Exception`, becomes:: + + import scrapy + + + class QuotesSpider(scrapy.Spider): + name = 'quotes' + + def __init__(self, *args, **kwargs): + self.timeout = int(kwargs.pop('timeout', '60')) + super(QuotesSpider, self).__init__(*args, **kwargs) + + def start_requests(self): + from twisted.internet import reactor + reactor.callLater(self.timeout, self.stop) + + urls = ['http://quotes.toscrape.com/page/1'] + for url in urls: + yield scrapy.Request(url=url, callback=self.parse) + + def parse(self, response): + for quote in response.css('div.quote'): + yield {'text': quote.css('span.text::text').get()} + + def stop(self): + self.crawler.engine.close_spider(self, 'timeout') + + +The default value of the :setting:`TWISTED_REACTOR` setting is ``None``, which +means that Scrapy will not attempt to install any specific reactor, and the +default reactor defined by Twisted for the current platform will be used. This +is to maintain backward compatibility and avoid possible problems caused by +using a non-default reactor. + +For additional information, see :doc:`core/howto/choosing-reactor`. .. setting:: URLLENGTH_LIMIT diff --git a/docs/topics/shell.rst b/docs/topics/shell.rst index 3cf8311a6..8f7518b19 100644 --- a/docs/topics/shell.rst +++ b/docs/topics/shell.rst @@ -41,7 +41,7 @@ variable; or by defining it in your :ref:`scrapy.cfg `:: .. _IPython: https://ipython.org/ .. _IPython installation guide: https://ipython.org/install.html -.. _bpython: https://www.bpython-interpreter.org/ +.. _bpython: https://bpython-interpreter.org/ Launch the shell ================ @@ -142,7 +142,7 @@ Example of shell session ======================== Here's an example of a typical shell session where we start by scraping the -https://scrapy.org page, and then proceed to scrape the https://reddit.com +https://scrapy.org page, and then proceed to scrape the https://old.reddit.com/ page. Finally, we modify the (Reddit) request method to POST and re-fetch it getting an error. We end the session by typing Ctrl-D (in Unix systems) or Ctrl-Z in Windows. @@ -182,7 +182,7 @@ After that, we can start playing with the objects: >>> response.xpath('//title/text()').get() 'Scrapy | A Fast and Powerful Scraping and Web Crawling Framework' ->>> fetch("https://reddit.com") +>>> fetch("https://old.reddit.com/") >>> response.xpath('//title/text()').get() 'reddit: the front page of the internet' diff --git a/docs/topics/signals.rst b/docs/topics/signals.rst index 886d1b866..2def53848 100644 --- a/docs/topics/signals.rst +++ b/docs/topics/signals.rst @@ -46,6 +46,7 @@ Here is a simple example showing how you can catch signals and perform some acti def parse(self, response): pass +.. _signal-deferred: Deferred signal handlers ======================== @@ -141,7 +142,7 @@ item_error .. signal:: item_error .. function:: item_error(item, response, spider, failure) - Sent when a :ref:`topics-item-pipeline` generates an error (ie. raises + Sent when a :ref:`topics-item-pipeline` generates an error (i.e. raises an exception), except :exc:`~scrapy.exceptions.DropItem` exception. This signal supports returning deferreds from their handlers. @@ -232,7 +233,7 @@ spider_error .. signal:: spider_error .. function:: spider_error(failure, response, spider) - Sent when a spider callback generates an error (ie. raises an exception). + Sent when a spider callback generates an error (i.e. raises an exception). This signal does not support returning deferreds from their handlers. @@ -301,6 +302,8 @@ request_left_downloader .. signal:: request_left_downloader .. function:: request_left_downloader(request, spider) + .. versionadded:: 2.0 + Sent when a :class:`~scrapy.http.Request` leaves the downloader, even in case of failure. diff --git a/docs/topics/spiders.rst b/docs/topics/spiders.rst index 406f50fb3..bf410c03f 100644 --- a/docs/topics/spiders.rst +++ b/docs/topics/spiders.rst @@ -299,8 +299,8 @@ The spider will not do any parsing on its own. If you were to set the ``start_urls`` attribute from the command line, you would have to parse it on your own into a list using something like -`ast.literal_eval `_ -or `json.loads `_ +`ast.literal_eval `_ +or `json.loads `_ and then set it as an attribute. Otherwise, you would cause iteration over a ``start_urls`` string (a very common python pitfall) @@ -422,6 +422,8 @@ Crawling rules callbacks for new requests when writing :class:`CrawlSpider`-based spiders; unexpected behaviour can occur otherwise. + .. versionadded:: 2.0 + The *errback* parameter. CrawlSpider example ~~~~~~~~~~~~~~~~~~~ @@ -824,6 +826,6 @@ Combine SitemapSpider with other sources of urls:: .. _Sitemaps: https://www.sitemaps.org/index.html .. _Sitemap index files: https://www.sitemaps.org/protocol.html#index -.. _robots.txt: http://www.robotstxt.org/ +.. _robots.txt: https://www.robotstxt.org/ .. _TLD: https://en.wikipedia.org/wiki/Top-level_domain .. _Scrapyd documentation: https://scrapyd.readthedocs.io/en/latest/ diff --git a/pytest.ini b/pytest.ini index 552829d4e..141a13a4f 100644 --- a/pytest.ini +++ b/pytest.ini @@ -37,7 +37,7 @@ flake8-ignore = scrapy/commands/edit.py E501 scrapy/commands/fetch.py E401 E501 E128 E731 scrapy/commands/genspider.py E128 E501 E502 - scrapy/commands/parse.py E128 E501 E731 E226 + scrapy/commands/parse.py E128 E501 E731 scrapy/commands/runspider.py E501 scrapy/commands/settings.py E128 scrapy/commands/shell.py E128 E501 E502 @@ -47,22 +47,22 @@ flake8-ignore = scrapy/contracts/__init__.py E501 W504 scrapy/contracts/default.py E128 # scrapy/core - scrapy/core/engine.py E501 E128 E127 E306 E502 + scrapy/core/engine.py E501 E128 E127 E502 scrapy/core/scheduler.py E501 - scrapy/core/scraper.py E501 E306 E128 W504 - scrapy/core/spidermw.py E501 E731 E126 E226 + scrapy/core/scraper.py E501 E128 W504 + scrapy/core/spidermw.py E501 E731 E126 scrapy/core/downloader/__init__.py E501 scrapy/core/downloader/contextfactory.py E501 E128 E126 scrapy/core/downloader/middleware.py E501 E502 - scrapy/core/downloader/tls.py E501 E305 E241 - scrapy/core/downloader/webclient.py E731 E501 E128 E126 E226 + scrapy/core/downloader/tls.py E501 E241 + scrapy/core/downloader/webclient.py E731 E501 E128 E126 scrapy/core/downloader/handlers/__init__.py E501 - scrapy/core/downloader/handlers/ftp.py E501 E305 E128 E127 + scrapy/core/downloader/handlers/ftp.py E501 E128 E127 scrapy/core/downloader/handlers/http10.py E501 scrapy/core/downloader/handlers/http11.py E501 scrapy/core/downloader/handlers/s3.py E501 E128 E126 # scrapy/downloadermiddlewares - scrapy/downloadermiddlewares/ajaxcrawl.py E501 E226 + scrapy/downloadermiddlewares/ajaxcrawl.py E501 scrapy/downloadermiddlewares/decompression.py E501 scrapy/downloadermiddlewares/defaultheaders.py E501 scrapy/downloadermiddlewares/httpcache.py E501 E126 @@ -76,7 +76,7 @@ flake8-ignore = scrapy/extensions/closespider.py E501 E128 E123 scrapy/extensions/corestats.py E501 scrapy/extensions/feedexport.py E128 E501 - scrapy/extensions/httpcache.py E128 E501 E303 + scrapy/extensions/httpcache.py E128 E501 scrapy/extensions/memdebug.py E501 scrapy/extensions/spiderstate.py E501 scrapy/extensions/telnet.py E501 W504 @@ -91,7 +91,7 @@ flake8-ignore = scrapy/http/response/text.py E501 E128 E124 # scrapy/linkextractors scrapy/linkextractors/__init__.py E731 E501 E402 W504 - scrapy/linkextractors/lxmlhtml.py E501 E731 E226 + scrapy/linkextractors/lxmlhtml.py E501 E731 # scrapy/loader scrapy/loader/__init__.py E501 E128 scrapy/loader/processors.py E501 @@ -105,7 +105,7 @@ flake8-ignore = scrapy/selector/unified.py E501 E111 # scrapy/settings scrapy/settings/__init__.py E501 - scrapy/settings/default_settings.py E501 E114 E116 E226 + scrapy/settings/default_settings.py E501 E114 E116 scrapy/settings/deprecated.py E501 # scrapy/spidermiddlewares scrapy/spidermiddlewares/httperror.py E501 @@ -121,28 +121,27 @@ flake8-ignore = scrapy/utils/asyncio.py E501 scrapy/utils/benchserver.py E501 scrapy/utils/conf.py E402 E501 - scrapy/utils/console.py E306 E305 - scrapy/utils/datatypes.py E501 E226 + scrapy/utils/datatypes.py E501 scrapy/utils/decorators.py E501 scrapy/utils/defer.py E501 E128 scrapy/utils/deprecate.py E128 E501 E127 E502 - scrapy/utils/gz.py E305 E501 W504 - scrapy/utils/http.py F403 E226 + scrapy/utils/gz.py E501 W504 + scrapy/utils/http.py F403 scrapy/utils/httpobj.py E501 - scrapy/utils/iterators.py E501 E701 + scrapy/utils/iterators.py E501 scrapy/utils/log.py E128 E501 scrapy/utils/markup.py F403 - scrapy/utils/misc.py E501 E226 + scrapy/utils/misc.py E501 scrapy/utils/multipart.py F403 scrapy/utils/project.py E501 scrapy/utils/python.py E501 - scrapy/utils/reactor.py E226 E501 + scrapy/utils/reactor.py E501 scrapy/utils/reqser.py E501 scrapy/utils/request.py E127 E501 scrapy/utils/response.py E501 E128 scrapy/utils/signal.py E501 E128 scrapy/utils/sitemap.py E501 - scrapy/utils/spider.py E271 E501 + scrapy/utils/spider.py E501 scrapy/utils/ssl.py E501 scrapy/utils/test.py E501 scrapy/utils/url.py E501 F403 E128 F405 @@ -152,7 +151,7 @@ flake8-ignore = scrapy/crawler.py E501 scrapy/dupefilters.py E501 E202 scrapy/exceptions.py E501 - scrapy/exporters.py E501 E226 + scrapy/exporters.py E501 scrapy/interfaces.py E501 scrapy/item.py E501 E128 scrapy/link.py E501 @@ -161,93 +160,91 @@ flake8-ignore = scrapy/middleware.py E128 E501 scrapy/pqueues.py E501 scrapy/resolver.py E501 - scrapy/responsetypes.py E128 E501 E305 + scrapy/responsetypes.py E128 E501 scrapy/robotstxt.py E501 scrapy/shell.py E501 scrapy/signalmanager.py E501 - scrapy/spiderloader.py E225 F841 E501 E126 + scrapy/spiderloader.py F841 E501 E126 scrapy/squeues.py E128 scrapy/statscollectors.py E501 # tests tests/__init__.py E402 E501 tests/mockserver.py E401 E501 E126 E123 - tests/pipelines.py F841 E226 + tests/pipelines.py F841 tests/spiders.py E501 E127 tests/test_closespider.py E501 E127 tests/test_command_fetch.py E501 - tests/test_command_parse.py E501 E128 E303 E226 + tests/test_command_parse.py E501 E128 tests/test_command_shell.py E501 E128 tests/test_commands.py E128 E501 tests/test_contracts.py E501 E128 tests/test_crawl.py E501 E741 E265 - tests/test_crawler.py F841 E306 E501 - tests/test_dependencies.py F841 E501 E305 - tests/test_downloader_handlers.py E124 E127 E128 E225 E265 E501 E701 E126 E226 E123 + tests/test_crawler.py F841 E501 + tests/test_dependencies.py F841 E501 + tests/test_downloader_handlers.py E124 E127 E128 E265 E501 E126 E123 tests/test_downloadermiddleware.py E501 tests/test_downloadermiddleware_ajaxcrawlable.py E501 - tests/test_downloadermiddleware_cookies.py E731 E741 E501 E128 E303 E265 E126 + tests/test_downloadermiddleware_cookies.py E731 E741 E501 E128 E265 E126 tests/test_downloadermiddleware_decompression.py E127 tests/test_downloadermiddleware_defaultheaders.py E501 tests/test_downloadermiddleware_downloadtimeout.py E501 - tests/test_downloadermiddleware_httpcache.py E501 E305 - tests/test_downloadermiddleware_httpcompression.py E501 E251 E126 E123 + tests/test_downloadermiddleware_httpcache.py E501 + tests/test_downloadermiddleware_httpcompression.py E501 E126 E123 tests/test_downloadermiddleware_httpproxy.py E501 E128 - tests/test_downloadermiddleware_redirect.py E501 E303 E128 E306 E127 E305 - tests/test_downloadermiddleware_retry.py E501 E128 E251 E303 E126 + tests/test_downloadermiddleware_redirect.py E501 E128 E127 + tests/test_downloadermiddleware_retry.py E501 E128 E126 tests/test_downloadermiddleware_robotstxt.py E501 tests/test_downloadermiddleware_stats.py E501 - tests/test_dupefilters.py E221 E501 E741 E128 E124 + tests/test_dupefilters.py E501 E741 E128 E124 tests/test_engine.py E401 E501 E128 - tests/test_exporters.py E501 E731 E306 E128 E124 + tests/test_exporters.py E501 E731 E128 E124 tests/test_extension_telnet.py F841 tests/test_feedexport.py E501 F841 E241 tests/test_http_cookies.py E501 tests/test_http_headers.py E501 tests/test_http_request.py E402 E501 E127 E128 E128 E126 E123 - tests/test_http_response.py E501 E301 E128 E265 - tests/test_item.py E701 E128 F841 E306 + tests/test_http_response.py E501 E128 E265 + tests/test_item.py E128 F841 tests/test_link.py E501 tests/test_linkextractors.py E501 E128 E124 - tests/test_loader.py E501 E731 E303 E741 E128 E117 E241 + tests/test_loader.py E501 E731 E741 E128 E117 E241 tests/test_logformatter.py E128 E501 E122 - tests/test_mail.py E128 E501 E305 + tests/test_mail.py E128 E501 tests/test_middleware.py E501 E128 - tests/test_pipeline_crawl.py E131 E501 E128 E126 - tests/test_pipeline_files.py E501 E303 E272 E226 - tests/test_pipeline_images.py F841 E501 E303 - tests/test_pipeline_media.py E501 E741 E731 E128 E306 E502 + tests/test_pipeline_crawl.py E501 E128 E126 + tests/test_pipeline_files.py E501 + tests/test_pipeline_images.py F841 E501 + tests/test_pipeline_media.py E501 E741 E731 E128 E502 tests/test_proxy_connect.py E501 E741 tests/test_request_cb_kwargs.py E501 - tests/test_responsetypes.py E501 E305 + tests/test_responsetypes.py E501 tests/test_robotstxt_interface.py E501 E501 tests/test_scheduler.py E501 E126 E123 tests/test_selector.py E501 E127 tests/test_spider.py E501 - tests/test_spidermiddleware.py E501 E226 + tests/test_spidermiddleware.py E501 tests/test_spidermiddleware_httperror.py E128 E501 E127 E121 tests/test_spidermiddleware_offsite.py E501 E128 E111 - tests/test_spidermiddleware_output_chain.py E501 E226 + tests/test_spidermiddleware_output_chain.py E501 tests/test_spidermiddleware_referer.py E501 F841 E125 E201 E124 E501 E241 E121 - tests/test_squeues.py E501 E701 E741 + tests/test_squeues.py E501 E741 tests/test_utils_asyncio.py E501 - tests/test_utils_conf.py E501 E303 E128 + tests/test_utils_conf.py E501 E128 tests/test_utils_curl.py E501 - tests/test_utils_datatypes.py E402 E501 E305 - tests/test_utils_defer.py E306 E501 F841 E226 - tests/test_utils_deprecate.py F841 E306 E501 + tests/test_utils_datatypes.py E402 E501 + tests/test_utils_defer.py E501 F841 + tests/test_utils_deprecate.py F841 E501 tests/test_utils_http.py E501 E128 W504 - tests/test_utils_iterators.py E501 E128 E129 E303 E241 - tests/test_utils_log.py E741 E226 - tests/test_utils_python.py E501 E303 E731 E701 E305 + tests/test_utils_iterators.py E501 E128 E129 E241 + tests/test_utils_log.py E741 + tests/test_utils_python.py E501 E731 tests/test_utils_reqser.py E501 E128 - tests/test_utils_request.py E501 E128 E305 + tests/test_utils_request.py E501 E128 tests/test_utils_response.py E501 - tests/test_utils_signal.py E741 F841 E731 E226 + tests/test_utils_signal.py E741 F841 E731 tests/test_utils_sitemap.py E128 E501 E124 - tests/test_utils_spider.py E305 - tests/test_utils_template.py E305 - tests/test_utils_url.py E501 E127 E305 E211 E125 E501 E226 E241 E126 E123 - tests/test_webclient.py E501 E128 E122 E303 E402 E306 E226 E241 E123 E126 + tests/test_utils_url.py E501 E127 E125 E501 E241 E126 E123 + tests/test_webclient.py E501 E128 E122 E402 E241 E123 E126 tests/test_cmdline/__init__.py E501 tests/test_settings/__init__.py E501 E128 tests/test_spiderloader/__init__.py E128 E501 diff --git a/scrapy/VERSION b/scrapy/VERSION index 27f9cd322..227cea215 100644 --- a/scrapy/VERSION +++ b/scrapy/VERSION @@ -1 +1 @@ -1.8.0 +2.0.0 diff --git a/scrapy/commands/crawl.py b/scrapy/commands/crawl.py index 8093fd402..4b2f9484b 100644 --- a/scrapy/commands/crawl.py +++ b/scrapy/commands/crawl.py @@ -1,7 +1,5 @@ -import os from scrapy.commands import ScrapyCommand -from scrapy.utils.conf import arglist_to_dict -from scrapy.utils.python import without_none_values +from scrapy.utils.conf import arglist_to_dict, feed_process_params_from_cli from scrapy.exceptions import UsageError @@ -19,7 +17,7 @@ class Command(ScrapyCommand): ScrapyCommand.add_options(self, parser) parser.add_option("-a", dest="spargs", action="append", default=[], metavar="NAME=VALUE", help="set spider argument (may be repeated)") - parser.add_option("-o", "--output", metavar="FILE", + parser.add_option("-o", "--output", metavar="FILE", action="append", help="dump scraped items into FILE (use - for stdout)") parser.add_option("-t", "--output-format", metavar="FORMAT", help="format to use for dumping items with -o") @@ -31,21 +29,8 @@ class Command(ScrapyCommand): except ValueError: raise UsageError("Invalid -a value, use -a NAME=VALUE", print_help=False) if opts.output: - if opts.output == '-': - self.settings.set('FEED_URI', 'stdout:', priority='cmdline') - else: - self.settings.set('FEED_URI', opts.output, priority='cmdline') - feed_exporters = without_none_values( - self.settings.getwithbase('FEED_EXPORTERS')) - valid_output_formats = feed_exporters.keys() - if not opts.output_format: - opts.output_format = os.path.splitext(opts.output)[1].replace(".", "") - if opts.output_format not in valid_output_formats: - raise UsageError("Unrecognized output format '%s', set one" - " using the '-t' switch or as a file extension" - " from the supported list %s" % (opts.output_format, - tuple(valid_output_formats))) - self.settings.set('FEED_FORMAT', opts.output_format, priority='cmdline') + feeds = feed_process_params_from_cli(self.settings, opts.output, opts.output_format) + self.settings.set('FEEDS', feeds, priority='cmdline') def run(self, args, opts): if len(args) < 1: @@ -54,8 +39,13 @@ class Command(ScrapyCommand): raise UsageError("running 'scrapy crawl' with more than one spider is no longer supported") spname = args[0] - self.crawler_process.crawl(spname, **opts.spargs) - self.crawler_process.start() + crawl_defer = self.crawler_process.crawl(spname, **opts.spargs) - if self.crawler_process.bootstrap_failed: + if getattr(crawl_defer, 'result', None) is not None and issubclass(crawl_defer.result.type, Exception): self.exitcode = 1 + else: + self.crawler_process.start() + + if self.crawler_process.bootstrap_failed or \ + (hasattr(self.crawler_process, 'has_exception') and self.crawler_process.has_exception): + self.exitcode = 1 diff --git a/scrapy/commands/parse.py b/scrapy/commands/parse.py index ff6f1d8cd..3ef8ddcb3 100644 --- a/scrapy/commands/parse.py +++ b/scrapy/commands/parse.py @@ -80,7 +80,7 @@ class Command(ScrapyCommand): else: items = self.items.get(lvl, []) - print("# Scraped Items ", "-"*60) + print("# Scraped Items ", "-" * 60) display.pprint([dict(x) for x in items], colorize=colour) def print_requests(self, lvl=None, colour=True): @@ -92,14 +92,14 @@ class Command(ScrapyCommand): else: requests = self.requests.get(lvl, []) - print("# Requests ", "-"*65) + print("# Requests ", "-" * 65) display.pprint(requests, colorize=colour) def print_results(self, opts): colour = not opts.nocolour if opts.verbose: - for level in range(1, self.max_level+1): + for level in range(1, self.max_level + 1): print('\n>>> DEPTH LEVEL: %s <<<' % level) if not opts.noitems: self.print_items(level, colour) diff --git a/scrapy/commands/runspider.py b/scrapy/commands/runspider.py index 57d8471ca..62510609a 100644 --- a/scrapy/commands/runspider.py +++ b/scrapy/commands/runspider.py @@ -5,8 +5,7 @@ from importlib import import_module from scrapy.utils.spider import iter_spider_classes from scrapy.commands import ScrapyCommand from scrapy.exceptions import UsageError -from scrapy.utils.conf import arglist_to_dict -from scrapy.utils.python import without_none_values +from scrapy.utils.conf import arglist_to_dict, feed_process_params_from_cli def _import_file(filepath): @@ -43,7 +42,7 @@ class Command(ScrapyCommand): ScrapyCommand.add_options(self, parser) parser.add_option("-a", dest="spargs", action="append", default=[], metavar="NAME=VALUE", help="set spider argument (may be repeated)") - parser.add_option("-o", "--output", metavar="FILE", + parser.add_option("-o", "--output", metavar="FILE", action="append", help="dump scraped items into FILE (use - for stdout)") parser.add_option("-t", "--output-format", metavar="FORMAT", help="format to use for dumping items with -o") @@ -55,19 +54,8 @@ class Command(ScrapyCommand): except ValueError: raise UsageError("Invalid -a value, use -a NAME=VALUE", print_help=False) if opts.output: - if opts.output == '-': - self.settings.set('FEED_URI', 'stdout:', priority='cmdline') - else: - self.settings.set('FEED_URI', opts.output, priority='cmdline') - feed_exporters = without_none_values(self.settings.getwithbase('FEED_EXPORTERS')) - if not opts.output_format: - opts.output_format = os.path.splitext(opts.output)[1].replace(".", "") - if opts.output_format not in feed_exporters: - raise UsageError("Unrecognized output format '%s', set one" - " using the '-t' switch or as a file extension" - " from the supported list %s" % (opts.output_format, - tuple(feed_exporters))) - self.settings.set('FEED_FORMAT', opts.output_format, priority='cmdline') + feeds = feed_process_params_from_cli(self.settings, opts.output, opts.output_format) + self.settings.set('FEEDS', feeds, priority='cmdline') def run(self, args, opts): if len(args) != 1: diff --git a/scrapy/core/downloader/__init__.py b/scrapy/core/downloader/__init__.py index 5a2fdadf5..644be121f 100644 --- a/scrapy/core/downloader/__init__.py +++ b/scrapy/core/downloader/__init__.py @@ -3,7 +3,7 @@ from time import time from datetime import datetime from collections import deque -from twisted.internet import reactor, defer, task +from twisted.internet import defer, task from scrapy.utils.defer import mustbe_deferred from scrapy.utils.httpobj import urlparse_cached @@ -133,6 +133,7 @@ class Downloader(object): return deferred def _process_queue(self, spider, slot): + from twisted.internet import reactor if slot.latercall and slot.latercall.active(): return diff --git a/scrapy/core/downloader/handlers/ftp.py b/scrapy/core/downloader/handlers/ftp.py index 1681c6df8..432cb1831 100644 --- a/scrapy/core/downloader/handlers/ftp.py +++ b/scrapy/core/downloader/handlers/ftp.py @@ -32,7 +32,6 @@ import re from io import BytesIO from urllib.parse import unquote -from twisted.internet import reactor from twisted.internet.protocol import ClientCreator, Protocol from twisted.protocols.ftp import CommandFailed, FTPClient @@ -81,6 +80,7 @@ class FTPDownloadHandler: return cls(crawler.settings) def download_request(self, request, spider): + from twisted.internet import reactor parsed_url = urlparse_cached(request) user = request.meta.get("ftp_user", self.default_user) password = request.meta.get("ftp_password", self.default_password) diff --git a/scrapy/core/downloader/handlers/http10.py b/scrapy/core/downloader/handlers/http10.py index d4aa51bd1..c0146a0a6 100644 --- a/scrapy/core/downloader/handlers/http10.py +++ b/scrapy/core/downloader/handlers/http10.py @@ -1,7 +1,5 @@ """Download handlers for http and https schemes """ -from twisted.internet import reactor - from scrapy.utils.misc import create_instance, load_object from scrapy.utils.python import to_unicode @@ -26,6 +24,7 @@ class HTTP10DownloadHandler: return factory.deferred def _connect(self, factory): + from twisted.internet import reactor host, port = to_unicode(factory.host), factory.port if factory.scheme == b'https': client_context_factory = create_instance( diff --git a/scrapy/core/downloader/handlers/http11.py b/scrapy/core/downloader/handlers/http11.py index 5a5f6cf0a..04a8d617a 100644 --- a/scrapy/core/downloader/handlers/http11.py +++ b/scrapy/core/downloader/handlers/http11.py @@ -3,11 +3,12 @@ import logging import re import warnings +from contextlib import suppress from io import BytesIO from time import time from urllib.parse import urldefrag -from twisted.internet import defer, protocol, reactor +from twisted.internet import defer, protocol, ssl from twisted.internet.endpoints import TCP4ClientEndpoint from twisted.internet.error import TimeoutError from twisted.web.client import Agent, HTTPConnectionPool, ResponseDone, ResponseFailed, URI @@ -32,6 +33,7 @@ class HTTP11DownloadHandler: lazy = False def __init__(self, settings, crawler=None): + from twisted.internet import reactor self._pool = HTTPConnectionPool(reactor, persistent=True) self._pool.maxPersistentPerHost = settings.getint('CONCURRENT_REQUESTS_PER_DOMAIN') self._pool._factory.noisy = False @@ -80,6 +82,7 @@ class HTTP11DownloadHandler: return agent.download_request(request) def close(self): + from twisted.internet import reactor d = self._pool.closeCachedConnections() # closeCachedConnections will hang on network or server issues, so # we'll manually timeout the deferred. @@ -283,6 +286,7 @@ class ScrapyAgent(object): self._txresponse = None def _get_agent(self, request, timeout): + from twisted.internet import reactor bindaddress = request.meta.get('bindaddress') or self._bindAddress proxy = request.meta.get('proxy') if proxy: @@ -325,6 +329,7 @@ class ScrapyAgent(object): ) def download_request(self, request): + from twisted.internet import reactor timeout = request.meta.get('download_timeout') or self._connectTimeout agent = self._get_agent(request, timeout) @@ -382,7 +387,7 @@ class ScrapyAgent(object): def _cb_bodyready(self, txresponse, request): # deliverBody hangs for responses without body if txresponse.length == 0: - return txresponse, b'', None + return txresponse, b'', None, None maxsize = request.meta.get('download_maxsize', self._maxsize) warnsize = request.meta.get('download_warnsize', self._warnsize) @@ -418,11 +423,12 @@ class ScrapyAgent(object): return d def _cb_bodydone(self, result, request, url): - txresponse, body, flags = result + txresponse, body, flags, certificate = result status = int(txresponse.code) headers = Headers(txresponse.headers.getAllRawHeaders()) respcls = responsetypes.from_args(headers=headers, url=url, body=body) - return respcls(url=url, status=status, headers=headers, body=body, flags=flags) + return respcls(url=url, status=status, headers=headers, body=body, + flags=flags, certificate=certificate) @implementer(IBodyProducer) @@ -456,6 +462,12 @@ class _ResponseReader(protocol.Protocol): self._fail_on_dataloss_warned = False self._reached_warnsize = False self._bytes_received = 0 + self._certificate = None + + def connectionMade(self): + if self._certificate is None: + with suppress(AttributeError): + self._certificate = ssl.Certificate(self.transport._producer.getPeerCertificate()) def dataReceived(self, bodyBytes): # This maybe called several times after cancel was called with buffered data. @@ -488,16 +500,16 @@ class _ResponseReader(protocol.Protocol): body = self._bodybuf.getvalue() if reason.check(ResponseDone): - self._finished.callback((self._txresponse, body, None)) + self._finished.callback((self._txresponse, body, None, self._certificate)) return if reason.check(PotentialDataLoss): - self._finished.callback((self._txresponse, body, ['partial'])) + self._finished.callback((self._txresponse, body, ['partial'], self._certificate)) return if reason.check(ResponseFailed) and any(r.check(_DataLoss) for r in reason.value.reasons): if not self._fail_on_dataloss: - self._finished.callback((self._txresponse, body, ['dataloss'])) + self._finished.callback((self._txresponse, body, ['dataloss'], self._certificate)) return elif not self._fail_on_dataloss_warned: diff --git a/scrapy/core/downloader/tls.py b/scrapy/core/downloader/tls.py index 4ed482058..a1c881d5e 100644 --- a/scrapy/core/downloader/tls.py +++ b/scrapy/core/downloader/tls.py @@ -89,4 +89,5 @@ class ScrapyClientTLSOptions(ClientTLSOptions): 'from host "{}" (exception: {})'.format( self._hostnameASCII, repr(e))) + DEFAULT_CIPHERS = AcceptableCiphers.fromOpenSSLCipherString('DEFAULT') diff --git a/scrapy/core/downloader/webclient.py b/scrapy/core/downloader/webclient.py index fc796e8bb..a71dc5fb3 100644 --- a/scrapy/core/downloader/webclient.py +++ b/scrapy/core/downloader/webclient.py @@ -140,7 +140,7 @@ class ScrapyHTTPClientFactory(HTTPClientFactory): self.headers['Content-Length'] = 0 def _build_response(self, body, request): - request.meta['download_latency'] = self.headers_time-self.start_time + request.meta['download_latency'] = self.headers_time - self.start_time status = int(self.status) headers = Headers(self.response_headers) respcls = responsetypes.from_args(headers=headers, url=self._url) diff --git a/scrapy/core/engine.py b/scrapy/core/engine.py index 829e69993..6ab8cde6b 100644 --- a/scrapy/core/engine.py +++ b/scrapy/core/engine.py @@ -230,6 +230,7 @@ class ExecutionEngine(object): def _download(self, request, spider): slot = self.slot slot.add_request(request) + def _on_success(response): assert isinstance(response, (Response, Request)) if isinstance(response, Response): diff --git a/scrapy/core/scheduler.py b/scrapy/core/scheduler.py index 975aede0c..c96b9b719 100644 --- a/scrapy/core/scheduler.py +++ b/scrapy/core/scheduler.py @@ -66,8 +66,7 @@ class Scheduler(object): dqclass = load_object(settings['SCHEDULER_DISK_QUEUE']) mqclass = load_object(settings['SCHEDULER_MEMORY_QUEUE']) - logunser = settings.getbool('LOG_UNSERIALIZABLE_REQUESTS', - settings.getbool('SCHEDULER_DEBUG')) + logunser = settings.getbool('SCHEDULER_DEBUG') return cls(dupefilter, jobdir=job_dir(settings), logunser=logunser, stats=crawler.stats, pqclass=pqclass, dqclass=dqclass, mqclass=mqclass, crawler=crawler) @@ -119,7 +118,7 @@ class Scheduler(object): if self.dqs is None: return try: - self.dqs.push(request, -request.priority) + self.dqs.push(request) except ValueError as e: # non serializable request if self.logunser: msg = ("Unable to serialize request: %(request)s - reason:" @@ -135,35 +134,29 @@ class Scheduler(object): return True def _mqpush(self, request): - self.mqs.push(request, -request.priority) + self.mqs.push(request) def _dqpop(self): if self.dqs: return self.dqs.pop() - def _newmq(self, priority): - """ Factory for creating memory queues. """ - return self.mqclass() - - def _newdq(self, priority): - """ Factory for creating disk queues. """ - path = join(self.dqdir, 'p%s' % (priority, )) - return self.dqclass(path) - def _mq(self): """ Create a new priority queue instance, with in-memory storage """ - return create_instance(self.pqclass, None, self.crawler, self._newmq, - serialize=False) + return create_instance(self.pqclass, + settings=None, + crawler=self.crawler, + downstream_queue_cls=self.mqclass, + key='') def _dq(self): """ Create a new priority queue instance, with disk storage """ state = self._read_dqs_state(self.dqdir) q = create_instance(self.pqclass, - None, - self.crawler, - self._newdq, - state, - serialize=True) + settings=None, + crawler=self.crawler, + downstream_queue_cls=self.dqclass, + key=self.dqdir, + startprios=state) if q: logger.info("Resuming crawl (%(queuesize)d requests scheduled)", {'queuesize': len(q)}, extra={'spider': self.spider}) diff --git a/scrapy/core/scraper.py b/scrapy/core/scraper.py index bf2e48080..bbc83fcd4 100644 --- a/scrapy/core/scraper.py +++ b/scrapy/core/scraper.py @@ -16,7 +16,6 @@ from scrapy import signals from scrapy.http import Request, Response from scrapy.item import BaseItem from scrapy.core.spidermw import SpiderMiddlewareManager -from scrapy.utils.request import referer_str logger = logging.getLogger(__name__) @@ -158,9 +157,9 @@ class Scraper(object): if isinstance(exc, CloseSpider): self.crawler.engine.close_spider(spider, exc.reason or 'cancelled') return - logger.error( - "Spider error processing %(request)s (referer: %(referer)s)", - {'request': request, 'referer': referer_str(request)}, + logkws = self.logformatter.spider_error(_failure, request, response, spider) + logger.log( + *logformatter_adapter(logkws), exc_info=failure_to_exc_info(_failure), extra={'spider': spider} ) @@ -208,16 +207,21 @@ class Scraper(object): """ if isinstance(download_failure, Failure) and not download_failure.check(IgnoreRequest): if download_failure.frames: - logger.error('Error downloading %(request)s', - {'request': request}, - exc_info=failure_to_exc_info(download_failure), - extra={'spider': spider}) + logkws = self.logformatter.download_error(download_failure, request, spider) + logger.log( + *logformatter_adapter(logkws), + extra={'spider': spider}, + exc_info=failure_to_exc_info(download_failure), + ) else: errmsg = download_failure.getErrorMessage() if errmsg: - logger.error('Error downloading %(request)s: %(errmsg)s', - {'request': request, 'errmsg': errmsg}, - extra={'spider': spider}) + logkws = self.logformatter.download_error( + download_failure, request, spider, errmsg) + logger.log( + *logformatter_adapter(logkws), + extra={'spider': spider}, + ) if spider_failure is not download_failure: return spider_failure @@ -236,7 +240,7 @@ class Scraper(object): signal=signals.item_dropped, item=item, response=response, spider=spider, exception=output.value) else: - logkws = self.logformatter.error(item, ex, response, spider) + logkws = self.logformatter.item_error(item, ex, response, spider) logger.log(*logformatter_adapter(logkws), extra={'spider': spider}, exc_info=failure_to_exc_info(output)) return self.signals.send_catch_log_deferred( diff --git a/scrapy/core/spidermw.py b/scrapy/core/spidermw.py index dd9b3c376..87d08cab7 100644 --- a/scrapy/core/spidermw.py +++ b/scrapy/core/spidermw.py @@ -82,7 +82,7 @@ class SpiderMiddlewareManager(MiddlewareManager): if _isiterable(result): # stop exception handling by handing control over to the # process_spider_output chain if an iterable has been returned - return process_spider_output(result, method_index+1) + return process_spider_output(result, method_index + 1) elif result is None: continue else: @@ -103,12 +103,12 @@ class SpiderMiddlewareManager(MiddlewareManager): # might fail directly if the output value is not a generator result = method(response=response, result=result, spider=spider) except Exception as ex: - exception_result = process_spider_exception(Failure(ex), method_index+1) + exception_result = process_spider_exception(Failure(ex), method_index + 1) if isinstance(exception_result, Failure): raise return exception_result if _isiterable(result): - result = _evaluate_iterable(result, method_index+1, recovered) + result = _evaluate_iterable(result, method_index + 1, recovered) else: msg = "Middleware {} must return an iterable, got {}" raise _InvalidOutput(msg.format(_fname(method), type(result))) diff --git a/scrapy/downloadermiddlewares/ajaxcrawl.py b/scrapy/downloadermiddlewares/ajaxcrawl.py index 7a140fcad..16b046e99 100644 --- a/scrapy/downloadermiddlewares/ajaxcrawl.py +++ b/scrapy/downloadermiddlewares/ajaxcrawl.py @@ -47,7 +47,7 @@ class AjaxCrawlMiddleware(object): return response # scrapy already handles #! links properly - ajax_crawl_request = request.replace(url=request.url+'#!') + ajax_crawl_request = request.replace(url=request.url + '#!') logger.debug("Downloading AJAX crawlable %(ajax_crawl_request)s instead of %(request)s", {'ajax_crawl_request': ajax_crawl_request, 'request': request}, extra={'spider': spider}) diff --git a/scrapy/downloadermiddlewares/redirect.py b/scrapy/downloadermiddlewares/redirect.py index 77cb5aa94..08cff8a55 100644 --- a/scrapy/downloadermiddlewares/redirect.py +++ b/scrapy/downloadermiddlewares/redirect.py @@ -93,8 +93,7 @@ class MetaRefreshMiddleware(BaseRedirectMiddleware): def __init__(self, settings): super(MetaRefreshMiddleware, self).__init__(settings) self._ignore_tags = settings.getlist('METAREFRESH_IGNORE_TAGS') - self._maxdelay = settings.getint('REDIRECT_MAX_METAREFRESH_DELAY', - settings.getint('METAREFRESH_MAXDELAY')) + self._maxdelay = settings.getint('METAREFRESH_MAXDELAY') def process_response(self, request, response, spider): if request.meta.get('dont_redirect', False) or request.method == 'HEAD' or \ diff --git a/scrapy/exporters.py b/scrapy/exporters.py index 1a3c9345f..2e20a7180 100644 --- a/scrapy/exporters.py +++ b/scrapy/exporters.py @@ -23,7 +23,7 @@ __all__ = ['BaseItemExporter', 'PprintItemExporter', 'PickleItemExporter', class BaseItemExporter(object): - def __init__(self, dont_fail=False, **kwargs): + def __init__(self, *, dont_fail=False, **kwargs): self._kwargs = kwargs self._configure(kwargs, dont_fail=dont_fail) @@ -173,12 +173,12 @@ class XmlItemExporter(BaseItemExporter): if hasattr(serialized_value, 'items'): self._beautify_newline() for subname, value in serialized_value.items(): - self._export_xml_field(subname, value, depth=depth+1) + self._export_xml_field(subname, value, depth=depth + 1) self._beautify_indent(depth=depth) elif is_listlike(serialized_value): self._beautify_newline() for value in serialized_value: - self._export_xml_field('value', value, depth=depth+1) + self._export_xml_field('value', value, depth=depth + 1) self._beautify_indent(depth=depth) elif isinstance(serialized_value, str): self.xg.characters(serialized_value) diff --git a/scrapy/extensions/closespider.py b/scrapy/extensions/closespider.py index afb2ed049..260b2e86e 100644 --- a/scrapy/extensions/closespider.py +++ b/scrapy/extensions/closespider.py @@ -6,8 +6,6 @@ See documentation in docs/topics/extensions.rst from collections import defaultdict -from twisted.internet import reactor - from scrapy import signals from scrapy.exceptions import NotConfigured @@ -54,6 +52,7 @@ class CloseSpider(object): self.crawler.engine.close_spider(spider, 'closespider_pagecount') def spider_opened(self, spider): + from twisted.internet import reactor self.task = reactor.callLater(self.close_on['timeout'], self.crawler.engine.close_spider, spider, reason='closespider_timeout') diff --git a/scrapy/extensions/feedexport.py b/scrapy/extensions/feedexport.py index f1b101780..108b6d35c 100644 --- a/scrapy/extensions/feedexport.py +++ b/scrapy/extensions/feedexport.py @@ -4,24 +4,27 @@ Feed Exports extension See documentation in docs/topics/feed-exports.rst """ +import logging import os import sys -import logging -from tempfile import NamedTemporaryFile +import warnings from datetime import datetime -from urllib.parse import urlparse, unquote +from tempfile import NamedTemporaryFile +from urllib.parse import unquote, urlparse -from zope.interface import Interface, implementer from twisted.internet import defer, threads from w3lib.url import file_uri_to_path +from zope.interface import implementer, Interface from scrapy import signals -from scrapy.utils.ftp import ftp_store_file -from scrapy.exceptions import NotConfigured -from scrapy.utils.misc import create_instance, load_object -from scrapy.utils.log import failure_to_exc_info -from scrapy.utils.python import without_none_values +from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning from scrapy.utils.boto import is_botocore +from scrapy.utils.conf import feed_complete_default_values_from_settings +from scrapy.utils.ftp import ftp_store_file +from scrapy.utils.log import failure_to_exc_info +from scrapy.utils.misc import create_instance, load_object +from scrapy.utils.python import without_none_values + logger = logging.getLogger(__name__) @@ -98,8 +101,6 @@ class S3FeedStorage(BlockingFeedStorage): from scrapy.utils.project import get_project_settings settings = get_project_settings() if 'AWS_ACCESS_KEY_ID' in settings or 'AWS_SECRET_ACCESS_KEY' in settings: - import warnings - from scrapy.exceptions import ScrapyDeprecationWarning warnings.warn( "Initialising `scrapy.extensions.feedexport.S3FeedStorage` " "without AWS keys is deprecated. Please supply credentials or " @@ -178,88 +179,117 @@ class FTPFeedStorage(BlockingFeedStorage): ) -class SpiderSlot(object): - def __init__(self, file, exporter, storage, uri): +class _FeedSlot(object): + def __init__(self, file, exporter, storage, uri, format, store_empty): self.file = file self.exporter = exporter self.storage = storage + # feed params self.uri = uri + self.format = format + self.store_empty = store_empty + # flags self.itemcount = 0 + self._exporting = False + + def start_exporting(self): + if not self._exporting: + self.exporter.start_exporting() + self._exporting = True + + def finish_exporting(self): + if self._exporting: + self.exporter.finish_exporting() + self._exporting = False class FeedExporter(object): - def __init__(self, settings): - self.settings = settings - if not settings['FEED_URI']: - raise NotConfigured - self.urifmt = str(settings['FEED_URI']) - self.format = settings['FEED_FORMAT'].lower() - self.export_encoding = settings['FEED_EXPORT_ENCODING'] - self.storages = self._load_components('FEED_STORAGES') - self.exporters = self._load_components('FEED_EXPORTERS') - if not self._storage_supported(self.urifmt): - raise NotConfigured - if not self._exporter_supported(self.format): - raise NotConfigured - self.store_empty = settings.getbool('FEED_STORE_EMPTY') - self._exporting = False - self.export_fields = settings.getlist('FEED_EXPORT_FIELDS') or None - self.indent = None - if settings.get('FEED_EXPORT_INDENT') is not None: - self.indent = settings.getint('FEED_EXPORT_INDENT') - uripar = settings['FEED_URI_PARAMS'] - self._uripar = load_object(uripar) if uripar else lambda x, y: None - @classmethod def from_crawler(cls, crawler): - o = cls(crawler.settings) - o.crawler = crawler - crawler.signals.connect(o.open_spider, signals.spider_opened) - crawler.signals.connect(o.close_spider, signals.spider_closed) - crawler.signals.connect(o.item_scraped, signals.item_scraped) - return o + exporter = cls(crawler) + crawler.signals.connect(exporter.open_spider, signals.spider_opened) + crawler.signals.connect(exporter.close_spider, signals.spider_closed) + crawler.signals.connect(exporter.item_scraped, signals.item_scraped) + return exporter + + def __init__(self, crawler): + self.crawler = crawler + self.settings = crawler.settings + self.feeds = {} + self.slots = [] + + if not self.settings['FEEDS'] and not self.settings['FEED_URI']: + raise NotConfigured + + # Begin: Backward compatibility for FEED_URI and FEED_FORMAT settings + if self.settings['FEED_URI']: + warnings.warn( + 'The `FEED_URI` and `FEED_FORMAT` settings have been deprecated in favor of ' + 'the `FEEDS` setting. Please see the `FEEDS` setting docs for more details', + category=ScrapyDeprecationWarning, stacklevel=2, + ) + uri = str(self.settings['FEED_URI']) # handle pathlib.Path objects + feed = {'format': self.settings.get('FEED_FORMAT', 'jsonlines')} + self.feeds[uri] = feed_complete_default_values_from_settings(feed, self.settings) + # End: Backward compatibility for FEED_URI and FEED_FORMAT settings + + # 'FEEDS' setting takes precedence over 'FEED_URI' + for uri, feed in self.settings.getdict('FEEDS').items(): + uri = str(uri) # handle pathlib.Path objects + self.feeds[uri] = feed_complete_default_values_from_settings(feed, self.settings) + + self.storages = self._load_components('FEED_STORAGES') + self.exporters = self._load_components('FEED_EXPORTERS') + for uri, feed in self.feeds.items(): + if not self._storage_supported(uri): + raise NotConfigured + if not self._exporter_supported(feed['format']): + raise NotConfigured def open_spider(self, spider): - uri = self.urifmt % self._get_uri_params(spider) - storage = self._get_storage(uri) - file = storage.open(spider) - exporter = self._get_exporter(file, fields_to_export=self.export_fields, - encoding=self.export_encoding, indent=self.indent) - if self.store_empty: - exporter.start_exporting() - self._exporting = True - self.slot = SpiderSlot(file, exporter, storage, uri) + for uri, feed in self.feeds.items(): + uri = uri % self._get_uri_params(spider, feed['uri_params']) + storage = self._get_storage(uri) + file = storage.open(spider) + exporter = self._get_exporter( + file=file, + format=feed['format'], + fields_to_export=feed['fields'], + encoding=feed['encoding'], + indent=feed['indent'], + ) + slot = _FeedSlot(file, exporter, storage, uri, feed['format'], feed['store_empty']) + self.slots.append(slot) + if slot.store_empty: + slot.start_exporting() def close_spider(self, spider): - slot = self.slot - if not slot.itemcount and not self.store_empty: - # We need to call slot.storage.store nonetheless to get the file - # properly closed. - return defer.maybeDeferred(slot.storage.store, slot.file) - if self._exporting: - slot.exporter.finish_exporting() - self._exporting = False - logfmt = "%s %%(format)s feed (%%(itemcount)d items) in: %%(uri)s" - log_args = {'format': self.format, - 'itemcount': slot.itemcount, - 'uri': slot.uri} - d = defer.maybeDeferred(slot.storage.store, slot.file) - d.addCallback(lambda _: logger.info(logfmt % "Stored", log_args, - extra={'spider': spider})) - d.addErrback(lambda f: logger.error(logfmt % "Error storing", log_args, - exc_info=failure_to_exc_info(f), - extra={'spider': spider})) - return d + deferred_list = [] + for slot in self.slots: + if not slot.itemcount and not slot.store_empty: + # We need to call slot.storage.store nonetheless to get the file + # properly closed. + return defer.maybeDeferred(slot.storage.store, slot.file) + slot.finish_exporting() + logfmt = "%s %%(format)s feed (%%(itemcount)d items) in: %%(uri)s" + log_args = {'format': slot.format, + 'itemcount': slot.itemcount, + 'uri': slot.uri} + d = defer.maybeDeferred(slot.storage.store, slot.file) + d.addCallback(lambda _: logger.info(logfmt % "Stored", log_args, + extra={'spider': spider})) + d.addErrback(lambda f: logger.error(logfmt % "Error storing", log_args, + exc_info=failure_to_exc_info(f), + extra={'spider': spider})) + deferred_list.append(d) + return defer.DeferredList(deferred_list) if deferred_list else None def item_scraped(self, item, spider): - slot = self.slot - if not self._exporting: - slot.exporter.start_exporting() - self._exporting = True - slot.exporter.export_item(item) - slot.itemcount += 1 - return item + for slot in self.slots: + slot.start_exporting() + slot.exporter.export_item(item) + slot.itemcount += 1 def _load_components(self, setting_prefix): conf = without_none_values(self.settings.getwithbase(setting_prefix)) @@ -295,17 +325,18 @@ class FeedExporter(object): objcls, self.settings, getattr(self, 'crawler', None), *args, **kwargs) - def _get_exporter(self, *args, **kwargs): - return self._get_instance(self.exporters[self.format], *args, **kwargs) + def _get_exporter(self, file, format, *args, **kwargs): + return self._get_instance(self.exporters[format], file, *args, **kwargs) def _get_storage(self, uri): return self._get_instance(self.storages[urlparse(uri).scheme], uri) - def _get_uri_params(self, spider): + def _get_uri_params(self, spider, uri_params): params = {} for k in dir(spider): params[k] = getattr(spider, k) ts = datetime.utcnow().replace(microsecond=0).isoformat().replace(':', '-') params['time'] = ts - self._uripar(params, spider) + uripar_function = load_object(uri_params) if uri_params else lambda x, y: None + uripar_function(params, spider) return params diff --git a/scrapy/extensions/memusage.py b/scrapy/extensions/memusage.py index c0570567e..14e0fb32d 100644 --- a/scrapy/extensions/memusage.py +++ b/scrapy/extensions/memusage.py @@ -47,7 +47,7 @@ class MemoryUsage(object): def get_virtual_size(self): size = self.resource.getrusage(self.resource.RUSAGE_SELF).ru_maxrss if sys.platform != 'darwin': - # on Mac OS X ru_maxrss is in bytes, on Linux it is in KB + # on macOS ru_maxrss is in bytes, on Linux it is in KB size *= 1024 return size diff --git a/scrapy/http/response/__init__.py b/scrapy/http/response/__init__.py index f92d0901c..682cec161 100644 --- a/scrapy/http/response/__init__.py +++ b/scrapy/http/response/__init__.py @@ -17,13 +17,24 @@ from scrapy.utils.trackref import object_ref class Response(object_ref): - def __init__(self, url, status=200, headers=None, body=b'', flags=None, request=None): + def __init__(self, url, status=200, headers=None, body=b'', flags=None, request=None, certificate=None): self.headers = Headers(headers or {}) self.status = int(status) self._set_body(body) self._set_url(url) self.request = request self.flags = [] if flags is None else list(flags) + self.certificate = certificate + + @property + def cb_kwargs(self): + try: + return self.request.cb_kwargs + except AttributeError: + raise AttributeError( + "Response.cb_kwargs not available, this response " + "is not tied to any request" + ) @property def meta(self): @@ -76,7 +87,7 @@ class Response(object_ref): """Create a new Response with the same attributes except for those given new values. """ - for x in ['url', 'status', 'headers', 'body', 'request', 'flags']: + for x in ['url', 'status', 'headers', 'body', 'request', 'flags', 'certificate']: kwargs.setdefault(x, getattr(self, x)) cls = kwargs.pop('cls', self.__class__) return cls(*args, **kwargs) @@ -107,7 +118,7 @@ class Response(object_ref): def follow(self, url, callback=None, method='GET', headers=None, body=None, cookies=None, meta=None, encoding='utf-8', priority=0, - dont_filter=False, errback=None, cb_kwargs=None): + dont_filter=False, errback=None, cb_kwargs=None, flags=None): # type: (...) -> Request """ Return a :class:`~.Request` instance to follow a link ``url``. @@ -118,12 +129,16 @@ class Response(object_ref): :class:`~.TextResponse` provides a :meth:`~.TextResponse.follow` method which supports selectors in addition to absolute/relative URLs and Link objects. + + .. versionadded:: 2.0 + The *flags* parameter. """ if isinstance(url, Link): url = url.url elif url is None: raise ValueError("url can't be None") url = self.urljoin(url) + return Request( url=url, callback=callback, @@ -137,13 +152,16 @@ class Response(object_ref): dont_filter=dont_filter, errback=errback, cb_kwargs=cb_kwargs, + flags=flags, ) def follow_all(self, urls, callback=None, method='GET', headers=None, body=None, cookies=None, meta=None, encoding='utf-8', priority=0, - dont_filter=False, errback=None, cb_kwargs=None): + dont_filter=False, errback=None, cb_kwargs=None, flags=None): # type: (...) -> Generator[Request, None, None] """ + .. versionadded:: 2.0 + Return an iterable of :class:`~.Request` instances to follow all links in ``urls``. It accepts the same arguments as ``Request.__init__`` method, but elements of ``urls`` can be relative URLs or :class:`~scrapy.link.Link` objects, @@ -169,6 +187,7 @@ class Response(object_ref): dont_filter=dont_filter, errback=errback, cb_kwargs=cb_kwargs, + flags=flags, ) for url in urls ) diff --git a/scrapy/http/response/text.py b/scrapy/http/response/text.py index 09049c157..2f0f3820c 100644 --- a/scrapy/http/response/text.py +++ b/scrapy/http/response/text.py @@ -121,7 +121,7 @@ class TextResponse(Response): def follow(self, url, callback=None, method='GET', headers=None, body=None, cookies=None, meta=None, encoding=None, priority=0, - dont_filter=False, errback=None, cb_kwargs=None): + dont_filter=False, errback=None, cb_kwargs=None, flags=None): # type: (...) -> Request """ Return a :class:`~.Request` instance to follow a link ``url``. @@ -157,11 +157,12 @@ class TextResponse(Response): dont_filter=dont_filter, errback=errback, cb_kwargs=cb_kwargs, + flags=flags, ) def follow_all(self, urls=None, callback=None, method='GET', headers=None, body=None, cookies=None, meta=None, encoding=None, priority=0, - dont_filter=False, errback=None, cb_kwargs=None, + dont_filter=False, errback=None, cb_kwargs=None, flags=None, css=None, xpath=None): # type: (...) -> Generator[Request, None, None] """ @@ -187,9 +188,11 @@ class TextResponse(Response): selectors from which links cannot be obtained (for instance, anchor tags without an ``href`` attribute) """ - arg_count = len(list(filter(None, (urls, css, xpath)))) - if arg_count != 1: - raise ValueError('Please supply exactly one of the following arguments: urls, css, xpath') + arguments = [x for x in (urls, css, xpath) if x is not None] + if len(arguments) != 1: + raise ValueError( + "Please supply exactly one of the following arguments: urls, css, xpath" + ) if not urls: if css: urls = self.css(css) @@ -214,6 +217,7 @@ class TextResponse(Response): dont_filter=dont_filter, errback=errback, cb_kwargs=cb_kwargs, + flags=flags, ) diff --git a/scrapy/linkextractors/lxmlhtml.py b/scrapy/linkextractors/lxmlhtml.py index fdfa92370..ab82e1915 100644 --- a/scrapy/linkextractors/lxmlhtml.py +++ b/scrapy/linkextractors/lxmlhtml.py @@ -5,11 +5,11 @@ from urllib.parse import urljoin import lxml.etree as etree from w3lib.html import strip_html5_whitespace -from w3lib.url import canonicalize_url +from w3lib.url import canonicalize_url, safe_url_string from scrapy.link import Link from scrapy.utils.misc import arg_to_iter, rel_has_nofollow -from scrapy.utils.python import unique as unique_list, to_unicode +from scrapy.utils.python import unique as unique_list from scrapy.utils.response import get_base_url from scrapy.linkextractors import FilteringLinkExtractor @@ -22,7 +22,7 @@ _collect_string_content = etree.XPath("string()") def _nons(tag): if isinstance(tag, str): - if tag[0] == '{' and tag[1:len(XHTML_NAMESPACE)+1] == XHTML_NAMESPACE: + if tag[0] == '{' and tag[1:len(XHTML_NAMESPACE) + 1] == XHTML_NAMESPACE: return tag.split('}')[-1] return tag @@ -66,7 +66,7 @@ class LxmlParserLinkExtractor(object): url = self.process_attr(attr_val) if url is None: continue - url = to_unicode(url, encoding=response_encoding) + url = safe_url_string(url, encoding=response_encoding) # to fix relative links after process_value url = urljoin(response_url, url) link = Link(url, _collect_string_content(el) or u'', diff --git a/scrapy/logformatter.py b/scrapy/logformatter.py index 4e5963e99..14cec44a6 100644 --- a/scrapy/logformatter.py +++ b/scrapy/logformatter.py @@ -5,10 +5,13 @@ from twisted.python.failure import Failure from scrapy.utils.request import referer_str -SCRAPEDMSG = u"Scraped from %(src)s" + os.linesep + "%(item)s" -DROPPEDMSG = u"Dropped: %(exception)s" + os.linesep + "%(item)s" -CRAWLEDMSG = u"Crawled (%(status)s) %(request)s%(request_flags)s (referer: %(referer)s)%(response_flags)s" -ERRORMSG = u"'Error processing %(item)s'" +SCRAPEDMSG = "Scraped from %(src)s" + os.linesep + "%(item)s" +DROPPEDMSG = "Dropped: %(exception)s" + os.linesep + "%(item)s" +CRAWLEDMSG = "Crawled (%(status)s) %(request)s%(request_flags)s (referer: %(referer)s)%(response_flags)s" +ITEMERRORMSG = "Error processing %(item)s" +SPIDERERRORMSG = "Spider error processing %(request)s (referer: %(referer)s)" +DOWNLOADERRORMSG_SHORT = "Error downloading %(request)s" +DOWNLOADERRORMSG_LONG = "Error downloading %(request)s: %(errmsg)s" class LogFormatter(object): @@ -93,16 +96,52 @@ class LogFormatter(object): } } - def error(self, item, exception, response, spider): - """Logs a message when an item causes an error while it is passing through the item pipeline.""" + def item_error(self, item, exception, response, spider): + """Logs a message when an item causes an error while it is passing + through the item pipeline. + + .. versionadded:: 2.0 + """ return { 'level': logging.ERROR, - 'msg': ERRORMSG, + 'msg': ITEMERRORMSG, 'args': { 'item': item, } } + def spider_error(self, failure, request, response, spider): + """Logs an error message from a spider. + + .. versionadded:: 2.0 + """ + return { + 'level': logging.ERROR, + 'msg': SPIDERERRORMSG, + 'args': { + 'request': request, + 'referer': referer_str(request), + } + } + + def download_error(self, failure, request, spider, errmsg=None): + """Logs a download error message from a spider (typically coming from + the engine). + + .. versionadded:: 2.0 + """ + args = {'request': request} + if errmsg: + msg = DOWNLOADERRORMSG_LONG + args['errmsg'] = errmsg + else: + msg = DOWNLOADERRORMSG_SHORT + return { + 'level': logging.ERROR, + 'msg': msg, + 'args': args, + } + @classmethod def from_crawler(cls, crawler): return cls() diff --git a/scrapy/mail.py b/scrapy/mail.py index 9655b8114..b2a24a3db 100644 --- a/scrapy/mail.py +++ b/scrapy/mail.py @@ -12,7 +12,7 @@ from email.mime.text import MIMEText from email.utils import COMMASPACE, formatdate from io import BytesIO -from twisted.internet import defer, reactor, ssl +from twisted.internet import defer, ssl from scrapy.utils.misc import arg_to_iter from scrapy.utils.python import to_bytes @@ -28,7 +28,6 @@ def _to_bytes_or_none(text): class MailSender(object): - def __init__(self, smtphost='localhost', mailfrom='scrapy@localhost', smtpuser=None, smtppass=None, smtpport=25, smtptls=False, smtpssl=False, debug=False): self.smtphost = smtphost @@ -47,6 +46,7 @@ class MailSender(object): settings.getbool('MAIL_TLS'), settings.getbool('MAIL_SSL')) def send(self, to, subject, body, cc=None, attachs=(), mimetype='text/plain', charset=None, _callback=None): + from twisted.internet import reactor if attachs: msg = MIMEMultipart() else: @@ -111,6 +111,7 @@ class MailSender(object): def _sendmail(self, to_addrs, msg): # Import twisted.mail here because it is not available in python3 + from twisted.internet import reactor from twisted.mail.smtp import ESMTPSenderFactory msg = BytesIO(msg) d = defer.Deferred() diff --git a/scrapy/pqueues.py b/scrapy/pqueues.py index 717ed4d27..1afe58dab 100644 --- a/scrapy/pqueues.py +++ b/scrapy/pqueues.py @@ -1,11 +1,7 @@ import hashlib import logging -from collections import namedtuple - -from queuelib import PriorityQueue - -from scrapy.utils.reqser import request_to_dict, request_from_dict +from scrapy.utils.misc import create_instance logger = logging.getLogger(__name__) @@ -29,88 +25,89 @@ def _path_safe(text): return '-'.join([pathable_slot, unique_slot]) -class _Priority(namedtuple("_Priority", ["priority", "slot"])): - """ Slot-specific priority. It is a hack - ``(priority, slot)`` tuple - which can be used instead of int priorities in queues: +class ScrapyPriorityQueue: + """A priority queue implemented using multiple internal queues (typically, + FIFO queues). It uses one internal queue for each priority value. The internal + queue must implement the following methods: + + * push(obj) + * pop() + * close() + * __len__() + + ``__init__`` method of ScrapyPriorityQueue receives a downstream_queue_cls + argument, which is a class used to instantiate a new (internal) queue when + a new priority is allocated. + + Only integer priorities should be used. Lower numbers are higher + priorities. + + startprios is a sequence of priorities to start with. If the queue was + previously closed leaving some priority buckets non-empty, those priorities + should be passed in startprios. - * they are ordered in the same way - order is still by priority value, - min(prios) works; - * str(p) representation is guaranteed to be different when slots - are different - this is important because str(p) is used to create - queue files on disk; - * they have readable str(p) representation which is safe - to use as a file name. """ - __slots__ = () - def __str__(self): - return '%s_%s' % (self.priority, _path_safe(str(self.slot))) + @classmethod + def from_crawler(cls, crawler, downstream_queue_cls, key, startprios=()): + return cls(crawler, downstream_queue_cls, key, startprios) + def __init__(self, crawler, downstream_queue_cls, key, startprios=()): + self.crawler = crawler + self.downstream_queue_cls = downstream_queue_cls + self.key = key + self.queues = {} + self.curprio = None + self.init_prios(startprios) -class _SlotPriorityQueues(object): - """ Container for multiple priority queues. """ - def __init__(self, pqfactory, slot_startprios=None): - """ - ``pqfactory`` is a factory for creating new PriorityQueues. - It must be a function which accepts a single optional ``startprios`` - argument, with a list of priorities to create queues for. + def init_prios(self, startprios): + if not startprios: + return - ``slot_startprios`` is a ``{slot: startprios}`` dict. - """ - self.pqfactory = pqfactory - self.pqueues = {} # slot -> priority queue - for slot, startprios in (slot_startprios or {}).items(): - self.pqueues[slot] = self.pqfactory(startprios) + for priority in startprios: + self.queues[priority] = self.qfactory(priority) - def pop_slot(self, slot): - """ Pop an object from a priority queue for this slot """ - queue = self.pqueues[slot] - request = queue.pop() - if len(queue) == 0: - del self.pqueues[slot] - return request + self.curprio = min(startprios) - def push_slot(self, slot, obj, priority): - """ Push an object to a priority queue for this slot """ - if slot not in self.pqueues: - self.pqueues[slot] = self.pqfactory() - queue = self.pqueues[slot] - queue.push(obj, priority) + def qfactory(self, key): + return create_instance(self.downstream_queue_cls, + None, + self.crawler, + self.key + '/' + str(key)) + + def priority(self, request): + return -request.priority + + def push(self, request): + priority = self.priority(request) + if priority not in self.queues: + self.queues[priority] = self.qfactory(priority) + q = self.queues[priority] + q.push(request) # this may fail (eg. serialization error) + if self.curprio is None or priority < self.curprio: + self.curprio = priority + + def pop(self): + if self.curprio is None: + return + q = self.queues[self.curprio] + m = q.pop() + if not q: + del self.queues[self.curprio] + q.close() + prios = [p for p, q in self.queues.items() if q] + self.curprio = min(prios) if prios else None + return m def close(self): - active = {slot: queue.close() - for slot, queue in self.pqueues.items()} - self.pqueues.clear() + active = [] + for p, q in self.queues.items(): + active.append(p) + q.close() return active def __len__(self): - return sum(len(x) for x in self.pqueues.values()) if self.pqueues else 0 - - -class ScrapyPriorityQueue(PriorityQueue): - """ - PriorityQueue which works with scrapy.Request instances and - can optionally convert them to/from dicts before/after putting to a queue. - """ - def __init__(self, crawler, qfactory, startprios=(), serialize=False): - super(ScrapyPriorityQueue, self).__init__(qfactory, startprios) - self.serialize = serialize - self.spider = crawler.spider - - @classmethod - def from_crawler(cls, crawler, qfactory, startprios=(), serialize=False): - return cls(crawler, qfactory, startprios, serialize) - - def push(self, request, priority=0): - if self.serialize: - request = request_to_dict(request, self.spider) - super(ScrapyPriorityQueue, self).push(request, priority) - - def pop(self): - request = super(ScrapyPriorityQueue, self).pop() - if request and self.serialize: - request = request_from_dict(request, self.spider) - return request + return sum(len(x) for x in self.queues.values()) if self.queues else 0 class DownloaderInterface(object): @@ -133,16 +130,16 @@ class DownloaderInterface(object): class DownloaderAwarePriorityQueue(object): - """ PriorityQueue which takes Downlaoder activity in account: + """ PriorityQueue which takes Downloader activity in account: domains (slots) with the least amount of active downloads are dequeued first. """ @classmethod - def from_crawler(cls, crawler, qfactory, slot_startprios=None, serialize=False): - return cls(crawler, qfactory, slot_startprios, serialize) + def from_crawler(cls, crawler, downstream_queue_cls, key, startprios=()): + return cls(crawler, downstream_queue_cls, key, startprios) - def __init__(self, crawler, qfactory, slot_startprios=None, serialize=False): + def __init__(self, crawler, downstream_queue_cls, key, slot_startprios=()): if crawler.settings.getint('CONCURRENT_REQUESTS_PER_IP') != 0: raise ValueError('"%s" does not support CONCURRENT_REQUESTS_PER_IP' % (self.__class__,)) @@ -156,35 +153,49 @@ class DownloaderAwarePriorityQueue(object): "queue class can be resumed." % slot_startprios.__class__) - slot_startprios = { - slot: [_Priority(p, slot) for p in startprios] - for slot, startprios in (slot_startprios or {}).items()} - - def pqfactory(startprios=()): - return ScrapyPriorityQueue(crawler, qfactory, startprios, serialize) - self._slot_pqueues = _SlotPriorityQueues(pqfactory, slot_startprios) - self.serialize = serialize self._downloader_interface = DownloaderInterface(crawler) + self.downstream_queue_cls = downstream_queue_cls + self.key = key + self.crawler = crawler + + self.pqueues = {} # slot -> priority queue + for slot, startprios in (slot_startprios or {}).items(): + self.pqueues[slot] = self.pqfactory(slot, startprios) + + def pqfactory(self, slot, startprios=()): + return ScrapyPriorityQueue(self.crawler, + self.downstream_queue_cls, + self.key + '/' + _path_safe(slot), + startprios) def pop(self): - stats = self._downloader_interface.stats(self._slot_pqueues.pqueues) + stats = self._downloader_interface.stats(self.pqueues) if not stats: return slot = min(stats)[1] - request = self._slot_pqueues.pop_slot(slot) + queue = self.pqueues[slot] + request = queue.pop() + if len(queue) == 0: + del self.pqueues[slot] return request - def push(self, request, priority): + def push(self, request): slot = self._downloader_interface.get_slot_key(request) - priority_slot = _Priority(priority=priority, slot=slot) - self._slot_pqueues.push_slot(slot, request, priority_slot) + if slot not in self.pqueues: + self.pqueues[slot] = self.pqfactory(slot) + queue = self.pqueues[slot] + queue.push(request) def close(self): - active = self._slot_pqueues.close() - return {slot: [p.priority for p in startprios] - for slot, startprios in active.items()} + active = {slot: queue.close() + for slot, queue in self.pqueues.items()} + self.pqueues.clear() + return active def __len__(self): - return len(self._slot_pqueues) + return sum(len(x) for x in self.pqueues.values()) if self.pqueues else 0 + + def __contains__(self, slot): + return slot in self.pqueues diff --git a/scrapy/resolver.py b/scrapy/resolver.py index 554a3a14d..f69894b1e 100644 --- a/scrapy/resolver.py +++ b/scrapy/resolver.py @@ -29,7 +29,7 @@ class CachingThreadedResolver(ThreadedResolver): cache_size = 0 return cls(reactor, cache_size, crawler.settings.getfloat('DNS_TIMEOUT')) - def install_on_reactor(self,): + def install_on_reactor(self): self.reactor.installResolver(self) def getHostByName(self, name, timeout=None): diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 91d309147..64bf93e86 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -116,4 +116,5 @@ class ResponseTypes(object): cls = self.from_body(body) return cls + responsetypes = ResponseTypes() diff --git a/scrapy/settings/default_settings.py b/scrapy/settings/default_settings.py index fc7b62e78..077317c81 100644 --- a/scrapy/settings/default_settings.py +++ b/scrapy/settings/default_settings.py @@ -75,8 +75,8 @@ DOWNLOAD_HANDLERS_BASE = { DOWNLOAD_TIMEOUT = 180 # 3mins -DOWNLOAD_MAXSIZE = 1024*1024*1024 # 1024m -DOWNLOAD_WARNSIZE = 32*1024*1024 # 32m +DOWNLOAD_MAXSIZE = 1024 * 1024 * 1024 # 1024m +DOWNLOAD_WARNSIZE = 32 * 1024 * 1024 # 32m DOWNLOAD_FAIL_ON_DATALOSS = True @@ -133,9 +133,8 @@ EXTENSIONS_BASE = { } FEED_TEMPDIR = None -FEED_URI = None +FEEDS = {} FEED_URI_PARAMS = None # a function to extend uri arguments -FEED_FORMAT = 'jsonlines' FEED_STORE_EMPTY = False FEED_EXPORT_ENCODING = None FEED_EXPORT_FIELDS = None diff --git a/scrapy/settings/deprecated.py b/scrapy/settings/deprecated.py index 91ed689e8..f6f878725 100644 --- a/scrapy/settings/deprecated.py +++ b/scrapy/settings/deprecated.py @@ -9,12 +9,8 @@ DEPRECATED_SETTINGS = [ ('ENCODING_ALIASES', 'no longer needed (encoding discovery uses w3lib now)'), ('STATS_ENABLED', 'no longer supported (change STATS_CLASS instead)'), ('SQLITE_DB', 'no longer supported'), - ('SELECTORS_BACKEND', 'use SCRAPY_SELECTORS_BACKEND environment variable instead'), ('AUTOTHROTTLE_MIN_DOWNLOAD_DELAY', 'use DOWNLOAD_DELAY instead'), ('AUTOTHROTTLE_MAX_CONCURRENCY', 'use CONCURRENT_REQUESTS_PER_DOMAIN instead'), - ('AUTOTHROTTLE_MAX_CONCURRENCY', 'use CONCURRENT_REQUESTS_PER_DOMAIN instead'), - ('REDIRECT_MAX_METAREFRESH_DELAY', 'use METAREFRESH_MAXDELAY instead'), - ('LOG_UNSERIALIZABLE_REQUESTS', 'use SCHEDULER_DEBUG instead'), ] diff --git a/scrapy/shell.py b/scrapy/shell.py index a23b04df9..a5e140484 100644 --- a/scrapy/shell.py +++ b/scrapy/shell.py @@ -5,14 +5,13 @@ See documentation in docs/topics/shell.rst """ import os import signal -import warnings from twisted.internet import threads, defer from twisted.python import threadable from w3lib.url import any_to_uri from scrapy.crawler import Crawler -from scrapy.exceptions import IgnoreRequest, ScrapyDeprecationWarning +from scrapy.exceptions import IgnoreRequest from scrapy.http import Request, Response from scrapy.item import BaseItem from scrapy.settings import Settings @@ -126,7 +125,6 @@ class Shell(object): self.vars['spider'] = spider self.vars['request'] = request self.vars['response'] = response - self.vars['sel'] = _SelectorProxy(response) if self.inthread: self.vars['fetch'] = self.fetch self.vars['view'] = open_in_browser @@ -173,7 +171,7 @@ def _request_deferred(request): This returns a Deferred whose first pair of callbacks are the request callback and errback. The Deferred also triggers when the request - callback/errback is executed (ie. when the request is downloaded) + callback/errback is executed (i.e. when the request is downloaded) WARNING: Do not call request.replace() until after the deferred is called. """ @@ -192,15 +190,3 @@ def _request_deferred(request): request.callback, request.errback = d.callback, d.errback return d - - -class _SelectorProxy(object): - - def __init__(self, response): - self._proxiedresponse = response - - def __getattr__(self, name): - warnings.warn('"sel" shortcut is deprecated. Use "response.xpath()", ' - '"response.css()" or "response.selector" instead', - category=ScrapyDeprecationWarning, stacklevel=2) - return getattr(self._proxiedresponse.selector, name) diff --git a/scrapy/spiderloader.py b/scrapy/spiderloader.py index 3beca4060..048e84e4f 100644 --- a/scrapy/spiderloader.py +++ b/scrapy/spiderloader.py @@ -28,7 +28,7 @@ class SpiderLoader(object): module=mod, cls=cls, name=name) for (mod, cls) in locations) for name, locations in self._found.items() - if len(locations)>1] + if len(locations) > 1] if dupes: msg = ("There are several spiders with the same name:\n\n" "{}\n\n This can cause unexpected behavior.".format( diff --git a/scrapy/spidermiddlewares/offsite.py b/scrapy/spidermiddlewares/offsite.py index 232e96cbb..2fab572e6 100644 --- a/scrapy/spidermiddlewares/offsite.py +++ b/scrapy/spidermiddlewares/offsite.py @@ -53,13 +53,22 @@ class OffsiteMiddleware(object): allowed_domains = getattr(spider, 'allowed_domains', None) if not allowed_domains: return re.compile('') # allow all by default - url_pattern = re.compile("^https?://.*$") + url_pattern = re.compile(r"^https?://.*$") + port_pattern = re.compile(r":\d+$") + domains = [] for domain in allowed_domains: - if url_pattern.match(domain): + if domain is None: + continue + elif url_pattern.match(domain): message = ("allowed_domains accepts only domains, not URLs. " "Ignoring URL entry %s in allowed_domains." % domain) warnings.warn(message, URLWarning) - domains = [re.escape(d) for d in allowed_domains if d is not None] + elif port_pattern.search(domain): + message = ("allowed_domains accepts only domains without ports. " + "Ignoring entry %s in allowed_domains." % domain) + warnings.warn(message, PortWarning) + else: + domains.append(re.escape(domain)) regex = r'^(.*\.)?(%s)$' % '|'.join(domains) return re.compile(regex) @@ -70,3 +79,7 @@ class OffsiteMiddleware(object): class URLWarning(Warning): pass + + +class PortWarning(Warning): + pass diff --git a/scrapy/spiders/__init__.py b/scrapy/spiders/__init__.py index 3e19f1e23..586948c24 100644 --- a/scrapy/spiders/__init__.py +++ b/scrapy/spiders/__init__.py @@ -78,6 +78,12 @@ class Spider(object_ref): def make_requests_from_url(self, url): """ This method is deprecated. """ + warnings.warn( + "Spider.make_requests_from_url method is deprecated: " + "it will be removed and not be called by the default " + "Spider.start_requests method in future Scrapy releases. " + "Please override Spider.start_requests method instead." + ) return Request(url, dont_filter=True) def _parse(self, response, **kwargs): diff --git a/scrapy/squeues.py b/scrapy/squeues.py index d5d3be67e..d0686dac3 100644 --- a/scrapy/squeues.py +++ b/scrapy/squeues.py @@ -3,10 +3,27 @@ Scheduler queues """ import marshal +import os import pickle from queuelib import queue +from scrapy.utils.reqser import request_to_dict, request_from_dict + + +def _with_mkdir(queue_class): + + class DirectoriesCreated(queue_class): + + def __init__(self, path, *args, **kwargs): + dirname = os.path.dirname(path) + if not os.path.exists(dirname): + os.makedirs(dirname, exist_ok=True) + + super(DirectoriesCreated, self).__init__(path, *args, **kwargs) + + return DirectoriesCreated + def _serializable_queue(queue_class, serialize, deserialize): @@ -24,6 +41,44 @@ def _serializable_queue(queue_class, serialize, deserialize): return SerializableQueue +def _scrapy_serialization_queue(queue_class): + + class ScrapyRequestQueue(queue_class): + + def __init__(self, crawler, key): + self.spider = crawler.spider + super(ScrapyRequestQueue, self).__init__(key) + + @classmethod + def from_crawler(cls, crawler, key, *args, **kwargs): + return cls(crawler, key) + + def push(self, request): + request = request_to_dict(request, self.spider) + return super(ScrapyRequestQueue, self).push(request) + + def pop(self): + request = super(ScrapyRequestQueue, self).pop() + + if not request: + return None + + request = request_from_dict(request, self.spider) + return request + + return ScrapyRequestQueue + + +def _scrapy_non_serialization_queue(queue_class): + + class ScrapyRequestQueue(queue_class): + @classmethod + def from_crawler(cls, crawler, *args, **kwargs): + return cls() + + return ScrapyRequestQueue + + def _pickle_serialize(obj): try: return pickle.dumps(obj, protocol=2) @@ -34,13 +89,38 @@ def _pickle_serialize(obj): raise ValueError(str(e)) -PickleFifoDiskQueue = _serializable_queue(queue.FifoDiskQueue, - _pickle_serialize, pickle.loads) -PickleLifoDiskQueue = _serializable_queue(queue.LifoDiskQueue, - _pickle_serialize, pickle.loads) -MarshalFifoDiskQueue = _serializable_queue(queue.FifoDiskQueue, - marshal.dumps, marshal.loads) -MarshalLifoDiskQueue = _serializable_queue(queue.LifoDiskQueue, - marshal.dumps, marshal.loads) -FifoMemoryQueue = queue.FifoMemoryQueue -LifoMemoryQueue = queue.LifoMemoryQueue +PickleFifoDiskQueueNonRequest = _serializable_queue( + _with_mkdir(queue.FifoDiskQueue), + _pickle_serialize, + pickle.loads +) +PickleLifoDiskQueueNonRequest = _serializable_queue( + _with_mkdir(queue.LifoDiskQueue), + _pickle_serialize, + pickle.loads +) +MarshalFifoDiskQueueNonRequest = _serializable_queue( + _with_mkdir(queue.FifoDiskQueue), + marshal.dumps, + marshal.loads +) +MarshalLifoDiskQueueNonRequest = _serializable_queue( + _with_mkdir(queue.LifoDiskQueue), + marshal.dumps, + marshal.loads +) + +PickleFifoDiskQueue = _scrapy_serialization_queue( + PickleFifoDiskQueueNonRequest +) +PickleLifoDiskQueue = _scrapy_serialization_queue( + PickleLifoDiskQueueNonRequest +) +MarshalFifoDiskQueue = _scrapy_serialization_queue( + MarshalFifoDiskQueueNonRequest +) +MarshalLifoDiskQueue = _scrapy_serialization_queue( + MarshalLifoDiskQueueNonRequest +) +FifoMemoryQueue = _scrapy_non_serialization_queue(queue.FifoMemoryQueue) +LifoMemoryQueue = _scrapy_non_serialization_queue(queue.LifoMemoryQueue) diff --git a/scrapy/utils/benchserver.py b/scrapy/utils/benchserver.py index cdbe21942..9d8d64612 100644 --- a/scrapy/utils/benchserver.py +++ b/scrapy/utils/benchserver.py @@ -3,7 +3,6 @@ from urllib.parse import urlencode from twisted.web.server import Site from twisted.web.resource import Resource -from twisted.internet import reactor class Root(Resource): @@ -34,6 +33,7 @@ def _getarg(request, name, default=None, type=str): if __name__ == '__main__': + from twisted.internet import reactor root = Root() factory = Site(root) httpPort = reactor.listenTCP(8998, Site(root)) diff --git a/scrapy/utils/conf.py b/scrapy/utils/conf.py index 23306ca28..5921f82bf 100644 --- a/scrapy/utils/conf.py +++ b/scrapy/utils/conf.py @@ -1,9 +1,12 @@ +import numbers import os import sys -import numbers +import warnings from configparser import ConfigParser from operator import itemgetter +from scrapy.exceptions import ScrapyDeprecationWarning, UsageError + from scrapy.settings import BaseSettings from scrapy.utils.deprecate import update_classpath from scrapy.utils.python import without_none_values @@ -106,3 +109,64 @@ def get_sources(use_closest=True): if use_closest: sources.append(closest_scrapy_cfg()) return sources + + +def feed_complete_default_values_from_settings(feed, settings): + out = feed.copy() + out.setdefault("encoding", settings["FEED_EXPORT_ENCODING"]) + out.setdefault("fields", settings.getlist("FEED_EXPORT_FIELDS") or None) + out.setdefault("store_empty", settings.getbool("FEED_STORE_EMPTY")) + out.setdefault("uri_params", settings["FEED_URI_PARAMS"]) + if settings["FEED_EXPORT_INDENT"] is None: + out.setdefault("indent", None) + else: + out.setdefault("indent", settings.getint("FEED_EXPORT_INDENT")) + return out + + +def feed_process_params_from_cli(settings, output, output_format=None): + """ + Receives feed export params (from the 'crawl' or 'runspider' commands), + checks for inconsistencies in their quantities and returns a dictionary + suitable to be used as the FEEDS setting. + """ + valid_output_formats = without_none_values( + settings.getwithbase('FEED_EXPORTERS') + ).keys() + + def check_valid_format(output_format): + if output_format not in valid_output_formats: + raise UsageError("Unrecognized output format '%s', set one after a" + " colon using the -o option (i.e. -o :)" + " or as a file extension, from the supported list %s" % + (output_format, tuple(valid_output_formats))) + + if output_format: + if len(output) == 1: + check_valid_format(output_format) + warnings.warn('The -t command line option is deprecated in favor' + ' of specifying the output format within the -o' + ' option, please check the -o option docs for more details', + category=ScrapyDeprecationWarning, stacklevel=2) + return {output[0]: {'format': output_format}} + else: + raise UsageError('The -t command line option cannot be used if multiple' + ' output files are specified with the -o option') + + result = {} + for element in output: + try: + feed_uri, feed_format = element.rsplit(':', 1) + except ValueError: + feed_uri = element + feed_format = os.path.splitext(element)[1].replace('.', '') + else: + if feed_uri == '-': + feed_uri = 'stdout:' + check_valid_format(feed_format) + result[feed_uri] = {'format': feed_format} + + # FEEDS setting should take precedence over the -o and -t CLI options + result.update(settings.getdict('FEEDS')) + + return result diff --git a/scrapy/utils/console.py b/scrapy/utils/console.py index 7eb40f0ce..c7a2ace88 100644 --- a/scrapy/utils/console.py +++ b/scrapy/utils/console.py @@ -54,6 +54,7 @@ def _embed_standard_shell(namespace={}, banner=''): else: import rlcompleter # noqa: F401 readline.parse_and_bind("tab:complete") + @wraps(_embed_standard_shell) def wrapper(namespace=namespace, banner=''): code.interact(banner=banner, local=namespace) diff --git a/scrapy/utils/datatypes.py b/scrapy/utils/datatypes.py index a52bbc70e..175f92d77 100644 --- a/scrapy/utils/datatypes.py +++ b/scrapy/utils/datatypes.py @@ -6,183 +6,9 @@ This module must not depend on any module outside the Standard Library. """ import collections -import copy -import warnings import weakref from collections.abc import Mapping -from scrapy.exceptions import ScrapyDeprecationWarning - - -class MultiValueDictKeyError(KeyError): - def __init__(self, *args, **kwargs): - warnings.warn( - "scrapy.utils.datatypes.MultiValueDictKeyError is deprecated " - "and will be removed in future releases.", - category=ScrapyDeprecationWarning, - stacklevel=2 - ) - super(MultiValueDictKeyError, self).__init__(*args, **kwargs) - - -class MultiValueDict(dict): - """ - A subclass of dictionary customized to handle multiple values for the same key. - - >>> d = MultiValueDict({'name': ['Adrian', 'Simon'], 'position': ['Developer']}) - >>> d['name'] - 'Simon' - >>> d.getlist('name') - ['Adrian', 'Simon'] - >>> d.get('lastname', 'nonexistent') - 'nonexistent' - >>> d.setlist('lastname', ['Holovaty', 'Willison']) - - This class exists to solve the irritating problem raised by cgi.parse_qs, - which returns a list for every key, even though most Web forms submit - single name-value pairs. - """ - def __init__(self, key_to_list_mapping=()): - warnings.warn("scrapy.utils.datatypes.MultiValueDict is deprecated " - "and will be removed in future releases.", - category=ScrapyDeprecationWarning, - stacklevel=2) - dict.__init__(self, key_to_list_mapping) - - def __repr__(self): - return "<%s: %s>" % (self.__class__.__name__, dict.__repr__(self)) - - def __getitem__(self, key): - """ - Returns the last data value for this key, or [] if it's an empty list; - raises KeyError if not found. - """ - try: - list_ = dict.__getitem__(self, key) - except KeyError: - raise MultiValueDictKeyError("Key %r not found in %r" % (key, self)) - try: - return list_[-1] - except IndexError: - return [] - - def __setitem__(self, key, value): - dict.__setitem__(self, key, [value]) - - def __copy__(self): - return self.__class__(dict.items(self)) - - def __deepcopy__(self, memo=None): - if memo is None: - memo = {} - result = self.__class__() - memo[id(self)] = result - for key, value in dict.items(self): - dict.__setitem__(result, copy.deepcopy(key, memo), copy.deepcopy(value, memo)) - return result - - def get(self, key, default=None): - "Returns the default value if the requested data doesn't exist" - try: - val = self[key] - except KeyError: - return default - if val == []: - return default - return val - - def getlist(self, key): - "Returns an empty list if the requested data doesn't exist" - try: - return dict.__getitem__(self, key) - except KeyError: - return [] - - def setlist(self, key, list_): - dict.__setitem__(self, key, list_) - - def setdefault(self, key, default=None): - if key not in self: - self[key] = default - return self[key] - - def setlistdefault(self, key, default_list=()): - if key not in self: - self.setlist(key, default_list) - return self.getlist(key) - - def appendlist(self, key, value): - "Appends an item to the internal list associated with key" - self.setlistdefault(key, []) - dict.__setitem__(self, key, self.getlist(key) + [value]) - - def items(self): - """ - Returns a list of (key, value) pairs, where value is the last item in - the list associated with the key. - """ - return [(key, self[key]) for key in self.keys()] - - def lists(self): - "Returns a list of (key, list) pairs." - return dict.items(self) - - def values(self): - "Returns a list of the last value on every key list." - return [self[key] for key in self.keys()] - - def copy(self): - "Returns a copy of this object." - return self.__deepcopy__() - - def update(self, *args, **kwargs): - "update() extends rather than replaces existing key lists. Also accepts keyword args." - if len(args) > 1: - raise TypeError("update expected at most 1 arguments, got %d" % len(args)) - if args: - other_dict = args[0] - if isinstance(other_dict, MultiValueDict): - for key, value_list in other_dict.lists(): - self.setlistdefault(key, []).extend(value_list) - else: - try: - for key, value in other_dict.items(): - self.setlistdefault(key, []).append(value) - except TypeError: - raise ValueError("MultiValueDict.update() takes either a MultiValueDict or dictionary") - for key, value in kwargs.items(): - self.setlistdefault(key, []).append(value) - - -class SiteNode(object): - """Class to represent a site node (page, image or any other file)""" - - def __init__(self, url): - warnings.warn( - "scrapy.utils.datatypes.SiteNode is deprecated " - "and will be removed in future releases.", - category=ScrapyDeprecationWarning, - stacklevel=2 - ) - - self.url = url - self.itemnames = [] - self.children = [] - self.parent = None - - def add_child(self, node): - self.children.append(node) - node.parent = self - - def to_string(self, level=0): - s = "%s%s\n" % (' '*level, self.url) - if self.itemnames: - for n in self.itemnames: - s += "%sScraped: %s\n" % (' '*(level+1), n) - for node in self.children: - s += node.to_string(level+1) - return s - class CaselessDict(dict): diff --git a/scrapy/utils/defer.py b/scrapy/utils/defer.py index 6a21a4903..34b8d9774 100644 --- a/scrapy/utils/defer.py +++ b/scrapy/utils/defer.py @@ -153,3 +153,20 @@ def deferred_f_from_coro_f(coro_f): def f(*coro_args, **coro_kwargs): return deferred_from_coro(coro_f(*coro_args, **coro_kwargs)) return f + + +def maybeDeferred_coro(f, *args, **kw): + """ Copy of defer.maybeDeferred that also converts coroutines to Deferreds. """ + try: + result = f(*args, **kw) + except: # noqa: E722 + return defer.fail(failure.Failure(captureVars=defer.Deferred.debug)) + + if isinstance(result, defer.Deferred): + return result + elif _isfuture(result) or inspect.isawaitable(result): + return deferred_from_coro(result) + elif isinstance(result, failure.Failure): + return defer.fail(result) + else: + return defer.succeed(result) diff --git a/scrapy/utils/gz.py b/scrapy/utils/gz.py index 9672e28da..c291ae237 100644 --- a/scrapy/utils/gz.py +++ b/scrapy/utils/gz.py @@ -42,6 +42,7 @@ def gunzip(data): raise return b''.join(output_list) + _is_gzipped = re.compile(br'^application/(x-)?gzip\b', re.I).search _is_octetstream = re.compile(br'^(application|binary)/octet-stream\b', re.I).search diff --git a/scrapy/utils/http.py b/scrapy/utils/http.py index bab262393..ceb3f0509 100644 --- a/scrapy/utils/http.py +++ b/scrapy/utils/http.py @@ -32,5 +32,5 @@ def decode_chunked_transfer(chunked_body): break size = int(h, 16) body += t[:size] - t = t[size+2:] + t = t[size + 2:] return body diff --git a/scrapy/utils/iterators.py b/scrapy/utils/iterators.py index 3c0cb68c3..7849174fb 100644 --- a/scrapy/utils/iterators.py +++ b/scrapy/utils/iterators.py @@ -101,8 +101,10 @@ def csviter(obj, delimiter=None, headers=None, encoding=None, quotechar=None): lines = StringIO(_body_or_str(obj, unicode=True)) kwargs = {} - if delimiter: kwargs["delimiter"] = delimiter - if quotechar: kwargs["quotechar"] = quotechar + if delimiter: + kwargs["delimiter"] = delimiter + if quotechar: + kwargs["quotechar"] = quotechar csv_r = csv.reader(lines, **kwargs) if not headers: diff --git a/scrapy/utils/misc.py b/scrapy/utils/misc.py index cb0ee5af3..52cfba208 100644 --- a/scrapy/utils/misc.py +++ b/scrapy/utils/misc.py @@ -37,8 +37,8 @@ def arg_to_iter(arg): def load_object(path): """Load an object given its absolute object path, and return it. - object can be a class, function, variable or an instance. - path ie: 'scrapy.downloadermiddlewares.redirect.RedirectMiddleware' + object can be the import path of a class, function, variable or an + instance, e.g. 'scrapy.downloadermiddlewares.redirect.RedirectMiddleware' """ try: @@ -46,7 +46,7 @@ def load_object(path): except ValueError: raise ValueError("Error loading object '%s': not a full path" % path) - module, name = path[:dot], path[dot+1:] + module, name = path[:dot], path[dot + 1:] mod = import_module(module) try: diff --git a/scrapy/utils/project.py b/scrapy/utils/project.py index f28c2eaa1..b8d3ebf9d 100644 --- a/scrapy/utils/project.py +++ b/scrapy/utils/project.py @@ -68,7 +68,6 @@ def get_project_settings(): if settings_module_path: settings.setmodule(settings_module_path, priority='project') - # XXX: remove this hack pickled_settings = os.environ.get("SCRAPY_PICKLED_SETTINGS_TO_OVERRIDE") if pickled_settings: warnings.warn("Use of environment variable " @@ -76,10 +75,24 @@ def get_project_settings(): "is deprecated.", ScrapyDeprecationWarning) settings.setdict(pickle.loads(pickled_settings), priority='project') - # XXX: deprecate and remove this functionality - env_overrides = {k[7:]: v for k, v in os.environ.items() if - k.startswith('SCRAPY_')} - if env_overrides: - settings.setdict(env_overrides, priority='project') + scrapy_envvars = {k[7:]: v for k, v in os.environ.items() if + k.startswith('SCRAPY_')} + valid_envvars = { + 'CHECK', + 'PICKLED_SETTINGS_TO_OVERRIDE', + 'PROJECT', + 'PYTHON_SHELL', + 'SETTINGS_MODULE', + } + setting_envvars = {k for k in scrapy_envvars if k not in valid_envvars} + if setting_envvars: + setting_envvar_list = ', '.join(sorted(setting_envvars)) + warnings.warn( + 'Use of environment variables prefixed with SCRAPY_ to override ' + 'settings is deprecated. The following environment variables are ' + 'currently defined: {}'.format(setting_envvar_list), + ScrapyDeprecationWarning + ) + settings.setdict(scrapy_envvars, priority='project') return settings diff --git a/scrapy/utils/py36.py b/scrapy/utils/py36.py new file mode 100644 index 000000000..c8c24076e --- /dev/null +++ b/scrapy/utils/py36.py @@ -0,0 +1,10 @@ +""" +Helpers using Python 3.6+ syntax (ignore SyntaxError on import). +""" + + +async def collect_asyncgen(result): + results = [] + async for x in result: + results.append(x) + return results diff --git a/scrapy/utils/python.py b/scrapy/utils/python.py index e5582cc18..e95a4648e 100644 --- a/scrapy/utils/python.py +++ b/scrapy/utils/python.py @@ -4,7 +4,6 @@ This module contains essential stuff that should've come with Python itself ;) import errno import gc import inspect -import os import re import sys import weakref @@ -165,14 +164,6 @@ _BINARYCHARS = {to_bytes(chr(i)) for i in range(32)} - {b"\0", b"\t", b"\n", b"\ _BINARYCHARS |= {ord(ch) for ch in _BINARYCHARS} -@deprecated("scrapy.utils.python.binary_is_text") -def isbinarytext(text): - """ This function is deprecated. - Please use scrapy.utils.python.binary_is_text, which was created to be more - clear about the functions behavior: it is behaving inverted to this one. """ - return not binary_is_text(text) - - def binary_is_text(data): """ Returns ``True`` if the given ``data`` argument (a ``bytes`` object) does not contain unprintable control characters. @@ -293,41 +284,6 @@ class WeakKeyCache(object): return self._weakdict[key] -@deprecated -def stringify_dict(dct_or_tuples, encoding='utf-8', keys_only=True): - """Return a (new) dict with unicode keys (and values when "keys_only" is - False) of the given dict converted to strings. ``dct_or_tuples`` can be a - dict or a list of tuples, like any dict ``__init__`` method supports. - """ - d = {} - for k, v in dict(dct_or_tuples).items(): - k = k.encode(encoding) if isinstance(k, str) else k - if not keys_only: - v = v.encode(encoding) if isinstance(v, str) else v - d[k] = v - return d - - -@deprecated -def is_writable(path): - """Return True if the given path can be written (if it exists) or created - (if it doesn't exist) - """ - if os.path.exists(path): - return os.access(path, os.W_OK) - else: - return os.access(os.path.dirname(path), os.W_OK) - - -@deprecated -def setattr_default(obj, name, value): - """Set attribute value, but only if it's not already set. Similar to - setdefault() for dicts. - """ - if not hasattr(obj, name): - setattr(obj, name, value) - - def retry_on_eintr(function, *args, **kw): """Run a function and retry it while getting EINTR errors""" while True: diff --git a/scrapy/utils/reactor.py b/scrapy/utils/reactor.py index 80f52a4ef..17d6b2857 100644 --- a/scrapy/utils/reactor.py +++ b/scrapy/utils/reactor.py @@ -16,7 +16,7 @@ def listen_tcp(portrange, host, factory): return reactor.listenTCP(portrange, factory, interface=host) if len(portrange) == 1: return reactor.listenTCP(portrange[0], factory, interface=host) - for x in range(portrange[0], portrange[1]+1): + for x in range(portrange[0], portrange[1] + 1): try: return reactor.listenTCP(x, factory, interface=host) except error.CannotListenError: @@ -50,6 +50,8 @@ class CallLaterOnce(object): def install_reactor(reactor_path): + """Installs the :mod:`~twisted.internet.reactor` with the specified + import path.""" reactor_class = load_object(reactor_path) if reactor_class is asyncioreactor.AsyncioSelectorReactor: with suppress(error.ReactorAlreadyInstalledError): @@ -63,6 +65,9 @@ def install_reactor(reactor_path): def verify_installed_reactor(reactor_path): + """Raises :exc:`Exception` if the installed + :mod:`~twisted.internet.reactor` does not match the specified import + path.""" from twisted.internet import reactor reactor_class = load_object(reactor_path) if not isinstance(reactor, reactor_class): diff --git a/scrapy/utils/request.py b/scrapy/utils/request.py index 356753ab5..b8c140a7e 100644 --- a/scrapy/utils/request.py +++ b/scrapy/utils/request.py @@ -28,7 +28,7 @@ def request_fingerprint(request, include_headers=None, keep_fragments=False): http://www.example.com/query?cat=222&id=111 Even though those are two different URLs both point to the same resource - and are equivalent (ie. they should return the same response). + and are equivalent (i.e. they should return the same response). Another example are cookies used to store session ids. Suppose the following page is only accessible to authenticated users: diff --git a/scrapy/utils/signal.py b/scrapy/utils/signal.py index de00bac49..60c561da6 100644 --- a/scrapy/utils/signal.py +++ b/scrapy/utils/signal.py @@ -2,12 +2,14 @@ import logging -from twisted.internet.defer import maybeDeferred, DeferredList, Deferred +from twisted.internet.defer import DeferredList, Deferred from twisted.python.failure import Failure from pydispatch.dispatcher import Any, Anonymous, liveReceivers, \ getAllReceivers, disconnect from pydispatch.robustapply import robustApply + +from scrapy.utils.defer import maybeDeferred_coro from scrapy.utils.log import failure_to_exc_info logger = logging.getLogger(__name__) @@ -61,7 +63,7 @@ def send_catch_log_deferred(signal=Any, sender=Anonymous, *arguments, **named): spider = named.get('spider', None) dfds = [] for receiver in liveReceivers(getAllReceivers(sender, signal)): - d = maybeDeferred(robustApply, receiver, signal=signal, sender=sender, + d = maybeDeferred_coro(robustApply, receiver, signal=signal, sender=sender, *arguments, **named) d.addErrback(logerror, receiver) d.addBoth(lambda result: (receiver, result)) diff --git a/scrapy/utils/spider.py b/scrapy/utils/spider.py index 72775df5c..1b8a82829 100644 --- a/scrapy/utils/spider.py +++ b/scrapy/utils/spider.py @@ -4,18 +4,26 @@ import inspect from scrapy.spiders import Spider from scrapy.utils.defer import deferred_from_coro from scrapy.utils.misc import arg_to_iter +try: + from scrapy.utils.py36 import collect_asyncgen +except SyntaxError: + collect_asyncgen = None logger = logging.getLogger(__name__) def iterate_spider_output(result): + if collect_asyncgen and hasattr(inspect, 'isasyncgen') and inspect.isasyncgen(result): + d = deferred_from_coro(collect_asyncgen(result)) + d.addCallback(iterate_spider_output) + return d return arg_to_iter(deferred_from_coro(result)) def iter_spider_classes(module): """Return an iterator over all spider classes defined in the given module - that can be instantiated (ie. which have name) + that can be instantiated (i.e. which have name) """ # this needs to be imported here until get rid of the spider manager # singleton in scrapy.spider.spiders diff --git a/scrapy/utils/testproc.py b/scrapy/utils/testproc.py index 0f15cf60a..37803b287 100644 --- a/scrapy/utils/testproc.py +++ b/scrapy/utils/testproc.py @@ -1,7 +1,7 @@ import sys import os -from twisted.internet import reactor, defer, protocol +from twisted.internet import defer, protocol class ProcessTest(object): @@ -11,6 +11,7 @@ class ProcessTest(object): cwd = os.getcwd() # trial chdirs to temp dir def execute(self, args, check_code=True, settings=None): + from twisted.internet import reactor env = os.environ.copy() if settings is not None: env['SCRAPY_SETTINGS_MODULE'] = settings diff --git a/scrapy/utils/testsite.py b/scrapy/utils/testsite.py index 6f5c21624..9e1598805 100644 --- a/scrapy/utils/testsite.py +++ b/scrapy/utils/testsite.py @@ -1,12 +1,12 @@ from urllib.parse import urljoin -from twisted.internet import reactor from twisted.web import server, resource, static, util class SiteTest(object): def setUp(self): + from twisted.internet import reactor super(SiteTest, self).setUp() self.site = reactor.listenTCP(0, test_site(), interface="127.0.0.1") self.baseurl = "http://localhost:%d/" % self.site.getHost().port @@ -38,6 +38,7 @@ def test_site(): if __name__ == '__main__': + from twisted.internet import reactor port = reactor.listenTCP(0, test_site(), interface="127.0.0.1") print("http://localhost:%d/" % port.getHost().port) reactor.run() diff --git a/sep/sep-003.rst b/sep/sep-003.rst index 184839525..e6357313d 100644 --- a/sep/sep-003.rst +++ b/sep/sep-003.rst @@ -18,7 +18,7 @@ Prerequisites This API proposal relies on the following API: -1. instantiating a item with an item instance as its first argument (ie. +1. instantiating a item with an item instance as its first argument (i.e. ``item2 = MyItem(item1)``) must return a **copy** of the first item instance) 2. items can be instantiated using this syntax: ``item = Item(attr1=value1, @@ -78,7 +78,7 @@ Defining an item containing ItemField's variants2 = ListField(ItemField(Variant), default=[]) It's important to note here that the (perhaps most intuitive) way of defining a -Product-Variant relationship (ie. defining a recursive !ItemField) doesn't +Product-Variant relationship (i.e. defining a recursive !ItemField) doesn't work. For example, this fails to compile: :: diff --git a/sep/sep-013.rst b/sep/sep-013.rst index 5b18b7501..4bc9abd30 100644 --- a/sep/sep-013.rst +++ b/sep/sep-013.rst @@ -59,7 +59,7 @@ Global changes to all middlewares To be discussed: -1. should we support returning deferreds (ie. ``maybeDeferred``) in middleware +1. should we support returning deferreds (i.e. ``maybeDeferred``) in middleware methods? 2. should we pass Twisted Failures instead of exceptions to error methods? diff --git a/sep/sep-021.rst b/sep/sep-021.rst index 628a95dd2..372429791 100644 --- a/sep/sep-021.rst +++ b/sep/sep-021.rst @@ -38,7 +38,7 @@ Goals: * simple to manage: adding or removing extensions should be just a matter of adding or removing lines in a ``scrapy.cfg`` file -* backward compatibility with enabling extension the "old way" (ie. modifying +* backward compatibility with enabling extension the "old way" (i.e. modifying settings directly) Non-goals: diff --git a/tests/pipelines.py b/tests/pipelines.py index d7d3b5259..de4894c32 100644 --- a/tests/pipelines.py +++ b/tests/pipelines.py @@ -6,7 +6,7 @@ Some pipelines used for testing class ZeroDivisionErrorPipeline(object): def open_spider(self, spider): - a = 1/0 + a = 1 / 0 def process_item(self, item, spider): return item @@ -15,4 +15,4 @@ class ZeroDivisionErrorPipeline(object): class ProcessWithZeroDivisionErrorPipiline(object): def process_item(self, item, spider): - 1/0 + 1 / 0 diff --git a/tests/py36/_test_crawl.py b/tests/py36/_test_crawl.py new file mode 100644 index 000000000..162a53760 --- /dev/null +++ b/tests/py36/_test_crawl.py @@ -0,0 +1,57 @@ +import asyncio + +from scrapy import Request +from tests.spiders import SimpleSpider + + +class AsyncDefAsyncioGenSpider(SimpleSpider): + + name = 'asyncdef_asyncio_gen' + + async def parse(self, response): + await asyncio.sleep(0.2) + yield {'foo': 42} + self.logger.info("Got response %d" % response.status) + + +class AsyncDefAsyncioGenLoopSpider(SimpleSpider): + + name = 'asyncdef_asyncio_gen_loop' + + async def parse(self, response): + for i in range(10): + await asyncio.sleep(0.1) + yield {'foo': i} + self.logger.info("Got response %d" % response.status) + + +class AsyncDefAsyncioGenComplexSpider(SimpleSpider): + + name = 'asyncdef_asyncio_gen_complex' + initial_reqs = 4 + following_reqs = 3 + depth = 2 + + def _get_req(self, index, cb=None): + return Request(self.mockserver.url("/status?n=200&request=%d" % index), + meta={'index': index}, + dont_filter=True, + callback=cb) + + def start_requests(self): + for i in range(1, self.initial_reqs + 1): + yield self._get_req(i) + + async def parse(self, response): + index = response.meta['index'] + yield {'index': index} + if index < 10 ** self.depth: + for new_index in range(10 * index, 10 * index + self.following_reqs): + yield self._get_req(new_index) + yield self._get_req(index, cb=self.parse2) + await asyncio.sleep(0.1) + yield {'index': index + 5} + + async def parse2(self, response): + await asyncio.sleep(0.1) + yield {'index2': response.meta['index']} diff --git a/tests/requirements-py3.txt b/tests/requirements-py3.txt index d97c4b8ee..d207c5fb0 100644 --- a/tests/requirements-py3.txt +++ b/tests/requirements-py3.txt @@ -2,7 +2,7 @@ jmespath mitmproxy; python_version >= '3.6' mitmproxy<4.0.0; python_version < '3.6' -pytest +pytest < 5.4 pytest-cov pytest-twisted >= 1.11 pytest-xdist diff --git a/tests/test_cmdline_crawl_with_pipeline/__init__.py b/tests/test_cmdline_crawl_with_pipeline/__init__.py new file mode 100644 index 000000000..d341888d3 --- /dev/null +++ b/tests/test_cmdline_crawl_with_pipeline/__init__.py @@ -0,0 +1,20 @@ +import os +import sys +import unittest +from subprocess import Popen, PIPE + + +class CmdlineCrawlPipelineTest(unittest.TestCase): + + def _execute(self, spname): + args = (sys.executable, '-m', 'scrapy.cmdline', 'crawl', spname) + cwd = os.path.dirname(os.path.abspath(__file__)) + proc = Popen(args, stdout=PIPE, stderr=PIPE, cwd=cwd) + proc.communicate() + return proc.returncode + + def test_open_spider_normally_in_pipeline(self): + self.assertEqual(self._execute('normal'), 0) + + def test_exception_at_open_spider_in_pipeline(self): + self.assertEqual(self._execute('exception'), 1) diff --git a/tests/test_cmdline_crawl_with_pipeline/scrapy.cfg b/tests/test_cmdline_crawl_with_pipeline/scrapy.cfg new file mode 100644 index 000000000..2f238dba3 --- /dev/null +++ b/tests/test_cmdline_crawl_with_pipeline/scrapy.cfg @@ -0,0 +1,2 @@ +[settings] +default = test_spider.settings diff --git a/tests/test_cmdline_crawl_with_pipeline/test_spider/__init__.py b/tests/test_cmdline_crawl_with_pipeline/test_spider/__init__.py new file mode 100644 index 000000000..e69de29bb diff --git a/tests/test_cmdline_crawl_with_pipeline/test_spider/pipelines.py b/tests/test_cmdline_crawl_with_pipeline/test_spider/pipelines.py new file mode 100644 index 000000000..ce916f699 --- /dev/null +++ b/tests/test_cmdline_crawl_with_pipeline/test_spider/pipelines.py @@ -0,0 +1,16 @@ +class TestSpiderPipeline(object): + + def open_spider(self, spider): + pass + + def process_item(self, item, spider): + return item + + +class TestSpiderExceptionPipeline(object): + + def open_spider(self, spider): + raise Exception('exception') + + def process_item(self, item, spider): + return item diff --git a/tests/test_cmdline_crawl_with_pipeline/test_spider/settings.py b/tests/test_cmdline_crawl_with_pipeline/test_spider/settings.py new file mode 100644 index 000000000..ae782c0d8 --- /dev/null +++ b/tests/test_cmdline_crawl_with_pipeline/test_spider/settings.py @@ -0,0 +1,2 @@ +BOT_NAME = 'test_spider' +SPIDER_MODULES = ['test_spider.spiders'] diff --git a/tests/test_cmdline_crawl_with_pipeline/test_spider/spiders/__init__.py b/tests/test_cmdline_crawl_with_pipeline/test_spider/spiders/__init__.py new file mode 100644 index 000000000..e69de29bb diff --git a/tests/test_cmdline_crawl_with_pipeline/test_spider/spiders/exception.py b/tests/test_cmdline_crawl_with_pipeline/test_spider/spiders/exception.py new file mode 100644 index 000000000..300f45ebf --- /dev/null +++ b/tests/test_cmdline_crawl_with_pipeline/test_spider/spiders/exception.py @@ -0,0 +1,14 @@ +import scrapy + + +class ExceptionSpider(scrapy.Spider): + name = 'exception' + + custom_settings = { + 'ITEM_PIPELINES': { + 'test_spider.pipelines.TestSpiderExceptionPipeline': 300 + } + } + + def parse(self, response): + pass diff --git a/tests/test_cmdline_crawl_with_pipeline/test_spider/spiders/normal.py b/tests/test_cmdline_crawl_with_pipeline/test_spider/spiders/normal.py new file mode 100644 index 000000000..87a40fdcb --- /dev/null +++ b/tests/test_cmdline_crawl_with_pipeline/test_spider/spiders/normal.py @@ -0,0 +1,14 @@ +import scrapy + + +class NormalSpider(scrapy.Spider): + name = 'normal' + + custom_settings = { + 'ITEM_PIPELINES': { + 'test_spider.pipelines.TestSpiderPipeline': 300 + } + } + + def parse(self, response): + pass diff --git a/tests/test_command_parse.py b/tests/test_command_parse.py index b7035fdff..5bf92b71a 100644 --- a/tests/test_command_parse.py +++ b/tests/test_command_parse.py @@ -147,7 +147,6 @@ ITEM_PIPELINES = {'%s.pipelines.MyPipeline': 1} self.url('/html')]) self.assertIn("DEBUG: It Works!", _textmode(stderr)) - @defer.inlineCallbacks def test_pipelines(self): _, _, stderr = yield self.execute(['--spider', self.spider_name, @@ -183,7 +182,7 @@ ITEM_PIPELINES = {'%s.pipelines.MyPipeline': 1} def test_crawlspider_matching_rule_callback_set(self): """If a rule matches the URL, use it's defined callback.""" status, out, stderr = yield self.execute( - ['--spider', 'goodcrawl'+self.spider_name, '-r', self.url('/html')] + ['--spider', 'goodcrawl' + self.spider_name, '-r', self.url('/html')] ) self.assertIn("""[{}, {'foo': 'bar'}]""", _textmode(out)) @@ -191,7 +190,7 @@ ITEM_PIPELINES = {'%s.pipelines.MyPipeline': 1} def test_crawlspider_matching_rule_default_callback(self): """If a rule match but it has no callback set, use the 'parse' callback.""" status, out, stderr = yield self.execute( - ['--spider', 'goodcrawl'+self.spider_name, '-r', self.url('/text')] + ['--spider', 'goodcrawl' + self.spider_name, '-r', self.url('/text')] ) self.assertIn("""[{}, {'nomatch': 'default'}]""", _textmode(out)) @@ -207,7 +206,7 @@ ITEM_PIPELINES = {'%s.pipelines.MyPipeline': 1} @defer.inlineCallbacks def test_crawlspider_missing_callback(self): status, out, stderr = yield self.execute( - ['--spider', 'badcrawl'+self.spider_name, '-r', self.url('/html')] + ['--spider', 'badcrawl' + self.spider_name, '-r', self.url('/html')] ) self.assertRegex(_textmode(out), r"""# Scraped Items -+\n\[\]""") @@ -215,7 +214,7 @@ ITEM_PIPELINES = {'%s.pipelines.MyPipeline': 1} def test_crawlspider_no_matching_rule(self): """The requested URL has no matching rule, so no items should be scraped""" status, out, stderr = yield self.execute( - ['--spider', 'badcrawl'+self.spider_name, '-r', self.url('/enc-gb18030')] + ['--spider', 'badcrawl' + self.spider_name, '-r', self.url('/enc-gb18030')] ) self.assertRegex(_textmode(out), r"""# Scraped Items -+\n\[\]""") self.assertIn("""Cannot find a rule that matches""", _textmode(stderr)) diff --git a/tests/test_commands.py b/tests/test_commands.py index 3612b70c9..24a341759 100644 --- a/tests/test_commands.py +++ b/tests/test_commands.py @@ -1,22 +1,46 @@ import inspect +import json +import optparse import os -import sys import subprocess +import sys import tempfile +from contextlib import contextmanager from os.path import exists, join, abspath from shutil import rmtree, copytree from tempfile import mkdtemp -from contextlib import contextmanager from threading import Timer from twisted.trial import unittest import scrapy +from scrapy.commands import ScrapyCommand +from scrapy.settings import Settings from scrapy.utils.python import to_unicode from scrapy.utils.test import get_testenv + from tests.test_crawler import ExceptionSpider, NoRequestsSpider +class CommandSettings(unittest.TestCase): + + def setUp(self): + self.command = ScrapyCommand() + self.command.settings = Settings() + self.parser = optparse.OptionParser( + formatter=optparse.TitledHelpFormatter(), + conflict_handler='resolve', + ) + self.command.add_options(self.parser) + + def test_settings_json_string(self): + feeds_json = '{"data.json": {"format": "json"}, "data.xml": {"format": "xml"}}' + opts, args = self.parser.parse_args(args=['-s', 'FEEDS={}'.format(feeds_json), 'spider.py']) + self.command.process_options(args, opts) + self.assertIsInstance(self.command.settings['FEEDS'], scrapy.settings.BaseSettings) + self.assertEqual(dict(self.command.settings['FEEDS']), json.loads(feeds_json)) + + class ProjectTest(unittest.TestCase): project_name = 'testproject' @@ -34,7 +58,7 @@ class ProjectTest(unittest.TestCase): with tempfile.TemporaryFile() as out: args = (sys.executable, '-m', 'scrapy.cmdline') + new_args return subprocess.call(args, stdout=out, stderr=out, cwd=self.cwd, - env=self.env, **kwargs) + env=self.env, **kwargs) def proc(self, *new_args, **popen_kwargs): args = (sys.executable, '-m', 'scrapy.cmdline') + new_args @@ -310,6 +334,6 @@ class BenchCommandTest(CommandTest): def test_run(self): _, _, log = self.proc('bench', '-s', 'LOGSTATS_INTERVAL=0.001', - '-s', 'CLOSESPIDER_TIMEOUT=0.01') + '-s', 'CLOSESPIDER_TIMEOUT=0.01') self.assertIn('INFO: Crawled', log) self.assertNotIn('Unhandled Error', log) diff --git a/tests/test_crawl.py b/tests/test_crawl.py index 5980b37f4..1d5075d22 100644 --- a/tests/test_crawl.py +++ b/tests/test_crawl.py @@ -1,9 +1,11 @@ import json import logging +import sys from pytest import mark from testfixtures import LogCapture from twisted.internet import defer +from twisted.internet.ssl import Certificate from twisted.trial.unittest import TestCase from scrapy import signals @@ -45,7 +47,7 @@ class CrawlTestCase(TestCase): @defer.inlineCallbacks def test_fixed_delay(self): - yield self._test_delay(total=3, delay=0.1) + yield self._test_delay(total=3, delay=0.2) @defer.inlineCallbacks def test_randomized_delay(self): @@ -357,7 +359,7 @@ class CrawlSpiderTestCase(TestCase): @mark.only_asyncio() @defer.inlineCallbacks def test_async_def_asyncio_parse(self): - runner = CrawlerRunner({"ASYNCIO_REACTOR": True}) + runner = CrawlerRunner({"TWISTED_REACTOR": "twisted.internet.asyncioreactor.AsyncioSelectorReactor"}) runner.crawl(AsyncDefAsyncioSpider, self.mockserver.url("/status?n=200"), mockserver=self.mockserver) with LogCapture() as log: yield runner.join() @@ -379,6 +381,59 @@ class CrawlSpiderTestCase(TestCase): self.assertIn({'id': 1}, items) self.assertIn({'id': 2}, items) + @mark.skipif(sys.version_info < (3, 6), reason="Async generators require Python 3.6 or higher") + @mark.only_asyncio() + @defer.inlineCallbacks + def test_async_def_asyncgen_parse(self): + from tests.py36._test_crawl import AsyncDefAsyncioGenSpider + crawler = self.runner.create_crawler(AsyncDefAsyncioGenSpider) + with LogCapture() as log: + yield crawler.crawl(self.mockserver.url("/status?n=200"), mockserver=self.mockserver) + self.assertIn("Got response 200", str(log)) + itemcount = crawler.stats.get_value('item_scraped_count') + self.assertEqual(itemcount, 1) + + @mark.skipif(sys.version_info < (3, 6), reason="Async generators require Python 3.6 or higher") + @mark.only_asyncio() + @defer.inlineCallbacks + def test_async_def_asyncgen_parse_loop(self): + items = [] + + def _on_item_scraped(item): + items.append(item) + + from tests.py36._test_crawl import AsyncDefAsyncioGenLoopSpider + crawler = self.runner.create_crawler(AsyncDefAsyncioGenLoopSpider) + crawler.signals.connect(_on_item_scraped, signals.item_scraped) + with LogCapture() as log: + yield crawler.crawl(self.mockserver.url("/status?n=200"), mockserver=self.mockserver) + self.assertIn("Got response 200", str(log)) + itemcount = crawler.stats.get_value('item_scraped_count') + self.assertEqual(itemcount, 10) + for i in range(10): + self.assertIn({'foo': i}, items) + + @mark.skipif(sys.version_info < (3, 6), reason="Async generators require Python 3.6 or higher") + @mark.only_asyncio() + @defer.inlineCallbacks + def test_async_def_asyncgen_parse_complex(self): + items = [] + + def _on_item_scraped(item): + items.append(item) + + from tests.py36._test_crawl import AsyncDefAsyncioGenComplexSpider + crawler = self.runner.create_crawler(AsyncDefAsyncioGenComplexSpider) + crawler.signals.connect(_on_item_scraped, signals.item_scraped) + yield crawler.crawl(mockserver=self.mockserver) + itemcount = crawler.stats.get_value('item_scraped_count') + self.assertEqual(itemcount, 156) + # some random items + for i in [1, 4, 21, 22, 207, 311]: + self.assertIn({'index': i}, items) + for i in [10, 30, 122]: + self.assertIn({'index2': i}, items) + @mark.only_asyncio() @defer.inlineCallbacks def test_async_def_asyncio_parse_reqs_list(self): @@ -387,3 +442,31 @@ class CrawlSpiderTestCase(TestCase): yield crawler.crawl(self.mockserver.url("/status?n=200"), mockserver=self.mockserver) for req_id in range(3): self.assertIn("Got response 200, req_id %d" % req_id, str(log)) + + @defer.inlineCallbacks + def test_response_ssl_certificate_none(self): + crawler = self.runner.create_crawler(SingleRequestSpider) + url = self.mockserver.url("/echo?body=test", is_secure=False) + yield crawler.crawl(seed=url, mockserver=self.mockserver) + self.assertIsNone(crawler.spider.meta['responses'][0].certificate) + + @defer.inlineCallbacks + def test_response_ssl_certificate(self): + crawler = self.runner.create_crawler(SingleRequestSpider) + url = self.mockserver.url("/echo?body=test", is_secure=True) + yield crawler.crawl(seed=url, mockserver=self.mockserver) + cert = crawler.spider.meta['responses'][0].certificate + self.assertIsInstance(cert, Certificate) + self.assertEqual(cert.getSubject().commonName, b"localhost") + self.assertEqual(cert.getIssuer().commonName, b"localhost") + + @mark.xfail(reason="Responses with no body return early and contain no certificate") + @defer.inlineCallbacks + def test_response_ssl_certificate_empty_response(self): + crawler = self.runner.create_crawler(SingleRequestSpider) + url = self.mockserver.url("/status?n=200", is_secure=True) + yield crawler.crawl(seed=url, mockserver=self.mockserver) + cert = crawler.spider.meta['responses'][0].certificate + self.assertIsInstance(cert, Certificate) + self.assertEqual(cert.getSubject().commonName, b"localhost") + self.assertEqual(cert.getIssuer().commonName, b"localhost") diff --git a/tests/test_crawler.py b/tests/test_crawler.py index f8fa26def..7bd76601d 100644 --- a/tests/test_crawler.py +++ b/tests/test_crawler.py @@ -107,6 +107,7 @@ class CrawlerLoggingTestCase(unittest.TestCase): def test_spider_custom_settings_log_level(self): log_file = self.mktemp() + class MySpider(scrapy.Spider): name = 'spider' custom_settings = { diff --git a/tests/test_dependencies.py b/tests/test_dependencies.py index e31ccd9b5..a169acbe6 100644 --- a/tests/test_dependencies.py +++ b/tests/test_dependencies.py @@ -13,5 +13,6 @@ class ScrapyUtilsTest(unittest.TestCase): installed_version = [int(x) for x in module.__version__.split('.')[:2]] assert installed_version >= [0, 6], "OpenSSL >= 0.6 required" + if __name__ == "__main__": unittest.main() diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py index 8d95d7cac..29d06bab4 100644 --- a/tests/test_downloader_handlers.py +++ b/tests/test_downloader_handlers.py @@ -348,7 +348,7 @@ class HttpTestCase(unittest.TestCase): return self.download_request(request, Spider('foo')).addCallback(_test) def test_payload(self): - body = b'1'*100 # PayloadResource requires body length to be 100 + body = b'1' * 100 # PayloadResource requires body length to be 100 request = Request(self.getURL('payload'), method='POST', body=body) d = self.download_request(request, Spider('foo')) d.addCallback(lambda r: r.body) @@ -812,7 +812,7 @@ class S3TestCase(unittest.TestCase): def test_request_signing1(self): # gets an object from the johnsmith bucket. - date ='Tue, 27 Mar 2007 19:36:42 +0000' + date = 'Tue, 27 Mar 2007 19:36:42 +0000' req = Request('s3://johnsmith/photos/puppy.jpg', headers={'Date': date}) with self._mocked_date(date): httpreq = self.download_request(req, self.spider) diff --git a/tests/test_downloadermiddleware_cookies.py b/tests/test_downloadermiddleware_cookies.py index 04884fb78..051f66680 100644 --- a/tests/test_downloadermiddleware_cookies.py +++ b/tests/test_downloadermiddleware_cookies.py @@ -145,7 +145,6 @@ class CookiesMiddlewareTest(TestCase): {'name': 'C3', 'value': 'value3', 'path': '/foo', 'domain': 'scrapytest.org'}, {'name': 'C4', 'value': 'value4', 'path': '/foo', 'domain': 'scrapy.org'}] - req = Request('http://scrapytest.org/', cookies=cookies) self.mw.process_request(req, self.spider) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index 9401dd66d..9b77c97a8 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -501,5 +501,6 @@ class RFC2616PolicyTest(DefaultStorageTest): self.assertEqualResponse(res1, res2) assert 'cached' in res2.flags + if __name__ == '__main__': unittest.main() diff --git a/tests/test_downloadermiddleware_httpcompression.py b/tests/test_downloadermiddleware_httpcompression.py index 64488841a..106ca3360 100644 --- a/tests/test_downloadermiddleware_httpcompression.py +++ b/tests/test_downloadermiddleware_httpcompression.py @@ -245,7 +245,7 @@ class HttpCompressionTest(TestCase): response.headers['Content-Type'] = 'application/gzip' request = response.request request.method = 'HEAD' - response = response.replace(body = None) + response = response.replace(body=None) newresponse = self.mw.process_response(request, response, self.spider) self.assertIs(newresponse, response) self.assertEqual(response.body, b'') diff --git a/tests/test_downloadermiddleware_redirect.py b/tests/test_downloadermiddleware_redirect.py index e0f145d0e..053e26fc3 100644 --- a/tests/test_downloadermiddleware_redirect.py +++ b/tests/test_downloadermiddleware_redirect.py @@ -68,7 +68,6 @@ class RedirectMiddlewareTest(unittest.TestCase): assert isinstance(r, Response) assert r is rsp - def test_redirect_302(self): url = 'http://www.example.com/302' url2 = 'http://www.example.com/redirected2' @@ -122,7 +121,6 @@ class RedirectMiddlewareTest(unittest.TestCase): del rsp.headers['Location'] assert self.mw.process_response(req, rsp, self.spider) is rsp - def test_max_redirect_times(self): self.mw.max_redirect_times = 1 req = Request('http://scrapytest.org/302') @@ -178,6 +176,7 @@ class RedirectMiddlewareTest(unittest.TestCase): def test_request_meta_handling(self): url = 'http://www.example.com/301' url2 = 'http://www.example.com/redirected' + def _test_passthrough(req): rsp = Response(url, headers={'Location': url2}, status=301, request=req) r = self.mw.process_response(req, rsp, self.spider) @@ -316,5 +315,6 @@ class MetaRefreshMiddlewareTest(unittest.TestCase): response = mw.process_response(req, rsp, self.spider) assert isinstance(response, Response) + if __name__ == "__main__": unittest.main() diff --git a/tests/test_dupefilters.py b/tests/test_dupefilters.py index 88ce9627f..9e24d86dd 100644 --- a/tests/test_dupefilters.py +++ b/tests/test_dupefilters.py @@ -43,7 +43,7 @@ class RFPDupeFilterTest(unittest.TestCase): def test_df_from_crawler_scheduler(self): settings = {'DUPEFILTER_DEBUG': True, - 'DUPEFILTER_CLASS': __name__ + '.FromCrawlerRFPDupeFilter'} + 'DUPEFILTER_CLASS': __name__ + '.FromCrawlerRFPDupeFilter'} crawler = get_crawler(settings_dict=settings) scheduler = Scheduler.from_crawler(crawler) self.assertTrue(scheduler.df.debug) @@ -51,14 +51,14 @@ class RFPDupeFilterTest(unittest.TestCase): def test_df_from_settings_scheduler(self): settings = {'DUPEFILTER_DEBUG': True, - 'DUPEFILTER_CLASS': __name__ + '.FromSettingsRFPDupeFilter'} + 'DUPEFILTER_CLASS': __name__ + '.FromSettingsRFPDupeFilter'} crawler = get_crawler(settings_dict=settings) scheduler = Scheduler.from_crawler(crawler) self.assertTrue(scheduler.df.debug) self.assertEqual(scheduler.df.method, 'from_settings') def test_df_direct_scheduler(self): - settings = {'DUPEFILTER_CLASS': __name__ + '.DirectDupeFilter'} + settings = {'DUPEFILTER_CLASS': __name__ + '.DirectDupeFilter'} crawler = get_crawler(settings_dict=settings) scheduler = Scheduler.from_crawler(crawler) self.assertEqual(scheduler.df.method, 'n/a') @@ -162,7 +162,7 @@ class RFPDupeFilterTest(unittest.TestCase): def test_log(self): with LogCapture() as l: settings = {'DUPEFILTER_DEBUG': False, - 'DUPEFILTER_CLASS': __name__ + '.FromCrawlerRFPDupeFilter'} + 'DUPEFILTER_CLASS': __name__ + '.FromCrawlerRFPDupeFilter'} crawler = get_crawler(SimpleSpider, settings_dict=settings) scheduler = Scheduler.from_crawler(crawler) spider = SimpleSpider.from_crawler(crawler) @@ -187,7 +187,7 @@ class RFPDupeFilterTest(unittest.TestCase): def test_log_debug(self): with LogCapture() as l: settings = {'DUPEFILTER_DEBUG': True, - 'DUPEFILTER_CLASS': __name__ + '.FromCrawlerRFPDupeFilter'} + 'DUPEFILTER_CLASS': __name__ + '.FromCrawlerRFPDupeFilter'} crawler = get_crawler(SimpleSpider, settings_dict=settings) scheduler = Scheduler.from_crawler(crawler) spider = SimpleSpider.from_crawler(crawler) diff --git a/tests/test_exporters.py b/tests/test_exporters.py index 5d1f5c182..6e2507508 100644 --- a/tests/test_exporters.py +++ b/tests/test_exporters.py @@ -312,6 +312,7 @@ class XmlItemExporterTest(BaseItemExporterTest): for child in children] else: return [(elem.tag, [(elem.text, ())])] + def xmlsplit(xmlcontent): doc = lxml.etree.fromstring(xmlcontent) return xmltuple(doc) diff --git a/tests/test_feedexport.py b/tests/test_feedexport.py index 2ca57c19d..08e8dfc41 100644 --- a/tests/test_feedexport.py +++ b/tests/test_feedexport.py @@ -1,32 +1,34 @@ -import os import csv import json -import warnings -import tempfile +import os +import random import shutil import string +import tempfile +import warnings from io import BytesIO from pathlib import Path +from string import ascii_letters, digits from unittest import mock from urllib.parse import urljoin, urlparse, quote from urllib.request import pathname2url -from zope.interface.verify import verifyObject -from twisted.trial import unittest +import lxml.etree from twisted.internet import defer -from scrapy.crawler import CrawlerRunner -from scrapy.settings import Settings -from tests.mockserver import MockServer +from twisted.trial import unittest from w3lib.url import path_to_file_uri +from zope.interface.verify import verifyObject import scrapy +from scrapy.crawler import CrawlerRunner from scrapy.exporters import CsvItemExporter -from scrapy.extensions.feedexport import ( - IFeedStorage, FileFeedStorage, FTPFeedStorage, - S3FeedStorage, StdoutFeedStorage, - BlockingFeedStorage) -from scrapy.utils.test import assert_aws_environ, get_s3_content_and_delete, get_crawler +from scrapy.extensions.feedexport import (BlockingFeedStorage, FileFeedStorage, FTPFeedStorage, + IFeedStorage, S3FeedStorage, StdoutFeedStorage) +from scrapy.settings import Settings from scrapy.utils.python import to_unicode +from scrapy.utils.test import assert_aws_environ, get_crawler, get_s3_content_and_delete + +from tests.mockserver import MockServer class FileFeedStorageTest(unittest.TestCase): @@ -395,29 +397,41 @@ class FeedExportTest(unittest.TestCase): egg = scrapy.Field() baz = scrapy.Field() + def setUp(self): + self.temp_dir = tempfile.mkdtemp() + + def tearDown(self): + shutil.rmtree(self.temp_dir, ignore_errors=True) + + def _random_temp_filename(self): + chars = [random.choice(ascii_letters + digits) for _ in range(15)] + filename = ''.join(chars) + return os.path.join(self.temp_dir, filename) + @defer.inlineCallbacks - def run_and_export(self, spider_cls, settings=None): + def run_and_export(self, spider_cls, settings): """ Run spider with specified settings; return exported data. """ - tmpdir = tempfile.mkdtemp() - res_path = os.path.join(tmpdir, 'res') - res_uri = urljoin('file:', pathname2url(res_path)) - defaults = { - 'FEED_URI': res_uri, - 'FEED_FORMAT': 'csv', - 'FEED_PATH': res_path + + FEEDS = settings.get('FEEDS') or {} + settings['FEEDS'] = { + urljoin('file:', pathname2url(str(file_path))): feed + for file_path, feed in FEEDS.items() } - defaults.update(settings or {}) + + content = {} try: with MockServer() as s: - runner = CrawlerRunner(Settings(defaults)) + runner = CrawlerRunner(Settings(settings)) spider_cls.start_urls = [s.url('/')] yield runner.crawl(spider_cls) - with open(str(defaults['FEED_PATH']), 'rb') as f: - content = f.read() + for file_path, feed in FEEDS.items(): + with open(str(file_path), 'rb') as f: + content[feed['format']] = f.read() finally: - shutil.rmtree(tmpdir) + for file_path in FEEDS.keys(): + os.remove(str(file_path)) defer.returnValue(content) @@ -453,10 +467,14 @@ class FeedExportTest(unittest.TestCase): @defer.inlineCallbacks def assertExportedCsv(self, items, header, rows, settings=None, ordered=True): settings = settings or {} - settings.update({'FEED_FORMAT': 'csv'}) + settings.update({ + 'FEEDS': { + self._random_temp_filename(): {'format': 'csv'}, + }, + }) data = yield self.exported_data(items, settings) - reader = csv.DictReader(to_unicode(data).splitlines()) + reader = csv.DictReader(to_unicode(data['csv']).splitlines()) got_rows = list(reader) if ordered: self.assertEqual(reader.fieldnames, header) @@ -468,51 +486,87 @@ class FeedExportTest(unittest.TestCase): @defer.inlineCallbacks def assertExportedJsonLines(self, items, rows, settings=None): settings = settings or {} - settings.update({'FEED_FORMAT': 'jl'}) + settings.update({ + 'FEEDS': { + self._random_temp_filename(): {'format': 'jl'}, + }, + }) data = yield self.exported_data(items, settings) - parsed = [json.loads(to_unicode(line)) for line in data.splitlines()] + parsed = [json.loads(to_unicode(line)) for line in data['jl'].splitlines()] rows = [{k: v for k, v in row.items() if v} for row in rows] self.assertEqual(rows, parsed) @defer.inlineCallbacks def assertExportedXml(self, items, rows, settings=None): settings = settings or {} - settings.update({'FEED_FORMAT': 'xml'}) + settings.update({ + 'FEEDS': { + self._random_temp_filename(): {'format': 'xml'}, + }, + }) data = yield self.exported_data(items, settings) rows = [{k: v for k, v in row.items() if v} for row in rows] - import lxml.etree - root = lxml.etree.fromstring(data) + root = lxml.etree.fromstring(data['xml']) got_rows = [{e.tag: e.text for e in it} for it in root.findall('item')] self.assertEqual(rows, got_rows) + @defer.inlineCallbacks + def assertExportedMultiple(self, items, rows, settings=None): + settings = settings or {} + settings.update({ + 'FEEDS': { + self._random_temp_filename(): {'format': 'xml'}, + self._random_temp_filename(): {'format': 'json'}, + }, + }) + data = yield self.exported_data(items, settings) + rows = [{k: v for k, v in row.items() if v} for row in rows] + # XML + root = lxml.etree.fromstring(data['xml']) + xml_rows = [{e.tag: e.text for e in it} for it in root.findall('item')] + self.assertEqual(rows, xml_rows) + # JSON + json_rows = json.loads(to_unicode(data['json'])) + self.assertEqual(rows, json_rows) + def _load_until_eof(self, data, load_func): - bytes_output = BytesIO(data) result = [] - while True: - try: - result.append(load_func(bytes_output)) - except EOFError: - break + with tempfile.TemporaryFile() as temp: + temp.write(data) + temp.seek(0) + while True: + try: + result.append(load_func(temp)) + except EOFError: + break return result @defer.inlineCallbacks def assertExportedPickle(self, items, rows, settings=None): settings = settings or {} - settings.update({'FEED_FORMAT': 'pickle'}) + settings.update({ + 'FEEDS': { + self._random_temp_filename(): {'format': 'pickle'}, + }, + }) data = yield self.exported_data(items, settings) expected = [{k: v for k, v in row.items() if v} for row in rows] import pickle - result = self._load_until_eof(data, load_func=pickle.load) + result = self._load_until_eof(data['pickle'], load_func=pickle.load) self.assertEqual(expected, result) @defer.inlineCallbacks def assertExportedMarshal(self, items, rows, settings=None): settings = settings or {} - settings.update({'FEED_FORMAT': 'marshal'}) + settings.update({ + 'FEEDS': { + self._random_temp_filename(): {'format': 'marshal'}, + }, + }) data = yield self.exported_data(items, settings) expected = [{k: v for k, v in row.items() if v} for row in rows] import marshal - result = self._load_until_eof(data, load_func=marshal.load) + result = self._load_until_eof(data['marshal'], load_func=marshal.load) self.assertEqual(expected, result) @defer.inlineCallbacks @@ -521,6 +575,8 @@ class FeedExportTest(unittest.TestCase): yield self.assertExportedJsonLines(items, rows, settings) yield self.assertExportedXml(items, rows, settings) yield self.assertExportedPickle(items, rows, settings) + yield self.assertExportedMarshal(items, rows, settings) + yield self.assertExportedMultiple(items, rows, settings) @defer.inlineCallbacks def test_export_items(self): @@ -538,15 +594,14 @@ class FeedExportTest(unittest.TestCase): @defer.inlineCallbacks def test_export_no_items_not_store_empty(self): - formats = ('json', - 'jsonlines', - 'xml', - 'csv',) - - for fmt in formats: - settings = {'FEED_FORMAT': fmt} + for fmt in ('json', 'jsonlines', 'xml', 'csv'): + settings = { + 'FEEDS': { + self._random_temp_filename(): {'format': fmt}, + }, + } data = yield self.exported_no_data(settings) - self.assertEqual(data, b'') + self.assertEqual(data[fmt], b'') @defer.inlineCallbacks def test_export_no_items_store_empty(self): @@ -558,9 +613,15 @@ class FeedExportTest(unittest.TestCase): ) for fmt, expctd in formats: - settings = {'FEED_FORMAT': fmt, 'FEED_STORE_EMPTY': True, 'FEED_EXPORT_INDENT': None} + settings = { + 'FEEDS': { + self._random_temp_filename(): {'format': fmt}, + }, + 'FEED_STORE_EMPTY': True, + 'FEED_EXPORT_INDENT': None, + } data = yield self.exported_no_data(settings) - self.assertEqual(data, expctd) + self.assertEqual(data[fmt], expctd) @defer.inlineCallbacks def test_export_multiple_item_classes(self): @@ -581,9 +642,9 @@ class FeedExportTest(unittest.TestCase): header = self.MyItem.fields.keys() rows_csv = [ {'egg': 'spam1', 'foo': 'bar1', 'baz': ''}, - {'egg': '', 'foo': 'bar2', 'baz': ''}, + {'egg': '', 'foo': 'bar2', 'baz': ''}, {'egg': 'spam3', 'foo': 'bar3', 'baz': 'quux3'}, - {'egg': 'spam4', 'foo': '', 'baz': ''}, + {'egg': 'spam4', 'foo': '', 'baz': ''}, ] rows_jl = [dict(row) for row in items] yield self.assertExportedCsv(items, header, rows_csv, ordered=False) @@ -598,10 +659,10 @@ class FeedExportTest(unittest.TestCase): header = ["foo", "baz", "hello"] settings = {'FEED_EXPORT_FIELDS': header} rows = [ - {'foo': 'bar1', 'baz': '', 'hello': ''}, - {'foo': 'bar2', 'baz': '', 'hello': 'world2'}, + {'foo': 'bar1', 'baz': '', 'hello': ''}, + {'foo': 'bar2', 'baz': '', 'hello': 'world2'}, {'foo': 'bar3', 'baz': 'quux3', 'hello': ''}, - {'foo': '', 'baz': '', 'hello': 'world4'}, + {'foo': '', 'baz': '', 'hello': 'world4'}, ] yield self.assertExported(items, header, rows, settings=settings, ordered=True) @@ -663,10 +724,15 @@ class FeedExportTest(unittest.TestCase): 'csv': u'foo\r\nTest\xd6\r\n'.encode('utf-8'), } - for format, expected in formats.items(): - settings = {'FEED_FORMAT': format, 'FEED_EXPORT_INDENT': None} + for fmt, expected in formats.items(): + settings = { + 'FEEDS': { + self._random_temp_filename(): {'format': fmt}, + }, + 'FEED_EXPORT_INDENT': None, + } data = yield self.exported_data(items, settings) - self.assertEqual(expected, data) + self.assertEqual(expected, data[fmt]) formats = { 'json': u'[{"foo": "Test\xd6"}]'.encode('latin-1'), @@ -675,11 +741,53 @@ class FeedExportTest(unittest.TestCase): 'csv': u'foo\r\nTest\xd6\r\n'.encode('latin-1'), } - settings = {'FEED_EXPORT_INDENT': None, 'FEED_EXPORT_ENCODING': 'latin-1'} - for format, expected in formats.items(): - settings['FEED_FORMAT'] = format + for fmt, expected in formats.items(): + settings = { + 'FEEDS': { + self._random_temp_filename(): {'format': fmt}, + }, + 'FEED_EXPORT_INDENT': None, + 'FEED_EXPORT_ENCODING': 'latin-1', + } data = yield self.exported_data(items, settings) - self.assertEqual(expected, data) + self.assertEqual(expected, data[fmt]) + + @defer.inlineCallbacks + def test_export_multiple_configs(self): + items = [dict({'foo': u'FOO', 'bar': u'BAR'})] + + formats = { + 'json': u'[\n{"bar": "BAR"}\n]'.encode('utf-8'), + 'xml': u'\n\n \n FOO\n \n'.encode('latin-1'), + 'csv': u'bar,foo\r\nBAR,FOO\r\n'.encode('utf-8'), + } + + settings = { + 'FEEDS': { + self._random_temp_filename(): { + 'format': 'json', + 'indent': 0, + 'fields': ['bar'], + 'encoding': 'utf-8', + }, + self._random_temp_filename(): { + 'format': 'xml', + 'indent': 2, + 'fields': ['foo'], + 'encoding': 'latin-1', + }, + self._random_temp_filename(): { + 'format': 'csv', + 'indent': None, + 'fields': ['bar', 'foo'], + 'encoding': 'utf-8', + }, + }, + } + + data = yield self.exported_data(items, settings) + for fmt, expected in formats.items(): + self.assertEqual(expected, data[fmt]) @defer.inlineCallbacks def test_export_indentation(self): @@ -827,33 +935,38 @@ class FeedExportTest(unittest.TestCase): ] for row in test_cases: - settings = {'FEED_FORMAT': row['format'], 'FEED_EXPORT_INDENT': row['indent']} + settings = { + 'FEEDS': { + self._random_temp_filename(): { + 'format': row['format'], + 'indent': row['indent'], + }, + }, + } data = yield self.exported_data(items, settings) - print(row['format'], row['indent']) - self.assertEqual(row['expected'], data) + self.assertEqual(row['expected'], data[row['format']]) @defer.inlineCallbacks def test_init_exporters_storages_with_crawler(self): settings = { - 'FEED_EXPORTERS': {'csv': 'tests.test_feedexport.' - 'FromCrawlerCsvItemExporter'}, - 'FEED_STORAGES': {'file': 'tests.test_feedexport.' - 'FromCrawlerFileFeedStorage'}, + 'FEED_EXPORTERS': {'csv': 'tests.test_feedexport.FromCrawlerCsvItemExporter'}, + 'FEED_STORAGES': {'file': 'tests.test_feedexport.FromCrawlerFileFeedStorage'}, + 'FEEDS': { + self._random_temp_filename(): {'format': 'csv'}, + }, } - yield self.exported_data({}, settings) + yield self.exported_data(items=[], settings=settings) self.assertTrue(FromCrawlerCsvItemExporter.init_with_crawler) self.assertTrue(FromCrawlerFileFeedStorage.init_with_crawler) @defer.inlineCallbacks def test_pathlib_uri(self): - tmpdir = tempfile.mkdtemp() - feed_uri = Path(tmpdir) / 'res' + feed_path = Path(self._random_temp_filename()) settings = { - 'FEED_FORMAT': 'csv', 'FEED_STORE_EMPTY': True, - 'FEED_URI': feed_uri, - 'FEED_PATH': feed_uri + 'FEEDS': { + feed_path: {'format': 'csv'} + }, } data = yield self.exported_no_data(settings) - self.assertEqual(data, b'') - shutil.rmtree(tmpdir, ignore_errors=True) + self.assertEqual(data['csv'], b'') diff --git a/tests/test_http_response.py b/tests/test_http_response.py index 4c1b2afc3..eafc3560e 100644 --- a/tests/test_http_response.py +++ b/tests/test_http_response.py @@ -72,6 +72,22 @@ class BaseResponseTest(unittest.TestCase): r1 = self.response_class("http://www.example.com", body=b"Some body", request=req) assert r1.meta is req.meta + def test_copy_cb_kwargs(self): + req = Request("http://www.example.com") + req.cb_kwargs['foo'] = 'bar' + r1 = self.response_class("http://www.example.com", body=b"Some body", request=req) + assert r1.cb_kwargs is req.cb_kwargs + + def test_unavailable_meta(self): + r1 = self.response_class("http://www.example.com", body=b"Some body") + with self.assertRaisesRegex(AttributeError, r'Response\.meta not available'): + r1.meta + + def test_unavailable_cb_kwargs(self): + r1 = self.response_class("http://www.example.com", body=b"Some body") + with self.assertRaisesRegex(AttributeError, r'Response\.cb_kwargs not available'): + r1.cb_kwargs + def test_copy_inherited_classes(self): """Test Response children copies preserve their class""" @@ -167,6 +183,11 @@ class BaseResponseTest(unittest.TestCase): self._assert_followed_url(Link('http://example.com/foo '), 'http://example.com/foo%20') + def test_follow_flags(self): + res = self.response_class('http://example.com/') + fol = res.follow('http://example.com/', flags=['cached', 'allowed']) + self.assertEqual(fol.flags, ['cached', 'allowed']) + # Response.follow_all def test_follow_all_absolute(self): @@ -194,6 +215,10 @@ class BaseResponseTest(unittest.TestCase): links = map(Link, absolute) self._assert_followed_all_urls(links, absolute) + def test_follow_all_empty(self): + r = self.response_class("http://example.com") + self.assertEqual([], list(r.follow_all([]))) + def test_follow_all_invalid(self): r = self.response_class("http://example.com") if self.response_class == Response: @@ -232,6 +257,17 @@ class BaseResponseTest(unittest.TestCase): expected = [u.replace(' ', '%20') for u in absolute] self._assert_followed_all_urls(links, expected) + def test_follow_all_flags(self): + re = self.response_class('http://www.example.com/') + urls = [ + 'http://www.example.com/', + 'http://www.example.com/2', + 'http://www.example.com/foo', + ] + fol = re.follow_all(urls, flags=['cached', 'allowed']) + for req in fol: + self.assertEqual(req.flags, ['cached', 'allowed']) + def _assert_followed_url(self, follow_obj, target_url, response=None): if response is None: response = self._links_response() @@ -562,6 +598,22 @@ class TextResponseTest(BaseResponseTest): ) self.assertEqual(req.encoding, 'cp1251') + def test_follow_flags(self): + res = self.response_class('http://example.com/') + fol = res.follow('http://example.com/', flags=['cached', 'allowed']) + self.assertEqual(fol.flags, ['cached', 'allowed']) + + def test_follow_all_flags(self): + re = self.response_class('http://www.example.com/') + urls = [ + 'http://www.example.com/', + 'http://www.example.com/2', + 'http://www.example.com/foo', + ] + fol = re.follow_all(urls, flags=['cached', 'allowed']) + for req in fol: + self.assertEqual(req.flags, ['cached', 'allowed']) + def test_follow_all_css(self): expected = [ 'http://example.com/sample3.html', diff --git a/tests/test_item.py b/tests/test_item.py index 30463a0f5..f70632d57 100644 --- a/tests/test_item.py +++ b/tests/test_item.py @@ -149,13 +149,15 @@ class ItemTest(unittest.TestCase): fields = {'load': Field(default='A')} save = Field(default='A') - class B(A): pass + class B(A): + pass class C(Item): fields = {'load': Field(default='C')} save = Field(default='C') - class D(B, C): pass + class D(B, C): + pass item = D(save='X', load='Y') self.assertEqual(item['save'], 'X') @@ -164,7 +166,8 @@ class ItemTest(unittest.TestCase): 'save': {'default': 'A'}}) # D class inverted - class E(C, B): pass + class E(C, B): + pass self.assertEqual(E(save='X')['save'], 'X') self.assertEqual(E(load='X')['load'], 'X') @@ -177,7 +180,8 @@ class ItemTest(unittest.TestCase): save = Field(default='A') load = Field(default='A') - class B(A): pass + class B(A): + pass class C(A): fields = {'update': Field(default='C')} @@ -206,14 +210,16 @@ class ItemTest(unittest.TestCase): fields = {'load': Field(default='A')} save = Field(default='A') - class B(A): pass + class B(A): + pass class C(object): fields = {'load': Field(default='C')} not_allowed = Field(default='not_allowed') save = Field(default='C') - class D(B, C): pass + class D(B, C): + pass self.assertRaises(KeyError, D, not_allowed='value') self.assertEqual(D(save='X')['save'], 'X') @@ -221,7 +227,8 @@ class ItemTest(unittest.TestCase): 'load': {'default': 'A'}}) # D class inverted - class E(C, B): pass + class E(C, B): + pass self.assertRaises(KeyError, E, not_allowed='value') self.assertEqual(E(save='X')['save'], 'X') @@ -259,6 +266,7 @@ class ItemTest(unittest.TestCase): with catch_warnings(record=True) as warnings: item = Item() self.assertEqual(len(warnings), 0) + class SubclassedItem(Item): pass subclassed_item = SubclassedItem() diff --git a/tests/test_linkextractors.py b/tests/test_linkextractors.py index 38fb8fb4a..53968e60e 100644 --- a/tests/test_linkextractors.py +++ b/tests/test_linkextractors.py @@ -2,8 +2,6 @@ import re import unittest from warnings import catch_warnings -import pytest - from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.http import HtmlResponse, XmlResponse from scrapy.link import Link @@ -16,7 +14,6 @@ from tests import get_testdata class Base: class LinkExtractorTestCase(unittest.TestCase): extractor_cls = None - escapes_whitespace = False def setUp(self): body = get_testdata('link_extractor', 'linkextractor.html') @@ -30,10 +27,7 @@ class Base: def test_extract_all_links(self): lx = self.extractor_cls() - if self.escapes_whitespace: - page4_url = 'http://example.com/page%204.html' - else: - page4_url = 'http://example.com/page 4.html' + page4_url = 'http://example.com/page%204.html' self.assertEqual([link for link in lx.extract_links(self.response)], [ Link(url='http://example.com/sample1.html', text=u''), @@ -214,7 +208,7 @@ class Base: response = HtmlResponse("http://example.org/somepage/index.html", body=html, encoding='iso8859-15') links = self.extractor_cls(restrict_xpaths='//p').extract_links(response) self.assertEqual(links, - [Link(url='http://example.org/%E2%99%A5/you?c=%E2%82%AC', text=u'text')]) + [Link(url='http://example.org/%E2%99%A5/you?c=%A4', text=u'text')]) def test_restrict_xpaths_concat_in_handle_data(self): """html entities cause SGMLParser to call handle_data hook twice""" @@ -310,10 +304,7 @@ class Base: def test_attrs(self): lx = self.extractor_cls(attrs="href") - if self.escapes_whitespace: - page4_url = 'http://example.com/page%204.html' - else: - page4_url = 'http://example.com/page 4.html' + page4_url = 'http://example.com/page%204.html' self.assertEqual(lx.extract_links(self.response), [ Link(url='http://example.com/sample1.html', text=u''), @@ -506,7 +497,6 @@ class LxmlLinkExtractorTestCase(Base.LinkExtractorTestCase): Link(url='http://example.org/item2.html', text=u'Pic of a dog', nofollow=False), ]) - @pytest.mark.xfail def test_restrict_xpaths_with_html_entities(self): super(LxmlLinkExtractorTestCase, self).test_restrict_xpaths_with_html_entities() diff --git a/tests/test_logformatter.py b/tests/test_logformatter.py index 7d8c6ec7f..bf9fbe5e4 100644 --- a/tests/test_logformatter.py +++ b/tests/test_logformatter.py @@ -2,6 +2,7 @@ import unittest from testfixtures import LogCapture from twisted.internet import defer +from twisted.python.failure import Failure from twisted.trial.unittest import TestCase as TwistedTestCase from scrapy.crawler import CrawlerRunner @@ -62,15 +63,46 @@ class LogFormatterTestCase(unittest.TestCase): assert all(isinstance(x, str) for x in lines) self.assertEqual(lines, [u"Dropped: \u2018", '{}']) - def test_error(self): + def test_item_error(self): # In practice, the complete traceback is shown by passing the # 'exc_info' argument to the logging function item = {'key': 'value'} exception = Exception() response = Response("http://www.example.com") - logkws = self.formatter.error(item, exception, response, self.spider) + logkws = self.formatter.item_error(item, exception, response, self.spider) logline = logkws['msg'] % logkws['args'] - self.assertEqual(logline, u"'Error processing {'key': 'value'}'") + self.assertEqual(logline, u"Error processing {'key': 'value'}") + + def test_spider_error(self): + # In practice, the complete traceback is shown by passing the + # 'exc_info' argument to the logging function + failure = Failure(Exception()) + request = Request("http://www.example.com", headers={'Referer': 'http://example.org'}) + response = Response("http://www.example.com", request=request) + logkws = self.formatter.spider_error(failure, request, response, self.spider) + logline = logkws['msg'] % logkws['args'] + self.assertEqual( + logline, + "Spider error processing (referer: http://example.org)" + ) + + def test_download_error_short(self): + # In practice, the complete traceback is shown by passing the + # 'exc_info' argument to the logging function + failure = Failure(Exception()) + request = Request("http://www.example.com") + logkws = self.formatter.download_error(failure, request, self.spider) + logline = logkws['msg'] % logkws['args'] + self.assertEqual(logline, "Error downloading ") + + def test_download_error_long(self): + # In practice, the complete traceback is shown by passing the + # 'exc_info' argument to the logging function + failure = Failure(Exception()) + request = Request("http://www.example.com") + logkws = self.formatter.download_error(failure, request, self.spider, "Some message") + logline = logkws['msg'] % logkws['args'] + self.assertEqual(logline, "Error downloading : Some message") def test_scraped(self): item = CustomItem() diff --git a/tests/test_mail.py b/tests/test_mail.py index ddb0f1e70..f5cb81a8b 100644 --- a/tests/test_mail.py +++ b/tests/test_mail.py @@ -121,5 +121,6 @@ class MailSenderTest(unittest.TestCase): self.assertEqual(text.get_charset(), Charset('utf-8')) self.assertEqual(attach.get_payload(decode=True).decode('utf-8'), body) + if __name__ == "__main__": unittest.main() diff --git a/tests/test_pipeline_crawl.py b/tests/test_pipeline_crawl.py index fb72c9d6d..962c33144 100644 --- a/tests/test_pipeline_crawl.py +++ b/tests/test_pipeline_crawl.py @@ -26,10 +26,9 @@ class MediaDownloadSpider(SimpleSpider): self.media_key: [], self.media_urls_key: [ self._process_url(response.urljoin(href)) - for href in response.xpath(''' - //table[thead/tr/th="Filename"] - /tbody//a/@href - ''').getall()], + for href in response.xpath( + '//table[thead/tr/th="Filename"]/tbody//a/@href' + ).getall()], } yield item @@ -99,8 +98,9 @@ class FileDownloadCrawlTestCase(TestCase): if self.expected_checksums is not None: checksums = set( i['checksum'] - for item in items - for i in item[self.media_key]) + for item in items + for i in item[self.media_key] + ) self.assertEqual(checksums, self.expected_checksums) # check that the image files where actually written to the media store diff --git a/tests/test_pipeline_files.py b/tests/test_pipeline_files.py index e5bad2ed0..f155db4ce 100644 --- a/tests/test_pipeline_files.py +++ b/tests/test_pipeline_files.py @@ -272,7 +272,7 @@ class FilesPipelineTestCaseCustomSettings(unittest.TestCase): prefix = pipeline_cls.__name__.upper() settings = self._generate_fake_settings(prefix=prefix) user_pipeline = pipeline_cls.from_settings(Settings(settings)) - for pipe_cls_attr, settings_attr, pipe_inst_attr in self.file_cls_attr_settings_map: + for pipe_cls_attr, settings_attr, pipe_inst_attr in self.file_cls_attr_settings_map: custom_value = settings.get(prefix + "_" + settings_attr) self.assertNotEqual(custom_value, self.default_cls_settings[pipe_cls_attr]) self.assertEqual(getattr(user_pipeline, pipe_inst_attr), custom_value) @@ -286,7 +286,6 @@ class FilesPipelineTestCaseCustomSettings(unittest.TestCase): self.assertEqual(pipeline.files_result_field, "this") self.assertEqual(pipeline.files_urls_field, "that") - def test_user_defined_subclass_default_key_names(self): """Test situation when user defines subclass of FilesPipeline, but uses attribute names for default pipeline (without prefixing @@ -360,7 +359,7 @@ class TestGCSFilesStore(unittest.TestCase): self.assertIn('checksum', s) self.assertEqual(s['checksum'], 'zc2oVgXkbQr2EQdSdw3OPA==') u = urlparse(uri) - content, acl, blob = get_gcs_content_and_delete(u.hostname, u.path[1:]+path) + content, acl, blob = get_gcs_content_and_delete(u.hostname, u.path[1:] + path) self.assertEqual(content, data) self.assertEqual(blob.metadata, {'foo': 'bar'}) self.assertEqual(blob.cache_control, GCSFilesStore.CACHE_CONTROL) diff --git a/tests/test_pipeline_images.py b/tests/test_pipeline_images.py index 7f1cb4a11..5018d6802 100644 --- a/tests/test_pipeline_images.py +++ b/tests/test_pipeline_images.py @@ -177,7 +177,6 @@ class ImagesPipelineTestCaseCustomSettings(unittest.TestCase): IMAGES_RESULT_FIELD='images' ) - def setUp(self): self.tempdir = mkdtemp() diff --git a/tests/test_pipeline_media.py b/tests/test_pipeline_media.py index 1fcc5799e..d369e147d 100644 --- a/tests/test_pipeline_media.py +++ b/tests/test_pipeline_media.py @@ -304,6 +304,7 @@ class MediaPipelineTestCase(BaseMediaPipelineTestCase): return response rsp1 = Response('http://url') + def rsp1_func(): dfd = Deferred().addCallback(_check_downloading) reactor.callLater(.1, dfd.callback, rsp1) diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index d5a3371ab..8cdf7a176 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -90,5 +90,6 @@ class ResponseTypesTest(unittest.TestCase): # check that mime.types files shipped with scrapy are loaded self.assertEqual(responsetypes.mimetypes.guess_type('x.scrapytest')[0], 'x-scrapy/test') + if __name__ == "__main__": unittest.main() diff --git a/tests/test_scheduler.py b/tests/test_scheduler.py index e0e3600e5..13c297084 100644 --- a/tests/test_scheduler.py +++ b/tests/test_scheduler.py @@ -46,7 +46,7 @@ class MockCrawler(Crawler): def __init__(self, priority_queue_cls, jobdir): settings = dict( - LOG_UNSERIALIZABLE_REQUESTS=False, + SCHEDULER_DEBUG=False, SCHEDULER_DISK_QUEUE='scrapy.squeues.PickleLifoDiskQueue', SCHEDULER_MEMORY_QUEUE='scrapy.squeues.LifoMemoryQueue', SCHEDULER_PRIORITY_QUEUE=priority_queue_cls, diff --git a/tests/test_signals.py b/tests/test_signals.py new file mode 100644 index 000000000..d6ae526be --- /dev/null +++ b/tests/test_signals.py @@ -0,0 +1,44 @@ +from pytest import mark +from twisted.internet import defer +from twisted.trial import unittest + +from scrapy import signals, Request, Spider +from scrapy.utils.test import get_crawler, get_from_asyncio_queue + +from tests.mockserver import MockServer + + +class ItemSpider(Spider): + name = 'itemspider' + + def start_requests(self): + for index in range(10): + yield Request(self.mockserver.url('/status?n=200&id=%d' % index), + meta={'index': index}) + + def parse(self, response): + return {'index': response.meta['index']} + + +class AsyncSignalTestCase(unittest.TestCase): + def setUp(self): + self.mockserver = MockServer() + self.mockserver.__enter__() + self.items = [] + + def tearDown(self): + self.mockserver.__exit__(None, None, None) + + async def _on_item_scraped(self, item): + item = await get_from_asyncio_queue(item) + self.items.append(item) + + @mark.only_asyncio() + @defer.inlineCallbacks + def test_simple_pipeline(self): + crawler = get_crawler(ItemSpider) + crawler.signals.connect(self._on_item_scraped, signals.item_scraped) + yield crawler.crawl(mockserver=self.mockserver) + self.assertEqual(len(self.items), 10) + for index in range(10): + self.assertIn({'index': index}, self.items) diff --git a/tests/test_spider.py b/tests/test_spider.py index 6fbec7e58..f0f128d38 100644 --- a/tests/test_spider.py +++ b/tests/test_spider.py @@ -602,13 +602,19 @@ class DeprecationTest(unittest.TestCase): self.assertEqual(len(list(spider1.start_requests())), 1) self.assertEqual(len(w), 0) + # spider without overridden make_requests_from_url method + # should issue a warning when called directly + request = spider1.make_requests_from_url("http://www.example.com") + self.assertTrue(isinstance(request, Request)) + self.assertEqual(len(w), 1) + # spider with overridden make_requests_from_url issues a warning, # but the method still works spider2 = MySpider5() requests = list(spider2.start_requests()) self.assertEqual(len(requests), 1) self.assertEqual(requests[0].url, 'http://example.com/foo') - self.assertEqual(len(w), 1) + self.assertEqual(len(w), 2) class NoParseMethodSpiderTest(unittest.TestCase): diff --git a/tests/test_spidermiddleware.py b/tests/test_spidermiddleware.py index 55d665e79..78e926adc 100644 --- a/tests/test_spidermiddleware.py +++ b/tests/test_spidermiddleware.py @@ -94,7 +94,7 @@ class ProcessSpiderExceptionReRaise(SpiderMiddlewareTestCase): class RaiseExceptionProcessSpiderOutputMiddleware: def process_spider_output(self, response, result, spider): - 1/0 + 1 / 0 self.mwman._add_middleware(ProcessSpiderExceptionReturnNoneMiddleware()) self.mwman._add_middleware(RaiseExceptionProcessSpiderOutputMiddleware()) diff --git a/tests/test_spidermiddleware_offsite.py b/tests/test_spidermiddleware_offsite.py index 7511aa568..b96807bc2 100644 --- a/tests/test_spidermiddleware_offsite.py +++ b/tests/test_spidermiddleware_offsite.py @@ -4,7 +4,7 @@ import warnings from scrapy.http import Response, Request from scrapy.spiders import Spider -from scrapy.spidermiddlewares.offsite import OffsiteMiddleware, URLWarning +from scrapy.spidermiddlewares.offsite import OffsiteMiddleware, URLWarning, PortWarning from scrapy.utils.test import get_crawler @@ -26,7 +26,8 @@ class TestOffsiteMiddleware(TestCase): Request('http://scrapy.org/1'), Request('http://sub.scrapy.org/1'), Request('http://offsite.tld/letmepass', dont_filter=True), - Request('http://scrapy.test.org/')] + Request('http://scrapy.test.org/'), + Request('http://scrapy.test.org:8000/')] offsite_reqs = [Request('http://scrapy2.org'), Request('http://offsite.tld/'), Request('http://offsite.tld/scrapytest.org'), @@ -55,21 +56,21 @@ class TestOffsiteMiddleware2(TestOffsiteMiddleware): class TestOffsiteMiddleware3(TestOffsiteMiddleware2): - def _get_spider(self): - return Spider('foo') + def _get_spiderargs(self): + return dict(name='foo') class TestOffsiteMiddleware4(TestOffsiteMiddleware3): - def _get_spider(self): - bad_hostname = urlparse('http:////scrapytest.org').hostname - return dict(name='foo', allowed_domains=['scrapytest.org', None, bad_hostname]) + def _get_spiderargs(self): + bad_hostname = urlparse('http:////scrapytest.org').hostname + return dict(name='foo', allowed_domains=['scrapytest.org', None, bad_hostname]) def test_process_spider_output(self): - res = Response('http://scrapytest.org') - reqs = [Request('http://scrapytest.org/1')] - out = list(self.mw.process_spider_output(res, reqs, self.spider)) - self.assertEqual(out, reqs) + res = Response('http://scrapytest.org') + reqs = [Request('http://scrapytest.org/1')] + out = list(self.mw.process_spider_output(res, reqs, self.spider)) + self.assertEqual(out, reqs) class TestOffsiteMiddleware5(TestOffsiteMiddleware4): @@ -80,3 +81,13 @@ class TestOffsiteMiddleware5(TestOffsiteMiddleware4): warnings.simplefilter("always") self.mw.get_host_regex(self.spider) assert issubclass(w[-1].category, URLWarning) + + +class TestOffsiteMiddleware6(TestOffsiteMiddleware4): + + def test_get_host_regex(self): + self.spider.allowed_domains = ['scrapytest.org:8000', 'scrapy.org', 'scrapy.test.org'] + with warnings.catch_warnings(record=True) as w: + warnings.simplefilter("always") + self.mw.get_host_regex(self.spider) + assert issubclass(w[-1].category, PortWarning) diff --git a/tests/test_spidermiddleware_output_chain.py b/tests/test_spidermiddleware_output_chain.py index b19a74609..b26353d6c 100644 --- a/tests/test_spidermiddleware_output_chain.py +++ b/tests/test_spidermiddleware_output_chain.py @@ -125,7 +125,7 @@ class NotGeneratorCallbackSpider(Spider): yield Request(self.mockserver.url('/status?n=200')) def parse(self, response): - return [{'test': 1}, {'test': 1/0}] + return [{'test': 1}, {'test': 1 / 0}] # ================================================================================ diff --git a/tests/test_squeues.py b/tests/test_squeues.py index d5fcf2f7f..f0f3dd4c6 100644 --- a/tests/test_squeues.py +++ b/tests/test_squeues.py @@ -1,7 +1,12 @@ import pickle from queuelib.tests import test_queue as t -from scrapy.squeues import MarshalFifoDiskQueue, MarshalLifoDiskQueue, PickleFifoDiskQueue, PickleLifoDiskQueue +from scrapy.squeues import ( + MarshalFifoDiskQueueNonRequest as MarshalFifoDiskQueue, + MarshalLifoDiskQueueNonRequest as MarshalLifoDiskQueue, + PickleFifoDiskQueueNonRequest as PickleFifoDiskQueue, + PickleLifoDiskQueueNonRequest as PickleLifoDiskQueue +) from scrapy.item import Item, Field from scrapy.http import Request from scrapy.loader import ItemLoader @@ -31,7 +36,9 @@ def nonserializable_object_test(self): self.assertRaises(ValueError, q.push, lambda x: x) else: # Use a different unpickleable object - class A(object): pass + class A(object): + pass + a = A() a.__reduce__ = a.__reduce_ex__ = None self.assertRaises(ValueError, q.push, a) diff --git a/tests/test_utils_conf.py b/tests/test_utils_conf.py index 02d8ba51e..332120021 100644 --- a/tests/test_utils_conf.py +++ b/tests/test_utils_conf.py @@ -1,7 +1,14 @@ import unittest +import warnings -from scrapy.settings import BaseSettings -from scrapy.utils.conf import build_component_list, arglist_to_dict +from scrapy.exceptions import UsageError, ScrapyDeprecationWarning +from scrapy.settings import BaseSettings, Settings +from scrapy.utils.conf import ( + arglist_to_dict, + build_component_list, + feed_complete_default_values_from_settings, + feed_process_params_from_cli +) class BuildComponentListTest(unittest.TestCase): @@ -83,7 +90,6 @@ class BuildComponentListTest(unittest.TestCase): self.assertRaises(ValueError, build_component_list, {}, d, convert=lambda x: x) - class UtilsConfTestCase(unittest.TestCase): def test_arglist_to_dict(self): @@ -91,5 +97,87 @@ class UtilsConfTestCase(unittest.TestCase): {'arg1': 'val1', 'arg2': 'val2'}) +class FeedExportConfigTestCase(unittest.TestCase): + + def test_feed_export_config_invalid_format(self): + settings = Settings() + self.assertRaises(UsageError, feed_process_params_from_cli, settings, ['items.dat'], 'noformat') + + def test_feed_export_config_mismatch(self): + settings = Settings() + self.assertRaises( + UsageError, + feed_process_params_from_cli, settings, ['items1.dat', 'items2.dat'], 'noformat' + ) + + def test_feed_export_config_backward_compatible(self): + with warnings.catch_warnings(record=True) as cw: + settings = Settings() + self.assertEqual( + {'items.dat': {'format': 'csv'}}, + feed_process_params_from_cli(settings, ['items.dat'], 'csv') + ) + self.assertEqual(cw[0].category, ScrapyDeprecationWarning) + + def test_feed_export_config_explicit_formats(self): + settings = Settings() + self.assertEqual( + {'items_1.dat': {'format': 'json'}, 'items_2.dat': {'format': 'xml'}, 'items_3.dat': {'format': 'csv'}}, + feed_process_params_from_cli(settings, ['items_1.dat:json', 'items_2.dat:xml', 'items_3.dat:csv']) + ) + + def test_feed_export_config_implicit_formats(self): + settings = Settings() + self.assertEqual( + {'items_1.json': {'format': 'json'}, 'items_2.xml': {'format': 'xml'}, 'items_3.csv': {'format': 'csv'}}, + feed_process_params_from_cli(settings, ['items_1.json', 'items_2.xml', 'items_3.csv']) + ) + + def test_feed_export_config_stdout(self): + settings = Settings() + self.assertEqual( + {'stdout:': {'format': 'pickle'}}, + feed_process_params_from_cli(settings, ['-:pickle']) + ) + + def test_feed_complete_default_values_from_settings_empty(self): + feed = {} + settings = Settings({ + "FEED_EXPORT_ENCODING": "custom encoding", + "FEED_EXPORT_FIELDS": ["f1", "f2", "f3"], + "FEED_EXPORT_INDENT": 42, + "FEED_STORE_EMPTY": True, + "FEED_URI_PARAMS": (1, 2, 3, 4), + }) + new_feed = feed_complete_default_values_from_settings(feed, settings) + self.assertEqual(new_feed, { + "encoding": "custom encoding", + "fields": ["f1", "f2", "f3"], + "indent": 42, + "store_empty": True, + "uri_params": (1, 2, 3, 4), + }) + + def test_feed_complete_default_values_from_settings_non_empty(self): + feed = { + "encoding": "other encoding", + "fields": None, + } + settings = Settings({ + "FEED_EXPORT_ENCODING": "custom encoding", + "FEED_EXPORT_FIELDS": ["f1", "f2", "f3"], + "FEED_EXPORT_INDENT": 42, + "FEED_STORE_EMPTY": True, + }) + new_feed = feed_complete_default_values_from_settings(feed, settings) + self.assertEqual(new_feed, { + "encoding": "other encoding", + "fields": None, + "indent": 42, + "store_empty": True, + "uri_params": None, + }) + + if __name__ == "__main__": unittest.main() diff --git a/tests/test_utils_defer.py b/tests/test_utils_defer.py index dfbe71ae2..a3b6e64f1 100644 --- a/tests/test_utils_defer.py +++ b/tests/test_utils_defer.py @@ -9,6 +9,7 @@ from scrapy.utils.defer import mustbe_deferred, process_chain, \ class MustbeDeferredTest(unittest.TestCase): def test_success_function(self): steps = [] + def _append(v): steps.append(v) return steps @@ -20,6 +21,7 @@ class MustbeDeferredTest(unittest.TestCase): def test_unfired_deferred(self): steps = [] + def _append(v): steps.append(v) dfd = defer.Deferred() @@ -102,7 +104,7 @@ class IterErrbackTest(unittest.TestCase): def iterbad(): for x in range(10): if x == 5: - a = 1/0 + a = 1 / 0 yield x errors = [] diff --git a/tests/test_utils_deprecate.py b/tests/test_utils_deprecate.py index 159ef8f25..b3a90d314 100644 --- a/tests/test_utils_deprecate.py +++ b/tests/test_utils_deprecate.py @@ -110,6 +110,7 @@ class WarnWhenSubclassedTest(unittest.TestCase): # ignore subclassing warnings with warnings.catch_warnings(): warnings.simplefilter('ignore', ScrapyDeprecationWarning) + class UserClass(Deprecated): pass @@ -233,6 +234,7 @@ class WarnWhenSubclassedTest(unittest.TestCase): with warnings.catch_warnings(record=True) as w: AlsoDeprecated() + class UserClass(AlsoDeprecated): pass @@ -247,6 +249,7 @@ class WarnWhenSubclassedTest(unittest.TestCase): with mock.patch('inspect.stack', side_effect=IndexError): with warnings.catch_warnings(record=True) as w: DeprecatedName = create_deprecated_class('DeprecatedName', NewName) + class SubClass(DeprecatedName): pass diff --git a/tests/test_utils_iterators.py b/tests/test_utils_iterators.py index 9776dfb2a..33fc4d570 100644 --- a/tests/test_utils_iterators.py +++ b/tests/test_utils_iterators.py @@ -387,7 +387,6 @@ class TestHelper(unittest.TestCase): self.assertTrue(type(r1) is type(r2)) self.assertTrue(type(r1) is not type(r3)) - def _assert_type_and_value(self, a, b, obj): self.assertTrue(type(a) is type(b), 'Got {}, expected {} for {!r}'.format(type(a), type(b), obj)) diff --git a/tests/test_utils_log.py b/tests/test_utils_log.py index 2c23f3616..21100aeb8 100644 --- a/tests/test_utils_log.py +++ b/tests/test_utils_log.py @@ -16,7 +16,7 @@ class FailureToExcInfoTest(unittest.TestCase): def test_failure(self): try: - 0/0 + 0 / 0 except ZeroDivisionError: exc_info = sys.exc_info() failure = Failure() diff --git a/tests/test_utils_project.py b/tests/test_utils_project.py index bd74b0c34..1ef4eeb14 100644 --- a/tests/test_utils_project.py +++ b/tests/test_utils_project.py @@ -3,7 +3,11 @@ import os import tempfile import shutil import contextlib -from scrapy.utils.project import data_path + +from pytest import warns + +from scrapy.exceptions import ScrapyDeprecationWarning +from scrapy.utils.project import data_path, get_project_settings @contextlib.contextmanager @@ -41,3 +45,53 @@ class ProjectUtilsTest(unittest.TestCase): ) abspath = os.path.join(os.path.sep, 'absolute', 'path') self.assertEqual(abspath, data_path(abspath)) + + +@contextlib.contextmanager +def set_env(**update): + modified = set(update.keys()) & set(os.environ.keys()) + update_after = {k: os.environ[k] for k in modified} + remove_after = frozenset(k for k in update if k not in os.environ) + try: + os.environ.update(update) + yield + finally: + os.environ.update(update_after) + for k in remove_after: + os.environ.pop(k) + + +class GetProjectSettingsTestCase(unittest.TestCase): + + def test_valid_envvar(self): + value = 'tests.test_cmdline.settings' + envvars = { + 'SCRAPY_SETTINGS_MODULE': value, + } + with set_env(**envvars), warns(None) as warnings: + settings = get_project_settings() + assert not warnings + assert settings.get('SETTINGS_MODULE') == value + + def test_invalid_envvar(self): + envvars = { + 'SCRAPY_FOO': 'bar', + } + with set_env(**envvars), warns(None) as warnings: + get_project_settings() + assert len(warnings) == 1 + assert warnings[0].category == ScrapyDeprecationWarning + assert str(warnings[0].message).endswith(': FOO') + + def test_valid_and_invalid_envvars(self): + value = 'tests.test_cmdline.settings' + envvars = { + 'SCRAPY_FOO': 'bar', + 'SCRAPY_SETTINGS_MODULE': value, + } + with set_env(**envvars), warns(None) as warnings: + settings = get_project_settings() + assert len(warnings) == 1 + assert warnings[0].category == ScrapyDeprecationWarning + assert str(warnings[0].message).endswith(': FOO') + assert settings.get('SETTINGS_MODULE') == value diff --git a/tests/test_utils_python.py b/tests/test_utils_python.py index b79e0ac1c..ec5b4c596 100644 --- a/tests/test_utils_python.py +++ b/tests/test_utils_python.py @@ -104,7 +104,6 @@ class BinaryIsTextTest(unittest.TestCase): assert not binary_is_text(b"\x02\xa3") - class UtilsPythonTestCase(unittest.TestCase): def test_equal_attributes(self): @@ -154,7 +153,9 @@ class UtilsPythonTestCase(unittest.TestCase): self.assertFalse(equal_attributes(a, b, [compare_z, 'x'])) def test_weakkeycache(self): - class _Weakme(object): pass + class _Weakme(object): + pass + _values = count() wk = WeakKeyCache(lambda k: next(_values)) k = _Weakme() @@ -215,7 +216,6 @@ class UtilsPythonTestCase(unittest.TestCase): self.assertEqual( get_func_args(operator.itemgetter(2), stripself=True), ['obj']) - def test_without_none_values(self): self.assertEqual(without_none_values([1, None, 3, 4]), [1, 3, 4]) self.assertEqual(without_none_values((1, None, 3, 4)), (1, 3, 4)) @@ -223,5 +223,6 @@ class UtilsPythonTestCase(unittest.TestCase): without_none_values({'one': 1, 'none': None, 'three': 3, 'four': 4}), {'one': 1, 'three': 3, 'four': 4}) + if __name__ == "__main__": unittest.main() diff --git a/tests/test_utils_request.py b/tests/test_utils_request.py index 3e664fc74..45f0f59e4 100644 --- a/tests/test_utils_request.py +++ b/tests/test_utils_request.py @@ -83,5 +83,6 @@ class UtilsRequestTest(unittest.TestCase): request_httprepr(Request("file:///tmp/foo.txt")) request_httprepr(Request("ftp://localhost/tmp/foo.txt")) + if __name__ == "__main__": unittest.main() diff --git a/tests/test_utils_signal.py b/tests/test_utils_signal.py index 16b7c5c68..9f6da09ed 100644 --- a/tests/test_utils_signal.py +++ b/tests/test_utils_signal.py @@ -1,3 +1,6 @@ +import asyncio + +from pytest import mark from testfixtures import LogCapture from twisted.trial import unittest from twisted.python.failure import Failure @@ -5,6 +8,7 @@ from twisted.internet import defer, reactor from pydispatch import dispatcher from scrapy.utils.signal import send_catch_log, send_catch_log_deferred +from scrapy.utils.test import get_from_asyncio_queue class SendCatchLogTest(unittest.TestCase): @@ -40,7 +44,7 @@ class SendCatchLogTest(unittest.TestCase): def error_handler(self, arg, handlers_called): handlers_called.add(self.error_handler) - a = 1/0 + a = 1 / 0 def ok_handler(self, arg, handlers_called): handlers_called.add(self.ok_handler) @@ -54,7 +58,7 @@ class SendCatchLogDeferredTest(SendCatchLogTest): return send_catch_log_deferred(signal, *a, **kw) -class SendCatchLogDeferredTest2(SendCatchLogTest): +class SendCatchLogDeferredTest2(SendCatchLogDeferredTest): def ok_handler(self, arg, handlers_called): handlers_called.add(self.ok_handler) @@ -63,8 +67,24 @@ class SendCatchLogDeferredTest2(SendCatchLogTest): reactor.callLater(0, d.callback, "OK") return d - def _get_result(self, signal, *a, **kw): - return send_catch_log_deferred(signal, *a, **kw) + +class SendCatchLogDeferredAsyncDefTest(SendCatchLogDeferredTest): + + async def ok_handler(self, arg, handlers_called): + handlers_called.add(self.ok_handler) + assert arg == 'test' + await defer.succeed(42) + return "OK" + + +@mark.only_asyncio() +class SendCatchLogDeferredAsyncioTest(SendCatchLogDeferredTest): + + async def ok_handler(self, arg, handlers_called): + handlers_called.add(self.ok_handler) + assert arg == 'test' + await asyncio.sleep(0.2) + return await get_from_asyncio_queue("OK") class SendCatchLogTest2(unittest.TestCase): diff --git a/tests/test_utils_template.py b/tests/test_utils_template.py index 40b733233..5a52dd695 100644 --- a/tests/test_utils_template.py +++ b/tests/test_utils_template.py @@ -38,5 +38,6 @@ class UtilsRenderTemplateFileTestCase(unittest.TestCase): os.remove(render_path) assert not os.path.exists(render_path) # Failure of test iself + if '__main__' == __name__: unittest.main() diff --git a/tests/test_utils_url.py b/tests/test_utils_url.py index 21e9a056a..7abff8281 100644 --- a/tests/test_utils_url.py +++ b/tests/test_utils_url.py @@ -30,7 +30,7 @@ class UrlUtilsTest(unittest.TestCase): url = 'javascript:%20document.orderform_2581_1190810811.mode.value=%27add%27;%20javascript:%20document.orderform_2581_1190810811.submit%28%29' self.assertFalse(url_is_from_any_domain(url, ['testdomain.com'])) - self.assertFalse(url_is_from_any_domain(url+'.testdomain.com', ['testdomain.com'])) + self.assertFalse(url_is_from_any_domain(url + '.testdomain.com', ['testdomain.com'])) def test_url_is_from_spider(self): spider = Spider(name='example.com') @@ -201,7 +201,8 @@ def create_skipped_scheme_t(args): assert url.startswith(args[1]) return do_expected -for k, args in enumerate ([ + +for k, args in enumerate([ ('/index', 'file://'), ('/index.html', 'file://'), ('./index.html', 'file://'), @@ -229,7 +230,7 @@ for k, args in enumerate ([ ], start=1): t_method = create_guess_scheme_t(args) t_method.__name__ = 'test_uri_%03d' % k - setattr (GuessSchemeTest, t_method.__name__, t_method) + setattr(GuessSchemeTest, t_method.__name__, t_method) # TODO: the following tests do not pass with current implementation for k, args in enumerate([ @@ -238,7 +239,7 @@ for k, args in enumerate([ ], start=1): t_method = create_skipped_scheme_t(args) t_method.__name__ = 'test_uri_skipped_%03d' % k - setattr (GuessSchemeTest, t_method.__name__, t_method) + setattr(GuessSchemeTest, t_method.__name__, t_method) class StripUrl(unittest.TestCase): diff --git a/tests/test_webclient.py b/tests/test_webclient.py index 746367b41..6253d5c3f 100644 --- a/tests/test_webclient.py +++ b/tests/test_webclient.py @@ -9,7 +9,12 @@ import OpenSSL.SSL from twisted.trial import unittest from twisted.web import server, static, util, resource from twisted.internet import reactor, defer -from twisted.test.proto_helpers import StringTransport +try: + from twisted.internet.testing import StringTransport +except ImportError: + # deprecated in Twisted 19.7.0 + # (remove once we bump our requirement past that version) + from twisted.test.proto_helpers import StringTransport from twisted.python.filepath import FilePath from twisted.protocols.policies import WrappingFactory from twisted.internet.defer import inlineCallbacks @@ -51,22 +56,22 @@ class ParseUrlTestCase(unittest.TestCase): ("http://127.0.0.1?c=v&c2=v2#fragment", ('http', lip, lip, 80, '/?c=v&c2=v2')), ("http://127.0.0.1/?c=v&c2=v2#fragment", ('http', lip, lip, 80, '/?c=v&c2=v2')), ("http://127.0.0.1/foo?c=v&c2=v2#frag", ('http', lip, lip, 80, '/foo?c=v&c2=v2')), - ("http://127.0.0.1:100?c=v&c2=v2#fragment", ('http', lip+':100', lip, 100, '/?c=v&c2=v2')), - ("http://127.0.0.1:100/?c=v&c2=v2#frag", ('http', lip+':100', lip, 100, '/?c=v&c2=v2')), - ("http://127.0.0.1:100/foo?c=v&c2=v2#frag", ('http', lip+':100', lip, 100, '/foo?c=v&c2=v2')), + ("http://127.0.0.1:100?c=v&c2=v2#fragment", ('http', lip + ':100', lip, 100, '/?c=v&c2=v2')), + ("http://127.0.0.1:100/?c=v&c2=v2#frag", ('http', lip + ':100', lip, 100, '/?c=v&c2=v2')), + ("http://127.0.0.1:100/foo?c=v&c2=v2#frag", ('http', lip + ':100', lip, 100, '/foo?c=v&c2=v2')), ("http://127.0.0.1", ('http', lip, lip, 80, '/')), ("http://127.0.0.1/", ('http', lip, lip, 80, '/')), ("http://127.0.0.1/foo", ('http', lip, lip, 80, '/foo')), ("http://127.0.0.1?param=value", ('http', lip, lip, 80, '/?param=value')), ("http://127.0.0.1/?param=value", ('http', lip, lip, 80, '/?param=value')), - ("http://127.0.0.1:12345/foo", ('http', lip+':12345', lip, 12345, '/foo')), + ("http://127.0.0.1:12345/foo", ('http', lip + ':12345', lip, 12345, '/foo')), ("http://spam:12345/foo", ('http', 'spam:12345', 'spam', 12345, '/foo')), ("http://spam.test.org/foo", ('http', 'spam.test.org', 'spam.test.org', 80, '/foo')), ("https://127.0.0.1/foo", ('https', lip, lip, 443, '/foo')), ("https://127.0.0.1/?param=value", ('https', lip, lip, 443, '/?param=value')), - ("https://127.0.0.1:12345/", ('https', lip+':12345', lip, 12345, '/')), + ("https://127.0.0.1:12345/", ('https', lip + ':12345', lip, 12345, '/')), ("http://scrapytest.org/foo ", ('http', 'scrapytest.org', 'scrapytest.org', 80, '/foo')), ("http://egg:7890 ", ('http', 'egg:7890', 'egg', 7890, '/')), @@ -294,6 +299,7 @@ class WebClientTestCase(unittest.TestCase): finished = self.assertFailure( getPage(self.getURL("wait"), timeout=0.000001), defer.TimeoutError) + def cleanup(passthrough): # Clean up the server which is hanging around not doing # anything.