diff --git a/.bandit.yml b/.bandit.yml new file mode 100644 index 000000000..243379b0b --- /dev/null +++ b/.bandit.yml @@ -0,0 +1,18 @@ +skips: +- B101 +- B105 +- B301 +- B303 +- B306 +- B307 +- B311 +- B320 +- B321 +- B402 # https://github.com/scrapy/scrapy/issues/4180 +- B403 +- B404 +- B406 +- B410 +- B503 +- B603 +- B605 diff --git a/.bumpversion.cfg b/.bumpversion.cfg index 70affe63f..c9f1abea5 100644 --- a/.bumpversion.cfg +++ b/.bumpversion.cfg @@ -1,5 +1,5 @@ [bumpversion] -current_version = 1.7.0 +current_version = 1.8.0 commit = True tag = True tag_name = {new_version} diff --git a/.coveragerc b/.coveragerc index 914d697a0..02acbff8e 100644 --- a/.coveragerc +++ b/.coveragerc @@ -3,4 +3,3 @@ branch = true include = scrapy/* omit = tests/* - scrapy/xlib/* diff --git a/.github/ISSUE_TEMPLATE/bug_report.md b/.github/ISSUE_TEMPLATE/bug_report.md new file mode 100644 index 000000000..8ca10109b --- /dev/null +++ b/.github/ISSUE_TEMPLATE/bug_report.md @@ -0,0 +1,41 @@ +--- +name: Bug report +about: Report a problem to help us improve +--- + + + +### Description + +[Description of the issue] + +### Steps to Reproduce + +1. [First Step] +2. [Second Step] +3. [and so on...] + +**Expected behavior:** [What you expect to happen] + +**Actual behavior:** [What actually happens] + +**Reproduces how often:** [What percentage of the time does it reproduce?] + +### Versions + +Please paste here the output of executing `scrapy version --verbose` in the command line. + +### Additional context + +Any additional information, configuration, data or output from commands that might be necessary to reproduce or understand the issue. Please try not to include screenshots of code or the command line, paste the contents as text instead. You can use [GitHub Flavored Markdown](https://help.github.com/en/articles/creating-and-highlighting-code-blocks) to make the text look better. diff --git a/.github/ISSUE_TEMPLATE/feature_request.md b/.github/ISSUE_TEMPLATE/feature_request.md new file mode 100644 index 000000000..e05273fe2 --- /dev/null +++ b/.github/ISSUE_TEMPLATE/feature_request.md @@ -0,0 +1,33 @@ +--- +name: Feature request +about: Suggest an idea for an enhancement or new feature +--- + + + +## Summary + +One paragraph explanation of the feature. + +## Motivation + +Why are we doing this? What use cases does it support? What is the expected outcome? + +## Describe alternatives you've considered + +A clear and concise description of the alternative solutions you've considered. Be sure to explain why Scrapy's existing customizability isn't suitable for this feature. + +## Additional context + +Any additional information about the feature request here. diff --git a/.readthedocs.yml b/.readthedocs.yml new file mode 100644 index 000000000..563add75f --- /dev/null +++ b/.readthedocs.yml @@ -0,0 +1,9 @@ +version: 2 +sphinx: + configuration: docs/conf.py +python: + # For available versions, see: + # https://docs.readthedocs.io/en/stable/config-file/v2.html#build-image + version: 3.7 # Keep in sync with .travis.yml + install: + - requirements: docs/requirements.txt diff --git a/.travis.yml b/.travis.yml index 08b0bf119..66e1a9617 100644 --- a/.travis.yml +++ b/.travis.yml @@ -1,4 +1,5 @@ language: python +dist: xenial branches: only: - master @@ -6,35 +7,31 @@ branches: - /^\d\.\d+\.\d+(rc\d+|\.dev\d+)?$/ matrix: include: - - python: 2.7 - env: TOXENV=py27 - - python: 2.7 - env: TOXENV=jessie - - python: 2.7 - env: TOXENV=pypy - - python: 2.7 - env: TOXENV=pypy3 - - python: 3.4 - env: TOXENV=py34 - - python: 3.5 - env: TOXENV=py35 - - python: 3.6 - env: TOXENV=py36 - - python: 3.7 - env: TOXENV=py37 - dist: xenial - sudo: true - - python: 3.6 - env: TOXENV=docs + - env: TOXENV=security + python: 3.8 + - env: TOXENV=flake8 + python: 3.8 + - env: TOXENV=pypy3 + - env: TOXENV=py35 + python: 3.5 + - env: TOXENV=pinned + python: 3.5 + - env: TOXENV=py35-asyncio + python: 3.5.2 + - env: TOXENV=py36 + python: 3.6 + - env: TOXENV=py37 + python: 3.7 + - env: TOXENV=py38 + python: 3.8 + - env: TOXENV=extra-deps + python: 3.8 + - env: TOXENV=py38-asyncio + python: 3.8 + - env: TOXENV=docs + python: 3.7 # Keep in sync with .readthedocs.yml install: - | - if [ "$TOXENV" = "pypy" ]; then - export PYPY_VERSION="pypy-6.0.0-linux_x86_64-portable" - wget "https://bitbucket.org/squeaky/portable-pypy/downloads/${PYPY_VERSION}.tar.bz2" - tar -jxf ${PYPY_VERSION}.tar.bz2 - virtualenv --python="$PYPY_VERSION/bin/pypy" "$HOME/virtualenvs/$PYPY_VERSION" - source "$HOME/virtualenvs/$PYPY_VERSION/bin/activate" - fi if [ "$TOXENV" = "pypy3" ]; then export PYPY_VERSION="pypy3.5-5.9-beta-linux_x86_64-portable" wget "https://bitbucket.org/squeaky/portable-pypy/downloads/${PYPY_VERSION}.tar.bz2" @@ -65,4 +62,4 @@ deploy: on: tags: true repo: scrapy/scrapy - condition: "$TOXENV == py27 && $TRAVIS_TAG =~ ^[0-9]+[.][0-9]+[.][0-9]+(rc[0-9]+|[.]dev[0-9]+)?$" + condition: "$TOXENV == py37 && $TRAVIS_TAG =~ ^[0-9]+[.][0-9]+[.][0-9]+(rc[0-9]+|[.]dev[0-9]+)?$" diff --git a/CODE_OF_CONDUCT.md b/CODE_OF_CONDUCT.md index d477168eb..d1cd3e517 100644 --- a/CODE_OF_CONDUCT.md +++ b/CODE_OF_CONDUCT.md @@ -68,7 +68,7 @@ members of the project's leadership. ## Attribution This Code of Conduct is adapted from the [Contributor Covenant][homepage], version 1.4, -available at [http://contributor-covenant.org/version/1/4][version] +available at [http://contributor-covenant.org/version/1/4][version]. [homepage]: http://contributor-covenant.org [version]: http://contributor-covenant.org/version/1/4/ diff --git a/README.rst b/README.rst index c28d217ff..7fefaeec9 100644 --- a/README.rst +++ b/README.rst @@ -34,13 +34,13 @@ Scrapy is a fast high-level web crawling and web scraping framework, used to crawl websites and extract structured data from their pages. It can be used for a wide range of purposes, from data mining to monitoring and automated testing. -For more information including a list of features check the Scrapy homepage at: -https://scrapy.org +Check the Scrapy homepage at https://scrapy.org for more information, +including a list of features. Requirements ============ -* Python 2.7 or Python 3.4+ +* Python 3.5+ * Works on Linux, Windows, Mac OSX, BSD Install @@ -50,8 +50,8 @@ The quick way:: pip install scrapy -For more details see the install section in the documentation: -https://docs.scrapy.org/en/latest/intro/install.html +See the install section in the documentation at +https://docs.scrapy.org/en/latest/intro/install.html for more details. Documentation ============= @@ -62,17 +62,17 @@ directory. Releases ======== -You can find release notes at https://docs.scrapy.org/en/latest/news.html +You can check https://docs.scrapy.org/en/latest/news.html for the release notes. Community (blog, twitter, mail list, IRC) ========================================= -See https://scrapy.org/community/ +See https://scrapy.org/community/ for details. Contributing ============ -See https://docs.scrapy.org/en/master/contributing.html +See https://docs.scrapy.org/en/master/contributing.html for details. Code of Conduct --------------- @@ -86,9 +86,9 @@ Please report unacceptable behavior to opensource@scrapinghub.com. Companies using Scrapy ====================== -See https://scrapy.org/companies/ +See https://scrapy.org/companies/ for a list. Commercial Support ================== -See https://scrapy.org/support/ +See https://scrapy.org/support/ for details. diff --git a/conftest.py b/conftest.py index d8531d6cc..c0de09909 100644 --- a/conftest.py +++ b/conftest.py @@ -1,31 +1,51 @@ -import glob -import six +from pathlib import Path + import pytest -from twisted import version as twisted_version def _py_files(folder): - return glob.glob(folder + "/*.py") + glob.glob(folder + "/*/*.py") + return (str(p) for p in Path(folder).rglob('*.py')) collect_ignore = [ # not a test, but looks like a test "scrapy/utils/testsite.py", - + # contains scripts to be run by tests/test_crawler.py::CrawlerProcessSubprocess + *_py_files("tests/CrawlerProcess") ] -if (twisted_version.major, twisted_version.minor, twisted_version.micro) >= (15, 5, 0): - collect_ignore += _py_files("scrapy/xlib/tx") - - -if six.PY3: - for line in open('tests/py3-ignores.txt'): - file_path = line.strip() - if file_path and file_path[0] != '#': - collect_ignore.append(file_path) +for line in open('tests/ignores.txt'): + file_path = line.strip() + if file_path and file_path[0] != '#': + collect_ignore.append(file_path) @pytest.fixture() def chdir(tmpdir): """Change to pytest-provided temporary directory""" tmpdir.chdir() + + +def pytest_collection_modifyitems(session, config, items): + # Avoid executing tests when executing `--flake8` flag (pytest-flake8) + try: + from pytest_flake8 import Flake8Item + if config.getoption('--flake8'): + items[:] = [item for item in items if isinstance(item, Flake8Item)] + except ImportError: + pass + + +@pytest.fixture(scope='class') +def reactor_pytest(request): + if not request.cls: + # doctests + return + request.cls.reactor_pytest = request.config.getoption("--reactor") + return request.cls.reactor_pytest + + +@pytest.fixture(autouse=True) +def only_asyncio(request, reactor_pytest): + if request.node.get_closest_marker('only_asyncio') and reactor_pytest != 'asyncio': + pytest.skip('This test is only run with --reactor=asyncio') diff --git a/debian/scrapy.lintian-overrides b/debian/scrapy.lintian-overrides index 955e7def0..b5de7f67d 100644 --- a/debian/scrapy.lintian-overrides +++ b/debian/scrapy.lintian-overrides @@ -1,2 +1 @@ new-package-should-close-itp-bug -extra-license-file usr/share/pyshared/scrapy/xlib/pydispatch/license.txt diff --git a/docs/_tests/quotes.html b/docs/_tests/quotes.html new file mode 100644 index 000000000..71aff8847 --- /dev/null +++ b/docs/_tests/quotes.html @@ -0,0 +1,281 @@ + + + + + Quotes to Scrape + + + + +
+
+ +
+

+ + Login + +

+
+
+ + +
+
+ +
+ “The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.” + by + (about) + +
+ Tags: + + + change + + deep-thoughts + + thinking + + world + +
+
+ +
+ “It is our choices, Harry, that show what we truly are, far more than our abilities.” + by + (about) + +
+ Tags: + + + abilities + + choices + +
+
+ +
+ “There are only two ways to live your life. One is as though nothing is a miracle. The other is as though everything is a miracle.” + by + (about) + +
+ Tags: + + + inspirational + + life + + live + + miracle + + miracles + +
+
+ +
+ “The person, be it gentleman or lady, who has not pleasure in a good novel, must be intolerably stupid.” + by + (about) + +
+ Tags: + + + aliteracy + + books + + classic + + humor + +
+
+ +
+ “Imperfection is beauty, madness is genius and it's better to be absolutely ridiculous than absolutely boring.” + by + (about) + +
+ Tags: + + + be-yourself + + inspirational + +
+
+ +
+ “Try not to become a man of success. Rather become a man of value.” + by + (about) + +
+ Tags: + + + adulthood + + success + + value + +
+
+ +
+ “It is better to be hated for what you are than to be loved for what you are not.” + by + (about) + +
+ Tags: + + + life + + love + +
+
+ +
+ “I have not failed. I've just found 10,000 ways that won't work.” + by + (about) + +
+ Tags: + + + edison + + failure + + inspirational + + paraphrased + +
+
+ +
+ “A woman is like a tea bag; you never know how strong it is until it's in hot water.” + by + (about) + + +
+ +
+ “A day without sunshine is like, you know, night.” + by + (about) + +
+ Tags: + + + humor + + obvious + + simile + +
+
+ + +
+
+ +

Top Ten tags

+ + + love + + + + inspirational + + + + life + + + + humor + + + + books + + + + reading + + + + friendship + + + + friends + + + + truth + + + + simile + + + +
+
+ +
+ + + \ No newline at end of file diff --git a/docs/_tests/quotes1.html b/docs/_tests/quotes1.html new file mode 100644 index 000000000..71aff8847 --- /dev/null +++ b/docs/_tests/quotes1.html @@ -0,0 +1,281 @@ + + + + + Quotes to Scrape + + + + +
+
+ +
+

+ + Login + +

+
+
+ + +
+
+ +
+ “The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.” + by + (about) + +
+ Tags: + + + change + + deep-thoughts + + thinking + + world + +
+
+ +
+ “It is our choices, Harry, that show what we truly are, far more than our abilities.” + by + (about) + +
+ Tags: + + + abilities + + choices + +
+
+ +
+ “There are only two ways to live your life. One is as though nothing is a miracle. The other is as though everything is a miracle.” + by + (about) + +
+ Tags: + + + inspirational + + life + + live + + miracle + + miracles + +
+
+ +
+ “The person, be it gentleman or lady, who has not pleasure in a good novel, must be intolerably stupid.” + by + (about) + +
+ Tags: + + + aliteracy + + books + + classic + + humor + +
+
+ +
+ “Imperfection is beauty, madness is genius and it's better to be absolutely ridiculous than absolutely boring.” + by + (about) + +
+ Tags: + + + be-yourself + + inspirational + +
+
+ +
+ “Try not to become a man of success. Rather become a man of value.” + by + (about) + +
+ Tags: + + + adulthood + + success + + value + +
+
+ +
+ “It is better to be hated for what you are than to be loved for what you are not.” + by + (about) + +
+ Tags: + + + life + + love + +
+
+ +
+ “I have not failed. I've just found 10,000 ways that won't work.” + by + (about) + +
+ Tags: + + + edison + + failure + + inspirational + + paraphrased + +
+
+ +
+ “A woman is like a tea bag; you never know how strong it is until it's in hot water.” + by + (about) + + +
+ +
+ “A day without sunshine is like, you know, night.” + by + (about) + +
+ Tags: + + + humor + + obvious + + simile + +
+
+ + +
+
+ +

Top Ten tags

+ + + love + + + + inspirational + + + + life + + + + humor + + + + books + + + + reading + + + + friendship + + + + friends + + + + truth + + + + simile + + + +
+
+ +
+ + + \ No newline at end of file diff --git a/docs/conf.py b/docs/conf.py index f49f79cd5..458bc6087 100644 --- a/docs/conf.py +++ b/docs/conf.py @@ -12,6 +12,7 @@ # serve to show the default. import sys +from datetime import datetime from os import path # If your extensions are in another directory, add it here. If the directory @@ -27,10 +28,13 @@ sys.path.insert(0, path.dirname(path.dirname(__file__))) # Add any Sphinx extension module names here, as strings. They can be extensions # coming with Sphinx (named 'sphinx.ext.*') or your custom ones. extensions = [ + 'hoverxref.extension', + 'notfound.extension', 'scrapydocs', 'sphinx.ext.autodoc', 'sphinx.ext.coverage', 'sphinx.ext.intersphinx', + 'sphinx.ext.viewcode', ] # Add any paths that contain templates here, relative to this directory. @@ -46,8 +50,8 @@ source_suffix = '.rst' master_doc = 'index' # General information about the project. -project = u'Scrapy' -copyright = u'2008–2018, Scrapy developers' +project = 'Scrapy' +copyright = '2008–{}, Scrapy developers'.format(datetime.now().year) # The version info for the project you're documenting, acts as replacement for # |version| and |release|, also used in various other places throughout the @@ -191,8 +195,8 @@ htmlhelp_basename = 'Scrapydoc' # Grouping the document tree into LaTeX files. List of tuples # (source start file, target name, title, author, document class [howto/manual]). latex_documents = [ - ('index', 'Scrapy.tex', u'Scrapy Documentation', - u'Scrapy developers', 'manual'), + ('index', 'Scrapy.tex', 'Scrapy Documentation', + 'Scrapy developers', 'manual'), ] # The name of an image file (relative to this directory) to place at the top of @@ -237,7 +241,7 @@ coverage_ignore_pyobjects = [ r'\bContractsManager\b$', # For default contracts we only want to document their general purpose in - # their constructor, the methods they reimplement to achieve that purpose + # their __init__ method, the methods they reimplement to achieve that purpose # should be irrelevant to developers using those contracts. r'\w+Contract\.(adjust_request_args|(pre|post)_process)$', @@ -252,6 +256,23 @@ coverage_ignore_pyobjects = [ # Private exception used by the command-line interface implementation. r'^scrapy\.exceptions\.UsageError', + + # Methods of BaseItemExporter subclasses are only documented in + # BaseItemExporter. + r'^scrapy\.exporters\.(?!BaseItemExporter\b)\w*?\.', + + # Extension behavior is only modified through settings. Methods of + # extension classes, as well as helper functions, are implementation + # details that are not documented. + r'^scrapy\.extensions\.[a-z]\w*?\.[A-Z]\w*?\.', # methods + r'^scrapy\.extensions\.[a-z]\w*?\.[a-z]', # helper functions + + # Never documented before, and deprecated now. + r'^scrapy\.item\.DictItem$', + r'^scrapy\.linkextractors\.FilteringLinkExtractor$', + + # Implementation detail of LxmlLinkExtractor + r'^scrapy\.linkextractors\.lxmlhtml\.LxmlParserLinkExtractor', ] @@ -259,5 +280,16 @@ coverage_ignore_pyobjects = [ # ------------------------------------- intersphinx_mapping = { + 'coverage': ('https://coverage.readthedocs.io/en/stable', None), + 'pytest': ('https://docs.pytest.org/en/latest', None), 'python': ('https://docs.python.org/3', None), + 'sphinx': ('https://www.sphinx-doc.org/en/master', None), + 'tox': ('https://tox.readthedocs.io/en/latest', None), + 'twisted': ('https://twistedmatrix.com/documents/current', None), } + + +# Options for sphinx-hoverxref options +# ------------------------------------ + +hoverxref_auto_ref = True diff --git a/docs/conftest.py b/docs/conftest.py new file mode 100644 index 000000000..8c735e838 --- /dev/null +++ b/docs/conftest.py @@ -0,0 +1,29 @@ +import os +from doctest import ELLIPSIS, NORMALIZE_WHITESPACE + +from scrapy.http.response.html import HtmlResponse +from sybil import Sybil +from sybil.parsers.codeblock import CodeBlockParser +from sybil.parsers.doctest import DocTestParser +from sybil.parsers.skip import skip + + +def load_response(url, filename): + input_path = os.path.join(os.path.dirname(__file__), '_tests', filename) + with open(input_path, 'rb') as input_file: + return HtmlResponse(url, body=input_file.read()) + + +def setup(namespace): + namespace['load_response'] = load_response + + +pytest_collect_file = Sybil( + parsers=[ + DocTestParser(optionflags=ELLIPSIS | NORMALIZE_WHITESPACE), + CodeBlockParser(future_imports=['print_function']), + skip, + ], + pattern='*.rst', + setup=setup, +).pytest() diff --git a/docs/contributing.rst b/docs/contributing.rst index 28dea74de..f40a6bba2 100644 --- a/docs/contributing.rst +++ b/docs/contributing.rst @@ -44,7 +44,7 @@ guidelines when you're going to report a new bug. * check the :ref:`FAQ ` first to see if your issue is addressed in a well-known question -* if you have a general question about scrapy usage, please ask it at +* if you have a general question about Scrapy usage, please ask it at `Stack Overflow `__ (use "scrapy" tag). @@ -177,79 +177,71 @@ Documentation policies ====================== For reference documentation of API members (classes, methods, etc.) use -docstrings and make sure that the Sphinx documentation uses the autodoc_ -extension to pull the docstrings. API reference documentation should follow -docstring conventions (`PEP 257`_) and be IDE-friendly: short, to the point, -and it may provide short examples. +docstrings and make sure that the Sphinx documentation uses the +:mod:`~sphinx.ext.autodoc` extension to pull the docstrings. API reference +documentation should follow docstring conventions (`PEP 257`_) and be +IDE-friendly: short, to the point, and it may provide short examples. Other types of documentation, such as tutorials or topics, should be covered in files within the ``docs/`` directory. This includes documentation that is specific to an API member, but goes beyond API reference documentation. -In any case, if something is covered in a docstring, use the autodoc_ -extension to pull the docstring into the documentation instead of duplicating -the docstring in files within the ``docs/`` directory. - -.. _autodoc: http://www.sphinx-doc.org/en/stable/ext/autodoc.html +In any case, if something is covered in a docstring, use the +:mod:`~sphinx.ext.autodoc` extension to pull the docstring into the +documentation instead of duplicating the docstring in files within the +``docs/`` directory. Tests ===== -Tests are implemented using the `Twisted unit-testing framework`_, running -tests requires `tox`_. +Tests are implemented using the :doc:`Twisted unit-testing framework +`. Running tests requires +:doc:`tox `. .. _running-tests: Running tests ------------- -Make sure you have a recent enough `tox`_ installation: +To run all tests:: - ``tox --version`` - -If your version is older than 1.7.0, please update it first: - - ``pip install -U tox`` - -To run all tests go to the root directory of Scrapy source code and run: - - ``tox`` + tox To run a specific test (say ``tests/test_loader.py``) use: ``tox -- tests/test_loader.py`` -To run the tests on a specific tox_ environment, use ``-e `` with an -environment name from ``tox.ini``. For example, to run the tests with Python -3.6 use:: +To run the tests on a specific :doc:`tox ` environment, use +``-e `` with an environment name from ``tox.ini``. For example, to run +the tests with Python 3.6 use:: tox -e py36 -You can also specify a comma-separated list of environmets, and use `tox’s -parallel mode`_ to run the tests on multiple environments in parallel:: +You can also specify a comma-separated list of environments, and use :ref:`tox’s +parallel mode ` to run the tests on multiple environments in +parallel:: - tox -e py27,py36 -p auto + tox -e py36,py38 -p auto -To pass command-line options to pytest_, add them after ``--`` in your call to -tox_. Using ``--`` overrides the default positional arguments defined in -``tox.ini``, so you must include those default positional arguments -(``scrapy tests``) after ``--`` as well:: +To pass command-line options to :doc:`pytest `, add them after +``--`` in your call to :doc:`tox `. Using ``--`` overrides the +default positional arguments defined in ``tox.ini``, so you must include those +default positional arguments (``scrapy tests``) after ``--`` as well:: tox -- scrapy tests -x # stop after first failure You can also use the `pytest-xdist`_ plugin. For example, to run all tests on -the Python 3.6 tox_ environment using all your CPU cores:: +the Python 3.6 :doc:`tox ` environment using all your CPU cores:: tox -e py36 -- scrapy tests -n auto -To see coverage report install `coverage`_ (``pip install coverage``) and run: +To see coverage report install :doc:`coverage ` +(``pip install coverage``) and run: ``coverage report`` see output of ``coverage --help`` for more options like html or xml report. -.. _coverage: https://pypi.python.org/pypi/coverage - Writing tests ------------- @@ -270,13 +262,9 @@ And their unit-tests are in:: .. _issue tracker: https://github.com/scrapy/scrapy/issues .. _scrapy-users: https://groups.google.com/forum/#!forum/scrapy-users .. _Scrapy subreddit: https://reddit.com/r/scrapy -.. _Twisted unit-testing framework: https://twistedmatrix.com/documents/current/core/development/policy/test-standard.html .. _AUTHORS: https://github.com/scrapy/scrapy/blob/master/AUTHORS .. _tests/: https://github.com/scrapy/scrapy/tree/master/tests .. _open issues: https://github.com/scrapy/scrapy/issues .. _PEP 257: https://www.python.org/dev/peps/pep-0257/ .. _pull request: https://help.github.com/en/articles/creating-a-pull-request -.. _pytest: https://docs.pytest.org/en/latest/usage.html -.. _pytest-xdist: https://docs.pytest.org/en/3.0.0/xdist.html -.. _tox: https://pypi.python.org/pypi/tox -.. _tox’s parallel mode: https://tox.readthedocs.io/en/latest/example/basic.html#parallel-mode +.. _pytest-xdist: https://github.com/pytest-dev/pytest-xdist diff --git a/docs/faq.rst b/docs/faq.rst index 44f2b97bd..169e1a479 100644 --- a/docs/faq.rst +++ b/docs/faq.rst @@ -69,11 +69,11 @@ Here's an example spider using BeautifulSoup API, with ``lxml`` as the HTML pars What Python versions does Scrapy support? ----------------------------------------- -Scrapy is supported under Python 2.7 and Python 3.4+ +Scrapy is supported under Python 3.5+ under CPython (default Python implementation) and PyPy (starting with PyPy 5.9). -Python 2.6 support was dropped starting at Scrapy 0.20. Python 3 support was added in Scrapy 1.1. PyPy support was added in Scrapy 1.4, PyPy3 support was added in Scrapy 1.5. +Python 2 support was dropped in Scrapy 2.0. .. note:: For Python 3 support on Windows, it is recommended to use @@ -140,7 +140,7 @@ setting the following settings:: While pending requests are below the configured values of :setting:`CONCURRENT_REQUESTS`, :setting:`CONCURRENT_REQUESTS_PER_DOMAIN` or -:setting:`CONCURRENT_REQUESTS_PER_DOMAIN`, those requests are sent +:setting:`CONCURRENT_REQUESTS_PER_IP`, those requests are sent concurrently. As a result, the first few requests of a crawl rarely follow the desired order. Lowering those settings to ``1`` enforces the desired order, but it significantly slows down the crawl as a whole. @@ -338,7 +338,7 @@ How to split an item into multiple items in an item pipeline? input item. :ref:`Create a spider middleware ` instead, and use its :meth:`~scrapy.spidermiddlewares.SpiderMiddleware.process_spider_output` -method for this puspose. For example:: +method for this purpose. For example:: from copy import deepcopy diff --git a/docs/index.rst b/docs/index.rst index 6d5f9e77d..a4343b7e0 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -170,7 +170,7 @@ Solving specific problems Get answers to most frequently asked questions. :doc:`topics/debug` - Learn how to debug common problems of your scrapy spider. + Learn how to debug common problems of your Scrapy spider. :doc:`topics/contracts` Learn how to use contracts for testing your spiders. diff --git a/docs/intro/install.rst b/docs/intro/install.rst index daec7fcb7..178be723c 100644 --- a/docs/intro/install.rst +++ b/docs/intro/install.rst @@ -7,7 +7,7 @@ Installation guide Installing Scrapy ================= -Scrapy runs on Python 2.7 and Python 3.4 or above +Scrapy runs on Python 3.5 or above under CPython (default Python implementation) and PyPy (starting with PyPy 5.9). If you're using `Anaconda`_ or `Miniconda`_, you can install the package from @@ -78,9 +78,9 @@ TL;DR: We recommend installing Scrapy inside a virtual environment on all platforms. Python packages can be installed either globally (a.k.a system wide), -or in user-space. We do not recommend installing scrapy system wide. +or in user-space. We do not recommend installing Scrapy system wide. -Instead, we recommend that you install scrapy within a so-called +Instead, we recommend that you install Scrapy within a so-called "virtual environment" (`virtualenv`_). Virtualenvs allow you to not conflict with already-installed Python system packages (which could break some of your system tools and scripts), @@ -97,15 +97,13 @@ Check this `user guide`_ on how to create your virtualenv. .. note:: If you use Linux or OS X, `virtualenvwrapper`_ is a handy tool to create virtualenvs. -Once you have created a virtualenv, you can install scrapy inside it with ``pip``, +Once you have created a virtualenv, you can install Scrapy inside it with ``pip``, just like any other Python package. (See :ref:`platform-specific guides ` below for non-Python dependencies that you may need to install beforehand). -Python virtualenvs can be created to use Python 2 by default, or Python 3 by default. - -* If you want to install scrapy with Python 3, install scrapy within a Python 3 virtualenv. -* And if you want to install scrapy with Python 2, install scrapy within a Python 2 virtualenv. +Python virtualenvs can be created to use Python 2 by default, or Python 3 by default. As Scrapy +only supports Python 3, make sure you created a Python 3 virtualenv. .. _virtualenv: https://virtualenv.pypa.io .. _virtualenv installation instructions: https://virtualenv.pypa.io/en/stable/installation/ @@ -146,19 +144,15 @@ albeit with potential issues with TLS connections. typically too old and slow to catch up with latest Scrapy. -To install scrapy on Ubuntu (or Ubuntu-based) systems, you need to install +To install Scrapy on Ubuntu (or Ubuntu-based) systems, you need to install these dependencies:: - sudo apt-get install python-dev python-pip libxml2-dev libxslt1-dev zlib1g-dev libffi-dev libssl-dev + sudo apt-get install python3 python3-dev python3-pip libxml2-dev libxslt1-dev zlib1g-dev libffi-dev libssl-dev -- ``python-dev``, ``zlib1g-dev``, ``libxml2-dev`` and ``libxslt1-dev`` +- ``python3-dev``, ``zlib1g-dev``, ``libxml2-dev`` and ``libxslt1-dev`` are required for ``lxml`` - ``libssl-dev`` and ``libffi-dev`` are required for ``cryptography`` -If you want to install scrapy on Python 3, you’ll also need Python 3 development headers:: - - sudo apt-get install python3 python3-dev - Inside a :ref:`virtualenv `, you can install Scrapy with ``pip`` after that:: @@ -231,17 +225,17 @@ PyPy We recommend using the latest PyPy version. The version tested is 5.9.0. For PyPy3, only Linux installation was tested. -Most scrapy dependencides now have binary wheels for CPython, but not for PyPy. +Most Scrapy dependencides now have binary wheels for CPython, but not for PyPy. This means that these dependecies will be built during installation. On OS X, you are likely to face an issue with building Cryptography dependency, solution to this problem is described `here `_, that is to ``brew install openssl`` and then export the flags that this command -recommends (only needed when installing scrapy). Installing on Linux has no special +recommends (only needed when installing Scrapy). Installing on Linux has no special issues besides installing build dependencies. -Installing scrapy with PyPy on Windows is not tested. +Installing Scrapy with PyPy on Windows is not tested. -You can check that scrapy is installed correctly by running ``scrapy bench``. +You can check that Scrapy is installed correctly by running ``scrapy bench``. If this command gives errors such as ``TypeError: ... got 2 unexpected keyword arguments``, this means that setuptools was unable to pick up one PyPy-specific dependency. @@ -278,7 +272,7 @@ For details, see `Issue #2473 `_. .. _Python: https://www.python.org/ .. _pip: https://pip.pypa.io/en/latest/installing/ -.. _lxml: http://lxml.de/ +.. _lxml: https://lxml.de/index.html .. _parsel: https://pypi.python.org/pypi/parsel .. _w3lib: https://pypi.python.org/pypi/w3lib .. _twisted: https://twistedmatrix.com/ @@ -290,5 +284,5 @@ For details, see `Issue #2473 `_. .. _zsh: https://www.zsh.org/ .. _Scrapinghub: https://scrapinghub.com .. _Anaconda: https://docs.anaconda.com/anaconda/ -.. _Miniconda: https://conda.io/docs/user-guide/install/index.html +.. _Miniconda: https://docs.conda.io/projects/conda/en/latest/user-guide/install/index.html .. _conda-forge: https://conda-forge.org/ diff --git a/docs/intro/overview.rst b/docs/intro/overview.rst index 8b2fef065..01986b594 100644 --- a/docs/intro/overview.rst +++ b/docs/intro/overview.rst @@ -34,8 +34,8 @@ http://quotes.toscrape.com, following the pagination:: def parse(self, response): for quote in response.css('div.quote'): yield { - 'text': quote.css('span.text::text').get(), 'author': quote.xpath('span/small/text()').get(), + 'text': quote.css('span.text::text').get(), } next_page = response.css('li.next a::attr("href")').get() diff --git a/docs/intro/tutorial.rst b/docs/intro/tutorial.rst index a190ce407..c9d00eb74 100644 --- a/docs/intro/tutorial.rst +++ b/docs/intro/tutorial.rst @@ -78,9 +78,9 @@ Our first Spider Spiders are classes that you define and that Scrapy uses to scrape information from a website (or a group of websites). They must subclass -:class:`scrapy.Spider` and define the initial requests to make, optionally how -to follow links in the pages, and how to parse the downloaded page content to -extract data. +:class:`~scrapy.spiders.Spider` and define the initial requests to make, +optionally how to follow links in the pages, and how to parse the downloaded +page content to extract data. This is the code for our first Spider. Save it in a file named ``quotes_spider.py`` under the ``tutorial/spiders`` directory in your project:: @@ -235,13 +235,16 @@ You will see something like:: [s] shelp() Shell help (print this help) [s] fetch(req_or_url) Fetch request (or URL) and update local objects [s] view(response) View response in a browser - >>> Using the shell, you can try selecting elements using `CSS`_ with the response -object:: +object: - >>> response.css('title') - [] +.. invisible-code-block: python + + response = load_response('http://quotes.toscrape.com/page/1/', 'quotes1.html') + +>>> response.css('title') +[] The result of running ``response.css('title')`` is a list-like object called :class:`~scrapy.selector.SelectorList`, which represents a list of @@ -249,30 +252,30 @@ The result of running ``response.css('title')`` is a list-like object called and allow you to run further queries to fine-grain the selection or extract the data. -To extract the text from the title above, you can do:: +To extract the text from the title above, you can do: - >>> response.css('title::text').getall() - ['Quotes to Scrape'] +>>> response.css('title::text').getall() +['Quotes to Scrape'] There are two things to note here: one is that we've added ``::text`` to the CSS query, to mean we want to select only the text elements directly inside ```` element. If we don't specify ``::text``, we'd get the full title -element, including its tags:: +element, including its tags: - >>> response.css('title').getall() - ['<title>Quotes to Scrape'] +>>> response.css('title').getall() +['Quotes to Scrape'] The other thing is that the result of calling ``.getall()`` is a list: it is possible that a selector returns more than one result, so we extract them all. -When you know you just want the first result, as in this case, you can do:: +When you know you just want the first result, as in this case, you can do: - >>> response.css('title::text').get() - 'Quotes to Scrape' +>>> response.css('title::text').get() +'Quotes to Scrape' -As an alternative, you could've written:: +As an alternative, you could've written: - >>> response.css('title::text')[0].get() - 'Quotes to Scrape' +>>> response.css('title::text')[0].get() +'Quotes to Scrape' However, using ``.get()`` directly on a :class:`~scrapy.selector.SelectorList` instance avoids an ``IndexError`` and returns ``None`` when it doesn't @@ -285,14 +288,14 @@ to be scraped, you can at least get **some** data. Besides the :meth:`~scrapy.selector.SelectorList.getall` and :meth:`~scrapy.selector.SelectorList.get` methods, you can also use the :meth:`~scrapy.selector.SelectorList.re` method to extract using `regular -expressions`_:: +expressions`_: - >>> response.css('title::text').re(r'Quotes.*') - ['Quotes to Scrape'] - >>> response.css('title::text').re(r'Q\w+') - ['Quotes'] - >>> response.css('title::text').re(r'(\w+) to (\w+)') - ['Quotes', 'Scrape'] +>>> response.css('title::text').re(r'Quotes.*') +['Quotes to Scrape'] +>>> response.css('title::text').re(r'Q\w+') +['Quotes'] +>>> response.css('title::text').re(r'(\w+) to (\w+)') +['Quotes', 'Scrape'] In order to find the proper CSS selectors to use, you might find useful opening the response page from the shell in your web browser using ``view(response)``. @@ -309,12 +312,12 @@ visually selected elements, which works in many browsers. XPath: a brief intro ^^^^^^^^^^^^^^^^^^^^ -Besides `CSS`_, Scrapy selectors also support using `XPath`_ expressions:: +Besides `CSS`_, Scrapy selectors also support using `XPath`_ expressions: - >>> response.xpath('//title') - [] - >>> response.xpath('//title/text()').get() - 'Quotes to Scrape' +>>> response.xpath('//title') +[] +>>> response.xpath('//title/text()').get() +'Quotes to Scrape' XPath expressions are very powerful, and are the foundation of Scrapy Selectors. In fact, CSS selectors are converted to XPath under-the-hood. You @@ -369,45 +372,53 @@ we want:: $ scrapy shell 'http://quotes.toscrape.com' -We get a list of selectors for the quote HTML elements with:: +We get a list of selectors for the quote HTML elements with: - >>> response.css("div.quote") +>>> response.css("div.quote") +[, + , + ...] Each of the selectors returned by the query above allows us to run further queries over their sub-elements. Let's assign the first selector to a -variable, so that we can run our CSS selectors directly on a particular quote:: +variable, so that we can run our CSS selectors directly on a particular quote: - >>> quote = response.css("div.quote")[0] +>>> quote = response.css("div.quote")[0] Now, let's extract ``text``, ``author`` and the ``tags`` from that quote -using the ``quote`` object we just created:: +using the ``quote`` object we just created: - >>> text = quote.css("span.text::text").get() - >>> text - '“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”' - >>> author = quote.css("small.author::text").get() - >>> author - 'Albert Einstein' +>>> text = quote.css("span.text::text").get() +>>> text +'“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”' +>>> author = quote.css("small.author::text").get() +>>> author +'Albert Einstein' Given that the tags are a list of strings, we can use the ``.getall()`` method -to get all of them:: +to get all of them: - >>> tags = quote.css("div.tags a.tag::text").getall() - >>> tags - ['change', 'deep-thoughts', 'thinking', 'world'] +>>> tags = quote.css("div.tags a.tag::text").getall() +>>> tags +['change', 'deep-thoughts', 'thinking', 'world'] + +.. invisible-code-block: python + + from sys import version_info + +.. skip: next if(version_info < (3, 6), reason="Only Python 3.6+ dictionaries match the output") Having figured out how to extract each bit, we can now iterate over all the -quotes elements and put them together into a Python dictionary:: +quotes elements and put them together into a Python dictionary: - >>> for quote in response.css("div.quote"): - ... text = quote.css("span.text::text").get() - ... author = quote.css("small.author::text").get() - ... tags = quote.css("div.tags a.tag::text").getall() - ... print(dict(text=text, author=author, tags=tags)) - {'tags': ['change', 'deep-thoughts', 'thinking', 'world'], 'author': 'Albert Einstein', 'text': '“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”'} - {'tags': ['abilities', 'choices'], 'author': 'J.K. Rowling', 'text': '“It is our choices, Harry, that show what we truly are, far more than our abilities.”'} - ... a few more of these, omitted for brevity - >>> +>>> for quote in response.css("div.quote"): +... text = quote.css("span.text::text").get() +... author = quote.css("small.author::text").get() +... tags = quote.css("div.tags a.tag::text").getall() +... print(dict(text=text, author=author, tags=tags)) +{'text': '“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”', 'author': 'Albert Einstein', 'tags': ['change', 'deep-thoughts', 'thinking', 'world']} +{'text': '“It is our choices, Harry, that show what we truly are, far more than our abilities.”', 'author': 'J.K. Rowling', 'tags': ['abilities', 'choices']} +... Extracting data in our spider ----------------------------- @@ -505,23 +516,23 @@ markup: -We can try extracting it in the shell:: +We can try extracting it in the shell: - >>> response.css('li.next a').get() - 'Next ' +>>> response.css('li.next a').get() +'Next ' This gets the anchor element, but we want the attribute ``href``. For that, Scrapy supports a CSS extension that lets you select the attribute contents, -like this:: +like this: - >>> response.css('li.next a::attr(href)').get() - '/page/2/' +>>> response.css('li.next a::attr(href)').get() +'/page/2/' There is also an ``attrib`` property available -(see :ref:`selecting-attributes` for more):: +(see :ref:`selecting-attributes` for more): - >>> response.css('li.next a').attrib['href'] - '/page/2' +>>> response.css('li.next a').attrib['href'] +'/page/2/' Let's see now our spider modified to recursively follow the link to the next page, extracting data from it:: @@ -605,21 +616,25 @@ instance; you still have to yield this Request. You can also pass a selector to ``response.follow`` instead of a string; this selector should extract necessary attributes:: - for href in response.css('li.next a::attr(href)'): + for href in response.css('ul.pager a::attr(href)'): yield response.follow(href, callback=self.parse) For ```` elements there is a shortcut: ``response.follow`` uses their href attribute automatically. So the code can be shortened further:: - for a in response.css('li.next a'): + for a in response.css('ul.pager a'): yield response.follow(a, callback=self.parse) -.. note:: +To create multiple requests from an iterable, you can use +:meth:`response.follow_all ` instead:: + + anchors = response.css('ul.pager a') + yield from response.follow_all(anchors, callback=self.parse) + +or, shortening it further:: + + yield from response.follow_all(css='ul.pager a', callback=self.parse) - ``response.follow(response.css('li.next a'))`` is not valid because - ``response.css`` returns a list-like object with selectors for all results, - not a single selector. A ``for`` loop like in the example above, or - ``response.follow(response.css('li.next a')[0])`` is fine. More examples and patterns -------------------------- @@ -636,13 +651,11 @@ this time for scraping author information:: start_urls = ['http://quotes.toscrape.com/'] def parse(self, response): - # follow links to author pages - for href in response.css('.author + a::attr(href)'): - yield response.follow(href, self.parse_author) + author_page_links = response.css('.author + a') + yield from response.follow_all(author_links, self.parse_author) - # follow pagination links - for href in response.css('li.next a::attr(href)'): - yield response.follow(href, self.parse) + pagination_links = response.css('li.next a') + yield from response.follow_all(pagination_links, self.parse) def parse_author(self, response): def extract_with_css(query): @@ -658,8 +671,10 @@ This spider will start from the main page, it will follow all the links to the authors pages calling the ``parse_author`` callback for each of them, and also the pagination links with the ``parse`` callback as we saw before. -Here we're passing callbacks to ``response.follow`` as positional arguments -to make the code shorter; it also works for ``scrapy.Request``. +Here we're passing callbacks to +:meth:`response.follow_all ` as positional +arguments to make the code shorter; it also works for +:class:`~scrapy.http.Request`. The ``parse_author`` callback defines a helper function to extract and cleanup the data from a CSS query and yields the Python dict with the author data. diff --git a/docs/news.rst b/docs/news.rst index a0f0c5697..6d0d4b4ee 100644 --- a/docs/news.rst +++ b/docs/news.rst @@ -6,11 +6,246 @@ Release notes .. note:: Scrapy 1.x will be the last series supporting Python 2. Scrapy 2.0, planned for Q4 2019 or Q1 2020, will support **Python 3 only**. +.. _release-1.8.0: + +Scrapy 1.8.0 (2019-10-28) +------------------------- + +Highlights: + +* Dropped Python 3.4 support and updated minimum requirements; made Python 3.8 + support official +* New :meth:`Request.from_curl ` class method +* New :setting:`ROBOTSTXT_PARSER` and :setting:`ROBOTSTXT_USER_AGENT` settings +* New :setting:`DOWNLOADER_CLIENT_TLS_CIPHERS` and + :setting:`DOWNLOADER_CLIENT_TLS_VERBOSE_LOGGING` settings + +Backward-incompatible changes +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +* Python 3.4 is no longer supported, and some of the minimum requirements of + Scrapy have also changed: + + * cssselect_ 0.9.1 + * cryptography_ 2.0 + * lxml_ 3.5.0 + * pyOpenSSL_ 16.2.0 + * queuelib_ 1.4.2 + * service_identity_ 16.0.0 + * six_ 1.10.0 + * Twisted_ 17.9.0 (16.0.0 with Python 2) + * zope.interface_ 4.1.3 + + (:issue:`3892`) + +* ``JSONRequest`` is now called :class:`~scrapy.http.JsonRequest` for + consistency with similar classes (:issue:`3929`, :issue:`3982`) + +* If you are using a custom context factory + (:setting:`DOWNLOADER_CLIENTCONTEXTFACTORY`), its ``__init__`` method must + accept two new parameters: ``tls_verbose_logging`` and ``tls_ciphers`` + (:issue:`2111`, :issue:`3392`, :issue:`3442`, :issue:`3450`) + +* :class:`~scrapy.loader.ItemLoader` now turns the values of its input item + into lists: + + >>> item = MyItem() + >>> item['field'] = 'value1' + >>> loader = ItemLoader(item=item) + >>> item['field'] + ['value1'] + + This is needed to allow adding values to existing fields + (``loader.add_value('field', 'value2')``). + + (:issue:`3804`, :issue:`3819`, :issue:`3897`, :issue:`3976`, :issue:`3998`, + :issue:`4036`) + +See also :ref:`1.8-deprecation-removals` below. + + +New features +~~~~~~~~~~~~ + +* A new :meth:`Request.from_curl ` class + method allows :ref:`creating a request from a cURL command + ` (:issue:`2985`, :issue:`3862`) + +* A new :setting:`ROBOTSTXT_PARSER` setting allows choosing which robots.txt_ + parser to use. It includes built-in support for + :ref:`RobotFileParser `, + :ref:`Protego ` (default), :ref:`Reppy `, and + :ref:`Robotexclusionrulesparser `, and allows you to + :ref:`implement support for additional parsers + ` (:issue:`754`, :issue:`2669`, + :issue:`3796`, :issue:`3935`, :issue:`3969`, :issue:`4006`) + +* A new :setting:`ROBOTSTXT_USER_AGENT` setting allows defining a separate + user agent string to use for robots.txt_ parsing (:issue:`3931`, + :issue:`3966`) + +* :class:`~scrapy.spiders.Rule` no longer requires a :class:`LinkExtractor + ` parameter + (:issue:`781`, :issue:`4016`) + +* Use the new :setting:`DOWNLOADER_CLIENT_TLS_CIPHERS` setting to customize + the TLS/SSL ciphers used by the default HTTP/1.1 downloader (:issue:`3392`, + :issue:`3442`) + +* Set the new :setting:`DOWNLOADER_CLIENT_TLS_VERBOSE_LOGGING` setting to + ``True`` to enable debug-level messages about TLS connection parameters + after establishing HTTPS connections (:issue:`2111`, :issue:`3450`) + +* Callbacks that receive keyword arguments + (see :attr:`Request.cb_kwargs `) can now be + tested using the new :class:`@cb_kwargs + ` + :ref:`spider contract ` (:issue:`3985`, :issue:`3988`) + +* When a :class:`@scrapes ` spider + contract fails, all missing fields are now reported (:issue:`766`, + :issue:`3939`) + +* :ref:`Custom log formats ` can now drop messages by + having the corresponding methods of the configured :setting:`LOG_FORMATTER` + return ``None`` (:issue:`3984`, :issue:`3987`) + +* A much improved completion definition is now available for Zsh_ + (:issue:`4069`) + + +Bug fixes +~~~~~~~~~ + +* :meth:`ItemLoader.load_item() ` no + longer makes later calls to :meth:`ItemLoader.get_output_value() + ` or + :meth:`ItemLoader.load_item() ` return + empty data (:issue:`3804`, :issue:`3819`, :issue:`3897`, :issue:`3976`, + :issue:`3998`, :issue:`4036`) + +* Fixed :class:`~scrapy.statscollectors.DummyStatsCollector` raising a + :exc:`TypeError` exception (:issue:`4007`, :issue:`4052`) + +* :meth:`FilesPipeline.file_path + ` and + :meth:`ImagesPipeline.file_path + ` no longer choose + file extensions that are not `registered with IANA`_ (:issue:`1287`, + :issue:`3953`, :issue:`3954`) + +* When using botocore_ to persist files in S3, all botocore-supported headers + are properly mapped now (:issue:`3904`, :issue:`3905`) + +* FTP passwords in :setting:`FEED_URI` containing percent-escaped characters + are now properly decoded (:issue:`3941`) + +* A memory-handling and error-handling issue in + :func:`scrapy.utils.ssl.get_temp_key_info` has been fixed (:issue:`3920`) + + +Documentation +~~~~~~~~~~~~~ + +* The documentation now covers how to define and configure a :ref:`custom log + format ` (:issue:`3616`, :issue:`3660`) + +* API documentation added for :class:`~scrapy.exporters.MarshalItemExporter` + and :class:`~scrapy.exporters.PythonItemExporter` (:issue:`3973`) + +* API documentation added for :class:`~scrapy.item.BaseItem` and + :class:`~scrapy.item.ItemMeta` (:issue:`3999`) + +* Minor documentation fixes (:issue:`2998`, :issue:`3398`, :issue:`3597`, + :issue:`3894`, :issue:`3934`, :issue:`3978`, :issue:`3993`, :issue:`4022`, + :issue:`4028`, :issue:`4033`, :issue:`4046`, :issue:`4050`, :issue:`4055`, + :issue:`4056`, :issue:`4061`, :issue:`4072`, :issue:`4071`, :issue:`4079`, + :issue:`4081`, :issue:`4089`, :issue:`4093`) + + +.. _1.8-deprecation-removals: + +Deprecation removals +~~~~~~~~~~~~~~~~~~~~ + +* ``scrapy.xlib`` has been removed (:issue:`4015`) + + +Deprecations +~~~~~~~~~~~~ + +* The LevelDB_ storage backend + (``scrapy.extensions.httpcache.LeveldbCacheStorage``) of + :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware` is + deprecated (:issue:`4085`, :issue:`4092`) + +* Use of the undocumented ``SCRAPY_PICKLED_SETTINGS_TO_OVERRIDE`` environment + variable is deprecated (:issue:`3910`) + +* ``scrapy.item.DictItem`` is deprecated, use :class:`~scrapy.item.Item` + instead (:issue:`3999`) + + +Other changes +~~~~~~~~~~~~~ + +* Minimum versions of optional Scrapy requirements that are covered by + continuous integration tests have been updated: + + * botocore_ 1.3.23 + * Pillow_ 3.4.2 + + Lower versions of these optional requirements may work, but it is not + guaranteed (:issue:`3892`) + +* GitHub templates for bug reports and feature requests (:issue:`3126`, + :issue:`3471`, :issue:`3749`, :issue:`3754`) + +* Continuous integration fixes (:issue:`3923`) + +* Code cleanup (:issue:`3391`, :issue:`3907`, :issue:`3946`, :issue:`3950`, + :issue:`4023`, :issue:`4031`) + + +.. _release-1.7.4: + +Scrapy 1.7.4 (2019-10-21) +------------------------- + +Revert the fix for :issue:`3804` (:issue:`3819`), which has a few undesired +side effects (:issue:`3897`, :issue:`3976`). + +As a result, when an item loader is initialized with an item, +:meth:`ItemLoader.load_item() ` once again +makes later calls to :meth:`ItemLoader.get_output_value() +` or :meth:`ItemLoader.load_item() +` return empty data. + + +.. _release-1.7.3: + +Scrapy 1.7.3 (2019-08-01) +------------------------- + +Enforce lxml 4.3.5 or lower for Python 3.4 (:issue:`3912`, :issue:`3918`). + + +.. _release-1.7.2: + +Scrapy 1.7.2 (2019-07-23) +------------------------- + +Fix Python 2 support (:issue:`3889`, :issue:`3893`, :issue:`3896`). + + +.. _release-1.7.1: + Scrapy 1.7.1 (2019-07-18) ------------------------- Re-packaging of Scrapy 1.7.0, which was missing some changes in PyPI. + .. _release-1.7.0: Scrapy 1.7.0 (2019-07-18) @@ -60,7 +295,7 @@ New features ~~~~~~~~~~~~ * A new scheduler priority queue, - :class:`scrapy.pqueues.DownloaderAwarePriorityQueue`, may be + ``scrapy.pqueues.DownloaderAwarePriorityQueue``, may be :ref:`enabled ` for a significant scheduling improvement on crawls targetting multiple web domains, at the cost of no :setting:`CONCURRENT_REQUESTS_PER_IP` support (:issue:`3520`) @@ -69,16 +304,16 @@ New features provides a cleaner way to pass keyword arguments to callback methods (:issue:`1138`, :issue:`3563`) -* A new :class:`~scrapy.http.JSONRequest` class offers a more convenient way - to build JSON requests (:issue:`3504`, :issue:`3505`) +* A new :class:`JSONRequest ` class offers a more + convenient way to build JSON requests (:issue:`3504`, :issue:`3505`) * A ``process_request`` callback passed to the :class:`~scrapy.spiders.Rule` - constructor now receives the :class:`~scrapy.http.Response` object that + ``__init__`` method now receives the :class:`~scrapy.http.Response` object that originated the request as its second argument (:issue:`3682`) * A new ``restrict_text`` parameter for the :attr:`LinkExtractor ` - constructor allows filtering links by linking text (:issue:`3622`, + ``__init__`` method allows filtering links by linking text (:issue:`3622`, :issue:`3635`) * A new :setting:`FEED_STORAGE_S3_ACL` setting allows defining a custom ACL @@ -139,9 +374,9 @@ Bug fixes :setting:`AWS_REGION_NAME`, :setting:`AWS_USE_SSL`, :setting:`AWS_VERIFY` (:issue:`3625`) -* Fixed a memory leak in :class:`~scrapy.pipelines.media.MediaPipeline` - affecting, for example, non-200 responses and exceptions from custom - middlewares (:issue:`3813`) +* Fixed a memory leak in ``scrapy.pipelines.media.MediaPipeline`` affecting, + for example, non-200 responses and exceptions from custom middlewares + (:issue:`3813`) * Requests with private callbacks are now correctly unserialized from disk (:issue:`3790`) @@ -244,7 +479,7 @@ The following deprecated APIs have been removed (:issue:`3578`): * From :class:`~scrapy.selector.Selector`: - * ``_root`` (both the constructor argument and the object property, use + * ``_root`` (both the ``__init__`` method argument and the object property, use ``root``) * ``extract_unquoted`` (use ``getall``) @@ -290,7 +525,7 @@ Deprecations * The ``queuelib.PriorityQueue`` value for the :setting:`SCHEDULER_PRIORITY_QUEUE` setting is deprecated. Use - :class:`scrapy.pqueues.ScrapyPriorityQueue` instead. + ``scrapy.pqueues.ScrapyPriorityQueue`` instead. * ``process_request`` callbacks passed to :class:`~scrapy.spiders.Rule` that do not accept two arguments are deprecated. @@ -443,7 +678,7 @@ Usability improvements * a message is added to IgnoreRequest in RobotsTxtMiddleware (:issue:`3113`) * better validation of ``url`` argument in ``Response.follow`` (:issue:`3131`) * non-zero exit code is returned from Scrapy commands when error happens - on spider inititalization (:issue:`3226`) + on spider initialization (:issue:`3226`) * Link extraction improvements: "ftp" is added to scheme list (:issue:`3152`); "flv" is added to common video extensions (:issue:`3165`) * better error message when an exporter is disabled (:issue:`3358`); @@ -540,12 +775,12 @@ Scrapy 1.5.2 (2019-01-22) *The fix is backward incompatible*, it enables telnet user-password authentication by default with a random generated password. If you can't - upgrade right away, please consider setting :setting:`TELNET_CONSOLE_PORT` + upgrade right away, please consider setting :setting:`TELNETCONSOLE_PORT` out of its default value. See :ref:`telnet console ` documentation for more info -* Backport CI build failure under GCE environemnt due to boto import error. +* Backport CI build failure under GCE environment due to boto import error. .. _release-1.5.1: @@ -747,7 +982,9 @@ Enjoy! (Or read on for the rest of changes in this release.) Deprecations and Backward Incompatible Changes ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ -- Default to ``canonicalize=False`` in :class:`scrapy.linkextractors.LinkExtractor` +- Default to ``canonicalize=False`` in + :class:`scrapy.linkextractors.LinkExtractor + ` (:issue:`2537`, fixes :issue:`1941` and :issue:`1982`): **warning, this is technically backward-incompatible** - Enable memusage extension by default (:issue:`2539`, fixes :issue:`2187`); @@ -783,10 +1020,13 @@ New Features - New ``data:`` URI download handler (:issue:`2334`, fixes :issue:`2156`) - Log cache directory when HTTP Cache is used (:issue:`2611`, fixes :issue:`2604`) - Warn users when project contains duplicate spider names (fixes :issue:`2181`) -- :class:`CaselessDict` now accepts ``Mapping`` instances and not only dicts (:issue:`2646`) -- :ref:`Media downloads `, with :class:`FilesPipelines` - or :class:`ImagesPipelines`, can now optionally handle HTTP redirects - using the new :setting:`MEDIA_ALLOW_REDIRECTS` setting (:issue:`2616`, fixes :issue:`2004`) +- ``scrapy.utils.datatypes.CaselessDict`` now accepts ``Mapping`` instances and + not only dicts (:issue:`2646`) +- :ref:`Media downloads `, with + :class:`~scrapy.pipelines.files.FilesPipeline` or + :class:`~scrapy.pipelines.images.ImagesPipeline`, can now optionally handle + HTTP redirects using the new :setting:`MEDIA_ALLOW_REDIRECTS` setting + (:issue:`2616`, fixes :issue:`2004`) - Accept non-complete responses from websites using a new :setting:`DOWNLOAD_FAIL_ON_DATALOSS` setting (:issue:`2590`, fixes :issue:`2586`) - Optional pretty-printing of JSON and XML items via @@ -806,8 +1046,8 @@ Bug fixes - LinkExtractor now strips leading and trailing whitespaces from attributes (:issue:`2547`, fixes :issue:`1614`) -- Properly handle whitespaces in action attribute in :class:`FormRequest` - (:issue:`2548`) +- Properly handle whitespaces in action attribute in + :class:`~scrapy.http.FormRequest` (:issue:`2548`) - Buffer CONNECT response bytes from proxy until all HTTP headers are received (:issue:`2495`, fixes :issue:`2491`) - FTP downloader now works on Python 3, provided you use Twisted>=17.1 @@ -840,7 +1080,8 @@ Cleanups & Refactoring fixes :issue:`2560`) - Add omitted ``self`` arguments in default project middleware template (:issue:`2595`) - Remove redundant ``slot.add_request()`` call in ExecutionEngine (:issue:`2617`) -- Catch more specific ``os.error`` exception in :class:`FSFilesStore` (:issue:`2644`) +- Catch more specific ``os.error`` exception in + ``scrapy.pipelines.files.FSFilesStore`` (:issue:`2644`) - Change "localhost" test server certificate (:issue:`2720`) - Remove unused ``MEMUSAGE_REPORT`` setting (:issue:`2576`) @@ -857,7 +1098,8 @@ Documentation (:issue:`2477`, fixes :issue:`2475`) - FAQ: rewrite note on Python 3 support on Windows (:issue:`2690`) - Rearrange selector sections (:issue:`2705`) -- Remove ``__nonzero__`` from :class:`SelectorList` docs (:issue:`2683`) +- Remove ``__nonzero__`` from :class:`~scrapy.selector.SelectorList` + docs (:issue:`2683`) - Mention how to disable request filtering in documentation of :setting:`DUPEFILTER_CLASS` setting (:issue:`2714`) - Add sphinx_rtd_theme to docs setup readme (:issue:`2668`) @@ -914,7 +1156,7 @@ Bug fixes - Fix :command:`view` command ; it was a regression in v1.3.0 (:issue:`2503`). - Fix tests regarding ``*_EXPIRES settings`` with Files/Images pipelines (:issue:`2460`). - Fix name of generated pipeline class when using basic project template (:issue:`2466`). -- Fix compatiblity with Twisted 17+ (:issue:`2496`, :issue:`2528`). +- Fix compatibility with Twisted 17+ (:issue:`2496`, :issue:`2528`). - Fix ``scrapy.Item`` inheritance on Python 3.6 (:issue:`2511`). - Enforce numeric values for components order in ``SPIDER_MIDDLEWARES``, ``DOWNLOADER_MIDDLEWARES``, ``EXTENIONS`` and ``SPIDER_CONTRACTS`` (:issue:`2420`). @@ -922,7 +1164,7 @@ Bug fixes Documentation ~~~~~~~~~~~~~ -- Reword Code of Coduct section and upgrade to Contributor Covenant v1.4 +- Reword Code of Conduct section and upgrade to Contributor Covenant v1.4 (:issue:`2469`). - Clarify that passing spider arguments converts them to spider attributes (:issue:`2483`). @@ -936,7 +1178,7 @@ Documentation Cleanups ~~~~~~~~ -- Remove reduntant check in ``MetaRefreshMiddleware`` (:issue:`2542`). +- Remove redundant check in ``MetaRefreshMiddleware`` (:issue:`2542`). - Faster checks in ``LinkExtractor`` for allow/deny patterns (:issue:`2538`). - Remove dead code supporting old Twisted versions (:issue:`2544`). @@ -962,7 +1204,7 @@ New Features - ``MailSender`` now accepts single strings as values for ``to`` and ``cc`` arguments (:issue:`2272`) - ``scrapy fetch url``, ``scrapy shell url`` and ``fetch(url)`` inside - scrapy shell now follow HTTP redirections by default (:issue:`2290`); + Scrapy shell now follow HTTP redirections by default (:issue:`2290`); See :command:`fetch` and :command:`shell` for details. - ``HttpErrorMiddleware`` now logs errors with ``INFO`` level instead of ``DEBUG``; this is technically **backward incompatible** so please check your log parsers. @@ -1118,7 +1360,7 @@ Documentation - Grammar fixes: :issue:`2128`, :issue:`1566`. - Download stats badge removed from README (:issue:`2160`). -- New scrapy :ref:`architecture diagram ` (:issue:`2165`). +- New Scrapy :ref:`architecture diagram ` (:issue:`2165`). - Updated ``Response`` parameters documentation (:issue:`2197`). - Reworded misleading :setting:`RANDOMIZE_DOWNLOAD_DELAY` description (:issue:`2190`). - Add StackOverflow as a support channel (:issue:`2257`). @@ -1208,7 +1450,7 @@ Documentation - Use "url" variable in downloader middleware example (:issue:`2015`) - Grammar fixes (:issue:`2054`, :issue:`2120`) - New FAQ entry on using BeautifulSoup in spider callbacks (:issue:`2048`) -- Add notes about scrapy not working on Windows with Python 3 (:issue:`2060`) +- Add notes about Scrapy not working on Windows with Python 3 (:issue:`2060`) - Encourage complete titles in pull requests (:issue:`2026`) Tests @@ -1258,8 +1500,8 @@ This 1.1 release brings a lot of interesting features and bug fixes: this behavior, update :setting:`ROBOTSTXT_OBEY` in ``settings.py`` file after creating a new project. - Exporters now work on unicode, instead of bytes by default (:issue:`1080`). - If you use ``PythonItemExporter``, you may want to update your code to - disable binary mode which is now deprecated. + If you use :class:`~scrapy.exporters.PythonItemExporter`, you may want to + update your code to disable binary mode which is now deprecated. - Accept XML node names containing dots as valid (:issue:`1533`). - When uploading files or images to S3 (with ``FilesPipeline`` or ``ImagesPipeline``), the default ACL policy is now "private" instead @@ -1267,7 +1509,7 @@ This 1.1 release brings a lot of interesting features and bug fixes: You can use :setting:`FILES_STORE_S3_ACL` to change it. - We've reimplemented ``canonicalize_url()`` for more correct output, especially for URLs with non-ASCII characters (:issue:`1947`). - This could change link extractors output compared to previous scrapy versions. + This could change link extractors output compared to previous Scrapy versions. This may also invalidate some cache entries you could still have from pre-1.1 runs. **Warning: backward incompatible!**. @@ -1397,8 +1639,8 @@ Bugfixes - Fixed bug on ``XMLItemExporter`` with non-string fields in items (:issue:`1738`). - Fixed startproject command in OS X (:issue:`1635`). -- Fixed PythonItemExporter and CSVExporter for non-string item - types (:issue:`1737`). +- Fixed :class:`~scrapy.exporters.PythonItemExporter` and CSVExporter for + non-string item types (:issue:`1737`). - Various logging related fixes (:issue:`1294`, :issue:`1419`, :issue:`1263`, :issue:`1624`, :issue:`1654`, :issue:`1722`, :issue:`1726` and :issue:`1303`). - Fixed bug in ``utils.template.render_templatefile()`` (:issue:`1212`). @@ -1463,7 +1705,7 @@ Scrapy 1.0.4 (2015-12-30) - fix ValueError: Invalid XPath: //div/[id="not-exists"]/text() on selectors.rst (:commit:`ca8d60f`) - Typos corrections (:commit:`7067117`) - fix typos in downloader-middleware.rst and exceptions.rst, middlware -> middleware (:commit:`32f115c`) -- Add note to ubuntu install section about debian compatibility (:commit:`23fda69`) +- Add note to Ubuntu install section about Debian compatibility (:commit:`23fda69`) - Replace alternative OSX install workaround with virtualenv (:commit:`98b63ee`) - Reference Homebrew's homepage for installation instructions (:commit:`1925db1`) - Add oldest supported tox version to contributing docs (:commit:`5d10d6d`) @@ -1480,7 +1722,7 @@ Scrapy 1.0.4 (2015-12-30) - Merge pull request #1513 from mgedmin/patch-2 (:commit:`5d4daf8`) - Typo (:commit:`f8d0682`) - Fix list formatting (:commit:`5f83a93`) -- fix scrapy squeue tests after recent changes to queuelib (:commit:`3365c01`) +- fix Scrapy squeue tests after recent changes to queuelib (:commit:`3365c01`) - Merge pull request #1475 from rweindl/patch-1 (:commit:`2d688cd`) - Update tutorial.rst (:commit:`fbc1f25`) - Merge pull request #1449 from rhoekman/patch-1 (:commit:`7d6538c`) @@ -1492,7 +1734,7 @@ Scrapy 1.0.4 (2015-12-30) Scrapy 1.0.3 (2015-08-11) ------------------------- -- add service_identity to scrapy install_requires (:commit:`cbc2501`) +- add service_identity to Scrapy install_requires (:commit:`cbc2501`) - Workaround for travis#296 (:commit:`66af9cd`) .. _release-1.0.2: @@ -1516,7 +1758,7 @@ Scrapy 1.0.1 (2015-07-01) - include tests/ to source distribution in MANIFEST.in (:commit:`eca227e`) - DOC Fix SelectJmes documentation (:commit:`b8567bc`) - DOC Bring Ubuntu and Archlinux outside of Windows subsection (:commit:`392233f`) -- DOC remove version suffix from ubuntu package (:commit:`5303c66`) +- DOC remove version suffix from Ubuntu package (:commit:`5303c66`) - DOC Update release date for 1.0 (:commit:`c89fa29`) .. _release-1.0.0: @@ -1969,7 +2211,7 @@ Scrapy 0.24.2 (2014-07-08) - Use a mutable mapping to proxy deprecated settings.overrides and settings.defaults attribute (:commit:`e5e8133`) - there is not support for python3 yet (:commit:`3cd6146`) -- Update python compatible version set to debian packages (:commit:`fa5d76b`) +- Update python compatible version set to Debian packages (:commit:`fa5d76b`) - DOC fix formatting in release notes (:commit:`c6a9e20`) Scrapy 0.24.1 (2014-06-27) @@ -1987,12 +2229,12 @@ Enhancements - Improve Scrapy top-level namespace (:issue:`494`, :issue:`684`) - Add selector shortcuts to responses (:issue:`554`, :issue:`690`) -- Add new lxml based LinkExtractor to replace unmantained SgmlLinkExtractor +- Add new lxml based LinkExtractor to replace unmaintained SgmlLinkExtractor (:issue:`559`, :issue:`761`, :issue:`763`) - Cleanup settings API - part of per-spider settings **GSoC project** (:issue:`737`) - Add UTF8 encoding header to templates (:issue:`688`, :issue:`762`) - Telnet console now binds to 127.0.0.1 by default (:issue:`699`) -- Update debian/ubuntu install instructions (:issue:`509`, :issue:`549`) +- Update Debian/Ubuntu install instructions (:issue:`509`, :issue:`549`) - Disable smart strings in lxml XPath evaluations (:issue:`535`) - Restore filesystem based cache as default for http cache middleware (:issue:`541`, :issue:`500`, :issue:`571`) @@ -2025,7 +2267,7 @@ Enhancements - Tests and docs for ``request_fingerprint`` function (:issue:`597`) - Update SEP-19 for GSoC project ``per-spider settings`` (:issue:`705`) - Set exit code to non-zero when contracts fails (:issue:`727`) -- Add a setting to control what class is instanciated as Downloader component +- Add a setting to control what class is instantiated as Downloader component (:issue:`738`) - Pass response in ``item_dropped`` signal (:issue:`724`) - Improve ``scrapy check`` contracts command (:issue:`733`, :issue:`752`) @@ -2034,7 +2276,7 @@ Enhancements - Add a note about reporting security issues (:issue:`697`) - Add LevelDB http cache storage backend (:issue:`626`, :issue:`500`) - Sort spider list output of ``scrapy list`` command (:issue:`742`) -- Multiple documentation enhancemens and fixes +- Multiple documentation enhancements and fixes (:issue:`575`, :issue:`587`, :issue:`590`, :issue:`596`, :issue:`610`, :issue:`617`, :issue:`618`, :issue:`627`, :issue:`613`, :issue:`643`, :issue:`654`, :issue:`675`, :issue:`663`, :issue:`711`, :issue:`714`) @@ -2079,19 +2321,19 @@ Scrapy 0.22.1 (released 2014-02-08) - BaseSgmlLinkExtractor: Added unit test of a link with an inner tag (:commit:`c1cb418`) - BaseSgmlLinkExtractor: Fixed unknown_endtag() so that it only set current_link=None when the end tag match the opening tag (:commit:`7e4d627`) - Fix tests for Travis-CI build (:commit:`76c7e20`) -- replace unencodeable codepoints with html entities. fixes #562 and #285 (:commit:`5f87b17`) +- replace unencodable codepoints with html entities. fixes #562 and #285 (:commit:`5f87b17`) - RegexLinkExtractor: encode URL unicode value when creating Links (:commit:`d0ee545`) - Updated the tutorial crawl output with latest output. (:commit:`8da65de`) - Updated shell docs with the crawler reference and fixed the actual shell output. (:commit:`875b9ab`) - PEP8 minor edits. (:commit:`f89efaf`) -- Expose current crawler in the scrapy shell. (:commit:`5349cec`) +- Expose current crawler in the Scrapy shell. (:commit:`5349cec`) - Unused re import and PEP8 minor edits. (:commit:`387f414`) - Ignore None's values when using the ItemLoader. (:commit:`0632546`) - DOC Fixed HTTPCACHE_STORAGE typo in the default value which is now Filesystem instead Dbm. (:commit:`cde9a8c`) -- show ubuntu setup instructions as literal code (:commit:`fb5c9c5`) +- show Ubuntu setup instructions as literal code (:commit:`fb5c9c5`) - Update Ubuntu installation instructions (:commit:`70fb105`) - Merge pull request #550 from stray-leone/patch-1 (:commit:`6f70b6a`) -- modify the version of scrapy ubuntu package (:commit:`725900d`) +- modify the version of Scrapy Ubuntu package (:commit:`725900d`) - fix 0.22.0 release date (:commit:`af0219a`) - fix typos in news.rst and remove (not released yet) header (:commit:`b7f58f4`) @@ -2112,7 +2354,7 @@ Enhancements - Improve test coverage and forthcoming Python 3 support (:issue:`525`) - Promote startup info on settings and middleware to INFO level (:issue:`520`) - Support partials in ``get_func_args`` util (:issue:`506`, issue:`504`) -- Allow running indiviual tests via tox (:issue:`503`) +- Allow running individual tests via tox (:issue:`503`) - Update extensions ignored by link extractors (:issue:`498`) - Add middleware methods to get files/images/thumbs paths (:issue:`490`) - Improve offsite middleware tests (:issue:`478`) @@ -2169,7 +2411,7 @@ Enhancements - scrapy.mail.MailSender now can connect over TLS or upgrade using STARTTLS (:issue:`327`) - New FilesPipeline with functionality factored out from ImagesPipeline (:issue:`370`, :issue:`409`) - Recommend Pillow instead of PIL for image handling (:issue:`317`) -- Added debian packages for Ubuntu quantal and raring (:commit:`86230c0`) +- Added Debian packages for Ubuntu Quantal and Raring (:commit:`86230c0`) - Mock server (used for tests) can listen for HTTPS requests (:issue:`410`) - Remove multi spider support from multiple core components (:issue:`422`, :issue:`421`, :issue:`420`, :issue:`419`, :issue:`423`, :issue:`418`) @@ -2188,7 +2430,7 @@ Bugfixes - Fix tests under Django 1.6 (:commit:`b6bed44c`) - Lot of bugfixes to retry middleware under disconnections using HTTP 1.1 download handler - Fix inconsistencies among Twisted releases (:issue:`406`) -- Fix scrapy shell bugs (:issue:`418`, :issue:`407`) +- Fix Scrapy shell bugs (:issue:`418`, :issue:`407`) - Fix invalid variable name in setup.py (:issue:`429`) - Fix tutorial references (:issue:`387`) - Improve request-response docs (:issue:`391`) @@ -2270,15 +2512,15 @@ Scrapy 0.18.1 (released 2013-08-27) - test PotentiaDataLoss errors on unbound responses (:commit:`b15470d`) - Treat responses without content-length or Transfer-Encoding as good responses (:commit:`c4bf324`) - do no include ResponseFailed if http11 handler is not enabled (:commit:`6cbe684`) -- New HTTP client wraps connection losts in ResponseFailed exception. fix #373 (:commit:`1a20bba`) +- New HTTP client wraps connection lost in ResponseFailed exception. fix #373 (:commit:`1a20bba`) - limit travis-ci build matrix (:commit:`3b01bb8`) - Merge pull request #375 from peterarenot/patch-1 (:commit:`fa766d7`) - Fixed so it refers to the correct folder (:commit:`3283809`) -- added quantal & raring to support ubuntu releases (:commit:`1411923`) +- added Quantal & Raring to support Ubuntu releases (:commit:`1411923`) - fix retry middleware which didn't retry certain connection errors after the upgrade to http1 client, closes GH-373 (:commit:`bb35ed0`) - fix XmlItemExporter in Python 2.7.4 and 2.7.5 (:commit:`de3e451`) - minor updates to 0.18 release notes (:commit:`c45e5f1`) -- fix contributters list format (:commit:`0b60031`) +- fix contributors list format (:commit:`0b60031`) Scrapy 0.18.0 (released 2013-08-09) ----------------------------------- @@ -2313,10 +2555,10 @@ Scrapy 0.18.0 (released 2013-08-09) - Collect idle downloader slots (:issue:`297`) - Add ``ftp://`` scheme downloader handler (:issue:`329`) - Added downloader benchmark webserver and spider tools :ref:`benchmarking` -- Moved persistent (on disk) queues to a separate project (queuelib_) which scrapy now depends on -- Add scrapy commands using external libraries (:issue:`260`) +- Moved persistent (on disk) queues to a separate project (queuelib_) which Scrapy now depends on +- Add Scrapy commands using external libraries (:issue:`260`) - Added ``--pdb`` option to ``scrapy`` command line tool -- Added :meth:`XPathSelector.remove_namespaces` which allows to remove all namespaces from XML documents for convenience (to work with namespace-less XPaths). Documented in :ref:`topics-selectors`. +- Added :meth:`XPathSelector.remove_namespaces ` which allows to remove all namespaces from XML documents for convenience (to work with namespace-less XPaths). Documented in :ref:`topics-selectors`. - Several improvements to spider contracts - New default middleware named MetaRefreshMiddldeware that handles meta-refresh html tag redirections, - MetaRefreshMiddldeware and RedirectMiddleware have different priorities to address #62 @@ -2326,7 +2568,7 @@ Scrapy 0.18.0 (released 2013-08-09) - several more cleanups to singletons and multi-spider support (thanks Nicolas Ramirez) - support custom download slots - added --spider option to "shell" command. -- log overridden settings when scrapy starts +- log overridden settings when Scrapy starts Thanks to everyone who contribute to this release. Here is a list of contributors sorted by number of commits:: @@ -2375,7 +2617,7 @@ contributors sorted by number of commits:: Scrapy 0.16.5 (released 2013-05-30) ----------------------------------- -- obey request method when scrapy deploy is redirected to a new endpoint (:commit:`8c4fcee`) +- obey request method when Scrapy deploy is redirected to a new endpoint (:commit:`8c4fcee`) - fix inaccurate downloader middleware documentation. refs #280 (:commit:`40667cb`) - doc: remove links to diveintopython.org, which is no longer available. closes #246 (:commit:`bd58bfa`) - Find form nodes in invalid html5 documents (:commit:`e3d6945`) @@ -2389,8 +2631,8 @@ Scrapy 0.16.4 (released 2013-01-23) - Fixed error message formatting. log.err() doesn't support cool formatting and when error occurred, the message was: "ERROR: Error processing %(item)s" (:commit:`c16150c`) - lint and improve images pipeline error logging (:commit:`56b45fc`) - fixed doc typos (:commit:`243be84`) -- add documentation topics: Broad Crawls & Common Practies (:commit:`1fbb715`) -- fix bug in scrapy parse command when spider is not specified explicitly. closes #209 (:commit:`c72e682`) +- add documentation topics: Broad Crawls & Common Practices (:commit:`1fbb715`) +- fix bug in Scrapy parse command when spider is not specified explicitly. closes #209 (:commit:`c72e682`) - Update docs/topics/commands.rst (:commit:`28eac7a`) Scrapy 0.16.3 (released 2012-12-07) @@ -2409,11 +2651,11 @@ Scrapy 0.16.3 (released 2012-12-07) Scrapy 0.16.2 (released 2012-11-09) ----------------------------------- -- scrapy contracts: python2.6 compat (:commit:`a4a9199`) -- scrapy contracts verbose option (:commit:`ec41673`) -- proper unittest-like output for scrapy contracts (:commit:`86635e4`) +- Scrapy contracts: python2.6 compat (:commit:`a4a9199`) +- Scrapy contracts verbose option (:commit:`ec41673`) +- proper unittest-like output for Scrapy contracts (:commit:`86635e4`) - added open_in_browser to debugging doc (:commit:`c9b690d`) -- removed reference to global scrapy stats from settings doc (:commit:`dd55067`) +- removed reference to global Scrapy stats from settings doc (:commit:`dd55067`) - Fix SpiderState bug in Windows platforms (:commit:`58998f4`) @@ -2423,7 +2665,7 @@ Scrapy 0.16.1 (released 2012-10-26) - fixed LogStats extension, which got broken after a wrong merge before the 0.16 release (:commit:`8c780fd`) - better backward compatibility for scrapy.conf.settings (:commit:`3403089`) - extended documentation on how to access crawler stats from extensions (:commit:`c4da0b5`) -- removed .hgtags (no longer needed now that scrapy uses git) (:commit:`d52c188`) +- removed .hgtags (no longer needed now that Scrapy uses git) (:commit:`d52c188`) - fix dashes under rst headers (:commit:`fa4f7f9`) - set release date for 0.16.0 in news (:commit:`e292246`) @@ -2437,9 +2679,8 @@ Scrapy changes: - added options ``-o`` and ``-t`` to the :command:`runspider` command - documented :doc:`topics/autothrottle` and added to extensions installed by default. You still need to enable it with :setting:`AUTOTHROTTLE_ENABLED` - major Stats Collection refactoring: removed separation of global/per-spider stats, removed stats-related signals (``stats_spider_opened``, etc). Stats are much simpler now, backward compatibility is kept on the Stats Collector API and signals. -- added :meth:`~scrapy.contrib.spidermiddleware.SpiderMiddleware.process_start_requests` method to spider middlewares -- dropped Signals singleton. Signals should now be accesed through the Crawler.signals attribute. See the signals documentation for more info. -- dropped Signals singleton. Signals should now be accesed through the Crawler.signals attribute. See the signals documentation for more info. +- added :meth:`~scrapy.spidermiddlewares.SpiderMiddleware.process_start_requests` method to spider middlewares +- dropped Signals singleton. Signals should now be accessed through the Crawler.signals attribute. See the signals documentation for more info. - dropped Stats Collector singleton. Stats can now be accessed through the Crawler.stats attribute. See the stats collection documentation for more info. - documented :ref:`topics-api` - ``lxml`` is now the default selectors backend instead of ``libxml2`` @@ -2461,7 +2702,7 @@ Scrapy changes: - removed ``ENCODING_ALIASES`` setting, as encoding auto-detection has been moved to the `w3lib`_ library - promoted :ref:`topics-djangoitem` to main contrib - LogFormatter method now return dicts(instead of strings) to support lazy formatting (:issue:`164`, :commit:`dcef7b0`) -- downloader handlers (:setting:`DOWNLOAD_HANDLERS` setting) now receive settings as the first argument of the constructor +- downloader handlers (:setting:`DOWNLOAD_HANDLERS` setting) now receive settings as the first argument of the ``__init__`` method - replaced memory usage acounting with (more portable) `resource`_ module, removed ``scrapy.utils.memory`` module - removed signal: ``scrapy.mail.mail_sent`` - removed ``TRACK_REFS`` setting, now :ref:`trackrefs ` is always enabled @@ -2473,7 +2714,7 @@ Scrapy changes: Scrapy 0.14.4 ------------- -- added precise to supported ubuntu distros (:commit:`b7e46df`) +- added precise to supported Ubuntu distros (:commit:`b7e46df`) - fixed bug in json-rpc webservice reported in https://groups.google.com/forum/#!topic/scrapy-users/qgVBmFybNAQ/discussion. also removed no longer supported 'run' command from extras/scrapy-ws.py (:commit:`340fbdb`) - meta tag attributes for content-type http equiv can be in any order. #123 (:commit:`0cb68af`) - replace "import Image" by more standard "from PIL import Image". closes #88 (:commit:`4d17048`) @@ -2486,11 +2727,11 @@ Scrapy 0.14.3 - include egg files used by testsuite in source distribution. #118 (:commit:`c897793`) - update docstring in project template to avoid confusion with genspider command, which may be considered as an advanced feature. refs #107 (:commit:`2548dcc`) - added note to docs/topics/firebug.rst about google directory being shut down (:commit:`668e352`) -- dont discard slot when empty, just save in another dict in order to recycle if needed again. (:commit:`8e9f607`) +- don't discard slot when empty, just save in another dict in order to recycle if needed again. (:commit:`8e9f607`) - do not fail handling unicode xpaths in libxml2 backed selectors (:commit:`b830e95`) - fixed minor mistake in Request objects documentation (:commit:`bf3c9ee`) - fixed minor defect in link extractors documentation (:commit:`ba14f38`) -- removed some obsolete remaining code related to sqlite support in scrapy (:commit:`0665175`) +- removed some obsolete remaining code related to sqlite support in Scrapy (:commit:`0665175`) Scrapy 0.14.2 ------------- @@ -2579,7 +2820,7 @@ Code rearranged and removed - `w3lib`_ (several functions from ``scrapy.utils.{http,markup,multipart,response,url}``, done in :rev:`2584`) - `scrapely`_ (was ``scrapy.contrib.ibl``, done in :rev:`2586`) - Removed unused function: ``scrapy.utils.request.request_info()`` (:rev:`2577`) -- Removed googledir project from ``examples/googledir``. There's now a new example project called ``dirbot`` available on github: https://github.com/scrapy/dirbot +- Removed googledir project from ``examples/googledir``. There's now a new example project called ``dirbot`` available on GitHub: https://github.com/scrapy/dirbot - Removed support for default field values in Scrapy items (:rev:`2616`) - Removed experimental crawlspider v2 (:rev:`2632`) - Removed scheduler middleware to simplify architecture. Duplicates filter is now done in the scheduler itself, using the same dupe fltering class as before (``DUPEFILTER_CLASS`` setting) (:rev:`2640`) @@ -2598,7 +2839,8 @@ The numbers like #NNN reference tickets in the old issue tracker (Trac) which is New features and improvements ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ -- Passed item is now sent in the ``item`` argument of the :signal:`item_passed` (#273) +- Passed item is now sent in the ``item`` argument of the :signal:`item_passed + ` (#273) - Added verbose option to ``scrapy version`` command, useful for bug reports (#298) - HTTP cache now stored by default in the project data dir (#279) - Added project data storage directory (#276, #277) @@ -2674,7 +2916,7 @@ API changes - ``Request.copy()`` and ``Request.replace()`` now also copies their ``callback`` and ``errback`` attributes (#231) - Removed ``UrlFilterMiddleware`` from ``scrapy.contrib`` (already disabled by default) - Offsite middelware doesn't filter out any request coming from a spider that doesn't have a allowed_domains attribute (#225) -- Removed Spider Manager ``load()`` method. Now spiders are loaded in the constructor itself. +- Removed Spider Manager ``load()`` method. Now spiders are loaded in the ``__init__`` method itself. - Changes to Scrapy Manager (now called "Crawler"): - ``scrapy.core.manager.ScrapyManager`` class renamed to ``scrapy.crawler.Crawler`` - ``scrapy.core.manager.scrapymanager`` singleton moved to ``scrapy.project.crawler`` @@ -2799,23 +3041,35 @@ First release of Scrapy. .. _AJAX crawleable urls: https://developers.google.com/webmasters/ajax-crawling/docs/getting-started?csw=1 +.. _botocore: https://github.com/boto/botocore .. _chunked transfer encoding: https://en.wikipedia.org/wiki/Chunked_transfer_encoding .. _ClientForm: http://wwwsearch.sourceforge.net/old/ClientForm/ .. _Creating a pull request: https://help.github.com/en/articles/creating-a-pull-request +.. _cryptography: https://cryptography.io/en/latest/ .. _cssselect: https://github.com/scrapy/cssselect/ .. _docstrings: https://docs.python.org/glossary.html#term-docstring .. _KeyboardInterrupt: https://docs.python.org/library/exceptions.html#KeyboardInterrupt +.. _LevelDB: https://github.com/google/leveldb .. _lxml: http://lxml.de/ .. _marshal: https://docs.python.org/2/library/marshal.html .. _parsel.csstranslator.GenericTranslator: https://parsel.readthedocs.io/en/latest/parsel.html#parsel.csstranslator.GenericTranslator .. _parsel.csstranslator.HTMLTranslator: https://parsel.readthedocs.io/en/latest/parsel.html#parsel.csstranslator.HTMLTranslator .. _parsel.csstranslator.XPathExpr: https://parsel.readthedocs.io/en/latest/parsel.html#parsel.csstranslator.XPathExpr .. _PEP 257: https://www.python.org/dev/peps/pep-0257/ +.. _Pillow: https://python-pillow.org/ +.. _pyOpenSSL: https://www.pyopenssl.org/en/stable/ .. _queuelib: https://github.com/scrapy/queuelib +.. _registered with IANA: https://www.iana.org/assignments/media-types/media-types.xhtml .. _resource: https://docs.python.org/2/library/resource.html +.. _robots.txt: http://www.robotstxt.org/ .. _scrapely: https://github.com/scrapy/scrapely +.. _service_identity: https://service-identity.readthedocs.io/en/stable/ +.. _six: https://six.readthedocs.io/ .. _tox: https://pypi.python.org/pypi/tox +.. _Twisted: https://twistedmatrix.com/trac/ .. _Twisted - hello, asynchronous programming: http://jessenoller.com/blog/2009/02/11/twisted-hello-asynchronous-programming/ .. _w3lib: https://github.com/scrapy/w3lib .. _w3lib.encoding: https://github.com/scrapy/w3lib/blob/master/w3lib/encoding.py .. _What is cacheable: https://www.w3.org/Protocols/rfc2616/rfc2616-sec14.html#sec14.9.1 +.. _zope.interface: https://zopeinterface.readthedocs.io/en/latest/ +.. _Zsh: https://www.zsh.org/ diff --git a/docs/requirements.txt b/docs/requirements.txt index 379da9994..773b92cea 100644 --- a/docs/requirements.txt +++ b/docs/requirements.txt @@ -1,2 +1,4 @@ Sphinx>=2.1 -sphinx_rtd_theme \ No newline at end of file +sphinx-hoverxref +sphinx-notfound-page +sphinx_rtd_theme diff --git a/docs/topics/api.rst b/docs/topics/api.rst index 7c8c40b5f..1c461a511 100644 --- a/docs/topics/api.rst +++ b/docs/topics/api.rst @@ -273,5 +273,3 @@ class (which they all inherit from). Close the given spider. After this is called, no more specific stats can be accessed or collected. - -.. _reactor: https://twistedmatrix.com/documents/current/core/howto/reactor-basics.html diff --git a/docs/topics/architecture.rst b/docs/topics/architecture.rst index 2effe94dc..ae25dfa2f 100644 --- a/docs/topics/architecture.rst +++ b/docs/topics/architecture.rst @@ -166,11 +166,10 @@ for concurrency. For more information about asynchronous programming and Twisted see these links: -* `Introduction to Deferreds in Twisted`_ +* :doc:`twisted:core/howto/defer-intro` * `Twisted - hello, asynchronous programming`_ * `Twisted Introduction - Krondo`_ .. _Twisted: https://twistedmatrix.com/trac/ -.. _Introduction to Deferreds in Twisted: https://twistedmatrix.com/documents/current/core/howto/defer-intro.html .. _Twisted - hello, asynchronous programming: http://jessenoller.com/blog/2009/02/11/twisted-hello-asynchronous-programming/ .. _Twisted Introduction - Krondo: http://krondo.com/an-introduction-to-asynchronous-programming-and-twisted/ diff --git a/docs/topics/autothrottle.rst b/docs/topics/autothrottle.rst index c9bece753..4317019fc 100644 --- a/docs/topics/autothrottle.rst +++ b/docs/topics/autothrottle.rst @@ -11,7 +11,7 @@ Design goals ============ 1. be nicer to sites instead of using default download delay of zero -2. automatically adjust scrapy to the optimum crawling speed, so the user +2. automatically adjust Scrapy to the optimum crawling speed, so the user doesn't have to tune the download delays to find the optimum one. The user only needs to specify the maximum concurrent requests it allows, and the extension does the rest. diff --git a/docs/topics/commands.rst b/docs/topics/commands.rst index a93bee06b..a0dcba90d 100644 --- a/docs/topics/commands.rst +++ b/docs/topics/commands.rst @@ -1,3 +1,5 @@ +.. highlight:: none + .. _topics-commands: ================= @@ -27,7 +29,7 @@ in standard locations: 1. ``/etc/scrapy.cfg`` or ``c:\scrapy\scrapy.cfg`` (system-wide), 2. ``~/.config/scrapy.cfg`` (``$XDG_CONFIG_HOME``) and ``~/.scrapy.cfg`` (``$HOME``) for global (user-wide) settings, and -3. ``scrapy.cfg`` inside a scrapy project's root (see next section). +3. ``scrapy.cfg`` inside a Scrapy project's root (see next section). Settings from these files are merged in the listed order of preference: user-defined values have higher priority than system-wide defaults @@ -66,7 +68,9 @@ structure by default, similar to this:: The directory where the ``scrapy.cfg`` file resides is known as the *project root directory*. That file contains the name of the python module that defines -the project settings. Here is an example:: +the project settings. Here is an example: + +.. code-block:: ini [settings] default = myproject.settings @@ -80,7 +84,9 @@ A project root directory, the one that contains the ``scrapy.cfg``, may be shared by multiple Scrapy projects, each with its own settings module. In that case, you must define one or more aliases for those settings modules -under ``[settings]`` in your ``scrapy.cfg`` file:: +under ``[settings]`` in your ``scrapy.cfg`` file: + +.. code-block:: ini [settings] default = myproject1.settings @@ -277,6 +283,8 @@ check Run contract checks. +.. skip: start + Usage examples:: $ scrapy check -l @@ -294,6 +302,8 @@ Usage examples:: [FAILED] first_spider:parse >>> Returned 92 requests, expected 0..4 +.. skip: end + .. command:: list list @@ -481,6 +491,8 @@ Supported options: * ``--verbose`` or ``-v``: display information for each depth level +.. skip: start + Usage example:: $ scrapy parse http://www.example.com/ -c parse_item @@ -495,6 +507,8 @@ Usage example:: # Requests ----------------------------------------------------------------- [] +.. skip: end + .. command:: settings @@ -573,7 +587,9 @@ Default: ``''`` (empty string) A module to use for looking up custom Scrapy commands. This is used to add custom commands for your Scrapy project. -Example:: +Example: + +.. code-block:: python COMMANDS_MODULE = 'mybot.commands' @@ -588,7 +604,11 @@ You can also add Scrapy commands from an external library by adding a ``scrapy.commands`` section in the entry points of the library ``setup.py`` file. -The following example adds ``my_command`` command:: +The following example adds ``my_command`` command: + +.. skip: next + +.. code-block:: python from setuptools import setup, find_packages diff --git a/docs/topics/contracts.rst b/docs/topics/contracts.rst index 957761b76..43db8f101 100644 --- a/docs/topics/contracts.rst +++ b/docs/topics/contracts.rst @@ -6,10 +6,6 @@ Spiders Contracts .. versionadded:: 0.15 -.. note:: This is a new feature (introduced in Scrapy 0.15) and may be subject - to minor functionality/API updates. Check the :ref:`release notes ` to - be notified of updates. - Testing spiders can get particularly annoying and while nothing prevents you from writing unit tests the task gets cumbersome quickly. Scrapy offers an integrated way of testing your spiders by the means of contracts. @@ -35,12 +31,20 @@ This callback is tested using three built-in contracts: .. class:: UrlContract - This contract (``@url``) sets the sample url used when checking other + This contract (``@url``) sets the sample URL used when checking other contract conditions for this spider. This contract is mandatory. All callbacks lacking this contract are ignored when running the checks:: @url url +.. class:: CallbackKeywordArgumentsContract + + This contract (``@cb_kwargs``) sets the :attr:`cb_kwargs ` + attribute for the sample request. It must be a valid JSON dictionary. + :: + + @cb_kwargs {"arg1": "value1", "arg2": "value2", ...} + .. class:: ReturnsContract This contract (``@returns``) sets lower and upper bounds for the items and @@ -60,7 +64,7 @@ Use the :command:`check` command to run the contract checks. Custom Contracts ================ -If you find you need more power than the built-in scrapy contracts you can +If you find you need more power than the built-in Scrapy contracts you can create and load your own contracts in the project by using the :setting:`SPIDER_CONTRACTS` setting:: diff --git a/docs/topics/debug.rst b/docs/topics/debug.rst index 0aaad0c77..d75f17301 100644 --- a/docs/topics/debug.rst +++ b/docs/topics/debug.rst @@ -5,7 +5,7 @@ Debugging Spiders ================= This document explains the most common techniques for debugging spiders. -Consider the following scrapy spider below:: +Consider the following Scrapy spider below:: import scrapy from myproject.items import MyItem @@ -48,6 +48,10 @@ The most basic way of checking the output of your spider is to use the of the spider at the method level. It has the advantage of being flexible and simple to use, but does not allow debugging code inside a method. +.. highlight:: none + +.. skip: start + In order to see the item scraped from a specific url:: $ scrapy parse --spider=myspider -c parse_item -d 2 @@ -85,6 +89,8 @@ using:: $ scrapy parse --spider=myspider -d 3 'http://example.com/page1' +.. skip: end + Scrapy Shell ============ @@ -94,6 +100,8 @@ spider, it is of little help to check what happens inside a callback, besides showing the response received and the output. How to debug the situation when ``parse_details`` sometimes receives no item? +.. highlight:: python + Fortunately, the :command:`shell` is your bread and butter in this case (see :ref:`topics-shell-inspect-response`):: diff --git a/docs/topics/developer-tools.rst b/docs/topics/developer-tools.rst index 82857c9da..b7d740c19 100644 --- a/docs/topics/developer-tools.rst +++ b/docs/topics/developer-tools.rst @@ -39,7 +39,7 @@ Therefore, you should keep in mind the following things: .. _topics-inspector: Inspecting a website -=================================== +==================== By far the most handy feature of the Developer Tools is the `Inspector` feature, which allows you to inspect the underlying HTML code of @@ -79,13 +79,23 @@ sections and tags of a webpage, which greatly improves readability. You can expand and collapse a tag by clicking on the arrow in front of it or by double clicking directly on the tag. If we expand the ``span`` tag with the ``class= "text"`` we will see the quote-text we clicked on. The `Inspector` lets you -copy XPaths to selected elements. Let's try it out: Right-click on the ``span`` -tag, select ``Copy > XPath`` and paste it in the scrapy shell like so:: +copy XPaths to selected elements. Let's try it out. + +First open the Scrapy shell at http://quotes.toscrape.com/ in a terminal: + +.. code-block:: none $ scrapy shell "http://quotes.toscrape.com/" - (...) - >>> response.xpath('/html/body/div/div[2]/div[1]/div[1]/span[1]/text()').getall() - ['"The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”] + +Then, back to your web browser, right-click on the ``span`` tag, select +``Copy > XPath`` and paste it in the Scrapy shell like so: + +.. invisible-code-block: python + + response = load_response('http://quotes.toscrape.com/', 'quotes.html') + +>>> response.xpath('/html/body/div/div[2]/div[1]/div[1]/span[1]/text()').getall() +['“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”'] Adding ``text()`` at the end we are able to extract the first quote with this basic selector. But this XPath is not really that clever. All it does is @@ -112,13 +122,13 @@ see each quote: With this knowledge we can refine our XPath: Instead of a path to follow, we'll simply select all ``span`` tags with the ``class="text"`` by using -the `has-class-extension`_:: +the `has-class-extension`_: - >>> response.xpath('//span[has-class("text")]/text()').getall() - ['"The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”, - '“It is our choices, Harry, that show what we truly are, far more than our abilities.”', - '“There are only two ways to live your life. One is as though nothing is a miracle. The other is as though everything is a miracle.”', - (...)] +>>> response.xpath('//span[has-class("text")]/text()').getall() +['“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”', +'“It is our choices, Harry, that show what we truly are, far more than our abilities.”', +'“There are only two ways to live your life. One is as though nothing is a miracle. The other is as though everything is a miracle.”', +...] And with one simple, cleverer XPath we are able to extract all quotes from the page. We could have constructed a loop over our first XPath to increase @@ -159,7 +169,11 @@ The page is quite similar to the basic `quotes.toscrape.com`_-page, but instead of the above-mentioned ``Next`` button, the page automatically loads new quotes when you scroll to the bottom. We could go ahead and try out different XPaths directly, but instead -we'll check another quite useful command from the scrapy shell:: +we'll check another quite useful command from the Scrapy shell: + +.. skip: next + +.. code-block:: none $ scrapy shell "quotes.toscrape.com/scroll" (...) @@ -203,7 +217,7 @@ where our quotes are coming from: First click on the request with the name ``scroll``. On the right you can now inspect the request. In ``Headers`` you'll find details about the request headers, such as the URL, the method, the IP-address, -and so on. We'll ignore the other tabs and click directly on ``Reponse``. +and so on. We'll ignore the other tabs and click directly on ``Response``. What you should see in the ``Preview`` pane is the rendered HTML-code, that is exactly what we saw when we called ``view(response)`` in the @@ -252,9 +266,33 @@ If the handy ``has_next`` element is ``true`` (try loading `quotes.toscrape.com/api/quotes?page=10`_ in your browser or a page-number greater than 10), we increment the ``page`` attribute and ``yield`` a new request, inserting the incremented page-number -into our ``url``. +into our ``url``. -You can see that with a few inspections in the `Network`-tool we +.. _requests-from-curl: + +In more complex websites, it could be difficult to easily reproduce the +requests, as we could need to add ``headers`` or ``cookies`` to make it work. +In those cases you can export the requests in `cURL `_ +format, by right-clicking on each of them in the network tool and using the +:meth:`~scrapy.http.Request.from_curl()` method to generate an equivalent +request:: + + from scrapy import Request + + request = Request.from_curl( + "curl 'http://quotes.toscrape.com/api/quotes?page=1' -H 'User-Agent: Mozil" + "la/5.0 (X11; Linux x86_64; rv:67.0) Gecko/20100101 Firefox/67.0' -H 'Acce" + "pt: */*' -H 'Accept-Language: ca,en-US;q=0.7,en;q=0.3' --compressed -H 'X" + "-Requested-With: XMLHttpRequest' -H 'Proxy-Authorization: Basic QFRLLTAzM" + "zEwZTAxLTk5MWUtNDFiNC1iZWRmLTJjNGI4M2ZiNDBmNDpAVEstMDMzMTBlMDEtOTkxZS00MW" + "I0LWJlZGYtMmM0YjgzZmI0MGY0' -H 'Connection: keep-alive' -H 'Referer: http" + "://quotes.toscrape.com/scroll' -H 'Cache-Control: max-age=0'") + +Alternatively, if you want to know the arguments needed to recreate that +request you can use the :func:`scrapy.utils.curl.curl_to_request_kwargs` +function to get a dictionary with the equivalent arguments. + +As you can see, with a few inspections in the `Network`-tool we were able to easily replicate the dynamic requests of the scrolling functionality of the page. Crawling dynamic pages can be quite daunting and pages can be very complex, but it (mostly) boils down @@ -262,7 +300,7 @@ to identifying the correct request and replicating it in your spider. .. _Developer Tools: https://en.wikipedia.org/wiki/Web_development_tools .. _quotes.toscrape.com: http://quotes.toscrape.com -.. _quotes.toscrape.com/scroll: quotes.toscrape.com/scroll/ +.. _quotes.toscrape.com/scroll: http://quotes.toscrape.com/scroll .. _quotes.toscrape.com/api/quotes?page=10: http://quotes.toscrape.com/api/quotes?page=10 .. _has-class-extension: https://parsel.readthedocs.io/en/latest/usage.html#other-xpath-extensions diff --git a/docs/topics/downloader-middleware.rst b/docs/topics/downloader-middleware.rst index 38a4fdb25..ae6d41809 100644 --- a/docs/topics/downloader-middleware.rst +++ b/docs/topics/downloader-middleware.rst @@ -348,7 +348,6 @@ HttpCacheMiddleware * :ref:`httpcache-storage-fs` * :ref:`httpcache-storage-dbm` - * :ref:`httpcache-storage-leveldb` You can change the HTTP cache storage backend with the :setting:`HTTPCACHE_STORAGE` setting. Or you can also :ref:`implement your own storage backend. ` @@ -365,24 +364,25 @@ HttpCacheMiddleware You can also avoid caching a response on every policy using :reqmeta:`dont_cache` meta key equals ``True``. +.. module:: scrapy.extensions.httpcache + :noindex: + .. _httpcache-policy-dummy: Dummy policy (default) ~~~~~~~~~~~~~~~~~~~~~~ -This policy has no awareness of any HTTP Cache-Control directives. -Every request and its corresponding response are cached. When the same -request is seen again, the response is returned without transferring -anything from the Internet. +.. class:: DummyPolicy -The Dummy policy is useful for testing spiders faster (without having -to wait for downloads every time) and for trying your spider offline, -when an Internet connection is not available. The goal is to be able to -"replay" a spider run *exactly as it ran before*. + This policy has no awareness of any HTTP Cache-Control directives. + Every request and its corresponding response are cached. When the same + request is seen again, the response is returned without transferring + anything from the Internet. -In order to use this policy, set: - -* :setting:`HTTPCACHE_POLICY` to ``scrapy.extensions.httpcache.DummyPolicy`` + The Dummy policy is useful for testing spiders faster (without having + to wait for downloads every time) and for trying your spider offline, + when an Internet connection is not available. The goal is to be able to + "replay" a spider run *exactly as it ran before*. .. _httpcache-policy-rfc2616: @@ -390,45 +390,44 @@ In order to use this policy, set: RFC2616 policy ~~~~~~~~~~~~~~ -This policy provides a RFC2616 compliant HTTP cache, i.e. with HTTP -Cache-Control awareness, aimed at production and used in continuous -runs to avoid downloading unmodified data (to save bandwidth and speed up crawls). +.. class:: RFC2616Policy -what is implemented: + This policy provides a RFC2616 compliant HTTP cache, i.e. with HTTP + Cache-Control awareness, aimed at production and used in continuous + runs to avoid downloading unmodified data (to save bandwidth and speed up + crawls). -* Do not attempt to store responses/requests with ``no-store`` cache-control directive set -* Do not serve responses from cache if ``no-cache`` cache-control directive is set even for fresh responses -* Compute freshness lifetime from ``max-age`` cache-control directive -* Compute freshness lifetime from ``Expires`` response header -* Compute freshness lifetime from ``Last-Modified`` response header (heuristic used by Firefox) -* Compute current age from ``Age`` response header -* Compute current age from ``Date`` header -* Revalidate stale responses based on ``Last-Modified`` response header -* Revalidate stale responses based on ``ETag`` response header -* Set ``Date`` header for any received response missing it -* Support ``max-stale`` cache-control directive in requests + What is implemented: - This allows spiders to be configured with the full RFC2616 cache policy, - but avoid revalidation on a request-by-request basis, while remaining - conformant with the HTTP spec. + * Do not attempt to store responses/requests with ``no-store`` cache-control directive set + * Do not serve responses from cache if ``no-cache`` cache-control directive is set even for fresh responses + * Compute freshness lifetime from ``max-age`` cache-control directive + * Compute freshness lifetime from ``Expires`` response header + * Compute freshness lifetime from ``Last-Modified`` response header (heuristic used by Firefox) + * Compute current age from ``Age`` response header + * Compute current age from ``Date`` header + * Revalidate stale responses based on ``Last-Modified`` response header + * Revalidate stale responses based on ``ETag`` response header + * Set ``Date`` header for any received response missing it + * Support ``max-stale`` cache-control directive in requests - Example: + This allows spiders to be configured with the full RFC2616 cache policy, + but avoid revalidation on a request-by-request basis, while remaining + conformant with the HTTP spec. - Add ``Cache-Control: max-stale=600`` to Request headers to accept responses that - have exceeded their expiration time by no more than 600 seconds. + Example: - See also: RFC2616, 14.9.3 + Add ``Cache-Control: max-stale=600`` to Request headers to accept responses that + have exceeded their expiration time by no more than 600 seconds. -what is missing: + See also: RFC2616, 14.9.3 -* ``Pragma: no-cache`` support https://www.w3.org/Protocols/rfc2616/rfc2616-sec14.html#sec14.9.1 -* ``Vary`` header support https://www.w3.org/Protocols/rfc2616/rfc2616-sec13.html#sec13.6 -* Invalidation after updates or deletes https://www.w3.org/Protocols/rfc2616/rfc2616-sec13.html#sec13.10 -* ... probably others .. + What is missing: -In order to use this policy, set: - -* :setting:`HTTPCACHE_POLICY` to ``scrapy.extensions.httpcache.RFC2616Policy`` + * ``Pragma: no-cache`` support https://www.w3.org/Protocols/rfc2616/rfc2616-sec14.html#sec14.9.1 + * ``Vary`` header support https://www.w3.org/Protocols/rfc2616/rfc2616-sec13.html#sec13.6 + * Invalidation after updates or deletes https://www.w3.org/Protocols/rfc2616/rfc2616-sec13.html#sec13.10 + * ... probably others .. .. _httpcache-storage-fs: @@ -436,67 +435,47 @@ In order to use this policy, set: Filesystem storage backend (default) ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ -File system storage backend is available for the HTTP cache middleware. +.. class:: FilesystemCacheStorage -In order to use this storage backend, set: + File system storage backend is available for the HTTP cache middleware. -* :setting:`HTTPCACHE_STORAGE` to ``scrapy.extensions.httpcache.FilesystemCacheStorage`` + Each request/response pair is stored in a different directory containing + the following files: -Each request/response pair is stored in a different directory containing -the following files: + * ``request_body`` - the plain request body - * ``request_body`` - the plain request body - * ``request_headers`` - the request headers (in raw HTTP format) - * ``response_body`` - the plain response body - * ``response_headers`` - the request headers (in raw HTTP format) - * ``meta`` - some metadata of this cache resource in Python ``repr()`` format - (grep-friendly format) - * ``pickled_meta`` - the same metadata in ``meta`` but pickled for more - efficient deserialization + * ``request_headers`` - the request headers (in raw HTTP format) -The directory name is made from the request fingerprint (see -``scrapy.utils.request.fingerprint``), and one level of subdirectories is -used to avoid creating too many files into the same directory (which is -inefficient in many file systems). An example directory could be:: + * ``response_body`` - the plain response body - /path/to/cache/dir/example.com/72/72811f648e718090f041317756c03adb0ada46c7 + * ``response_headers`` - the request headers (in raw HTTP format) + + * ``meta`` - some metadata of this cache resource in Python ``repr()`` + format (grep-friendly format) + + * ``pickled_meta`` - the same metadata in ``meta`` but pickled for more + efficient deserialization + + The directory name is made from the request fingerprint (see + ``scrapy.utils.request.fingerprint``), and one level of subdirectories is + used to avoid creating too many files into the same directory (which is + inefficient in many file systems). An example directory could be:: + + /path/to/cache/dir/example.com/72/72811f648e718090f041317756c03adb0ada46c7 .. _httpcache-storage-dbm: DBM storage backend ~~~~~~~~~~~~~~~~~~~ -.. versionadded:: 0.13 +.. class:: DbmCacheStorage -A DBM_ storage backend is also available for the HTTP cache middleware. + .. versionadded:: 0.13 -By default, it uses the anydbm_ module, but you can change it with the -:setting:`HTTPCACHE_DBM_MODULE` setting. + A DBM_ storage backend is also available for the HTTP cache middleware. -In order to use this storage backend, set: - -* :setting:`HTTPCACHE_STORAGE` to ``scrapy.extensions.httpcache.DbmCacheStorage`` - -.. _httpcache-storage-leveldb: - -LevelDB storage backend -~~~~~~~~~~~~~~~~~~~~~~~ - -.. versionadded:: 0.23 - -A LevelDB_ storage backend is also available for the HTTP cache middleware. - -This backend is not recommended for development because only one process can -access LevelDB databases at the same time, so you can't run a crawl and open -the scrapy shell in parallel for the same spider. - -In order to use this storage backend: - -* set :setting:`HTTPCACHE_STORAGE` to ``scrapy.extensions.httpcache.LeveldbCacheStorage`` -* install `LevelDB python bindings`_ like ``pip install leveldb`` - -.. _LevelDB: https://github.com/google/leveldb -.. _leveldb python bindings: https://pypi.python.org/pypi/leveldb + By default, it uses the :mod:`dbm`, but you can change it with the + :setting:`HTTPCACHE_DBM_MODULE` setting. .. _httpcache-storage-custom: @@ -512,7 +491,7 @@ defines the methods described below. .. method:: open_spider(spider) - This method gets called after a spider has been opened for crawling. It handles + This method gets called after a spider has been opened for crawling. It handles the :signal:`open_spider ` signal. :param spider: the spider which has been opened @@ -520,8 +499,8 @@ defines the methods described below. .. method:: close_spider(spider) - This method gets called after a spider has been closed. It handles - the :signal:`close_spider ` signal. + This method gets called after a spider has been closed. It handles + the :signal:`close_spider ` signal. :param spider: the spider which has been closed :type spider: :class:`~scrapy.spiders.Spider` object @@ -533,7 +512,7 @@ defines the methods described below. :param spider: the spider which generated the request :type spider: :class:`~scrapy.spiders.Spider` object - :param request: the request to find cached reponse for + :param request: the request to find cached response for :type request: :class:`~scrapy.http.Request` object .. method:: store_response(spider, request, response) @@ -647,7 +626,7 @@ HTTPCACHE_DBM_MODULE .. versionadded:: 0.13 -Default: ``'anydbm'`` +Default: ``'dbm'`` The database module to use in the :ref:`DBM storage backend `. This setting is specific to the DBM backend. @@ -963,7 +942,7 @@ precedence over the :setting:`RETRY_TIMES` setting. RETRY_HTTP_CODES ^^^^^^^^^^^^^^^^ -Default: ``[500, 502, 503, 504, 522, 524, 408]`` +Default: ``[500, 502, 503, 504, 522, 524, 408, 429]`` Which HTTP response codes to retry. Other errors (DNS lookup issues, connections lost, etc) are always retried. @@ -989,6 +968,24 @@ RobotsTxtMiddleware To make sure Scrapy respects robots.txt make sure the middleware is enabled and the :setting:`ROBOTSTXT_OBEY` setting is enabled. + The :setting:`ROBOTSTXT_USER_AGENT` setting can be used to specify the + user agent string to use for matching in the robots.txt_ file. If it + is ``None``, the User-Agent header you are sending with the request or the + :setting:`USER_AGENT` setting (in that order) will be used for determining + the user agent to use in the robots.txt_ file. + + This middleware has to be combined with a robots.txt_ parser. + + Scrapy ships with support for the following robots.txt_ parsers: + + * :ref:`Protego ` (default) + * :ref:`RobotFileParser ` + * :ref:`Reppy ` + * :ref:`Robotexclusionrulesparser ` + + You can change the robots.txt_ parser with the :setting:`ROBOTSTXT_PARSER` + setting. Or you can also :ref:`implement support for a new parser `. + .. reqmeta:: dont_obey_robotstxt If :attr:`Request.meta ` has @@ -996,6 +993,129 @@ If :attr:`Request.meta ` has the request will be ignored by this middleware even if :setting:`ROBOTSTXT_OBEY` is enabled. +Parsers vary in several aspects: + +* Language of implementation + +* Supported specification + +* Support for wildcard matching + +* Usage of `length based rule `_: + in particular for ``Allow`` and ``Disallow`` directives, where the most + specific rule based on the length of the path trumps the less specific + (shorter) rule + +Performance comparison of different parsers is available at `the following link +`_. + +.. _protego-parser: + +Protego parser +~~~~~~~~~~~~~~ + +Based on `Protego `_: + +* implemented in Python + +* is compliant with `Google's Robots.txt Specification + `_ + +* supports wildcard matching + +* uses the length based rule + +Scrapy uses this parser by default. + +.. _python-robotfileparser: + +RobotFileParser +~~~~~~~~~~~~~~~ + +Based on `RobotFileParser +`_: + +* is Python's built-in robots.txt_ parser + +* is compliant with `Martijn Koster's 1996 draft specification + `_ + +* lacks support for wildcard matching + +* doesn't use the length based rule + +It is faster than Protego and backward-compatible with versions of Scrapy before 1.8.0. + +In order to use this parser, set: + +* :setting:`ROBOTSTXT_PARSER` to ``scrapy.robotstxt.PythonRobotParser`` + +.. _reppy-parser: + +Reppy parser +~~~~~~~~~~~~ + +Based on `Reppy `_: + +* is a Python wrapper around `Robots Exclusion Protocol Parser for C++ + `_ + +* is compliant with `Martijn Koster's 1996 draft specification + `_ + +* supports wildcard matching + +* uses the length based rule + +Native implementation, provides better speed than Protego. + +In order to use this parser: + +* Install `Reppy `_ by running ``pip install reppy`` + +* Set :setting:`ROBOTSTXT_PARSER` setting to + ``scrapy.robotstxt.ReppyRobotParser`` + +.. _rerp-parser: + +Robotexclusionrulesparser +~~~~~~~~~~~~~~~~~~~~~~~~~ + +Based on `Robotexclusionrulesparser `_: + +* implemented in Python + +* is compliant with `Martijn Koster's 1996 draft specification + `_ + +* supports wildcard matching + +* doesn't use the length based rule + +In order to use this parser: + +* Install `Robotexclusionrulesparser `_ by running + ``pip install robotexclusionrulesparser`` + +* Set :setting:`ROBOTSTXT_PARSER` setting to + ``scrapy.robotstxt.RerpRobotParser`` + +.. _support-for-new-robots-parser: + +Implementing support for a new parser +------------------------------------- + +You can implement support for a new robots.txt_ parser by subclassing +the abstract base class :class:`~scrapy.robotstxt.RobotParser` and +implementing the methods described below. + +.. module:: scrapy.robotstxt + :synopsis: robots.txt parser interface and implementations + +.. autoclass:: RobotParser + :members: + +.. _robots.txt: http://www.robotstxt.org/ DownloaderStats --------------- @@ -1082,4 +1202,3 @@ The default encoding for proxy authentication on :class:`HttpProxyMiddleware`. .. _DBM: https://en.wikipedia.org/wiki/Dbm -.. _anydbm: https://docs.python.org/2/library/anydbm.html diff --git a/docs/topics/dynamic-content.rst b/docs/topics/dynamic-content.rst index 8b5dacf56..1c3607860 100644 --- a/docs/topics/dynamic-content.rst +++ b/docs/topics/dynamic-content.rst @@ -85,6 +85,13 @@ It might be enough to yield a :class:`~scrapy.http.Request` with the same HTTP method and URL. However, you may also need to reproduce the body, headers and form parameters (see :class:`~scrapy.http.FormRequest`) of that request. +As all major browsers allow to export the requests in `cURL +`_ format, Scrapy incorporates the method +:meth:`~scrapy.http.Request.from_curl()` to generate an equivalent +:class:`~scrapy.http.Request` from a cURL command. To get more information +visit :ref:`request from curl ` inside the network +tool section. + Once you get the expected response, you can :ref:`extract the desired data from it `. @@ -165,27 +172,27 @@ data from it: data in JSON format, which you can then parse with `json.loads`_. For example, if the JavaScript code contains a separate line like - ``var data = {"field": "value"};`` you can extract that data as follows:: + ``var data = {"field": "value"};`` you can extract that data as follows: - >>> pattern = r'\bvar\s+data\s*=\s*(\{.*?\})\s*;\s*\n' - >>> json_data = response.css('script::text').re_first(pattern) - >>> json.loads(json_data) - {'field': 'value'} + >>> pattern = r'\bvar\s+data\s*=\s*(\{.*?\})\s*;\s*\n' + >>> json_data = response.css('script::text').re_first(pattern) + >>> json.loads(json_data) + {'field': 'value'} - Otherwise, use js2xml_ to convert the JavaScript code into an XML document that you can parse using :ref:`selectors `. For example, if the JavaScript code contains - ``var data = {field: "value"};`` you can extract that data as follows:: + ``var data = {field: "value"};`` you can extract that data as follows: - >>> import js2xml - >>> import lxml.etree - >>> from parsel import Selector - >>> javascript = response.css('script::text').get() - >>> xml = lxml.etree.tostring(js2xml.parse(javascript), encoding='unicode') - >>> selector = Selector(text=xml) - >>> selector.css('var[name="data"]').get() - 'value' + >>> import js2xml + >>> import lxml.etree + >>> from parsel import Selector + >>> javascript = response.css('script::text').get() + >>> xml = lxml.etree.tostring(js2xml.parse(javascript), encoding='unicode') + >>> selector = Selector(text=xml) + >>> selector.css('var[name="data"]').get() + 'value' .. _topics-javascript-rendering: diff --git a/docs/topics/email.rst b/docs/topics/email.rst index 949cdc638..72bf52227 100644 --- a/docs/topics/email.rst +++ b/docs/topics/email.rst @@ -9,19 +9,19 @@ Sending e-mail Although Python makes sending e-mails relatively easy via the `smtplib`_ library, Scrapy provides its own facility for sending e-mails which is very -easy to use and it's implemented using `Twisted non-blocking IO`_, to avoid -interfering with the non-blocking IO of the crawler. It also provides a -simple API for sending attachments and it's very easy to configure, with a few -:ref:`settings `. +easy to use and it's implemented using :doc:`Twisted non-blocking IO +`, to avoid interfering with the non-blocking +IO of the crawler. It also provides a simple API for sending attachments and +it's very easy to configure, with a few :ref:`settings +`. .. _smtplib: https://docs.python.org/2/library/smtplib.html -.. _Twisted non-blocking IO: https://twistedmatrix.com/documents/current/core/howto/defer-intro.html Quick example ============= There are two ways to instantiate the mail sender. You can instantiate it using -the standard constructor:: +the standard ``__init__`` method:: from scrapy.mail import MailSender mailer = MailSender() @@ -39,7 +39,8 @@ MailSender class reference ========================== MailSender is the preferred class to use for sending emails from Scrapy, as it -uses `Twisted non-blocking IO`_, like the rest of the framework. +uses :doc:`Twisted non-blocking IO `, like the +rest of the framework. .. class:: MailSender(smtphost=None, mailfrom=None, smtpuser=None, smtppass=None, smtpport=None) @@ -111,7 +112,7 @@ uses `Twisted non-blocking IO`_, like the rest of the framework. Mail settings ============= -These settings define the default constructor values of the :class:`MailSender` +These settings define the default ``__init__`` method values of the :class:`MailSender` class, and can be used to configure e-mail notifications in your project without writing any code (for those extensions and code that uses :class:`MailSender`). diff --git a/docs/topics/exporters.rst b/docs/topics/exporters.rst index f5048d2da..b8d898022 100644 --- a/docs/topics/exporters.rst +++ b/docs/topics/exporters.rst @@ -51,7 +51,6 @@ value of one of their fields:: def close_spider(self, spider): for exporter in self.year_to_exporter.values(): exporter.finish_exporting() - exporter.file.close() def _exporter_for_item(self, item): year = item['year'] @@ -88,8 +87,8 @@ described next. 1. Declaring a serializer in the field -------------------------------------- -If you use :class:`~.Item` you can declare a serializer in the -:ref:`field metadata `. The serializer must be +If you use :class:`~.Item` you can declare a serializer in the +:ref:`field metadata `. The serializer must be a callable which receives a value and returns its serialized form. Example:: @@ -145,7 +144,7 @@ BaseItemExporter defining what fields to export, whether to export empty fields, or which encoding to use. - These features can be configured through the constructor arguments which + These features can be configured through the ``__init__`` method arguments which populate their respective instance attributes: :attr:`fields_to_export`, :attr:`export_empty_fields`, :attr:`encoding`, :attr:`indent`. @@ -165,9 +164,9 @@ BaseItemExporter value unchanged except for ``unicode`` values which are encoded to ``str`` using the encoding declared in the :attr:`encoding` attribute. - :param field: the field being serialized. If a raw dict is being + :param field: the field being serialized. If a raw dict is being exported (not :class:`~.Item`) *field* value is an empty dict. - :type field: :class:`~scrapy.item.Field` object or an empty dict + :type field: :class:`~scrapy.item.Field` object or an empty dict :param name: the name of the field being serialized :type name: str @@ -223,6 +222,12 @@ BaseItemExporter * ``indent<=0`` each item on its own line, no indentation * ``indent>0`` each item on its own line, indented with the provided numeric value +PythonItemExporter +------------------ + +.. autoclass:: PythonItemExporter + + .. highlight:: none XmlItemExporter @@ -241,8 +246,8 @@ XmlItemExporter :param item_element: The name of each item element in the exported XML. :type item_element: str - The additional keyword arguments of this constructor are passed to the - :class:`BaseItemExporter` constructor. + The additional keyword arguments of this ``__init__`` method are passed to the + :class:`BaseItemExporter` ``__init__`` method. A typical output of this exporter would be:: @@ -301,9 +306,9 @@ CsvItemExporter multi-valued fields, if found. :type include_headers_line: str - The additional keyword arguments of this constructor are passed to the - :class:`BaseItemExporter` constructor, and the leftover arguments to the - `csv.writer`_ constructor, so you can use any ``csv.writer`` constructor + The additional keyword arguments of this ``__init__`` method are passed to the + :class:`BaseItemExporter` ``__init__`` method, and the leftover arguments to the + `csv.writer`_ ``__init__`` method, so you can use any ``csv.writer`` ``__init__`` method argument to customize this exporter. A typical output of this exporter would be:: @@ -329,8 +334,8 @@ PickleItemExporter For more information, refer to the `pickle module documentation`_. - The additional keyword arguments of this constructor are passed to the - :class:`BaseItemExporter` constructor. + The additional keyword arguments of this ``__init__`` method are passed to the + :class:`BaseItemExporter` ``__init__`` method. Pickle isn't a human readable format, so no output examples are provided. @@ -346,8 +351,8 @@ PprintItemExporter :param file: the file-like object to use for exporting the data. Its ``write`` method should accept ``bytes`` (a disk file opened in binary mode, a ``io.BytesIO`` object, etc) - The additional keyword arguments of this constructor are passed to the - :class:`BaseItemExporter` constructor. + The additional keyword arguments of this ``__init__`` method are passed to the + :class:`BaseItemExporter` ``__init__`` method. A typical output of this exporter would be:: @@ -362,10 +367,10 @@ JsonItemExporter .. class:: JsonItemExporter(file, \**kwargs) Exports Items in JSON format to the specified file-like object, writing all - objects as a list of objects. The additional constructor arguments are - passed to the :class:`BaseItemExporter` constructor, and the leftover - arguments to the `JSONEncoder`_ constructor, so you can use any - `JSONEncoder`_ constructor argument to customize this exporter. + objects as a list of objects. The additional ``__init__`` method arguments are + passed to the :class:`BaseItemExporter` ``__init__`` method, and the leftover + arguments to the `JSONEncoder`_ ``__init__`` method, so you can use any + `JSONEncoder`_ ``__init__`` method argument to customize this exporter. :param file: the file-like object to use for exporting the data. Its ``write`` method should accept ``bytes`` (a disk file opened in binary mode, a ``io.BytesIO`` object, etc) @@ -393,10 +398,10 @@ JsonLinesItemExporter .. class:: JsonLinesItemExporter(file, \**kwargs) Exports Items in JSON format to the specified file-like object, writing one - JSON-encoded item per line. The additional constructor arguments are passed - to the :class:`BaseItemExporter` constructor, and the leftover arguments to - the `JSONEncoder`_ constructor, so you can use any `JSONEncoder`_ - constructor argument to customize this exporter. + JSON-encoded item per line. The additional ``__init__`` method arguments are passed + to the :class:`BaseItemExporter` ``__init__`` method, and the leftover arguments to + the `JSONEncoder`_ ``__init__`` method, so you can use any `JSONEncoder`_ + ``__init__`` method argument to customize this exporter. :param file: the file-like object to use for exporting the data. Its ``write`` method should accept ``bytes`` (a disk file opened in binary mode, a ``io.BytesIO`` object, etc) @@ -410,3 +415,8 @@ JsonLinesItemExporter this exporter is well suited for serializing large amounts of data. .. _JSONEncoder: https://docs.python.org/2/library/json.html#json.JSONEncoder + +MarshalItemExporter +------------------- + +.. autoclass:: MarshalItemExporter diff --git a/docs/topics/extensions.rst b/docs/topics/extensions.rst index d6e7452a1..0a7455ec9 100644 --- a/docs/topics/extensions.rst +++ b/docs/topics/extensions.rst @@ -28,7 +28,7 @@ Loading & activating extensions Extensions are loaded and activated at startup by instantiating a single instance of the extension class. Therefore, all the extension initialization -code must be performed in the class constructor (``__init__`` method). +code must be performed in the class ``__init__`` method. To make an extension available, add it to the :setting:`EXTENSIONS` setting in your Scrapy settings. In :setting:`EXTENSIONS`, each extension is represented @@ -183,7 +183,7 @@ Telnet console extension .. module:: scrapy.extensions.telnet :synopsis: Telnet console -.. class:: scrapy.extensions.telnet.TelnetConsole +.. class:: TelnetConsole Provides a telnet console for getting into a Python interpreter inside the currently running Scrapy process, which can be very useful for debugging. @@ -200,7 +200,7 @@ Memory usage extension .. module:: scrapy.extensions.memusage :synopsis: Memory usage extension -.. class:: scrapy.extensions.memusage.MemoryUsage +.. class:: MemoryUsage .. note:: This extension does not work in Windows. @@ -228,7 +228,7 @@ Memory debugger extension .. module:: scrapy.extensions.memdebug :synopsis: Memory debugger extension -.. class:: scrapy.extensions.memdebug.MemoryDebugger +.. class:: MemoryDebugger An extension for debugging memory usage. It collects information about: @@ -244,7 +244,7 @@ Close spider extension .. module:: scrapy.extensions.closespider :synopsis: Close spider extension -.. class:: scrapy.extensions.closespider.CloseSpider +.. class:: CloseSpider Closes a spider automatically when some conditions are met, using a specific closing reason for each condition. @@ -317,7 +317,7 @@ StatsMailer extension .. module:: scrapy.extensions.statsmailer :synopsis: StatsMailer extension -.. class:: scrapy.extensions.statsmailer.StatsMailer +.. class:: StatsMailer This simple extension can be used to send a notification e-mail every time a domain has finished scraping, including the Scrapy stats collected. The email @@ -333,7 +333,7 @@ Debugging extensions Stack trace dump extension ~~~~~~~~~~~~~~~~~~~~~~~~~~ -.. class:: scrapy.extensions.debug.StackTraceDump +.. class:: StackTraceDump Dumps information about the running process when a `SIGQUIT`_ or `SIGUSR2`_ signal is received. The information dumped is the following: @@ -362,7 +362,7 @@ There are at least two ways to send Scrapy the `SIGQUIT`_ signal: Debugger extension ~~~~~~~~~~~~~~~~~~ -.. class:: scrapy.extensions.debug.Debugger +.. class:: Debugger Invokes a `Python debugger`_ inside a running Scrapy process when a `SIGUSR2`_ signal is received. After the debugger is exited, the Scrapy process continues diff --git a/docs/topics/feed-exports.rst b/docs/topics/feed-exports.rst index af541db78..7481b1a99 100644 --- a/docs/topics/feed-exports.rst +++ b/docs/topics/feed-exports.rst @@ -99,12 +99,12 @@ The storages backends supported out of the box are: * :ref:`topics-feed-storage-fs` * :ref:`topics-feed-storage-ftp` - * :ref:`topics-feed-storage-s3` (requires botocore_ or boto_) + * :ref:`topics-feed-storage-s3` (requires botocore_) * :ref:`topics-feed-storage-stdout` Some storage backends may be unavailable if the required external libraries are not available. For example, the S3 backend is only available if the botocore_ -or boto_ library is installed (Scrapy supports boto_ only on Python 2). +library is installed. .. _topics-feed-uri-params: @@ -182,7 +182,7 @@ The feeds are stored on `Amazon S3`_. * ``s3://mybucket/path/to/export.csv`` * ``s3://aws_key:aws_secret@mybucket/path/to/export.csv`` - * Required external libraries: `botocore`_ (Python 2 and Python 3) or `boto`_ (Python 2 only) + * Required external libraries: `botocore`_ The AWS credentials can be passed as user/password in the URI, or they can be passed through the following settings: @@ -399,6 +399,5 @@ format in :setting:`FEED_EXPORTERS`. E.g., to disable the built-in CSV exporter .. _URI: https://en.wikipedia.org/wiki/Uniform_Resource_Identifier .. _Amazon S3: https://aws.amazon.com/s3/ -.. _boto: https://github.com/boto/boto .. _botocore: https://github.com/boto/botocore .. _Canned ACL: https://docs.aws.amazon.com/AmazonS3/latest/dev/acl-overview.html#canned-acl diff --git a/docs/topics/item-pipeline.rst b/docs/topics/item-pipeline.rst index fae18200a..cdc4953c2 100644 --- a/docs/topics/item-pipeline.rst +++ b/docs/topics/item-pipeline.rst @@ -29,7 +29,8 @@ Each item pipeline component is a Python class that must implement the following This method is called for every item pipeline component. :meth:`process_item` must either: return a dict with data, return an :class:`~scrapy.item.Item` - (or any descendant class) object, return a `Twisted Deferred`_ or raise + (or any descendant class) object, return a + :class:`~twisted.internet.defer.Deferred` or raise :exc:`~scrapy.exceptions.DropItem` exception. Dropped items are no longer processed by further pipeline components. @@ -67,8 +68,6 @@ Additionally, they may also implement the following methods: :type crawler: :class:`~scrapy.crawler.Crawler` object -.. _Twisted Deferred: https://twistedmatrix.com/documents/current/core/howto/defer.html - Item pipeline example ===================== @@ -166,7 +165,8 @@ method and how to clean up the resources properly.:: Take screenshot of item ----------------------- -This example demonstrates how to return Deferred_ from :meth:`process_item` method. +This example demonstrates how to return a +:class:`~twisted.internet.defer.Deferred` from the :meth:`process_item` method. It uses Splash_ to render screenshot of item url. Pipeline makes request to locally running instance of Splash_. After request is downloaded and Deferred callback fires, it saves item to a file and adds filename to an item. @@ -209,7 +209,6 @@ and Deferred callback fires, it saves item to a file and adds filename to an ite return item .. _Splash: https://splash.readthedocs.io/en/stable/ -.. _Deferred: https://twistedmatrix.com/documents/current/core/howto/defer.html Duplicates filter ----------------- diff --git a/docs/topics/items.rst b/docs/topics/items.rst index 60fbc82f8..15313775b 100644 --- a/docs/topics/items.rst +++ b/docs/topics/items.rst @@ -16,12 +16,12 @@ especially in a larger project with many spiders. To define common output data format Scrapy provides the :class:`Item` class. :class:`Item` objects are simple containers used to collect the scraped data. They provide a `dictionary-like`_ API with a convenient syntax for declaring -their available fields. +their available fields. -Various Scrapy components use extra information provided by Items: +Various Scrapy components use extra information provided by Items: exporters look at declared fields to figure out columns to export, serialization can be customized using Item fields metadata, :mod:`trackref` -tracks Item instances to help find memory leaks +tracks Item instances to help find memory leaks (see :ref:`topics-leaks-trackrefs`), etc. .. _dictionary-like: https://docs.python.org/2/library/stdtypes.html#dict @@ -84,77 +84,74 @@ notice the API is very similar to the `dict API`_. Creating items -------------- -:: +>>> product = Product(name='Desktop PC', price=1000) +>>> print(product) +Product(name='Desktop PC', price=1000) - >>> product = Product(name='Desktop PC', price=1000) - >>> print(product) - Product(name='Desktop PC', price=1000) Getting field values -------------------- -:: +>>> product['name'] +Desktop PC +>>> product.get('name') +Desktop PC - >>> product['name'] - Desktop PC - >>> product.get('name') - Desktop PC +>>> product['price'] +1000 - >>> product['price'] - 1000 +>>> product['last_updated'] +Traceback (most recent call last): + ... +KeyError: 'last_updated' - >>> product['last_updated'] - Traceback (most recent call last): - ... - KeyError: 'last_updated' +>>> product.get('last_updated', 'not set') +not set - >>> product.get('last_updated', 'not set') - not set +>>> product['lala'] # getting unknown field +Traceback (most recent call last): + ... +KeyError: 'lala' - >>> product['lala'] # getting unknown field - Traceback (most recent call last): - ... - KeyError: 'lala' +>>> product.get('lala', 'unknown field') +'unknown field' - >>> product.get('lala', 'unknown field') - 'unknown field' +>>> 'name' in product # is name field populated? +True - >>> 'name' in product # is name field populated? - True +>>> 'last_updated' in product # is last_updated populated? +False - >>> 'last_updated' in product # is last_updated populated? - False +>>> 'last_updated' in product.fields # is last_updated a declared field? +True - >>> 'last_updated' in product.fields # is last_updated a declared field? - True +>>> 'lala' in product.fields # is lala a declared field? +False - >>> 'lala' in product.fields # is lala a declared field? - False Setting field values -------------------- -:: +>>> product['last_updated'] = 'today' +>>> product['last_updated'] +today - >>> product['last_updated'] = 'today' - >>> product['last_updated'] - today +>>> product['lala'] = 'test' # setting unknown field +Traceback (most recent call last): + ... +KeyError: 'Product does not support field: lala' - >>> product['lala'] = 'test' # setting unknown field - Traceback (most recent call last): - ... - KeyError: 'Product does not support field: lala' Accessing all populated values ------------------------------ -To access all populated values, just use the typical `dict API`_:: +To access all populated values, just use the typical `dict API`_: - >>> product.keys() - ['price', 'name'] +>>> product.keys() +['price', 'name'] - >>> product.items() - [('price', 1000), ('name', 'Desktop PC')] +>>> product.items() +[('price', 1000), ('name', 'Desktop PC')] .. _copying-items: @@ -194,20 +191,21 @@ To create a deep copy, call :meth:`~scrapy.item.Item.deepcopy` instead Other common tasks ------------------ -Creating dicts from items:: +Creating dicts from items: - >>> dict(product) # create a dict from all populated values - {'price': 1000, 'name': 'Desktop PC'} +>>> dict(product) # create a dict from all populated values +{'price': 1000, 'name': 'Desktop PC'} -Creating items from dicts:: +Creating items from dicts: - >>> Product({'name': 'Laptop PC', 'price': 1500}) - Product(price=1500, name='Laptop PC') +>>> Product({'name': 'Laptop PC', 'price': 1500}) +Product(price=1500, name='Laptop PC') + +>>> Product({'name': 'Laptop PC', 'lala': 1500}) # warning: unknown field in dict +Traceback (most recent call last): + ... +KeyError: 'Product does not support field: lala' - >>> Product({'name': 'Laptop PC', 'lala': 1500}) # warning: unknown field in dict - Traceback (most recent call last): - ... - KeyError: 'Product does not support field: lala' Extending Items =============== @@ -237,8 +235,12 @@ Item objects Return a new Item optionally initialized from the given argument. - Items replicate the standard `dict API`_, including its constructor. The - only additional attribute provided by Items is: + Items replicate the standard `dict API`_, including its ``__init__`` method, and + also provide the following additional API members: + + .. automethod:: copy + + .. automethod:: deepcopy .. attribute:: fields @@ -263,3 +265,9 @@ Field objects .. _dict: https://docs.python.org/2/library/stdtypes.html#dict +Other classes related to Item +============================= + +.. autoclass:: BaseItem + +.. autoclass:: ItemMeta diff --git a/docs/topics/jobs.rst b/docs/topics/jobs.rst index 9fd311c69..8816a028c 100644 --- a/docs/topics/jobs.rst +++ b/docs/topics/jobs.rst @@ -71,34 +71,11 @@ on cookies. Request serialization --------------------- -Requests must be serializable by the ``pickle`` module, in order for persistence -to work, so you should make sure that your requests are serializable. - -The most common issue here is to use ``lambda`` functions on request callbacks that -can't be persisted. - -So, for example, this won't work:: - - def some_callback(self, response): - somearg = 'test' - return scrapy.Request('http://www.example.com', - callback=lambda r: self.other_callback(r, somearg)) - - def other_callback(self, response, somearg): - print("the argument passed is: %s" % somearg) - -But this will:: - - def some_callback(self, response): - somearg = 'test' - return scrapy.Request('http://www.example.com', - callback=self.other_callback, cb_kwargs={'somearg': somearg}) - - def other_callback(self, response, somearg): - print("the argument passed is: %s" % somearg) +For persistence to work, :class:`~scrapy.http.Request` objects must be +serializable with :mod:`pickle`, except for the ``callback`` and ``errback`` +values passed to their ``__init__`` method, which must be methods of the +running :class:`~scrapy.spiders.Spider` class. If you wish to log the requests that couldn't be serialized, you can set the :setting:`SCHEDULER_DEBUG` setting to ``True`` in the project's settings page. It is ``False`` by default. - -.. _pickle: https://docs.python.org/library/pickle.html diff --git a/docs/topics/leaks.rst b/docs/topics/leaks.rst index 8278e9849..9fee333ac 100644 --- a/docs/topics/leaks.rst +++ b/docs/topics/leaks.rst @@ -110,7 +110,7 @@ ties the response lifetime to the requests' one, and that would definitely cause memory leaks. Let's see how we can discover the cause (without knowing it -a-priori, of course) by using the ``trackref`` tool. +a priori, of course) by using the ``trackref`` tool. After the crawler is running for a few minutes and we notice its memory usage has grown a lot, we can enter its telnet console and check the live @@ -132,21 +132,21 @@ and check the code of the spider to discover the nasty line that is generating the leaks (passing response references inside requests). Sometimes extra information about live objects can be helpful. -Let's check the oldest response:: +Let's check the oldest response: - >>> from scrapy.utils.trackref import get_oldest - >>> r = get_oldest('HtmlResponse') - >>> r.url - 'http://www.somenastyspider.com/product.php?pid=123' +>>> from scrapy.utils.trackref import get_oldest +>>> r = get_oldest('HtmlResponse') +>>> r.url +'http://www.somenastyspider.com/product.php?pid=123' If you want to iterate over all objects, instead of getting the oldest one, you -can use the :func:`scrapy.utils.trackref.iter_all` function:: +can use the :func:`scrapy.utils.trackref.iter_all` function: - >>> from scrapy.utils.trackref import iter_all - >>> [r.url for r in iter_all('HtmlResponse')] - ['http://www.somenastyspider.com/product.php?pid=123', - 'http://www.somenastyspider.com/product.php?pid=584', - ... +>>> from scrapy.utils.trackref import iter_all +>>> [r.url for r in iter_all('HtmlResponse')] +['http://www.somenastyspider.com/product.php?pid=123', + 'http://www.somenastyspider.com/product.php?pid=584', +...] Too many spiders? ----------------- @@ -155,10 +155,10 @@ If your project has too many spiders executed in parallel, the output of :func:`prefs()` can be difficult to read. For this reason, that function has a ``ignore`` argument which can be used to ignore a particular class (and all its subclases). For -example, this won't show any live references to spiders:: +example, this won't show any live references to spiders: - >>> from scrapy.spiders import Spider - >>> prefs(ignore=Spider) +>>> from scrapy.spiders import Spider +>>> prefs(ignore=Spider) .. module:: scrapy.utils.trackref :synopsis: Track references of live objects @@ -214,41 +214,41 @@ If you use ``pip``, you can install Guppy with the following command:: The telnet console also comes with a built-in shortcut (``hpy``) for accessing Guppy heap objects. Here's an example to view all Python objects available in -the heap using Guppy:: +the heap using Guppy: - >>> x = hpy.heap() - >>> x.bytype - Partition of a set of 297033 objects. Total size = 52587824 bytes. - Index Count % Size % Cumulative % Type - 0 22307 8 16423880 31 16423880 31 dict - 1 122285 41 12441544 24 28865424 55 str - 2 68346 23 5966696 11 34832120 66 tuple - 3 227 0 5836528 11 40668648 77 unicode - 4 2461 1 2222272 4 42890920 82 type - 5 16870 6 2024400 4 44915320 85 function - 6 13949 5 1673880 3 46589200 89 types.CodeType - 7 13422 5 1653104 3 48242304 92 list - 8 3735 1 1173680 2 49415984 94 _sre.SRE_Pattern - 9 1209 0 456936 1 49872920 95 scrapy.http.headers.Headers - <1676 more rows. Type e.g. '_.more' to view.> +>>> x = hpy.heap() +>>> x.bytype +Partition of a set of 297033 objects. Total size = 52587824 bytes. + Index Count % Size % Cumulative % Type + 0 22307 8 16423880 31 16423880 31 dict + 1 122285 41 12441544 24 28865424 55 str + 2 68346 23 5966696 11 34832120 66 tuple + 3 227 0 5836528 11 40668648 77 unicode + 4 2461 1 2222272 4 42890920 82 type + 5 16870 6 2024400 4 44915320 85 function + 6 13949 5 1673880 3 46589200 89 types.CodeType + 7 13422 5 1653104 3 48242304 92 list + 8 3735 1 1173680 2 49415984 94 _sre.SRE_Pattern + 9 1209 0 456936 1 49872920 95 scrapy.http.headers.Headers +<1676 more rows. Type e.g. '_.more' to view.> You can see that most space is used by dicts. Then, if you want to see from -which attribute those dicts are referenced, you could do:: +which attribute those dicts are referenced, you could do: - >>> x.bytype[0].byvia - Partition of a set of 22307 objects. Total size = 16423880 bytes. - Index Count % Size % Cumulative % Referred Via: - 0 10982 49 9416336 57 9416336 57 '.__dict__' - 1 1820 8 2681504 16 12097840 74 '.__dict__', '.func_globals' - 2 3097 14 1122904 7 13220744 80 - 3 990 4 277200 2 13497944 82 "['cookies']" - 4 987 4 276360 2 13774304 84 "['cache']" - 5 985 4 275800 2 14050104 86 "['meta']" - 6 897 4 251160 2 14301264 87 '[2]' - 7 1 0 196888 1 14498152 88 "['moduleDict']", "['modules']" - 8 672 3 188160 1 14686312 89 "['cb_kwargs']" - 9 27 0 155016 1 14841328 90 '[1]' - <333 more rows. Type e.g. '_.more' to view.> +>>> x.bytype[0].byvia +Partition of a set of 22307 objects. Total size = 16423880 bytes. + Index Count % Size % Cumulative % Referred Via: + 0 10982 49 9416336 57 9416336 57 '.__dict__' + 1 1820 8 2681504 16 12097840 74 '.__dict__', '.func_globals' + 2 3097 14 1122904 7 13220744 80 + 3 990 4 277200 2 13497944 82 "['cookies']" + 4 987 4 276360 2 13774304 84 "['cache']" + 5 985 4 275800 2 14050104 86 "['meta']" + 6 897 4 251160 2 14301264 87 '[2]' + 7 1 0 196888 1 14498152 88 "['moduleDict']", "['modules']" + 8 672 3 188160 1 14686312 89 "['cb_kwargs']" + 9 27 0 155016 1 14841328 90 '[1]' +<333 more rows. Type e.g. '_.more' to view.> As you can see, the Guppy module is very powerful but also requires some deep knowledge about Python internals. For more info about Guppy, refer to the @@ -260,7 +260,7 @@ knowledge about Python internals. For more info about Guppy, refer to the Debugging memory leaks with muppy ================================= -If you're using Python 3, you can use muppy from `Pympler`_. +You can use muppy from `Pympler`_. .. _Pympler: https://pypi.org/project/Pympler/ @@ -269,32 +269,32 @@ If you use ``pip``, you can install muppy with the following command:: pip install Pympler Here's an example to view all Python objects available in -the heap using muppy:: +the heap using muppy: - >>> from pympler import muppy - >>> all_objects = muppy.get_objects() - >>> len(all_objects) - 28667 - >>> from pympler import summary - >>> suml = summary.summarize(all_objects) - >>> summary.print_(suml) - types | # objects | total size - ==================================== | =========== | ============ - >> from pympler import muppy +>>> all_objects = muppy.get_objects() +>>> len(all_objects) +28667 +>>> from pympler import summary +>>> suml = summary.summarize(all_objects) +>>> summary.print_(suml) + types | # objects | total size +==================================== | =========== | ============ + ` returns a +list of matching :class:`scrapy.link.Link` objects from a +:class:`~scrapy.http.Response` object. +Link extractors are used in :class:`~scrapy.spiders.CrawlSpider` spiders +through a set of :class:`~scrapy.spiders.Rule` objects. You can also use link +extractors in regular spiders. .. _topics-link-extractors-ref: -Built-in link extractors reference -================================== +Link extractor reference +======================== .. module:: scrapy.linkextractors :synopsis: Link extractors classes -Link extractors classes bundled with Scrapy are provided in the -:mod:`scrapy.linkextractors` module. - -The default link extractor is ``LinkExtractor``, which is the same as -:class:`~.LxmlLinkExtractor`:: +The link extractor class is +:class:`scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor`. For convenience it +can also be imported as ``scrapy.linkextractors.LinkExtractor``:: from scrapy.linkextractors import LinkExtractor -There used to be other link extractor classes in previous Scrapy versions, -but they are deprecated now. - LxmlLinkExtractor ----------------- @@ -152,4 +139,6 @@ LxmlLinkExtractor from elements or attributes which allow leading/trailing whitespaces). :type strip: boolean + .. automethod:: extract_links + .. _scrapy.linkextractors: https://github.com/scrapy/scrapy/blob/master/scrapy/linkextractors/__init__.py diff --git a/docs/topics/loaders.rst b/docs/topics/loaders.rst index 1c2f1da4d..9d5fccbbc 100644 --- a/docs/topics/loaders.rst +++ b/docs/topics/loaders.rst @@ -26,7 +26,7 @@ Using Item Loaders to populate items To use an Item Loader, you must first instantiate it. You can either instantiate it with a dict-like object (e.g. Item or dict) or without one, in -which case an Item is automatically instantiated in the Item Loader constructor +which case an Item is automatically instantiated in the Item Loader ``__init__`` method using the Item class specified in the :attr:`ItemLoader.default_item_class` attribute. @@ -35,6 +35,12 @@ Then, you start collecting values into the Item Loader, typically using the same item field; the Item Loader will know how to "join" those values later using a proper processing function. +.. note:: Collected data is internally stored as lists, + allowing to add several values to the same field. + If an ``item`` argument is passed when creating a loader, + each of the item's values will be stored as-is if it's already + an iterable, or wrapped with a list if it's a single value. + Here is a typical Item Loader usage in a :ref:`Spider `, using the :ref:`Product item ` declared in the :ref:`Items chapter `:: @@ -128,28 +134,14 @@ So what happens is: It's worth noticing that processors are just callable objects, which are called with the data to be parsed, and return a parsed value. So you can use any function as input or output processor. The only requirement is that they must -accept one (and only one) positional argument, which will be an iterator. +accept one (and only one) positional argument, which will be an iterable. -.. note:: Both input and output processors must receive an iterator as their +.. note:: Both input and output processors must receive an iterable as their first argument. The output of those functions can be anything. The result of input processors will be appended to an internal list (in the Loader) containing the collected values (for that field). The result of the output processors is the value that will be finally assigned to the item. -If you want to use a plain function as a processor, make sure it receives -``self`` as the first argument:: - - def lowercase_processor(self, values): - for v in values: - yield v.lower() - - class MyItemLoader(ItemLoader): - name_in = lowercase_processor - -This is because whenever a function is assigned as a class variable, it becomes -a method and would be passed the instance as the the first argument when being -called. See `this answer on stackoverflow`_ for more details. - The other thing you need to keep in mind is that the values returned by input processors are collected internally (in lists) and then passed to output processors to populate the fields. @@ -157,7 +149,7 @@ processors to populate the fields. Last, but not least, Scrapy comes with some :ref:`commonly used processors ` built-in for convenience. -.. _this answer on stackoverflow: https://stackoverflow.com/a/35322635 + Declaring Item Loaders ====================== @@ -214,14 +206,12 @@ metadata. Here is an example:: output_processor=TakeFirst(), ) -:: - - >>> from scrapy.loader import ItemLoader - >>> il = ItemLoader(item=Product()) - >>> il.add_value('name', [u'Welcome to my', u'website']) - >>> il.add_value('price', [u'€', u'1000']) - >>> il.load_item() - {'name': u'Welcome to my website', 'price': u'1000'} +>>> from scrapy.loader import ItemLoader +>>> il = ItemLoader(item=Product()) +>>> il.add_value('name', [u'Welcome to my', u'website']) +>>> il.add_value('price', [u'€', u'1000']) +>>> il.load_item() +{'name': u'Welcome to my website', 'price': u'1000'} The precedence order, for both input and output processors, is as follows: @@ -265,7 +255,7 @@ There are several ways to modify Item Loader context values: loader.context['unit'] = 'cm' 2. On Item Loader instantiation (the keyword arguments of Item Loader - constructor are stored in the Item Loader context):: + ``__init__`` method are stored in the Item Loader context):: loader = ItemLoader(product, unit='cm') @@ -322,11 +312,11 @@ ItemLoader objects applied before processors :type re: str or compiled regex - Examples:: + Examples: - >>> from scrapy.loader.processors import TakeFirst - >>> loader.get_value(u'name: foo', TakeFirst(), unicode.upper, re='name: (.+)') - 'FOO` + >>> from scrapy.loader.processors import TakeFirst + >>> loader.get_value(u'name: foo', TakeFirst(), unicode.upper, re='name: (.+)') + 'FOO` .. method:: add_value(field_name, value, \*processors, \**kwargs) @@ -485,6 +475,8 @@ ItemLoader objects .. attribute:: item The :class:`~scrapy.item.Item` object being parsed by this Item Loader. + This is mostly used as a property so when attempting to override this + value, you may want to check out :attr:`default_item_class` first. .. attribute:: context @@ -494,7 +486,7 @@ ItemLoader objects .. attribute:: default_item_class An Item class (or factory), used to instantiate items when not given in - the constructor. + the ``__init__`` method. .. attribute:: default_input_processor @@ -509,15 +501,15 @@ ItemLoader objects .. attribute:: default_selector_class The class used to construct the :attr:`selector` of this - :class:`ItemLoader`, if only a response is given in the constructor. - If a selector is given in the constructor this attribute is ignored. + :class:`ItemLoader`, if only a response is given in the ``__init__`` method. + If a selector is given in the ``__init__`` method this attribute is ignored. This attribute is sometimes overridden in subclasses. .. attribute:: selector The :class:`~scrapy.selector.Selector` object to extract data from. - It's either the selector given in the constructor or one created from - the response given in the constructor using the + It's either the selector given in the ``__init__`` method or one created from + the response given in the ``__init__`` method using the :attr:`default_selector_class`. This attribute is meant to be read-only. @@ -642,46 +634,46 @@ Here is a list of all built-in processors: .. class:: Identity The simplest processor, which doesn't do anything. It returns the original - values unchanged. It doesn't receive any constructor arguments, nor does it + values unchanged. It doesn't receive any ``__init__`` method arguments, nor does it accept Loader contexts. - Example:: + Example: - >>> from scrapy.loader.processors import Identity - >>> proc = Identity() - >>> proc(['one', 'two', 'three']) - ['one', 'two', 'three'] + >>> from scrapy.loader.processors import Identity + >>> proc = Identity() + >>> proc(['one', 'two', 'three']) + ['one', 'two', 'three'] .. class:: TakeFirst Returns the first non-null/non-empty value from the values received, so it's typically used as an output processor to single-valued fields. - It doesn't receive any constructor arguments, nor does it accept Loader contexts. + It doesn't receive any ``__init__`` method arguments, nor does it accept Loader contexts. - Example:: + Example: - >>> from scrapy.loader.processors import TakeFirst - >>> proc = TakeFirst() - >>> proc(['', 'one', 'two', 'three']) - 'one' + >>> from scrapy.loader.processors import TakeFirst + >>> proc = TakeFirst() + >>> proc(['', 'one', 'two', 'three']) + 'one' .. class:: Join(separator=u' ') - Returns the values joined with the separator given in the constructor, which + Returns the values joined with the separator given in the ``__init__`` method, which defaults to ``u' '``. It doesn't accept Loader contexts. When using the default separator, this processor is equivalent to the function: ``u' '.join`` - Examples:: + Examples: - >>> from scrapy.loader.processors import Join - >>> proc = Join() - >>> proc(['one', 'two', 'three']) - 'one two three' - >>> proc = Join('
') - >>> proc(['one', 'two', 'three']) - 'one
two
three' + >>> from scrapy.loader.processors import Join + >>> proc = Join() + >>> proc(['one', 'two', 'three']) + 'one two three' + >>> proc = Join('
') + >>> proc(['one', 'two', 'three']) + 'one
two
three' .. class:: Compose(\*functions, \**default_loader_context) @@ -694,18 +686,18 @@ Here is a list of all built-in processors: By default, stop process on ``None`` value. This behaviour can be changed by passing keyword argument ``stop_on_none=False``. - Example:: + Example: - >>> from scrapy.loader.processors import Compose - >>> proc = Compose(lambda v: v[0], str.upper) - >>> proc(['hello', 'world']) - 'HELLO' + >>> from scrapy.loader.processors import Compose + >>> proc = Compose(lambda v: v[0], str.upper) + >>> proc(['hello', 'world']) + 'HELLO' Each function can optionally receive a ``loader_context`` parameter. For those which do, this processor will pass the currently active :ref:`Loader context ` through that parameter. - The keyword arguments passed in the constructor are used as the default + The keyword arguments passed in the ``__init__`` method are used as the default Loader context values passed to each function call. However, the final Loader context values passed to functions are overridden with the currently active Loader context accessible through the :meth:`ItemLoader.context` @@ -738,41 +730,41 @@ Here is a list of all built-in processors: :meth:`~scrapy.selector.Selector.extract` method of :ref:`selectors `, which returns a list of unicode strings. - The example below should clarify how it works:: + The example below should clarify how it works: - >>> def filter_world(x): - ... return None if x == 'world' else x - ... - >>> from scrapy.loader.processors import MapCompose - >>> proc = MapCompose(filter_world, str.upper) - >>> proc(['hello', 'world', 'this', 'is', 'scrapy']) - ['HELLO, 'THIS', 'IS', 'SCRAPY'] + >>> def filter_world(x): + ... return None if x == 'world' else x + ... + >>> from scrapy.loader.processors import MapCompose + >>> proc = MapCompose(filter_world, str.upper) + >>> proc(['hello', 'world', 'this', 'is', 'scrapy']) + ['HELLO, 'THIS', 'IS', 'SCRAPY'] As with the Compose processor, functions can receive Loader contexts, and - constructor keyword arguments are used as default context values. See + ``__init__`` method keyword arguments are used as default context values. See :class:`Compose` processor for more info. .. class:: SelectJmes(json_path) - Queries the value using the json path provided to the constructor and returns the output. + Queries the value using the json path provided to the ``__init__`` method and returns the output. Requires jmespath (https://github.com/jmespath/jmespath.py) to run. This processor takes only one input at a time. - Example:: + Example: - >>> from scrapy.loader.processors import SelectJmes, Compose, MapCompose - >>> proc = SelectJmes("foo") #for direct use on lists and dictionaries - >>> proc({'foo': 'bar'}) - 'bar' - >>> proc({'foo': {'bar': 'baz'}}) - {'bar': 'baz'} + >>> from scrapy.loader.processors import SelectJmes, Compose, MapCompose + >>> proc = SelectJmes("foo") #for direct use on lists and dictionaries + >>> proc({'foo': 'bar'}) + 'bar' + >>> proc({'foo': {'bar': 'baz'}}) + {'bar': 'baz'} - Working with Json:: + Working with Json: - >>> import json - >>> proc_single_json_str = Compose(json.loads, SelectJmes("foo")) - >>> proc_single_json_str('{"foo": "bar"}') - 'bar' - >>> proc_json_list = Compose(json.loads, MapCompose(SelectJmes('foo'))) - >>> proc_json_list('[{"foo":"bar"}, {"baz":"tar"}]') - ['bar'] + >>> import json + >>> proc_single_json_str = Compose(json.loads, SelectJmes("foo")) + >>> proc_single_json_str('{"foo": "bar"}') + 'bar' + >>> proc_json_list = Compose(json.loads, MapCompose(SelectJmes('foo'))) + >>> proc_json_list('[{"foo":"bar"}, {"baz":"tar"}]') + ['bar'] diff --git a/docs/topics/logging.rst b/docs/topics/logging.rst index dea0528db..d4d22d889 100644 --- a/docs/topics/logging.rst +++ b/docs/topics/logging.rst @@ -171,9 +171,9 @@ listed in `logging's logrecord attributes docs `_ respectively. -If :setting:`LOG_SHORT_NAMES` is set, then the logs will not display the scrapy +If :setting:`LOG_SHORT_NAMES` is set, then the logs will not display the Scrapy component that prints the log. It is unset by default, hence logs contain the -scrapy component responsible for that log output. +Scrapy component responsible for that log output. Command-line options -------------------- @@ -193,6 +193,18 @@ to override some of the Scrapy settings regarding logging. Module `logging.handlers `_ Further documentation on available handlers +.. _custom-log-formats: + +Custom Log Formats +------------------ + +A custom log format can be set for different actions by extending +:class:`~scrapy.logformatter.LogFormatter` class and making +:setting:`LOG_FORMATTER` point to your new class. + +.. autoclass:: scrapy.logformatter.LogFormatter + :members: + Advanced customization ---------------------- @@ -243,18 +255,18 @@ scrapy.utils.log module when running custom scripts using :class:`~scrapy.crawler.CrawlerRunner`. In that case, its usage is not required but it's recommended. - If you plan on configuring the handlers yourself is still recommended you - call this function, passing ``install_root_handler=False``. Bear in mind - there won't be any log output set by default in that case. + Another option when running custom scripts is to manually configure the logging. + To do this you can use `logging.basicConfig()`_ to set a basic root handler. - To get you started on manually configuring logging's output, you can use - `logging.basicConfig()`_ to set a basic root handler. This is an example - on how to redirect ``INFO`` or higher messages to a file:: + Note that :class:`~scrapy.crawler.CrawlerProcess` automatically calls ``configure_logging``, + so it is recommended to only use `logging.basicConfig()`_ together with + :class:`~scrapy.crawler.CrawlerRunner`. + + This is an example on how to redirect ``INFO`` or higher messages to a file:: import logging from scrapy.utils.log import configure_logging - configure_logging(install_root_handler=False) logging.basicConfig( filename='log.txt', format='%(levelname)s: %(message)s', diff --git a/docs/topics/media-pipeline.rst b/docs/topics/media-pipeline.rst index 327517996..1e0e0f18f 100644 --- a/docs/topics/media-pipeline.rst +++ b/docs/topics/media-pipeline.rst @@ -97,7 +97,6 @@ For Files Pipeline, use:: ITEM_PIPELINES = {'scrapy.pipelines.files.FilesPipeline': 1} - .. note:: You can also use both the Files and Images Pipeline at the same time. @@ -190,7 +189,7 @@ policy:: For more information, see `canned ACLs`_ in the Amazon S3 Developer Guide. -Because Scrapy uses ``boto`` / ``botocore`` internally you can also use other S3-like storages. Storages like +Because Scrapy uses ``botocore`` internally you can also use other S3-like storages. Storages like self-hosted `Minio`_ or `s3.scality`_. All you need to do is set endpoint option in you Scrapy settings:: AWS_ENDPOINT_URL = 'http://minio.example.com:9000' @@ -460,8 +459,9 @@ See here the methods that you can override in your custom Files Pipeline: * ``success`` is a boolean which is ``True`` if the image was downloaded successfully or ``False`` if it failed for some reason - * ``file_info_or_error`` is a dict containing the following keys (if success - is ``True``) or a `Twisted Failure`_ if there was a problem. + * ``file_info_or_error`` is a dict containing the following keys (if + success is ``True``) or a :exc:`~twisted.python.failure.Failure` if + there was a problem. * ``url`` - the url where the file was downloaded from. This is the url of the request returned from the :meth:`~get_media_requests` @@ -576,7 +576,7 @@ See here the methods that you can override in your custom Images Pipeline: Custom Images pipeline example ============================== -Here is a full example of the Images Pipeline whose methods are examplified +Here is a full example of the Images Pipeline whose methods are exemplified above:: import scrapy @@ -596,5 +596,12 @@ above:: item['image_paths'] = image_paths return item -.. _Twisted Failure: https://twistedmatrix.com/documents/current/api/twisted.python.failure.Failure.html + +To enable your custom media pipeline component you must add its class import path to the +:setting:`ITEM_PIPELINES` setting, like in the following example:: + + ITEM_PIPELINES = { + 'myproject.pipelines.MyImagesPipeline': 300 + } + .. _MD5 hash: https://en.wikipedia.org/wiki/MD5 diff --git a/docs/topics/practices.rst b/docs/topics/practices.rst index a6d4f0d6d..e3e8fdc72 100644 --- a/docs/topics/practices.rst +++ b/docs/topics/practices.rst @@ -101,7 +101,7 @@ reactor after ``MySpider`` has finished running. d.addBoth(lambda _: reactor.stop()) reactor.run() # the script will block here until the crawling is finished -.. seealso:: `Twisted Reactor Overview`_. +.. seealso:: :doc:`twisted:core/howto/reactor-basics` .. _run-multiple-spiders: @@ -253,6 +253,5 @@ If you are still unable to prevent your bot getting banned, consider contacting .. _ProxyMesh: https://proxymesh.com/ .. _Google cache: http://www.googleguide.com/cached_pages.html .. _testspiders: https://github.com/scrapinghub/testspiders -.. _Twisted Reactor Overview: https://twistedmatrix.com/documents/current/core/howto/reactor-basics.html .. _Crawlera: https://scrapinghub.com/crawlera .. _scrapoxy: https://scrapoxy.io/ diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index 4e81ce878..8997a7f19 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -24,7 +24,7 @@ below in :ref:`topics-request-response-ref-request-subclasses` and Request objects =============== -.. class:: Request(url[, callback, method='GET', headers, body, cookies, meta, encoding='utf-8', priority=0, dont_filter=False, errback, flags, cb_kwargs]) +.. autoclass:: Request A :class:`Request` object represents an HTTP request, which is usually generated in the Spider and executed by the Downloader, and thus generating @@ -83,17 +83,21 @@ Request objects .. reqmeta:: dont_merge_cookies When some site returns cookies (in a response) those are stored in the - cookies for that domain and will be sent again in future requests. That's - the typical behaviour of any regular web browser. However, if, for some - reason, you want to avoid merging with existing cookies you can instruct - Scrapy to do so by setting the ``dont_merge_cookies`` key to True in the - :attr:`Request.meta`. + cookies for that domain and will be sent again in future requests. + That's the typical behaviour of any regular web browser. - Example of request without merging cookies:: + To create a request that does not send stored cookies and does not + store received cookies, set the ``dont_merge_cookies`` key to ``True`` + in :attr:`request.meta `. - request_with_cookies = Request(url="http://www.example.com", - cookies={'currency': 'USD', 'country': 'UY'}, - meta={'dont_merge_cookies': True}) + Example of a request that sends manually-defined cookies and ignores + cookie storage:: + + Request( + url="http://www.example.com", + cookies={'currency': 'USD', 'country': 'UY'}, + meta={'dont_merge_cookies': True}, + ) For more info see :ref:`cookies-mw`. :type cookies: dict or list @@ -117,8 +121,8 @@ Request objects :param errback: a function that will be called if any exception was raised while processing the request. This includes pages that failed - with 404 HTTP errors and such. It receives a `Twisted Failure`_ instance - as first parameter. + with 404 HTTP errors and such. It receives a + :exc:`~twisted.python.failure.Failure` as first parameter. For more information, see :ref:`topics-request-response-ref-errbacks` below. :type errback: callable @@ -133,7 +137,7 @@ Request objects A string containing the URL of this request. Keep in mind that this attribute contains the escaped URL, so it can differ from the URL passed in - the constructor. + the ``__init__`` method. This attribute is read-only. To change the URL of a Request use :meth:`replace`. @@ -194,6 +198,8 @@ Request objects copied by default (unless new values are given as arguments). See also :ref:`topics-request-response-ref-request-callback-arguments`. + .. automethod:: from_curl + .. _topics-request-response-ref-request-callback-arguments: Passing additional data to callback functions @@ -248,8 +254,8 @@ Using errbacks to catch exceptions in request processing The errback of a request is a function that will be called when an exception is raise while processing it. -It receives a `Twisted Failure`_ instance as first parameter and can be -used to track connection establishment timeouts, DNS errors etc. +It receives a :exc:`~twisted.python.failure.Failure` as first parameter and can +be used to track connection establishment timeouts, DNS errors etc. Here's an example spider logging all errors and catching some specific errors if needed:: @@ -394,7 +400,7 @@ fields with form data from :class:`Response` objects. .. class:: FormRequest(url, [formdata, ...]) - The :class:`FormRequest` class adds a new argument to the constructor. The + The :class:`FormRequest` class adds a new keyword parameter to the ``__init__`` method. The remaining arguments are the same as for the :class:`Request` class and are not documented here. @@ -467,7 +473,7 @@ fields with form data from :class:`Response` objects. :type dont_click: boolean The other parameters of this class method are passed directly to the - :class:`FormRequest` constructor. + :class:`FormRequest` ``__init__`` method. .. versionadded:: 0.10.3 The ``formname`` parameter. @@ -533,24 +539,24 @@ method for this job. Here's an example spider which uses it:: # continue scraping with authenticated session... -JSONRequest +JsonRequest ----------- -The JSONRequest class extends the base :class:`Request` class with functionality for +The JsonRequest class extends the base :class:`Request` class with functionality for dealing with JSON requests. -.. class:: JSONRequest(url, [... data, dumps_kwargs]) +.. class:: JsonRequest(url, [... data, dumps_kwargs]) - The :class:`JSONRequest` class adds two new argument to the constructor. The + The :class:`JsonRequest` class adds two new keyword parameters to the ``__init__`` method. The remaining arguments are the same as for the :class:`Request` class and are not documented here. - Using the :class:`JSONRequest` will set the ``Content-Type`` header to ``application/json`` + Using the :class:`JsonRequest` will set the ``Content-Type`` header to ``application/json`` and ``Accept`` header to ``application/json, text/javascript, */*; q=0.01`` :param data: is any JSON serializable object that needs to be JSON encoded and assigned to body. if :attr:`Request.body` argument is provided this parameter will be ignored. - if :attr:`Request.body` argument is not provided and data argument is provided :attr:`Request.method` will be + if :attr:`Request.body` argument is not provided and data argument is provided :attr:`Request.method` will be set to ``'POST'`` automatically. :type data: JSON serializable object @@ -560,7 +566,7 @@ dealing with JSON requests. .. _json.dumps: https://docs.python.org/3/library/json.html#json.dumps -JSONRequest usage example +JsonRequest usage example ------------------------- Sending a JSON POST request with a JSON payload:: @@ -569,13 +575,13 @@ Sending a JSON POST request with a JSON payload:: 'name1': 'value1', 'name2': 'value2', } - yield JSONRequest(url='http://www.example.com/post/action', data=data) + yield JsonRequest(url='http://www.example.com/post/action', data=data) Response objects ================ -.. class:: Response(url, [status=200, headers=None, body=b'', flags=None, request=None]) +.. autoclass:: Response A :class:`Response` object represents an HTTP response, which is usually downloaded (by the Downloader) and fed to the Spiders for processing. @@ -590,8 +596,8 @@ Response objects (for single valued headers) or lists (for multi-valued headers). :type headers: dict - :param body: the response body. To access the decoded text as str (unicode - in Python 2) you can use ``response.text`` from an encoding-aware + :param body: the response body. To access the decoded text as str you can use + ``response.text`` from an encoding-aware :ref:`Response subclass `, such as :class:`TextResponse`. :type body: bytes @@ -695,6 +701,8 @@ Response objects .. automethod:: Response.follow + .. automethod:: Response.follow_all + .. _urlparse.urljoin: https://docs.python.org/2/library/urlparse.html#urlparse.urljoin @@ -715,7 +723,7 @@ TextResponse objects :class:`Response` class, which is meant to be used only for binary data, such as images, sounds or any media file. - :class:`TextResponse` objects support a new constructor argument, in + :class:`TextResponse` objects support a new ``__init__`` method argument, in addition to the base :class:`Response` objects. The remaining functionality is the same as for the :class:`Response` class and is not documented here. @@ -749,7 +757,7 @@ TextResponse objects A string with the encoding of this response. The encoding is resolved by trying the following mechanisms, in order: - 1. the encoding passed in the constructor ``encoding`` argument + 1. the encoding passed in the ``__init__`` method ``encoding`` argument 2. the encoding declared in the Content-Type HTTP header. If this encoding is not valid (ie. unknown), it is ignored and the next @@ -784,6 +792,8 @@ TextResponse objects .. automethod:: TextResponse.follow + .. automethod:: TextResponse.follow_all + .. method:: TextResponse.body_as_unicode() The same as :attr:`text`, but available as a method. This method is @@ -810,5 +820,4 @@ XmlResponse objects adds encoding auto-discovering support by looking into the XML declaration line. See :attr:`TextResponse.encoding`. -.. _Twisted Failure: https://twistedmatrix.com/documents/current/api/twisted.python.failure.Failure.html .. _bug in lxml: https://bugs.launchpad.net/lxml/+bug/1665241 diff --git a/docs/topics/selectors.rst b/docs/topics/selectors.rst index 282a585d4..8ec758b0e 100644 --- a/docs/topics/selectors.rst +++ b/docs/topics/selectors.rst @@ -51,18 +51,18 @@ Constructing selectors .. highlight:: python Response objects expose a :class:`~scrapy.selector.Selector` instance -on ``.selector`` attribute:: +on ``.selector`` attribute: - >>> response.selector.xpath('//span/text()').get() - 'good' +>>> response.selector.xpath('//span/text()').get() +'good' Querying responses using XPath and CSS is so common that responses include two -more shortcuts: ``response.xpath()`` and ``response.css()``:: +more shortcuts: ``response.xpath()`` and ``response.css()``: - >>> response.xpath('//span/text()').get() - 'good' - >>> response.css('span::text').get() - 'good' +>>> response.xpath('//span/text()').get() +'good' +>>> response.css('span::text').get() +'good' Scrapy selectors are instances of :class:`~scrapy.selector.Selector` class constructed by passing either :class:`~scrapy.http.TextResponse` object or @@ -74,21 +74,21 @@ shortcuts. By using ``response.selector`` or one of these shortcuts you can also ensure the response body is parsed only once. But if required, it is possible to use ``Selector`` directly. -Constructing from text:: +Constructing from text: - >>> from scrapy.selector import Selector - >>> body = 'good' - >>> Selector(text=body).xpath('//span/text()').get() - 'good' +>>> from scrapy.selector import Selector +>>> body = 'good' +>>> Selector(text=body).xpath('//span/text()').get() +'good' Constructing from response - :class:`~scrapy.http.HtmlResponse` is one of -:class:`~scrapy.http.TextResponse` subclasses:: +:class:`~scrapy.http.TextResponse` subclasses: - >>> from scrapy.selector import Selector - >>> from scrapy.http import HtmlResponse - >>> response = HtmlResponse(url='http://example.com', body=body) - >>> Selector(response=response).xpath('//span/text()').get() - 'good' +>>> from scrapy.selector import Selector +>>> from scrapy.http import HtmlResponse +>>> response = HtmlResponse(url='http://example.com', body=body) +>>> Selector(response=response).xpath('//span/text()').get() +'good' ``Selector`` automatically chooses the best parsing rules (XML vs HTML) based on input type. @@ -123,118 +123,118 @@ Since we're dealing with HTML, the selector will automatically use an HTML parse .. highlight:: python So, by looking at the :ref:`HTML code ` of that -page, let's construct an XPath for selecting the text inside the title tag:: +page, let's construct an XPath for selecting the text inside the title tag: - >>> response.xpath('//title/text()') - [] +>>> response.xpath('//title/text()') +[] To actually extract the textual data, you must call the selector ``.get()`` -or ``.getall()`` methods, as follows:: +or ``.getall()`` methods, as follows: - >>> response.xpath('//title/text()').getall() - ['Example website'] - >>> response.xpath('//title/text()').get() - 'Example website' +>>> response.xpath('//title/text()').getall() +['Example website'] +>>> response.xpath('//title/text()').get() +'Example website' ``.get()`` always returns a single result; if there are several matches, content of a first match is returned; if there are no matches, None is returned. ``.getall()`` returns a list with all results. Notice that CSS selectors can select text or attribute nodes using CSS3 -pseudo-elements:: +pseudo-elements: - >>> response.css('title::text').get() - 'Example website' +>>> response.css('title::text').get() +'Example website' As you can see, ``.xpath()`` and ``.css()`` methods return a :class:`~scrapy.selector.SelectorList` instance, which is a list of new -selectors. This API can be used for quickly selecting nested data:: +selectors. This API can be used for quickly selecting nested data: - >>> response.css('img').xpath('@src').getall() - ['image1_thumb.jpg', - 'image2_thumb.jpg', - 'image3_thumb.jpg', - 'image4_thumb.jpg', - 'image5_thumb.jpg'] +>>> response.css('img').xpath('@src').getall() +['image1_thumb.jpg', + 'image2_thumb.jpg', + 'image3_thumb.jpg', + 'image4_thumb.jpg', + 'image5_thumb.jpg'] If you want to extract only the first matched element, you can call the selector ``.get()`` (or its alias ``.extract_first()`` commonly used in -previous Scrapy versions):: +previous Scrapy versions): - >>> response.xpath('//div[@id="images"]/a/text()').get() - 'Name: My image 1 ' +>>> response.xpath('//div[@id="images"]/a/text()').get() +'Name: My image 1 ' -It returns ``None`` if no element was found:: +It returns ``None`` if no element was found: - >>> response.xpath('//div[@id="not-exists"]/text()').get() is None - True +>>> response.xpath('//div[@id="not-exists"]/text()').get() is None +True A default return value can be provided as an argument, to be used instead of ``None``: - >>> response.xpath('//div[@id="not-exists"]/text()').get(default='not-found') - 'not-found' +>>> response.xpath('//div[@id="not-exists"]/text()').get(default='not-found') +'not-found' Instead of using e.g. ``'@src'`` XPath it is possible to query for attributes -using ``.attrib`` property of a :class:`~scrapy.selector.Selector`:: +using ``.attrib`` property of a :class:`~scrapy.selector.Selector`: - >>> [img.attrib['src'] for img in response.css('img')] - ['image1_thumb.jpg', - 'image2_thumb.jpg', - 'image3_thumb.jpg', - 'image4_thumb.jpg', - 'image5_thumb.jpg'] +>>> [img.attrib['src'] for img in response.css('img')] +['image1_thumb.jpg', + 'image2_thumb.jpg', + 'image3_thumb.jpg', + 'image4_thumb.jpg', + 'image5_thumb.jpg'] As a shortcut, ``.attrib`` is also available on SelectorList directly; -it returns attributes for the first matching element:: +it returns attributes for the first matching element: - >>> response.css('img').attrib['src'] - 'image1_thumb.jpg' +>>> response.css('img').attrib['src'] +'image1_thumb.jpg' This is most useful when only a single result is expected, e.g. when selecting -by id, or selecting unique elements on a web page:: +by id, or selecting unique elements on a web page: - >>> response.css('base').attrib['href'] - 'http://example.com/' +>>> response.css('base').attrib['href'] +'http://example.com/' -Now we're going to get the base URL and some image links:: +Now we're going to get the base URL and some image links: - >>> response.xpath('//base/@href').get() - 'http://example.com/' +>>> response.xpath('//base/@href').get() +'http://example.com/' - >>> response.css('base::attr(href)').get() - 'http://example.com/' +>>> response.css('base::attr(href)').get() +'http://example.com/' - >>> response.css('base').attrib['href'] - 'http://example.com/' +>>> response.css('base').attrib['href'] +'http://example.com/' - >>> response.xpath('//a[contains(@href, "image")]/@href').getall() - ['image1.html', - 'image2.html', - 'image3.html', - 'image4.html', - 'image5.html'] +>>> response.xpath('//a[contains(@href, "image")]/@href').getall() +['image1.html', + 'image2.html', + 'image3.html', + 'image4.html', + 'image5.html'] - >>> response.css('a[href*=image]::attr(href)').getall() - ['image1.html', - 'image2.html', - 'image3.html', - 'image4.html', - 'image5.html'] +>>> response.css('a[href*=image]::attr(href)').getall() +['image1.html', + 'image2.html', + 'image3.html', + 'image4.html', + 'image5.html'] - >>> response.xpath('//a[contains(@href, "image")]/img/@src').getall() - ['image1_thumb.jpg', - 'image2_thumb.jpg', - 'image3_thumb.jpg', - 'image4_thumb.jpg', - 'image5_thumb.jpg'] +>>> response.xpath('//a[contains(@href, "image")]/img/@src').getall() +['image1_thumb.jpg', + 'image2_thumb.jpg', + 'image3_thumb.jpg', + 'image4_thumb.jpg', + 'image5_thumb.jpg'] - >>> response.css('a[href*=image] img::attr(src)').getall() - ['image1_thumb.jpg', - 'image2_thumb.jpg', - 'image3_thumb.jpg', - 'image4_thumb.jpg', - 'image5_thumb.jpg'] +>>> response.css('a[href*=image] img::attr(src)').getall() +['image1_thumb.jpg', + 'image2_thumb.jpg', + 'image3_thumb.jpg', + 'image4_thumb.jpg', + 'image5_thumb.jpg'] .. _topics-selectors-css-extensions: @@ -259,47 +259,47 @@ that Scrapy (parsel) implements a couple of **non-standard pseudo-elements**: Examples: -* ``title::text`` selects children text nodes of a descendant ```` element:: +* ``title::text`` selects children text nodes of a descendant ``<title>`` element: - >>> response.css('title::text').get() - 'Example website' +>>> response.css('title::text').get() +'Example website' -* ``*::text`` selects all descendant text nodes of the current selector context:: +* ``*::text`` selects all descendant text nodes of the current selector context: - >>> response.css('#images *::text').getall() - ['\n ', - 'Name: My image 1 ', - '\n ', - 'Name: My image 2 ', - '\n ', - 'Name: My image 3 ', - '\n ', - 'Name: My image 4 ', - '\n ', - 'Name: My image 5 ', - '\n '] +>>> response.css('#images *::text').getall() +['\n ', + 'Name: My image 1 ', + '\n ', + 'Name: My image 2 ', + '\n ', + 'Name: My image 3 ', + '\n ', + 'Name: My image 4 ', + '\n ', + 'Name: My image 5 ', + '\n '] * ``foo::text`` returns no results if ``foo`` element exists, but contains - no text (i.e. text is empty):: + no text (i.e. text is empty): - >>> response.css('img::text').getall() - [] +>>> response.css('img::text').getall() +[] This means ``.css('foo::text').get()`` could return None even if an element - exists. Use ``default=''`` if you always want a string:: + exists. Use ``default=''`` if you always want a string: - >>> response.css('img::text').get() - >>> response.css('img::text').get(default='') - '' +>>> response.css('img::text').get() +>>> response.css('img::text').get(default='') +'' -* ``a::attr(href)`` selects the *href* attribute value of descendant links:: +* ``a::attr(href)`` selects the *href* attribute value of descendant links: - >>> response.css('a::attr(href)').getall() - ['image1.html', - 'image2.html', - 'image3.html', - 'image4.html', - 'image5.html'] +>>> response.css('a::attr(href)').getall() +['image1.html', + 'image2.html', + 'image3.html', + 'image4.html', + 'image5.html'] .. note:: See also: :ref:`selecting-attributes`. @@ -318,25 +318,24 @@ Nesting selectors The selection methods (``.xpath()`` or ``.css()``) return a list of selectors of the same type, so you can call the selection methods for those selectors -too. Here's an example:: +too. Here's an example: - >>> links = response.xpath('//a[contains(@href, "image")]') - >>> links.getall() - ['<a href="image1.html">Name: My image 1 <br><img src="image1_thumb.jpg"></a>', - '<a href="image2.html">Name: My image 2 <br><img src="image2_thumb.jpg"></a>', - '<a href="image3.html">Name: My image 3 <br><img src="image3_thumb.jpg"></a>', - '<a href="image4.html">Name: My image 4 <br><img src="image4_thumb.jpg"></a>', - '<a href="image5.html">Name: My image 5 <br><img src="image5_thumb.jpg"></a>'] +>>> links = response.xpath('//a[contains(@href, "image")]') +>>> links.getall() +['<a href="image1.html">Name: My image 1 <br><img src="image1_thumb.jpg"></a>', + '<a href="image2.html">Name: My image 2 <br><img src="image2_thumb.jpg"></a>', + '<a href="image3.html">Name: My image 3 <br><img src="image3_thumb.jpg"></a>', + '<a href="image4.html">Name: My image 4 <br><img src="image4_thumb.jpg"></a>', + '<a href="image5.html">Name: My image 5 <br><img src="image5_thumb.jpg"></a>'] - >>> for index, link in enumerate(links): - ... args = (index, link.xpath('@href').get(), link.xpath('img/@src').get()) - ... print('Link number %d points to url %r and image %r' % args) - - Link number 0 points to url 'image1.html' and image 'image1_thumb.jpg' - Link number 1 points to url 'image2.html' and image 'image2_thumb.jpg' - Link number 2 points to url 'image3.html' and image 'image3_thumb.jpg' - Link number 3 points to url 'image4.html' and image 'image4_thumb.jpg' - Link number 4 points to url 'image5.html' and image 'image5_thumb.jpg' +>>> for index, link in enumerate(links): +... args = (index, link.xpath('@href').get(), link.xpath('img/@src').get()) +... print('Link number %d points to url %r and image %r' % args) +Link number 0 points to url 'image1.html' and image 'image1_thumb.jpg' +Link number 1 points to url 'image2.html' and image 'image2_thumb.jpg' +Link number 2 points to url 'image3.html' and image 'image3_thumb.jpg' +Link number 3 points to url 'image4.html' and image 'image4_thumb.jpg' +Link number 4 points to url 'image5.html' and image 'image5_thumb.jpg' .. _selecting-attributes: @@ -344,42 +343,42 @@ Selecting element attributes ---------------------------- There are several ways to get a value of an attribute. First, one can use -XPath syntax:: +XPath syntax: - >>> response.xpath("//a/@href").getall() - ['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html'] +>>> response.xpath("//a/@href").getall() +['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html'] XPath syntax has a few advantages: it is a standard XPath feature, and ``@attributes`` can be used in other parts of an XPath expression - e.g. it is possible to filter by attribute value. Scrapy also provides an extension to CSS selectors (``::attr(...)``) -which allows to get attribute values:: +which allows to get attribute values: - >>> response.css('a::attr(href)').getall() - ['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html'] +>>> response.css('a::attr(href)').getall() +['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html'] In addition to that, there is a ``.attrib`` property of Selector. You can use it if you prefer to lookup attributes in Python -code, without using XPaths or CSS extensions:: +code, without using XPaths or CSS extensions: - >>> [a.attrib['href'] for a in response.css('a')] - ['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html'] +>>> [a.attrib['href'] for a in response.css('a')] +['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html'] This property is also available on SelectorList; it returns a dictionary with attributes of a first matching element. It is convenient to use when a selector is expected to give a single result (e.g. when selecting by element -ID, or when selecting an unique element on a page):: +ID, or when selecting an unique element on a page): - >>> response.css('base').attrib - {'href': 'http://example.com/'} - >>> response.css('base').attrib['href'] - 'http://example.com/' +>>> response.css('base').attrib +{'href': 'http://example.com/'} +>>> response.css('base').attrib['href'] +'http://example.com/' -``.attrib`` property of an empty SelectorList is empty:: +``.attrib`` property of an empty SelectorList is empty: - >>> response.css('foo').attrib - {} +>>> response.css('foo').attrib +{} Using selectors with regular expressions ---------------------------------------- @@ -390,21 +389,21 @@ data using regular expressions. However, unlike using ``.xpath()`` or can't construct nested ``.re()`` calls. Here's an example used to extract image names from the :ref:`HTML code -<topics-selectors-htmlcode>` above:: +<topics-selectors-htmlcode>` above: - >>> response.xpath('//a[contains(@href, "image")]/text()').re(r'Name:\s*(.*)') - ['My image 1', - 'My image 2', - 'My image 3', - 'My image 4', - 'My image 5'] +>>> response.xpath('//a[contains(@href, "image")]/text()').re(r'Name:\s*(.*)') +['My image 1', + 'My image 2', + 'My image 3', + 'My image 4', + 'My image 5'] There's an additional helper reciprocating ``.get()`` (and its alias ``.extract_first()``) for ``.re()``, named ``.re_first()``. -Use it to extract just the first matching string:: +Use it to extract just the first matching string: - >>> response.xpath('//a[contains(@href, "image")]/text()').re_first(r'Name:\s*(.*)') - 'My image 1' +>>> response.xpath('//a[contains(@href, "image")]/text()').re_first(r'Name:\s*(.*)') +'My image 1' .. _old-extraction-api: @@ -422,28 +421,28 @@ and readable code. The following examples show how these methods map to each other. -1. ``SelectorList.get()`` is the same as ``SelectorList.extract_first()``:: +1. ``SelectorList.get()`` is the same as ``SelectorList.extract_first()``: - >>> response.css('a::attr(href)').get() - 'image1.html' - >>> response.css('a::attr(href)').extract_first() - 'image1.html' + >>> response.css('a::attr(href)').get() + 'image1.html' + >>> response.css('a::attr(href)').extract_first() + 'image1.html' -2. ``SelectorList.getall()`` is the same as ``SelectorList.extract()``:: +2. ``SelectorList.getall()`` is the same as ``SelectorList.extract()``: - >>> response.css('a::attr(href)').getall() - ['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html'] - >>> response.css('a::attr(href)').extract() - ['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html'] + >>> response.css('a::attr(href)').getall() + ['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html'] + >>> response.css('a::attr(href)').extract() + ['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html'] -3. ``Selector.get()`` is the same as ``Selector.extract()``:: +3. ``Selector.get()`` is the same as ``Selector.extract()``: - >>> response.css('a::attr(href)')[0].get() - 'image1.html' - >>> response.css('a::attr(href)')[0].extract() - 'image1.html' + >>> response.css('a::attr(href)')[0].get() + 'image1.html' + >>> response.css('a::attr(href)')[0].extract() + 'image1.html' -4. For consistency, there is also ``Selector.getall()``, which returns a list:: +4. For consistency, there is also ``Selector.getall()``, which returns a list: >>> response.css('a::attr(href)')[0].getall() ['image1.html'] @@ -481,26 +480,26 @@ with ``/``, that XPath will be absolute to the document and not relative to the ``Selector`` you're calling it from. For example, suppose you want to extract all ``<p>`` elements inside ``<div>`` -elements. First, you would get all ``<div>`` elements:: +elements. First, you would get all ``<div>`` elements: - >>> divs = response.xpath('//div') +>>> divs = response.xpath('//div') At first, you may be tempted to use the following approach, which is wrong, as it actually extracts all ``<p>`` elements from the document, not only those -inside ``<div>`` elements:: +inside ``<div>`` elements: - >>> for p in divs.xpath('//p'): # this is wrong - gets all <p> from the whole document - ... print(p.get()) +>>> for p in divs.xpath('//p'): # this is wrong - gets all <p> from the whole document +... print(p.get()) -This is the proper way to do it (note the dot prefixing the ``.//p`` XPath):: +This is the proper way to do it (note the dot prefixing the ``.//p`` XPath): - >>> for p in divs.xpath('.//p'): # extracts all <p> inside - ... print(p.get()) +>>> for p in divs.xpath('.//p'): # extracts all <p> inside +... print(p.get()) -Another common case would be to extract all direct ``<p>`` children:: +Another common case would be to extract all direct ``<p>`` children: - >>> for p in divs.xpath('p'): - ... print(p.get()) +>>> for p in divs.xpath('p'): +... print(p.get()) For more details about relative XPaths see the `Location Paths`_ section in the XPath specification. @@ -521,12 +520,12 @@ for that you may end up with more elements that you want, if they have a differe class name that shares the string ``someclass``. As it turns out, Scrapy selectors allow you to chain selectors, so most of the time -you can just select by class using CSS and then switch to XPath when needed:: +you can just select by class using CSS and then switch to XPath when needed: - >>> from scrapy import Selector - >>> sel = Selector(text='<div class="hero shout"><time datetime="2014-07-23 19:00">Special date</time></div>') - >>> sel.css('.shout').xpath('./time/@datetime').getall() - ['2014-07-23 19:00'] +>>> from scrapy import Selector +>>> sel = Selector(text='<div class="hero shout"><time datetime="2014-07-23 19:00">Special date</time></div>') +>>> sel.css('.shout').xpath('./time/@datetime').getall() +['2014-07-23 19:00'] This is cleaner than using the verbose XPath trick shown above. Just remember to use the ``.`` in the XPath expressions that will follow. @@ -538,41 +537,41 @@ Beware of the difference between //node[1] and (//node)[1] ``(//node)[1]`` selects all the nodes in the document, and then gets only the first of them. -Example:: +Example: - >>> from scrapy import Selector - >>> sel = Selector(text=""" - ....: <ul class="list"> - ....: <li>1</li> - ....: <li>2</li> - ....: <li>3</li> - ....: </ul> - ....: <ul class="list"> - ....: <li>4</li> - ....: <li>5</li> - ....: <li>6</li> - ....: </ul>""") - >>> xp = lambda x: sel.xpath(x).getall() +>>> from scrapy import Selector +>>> sel = Selector(text=""" +....: <ul class="list"> +....: <li>1</li> +....: <li>2</li> +....: <li>3</li> +....: </ul> +....: <ul class="list"> +....: <li>4</li> +....: <li>5</li> +....: <li>6</li> +....: </ul>""") +>>> xp = lambda x: sel.xpath(x).getall() -This gets all first ``<li>`` elements under whatever it is its parent:: +This gets all first ``<li>`` elements under whatever it is its parent: - >>> xp("//li[1]") - ['<li>1</li>', '<li>4</li>'] +>>> xp("//li[1]") +['<li>1</li>', '<li>4</li>'] -And this gets the first ``<li>`` element in the whole document:: +And this gets the first ``<li>`` element in the whole document: - >>> xp("(//li)[1]") - ['<li>1</li>'] +>>> xp("(//li)[1]") +['<li>1</li>'] -This gets all first ``<li>`` elements under an ``<ul>`` parent:: +This gets all first ``<li>`` elements under an ``<ul>`` parent: - >>> xp("//ul/li[1]") - ['<li>1</li>', '<li>4</li>'] +>>> xp("//ul/li[1]") +['<li>1</li>', '<li>4</li>'] -And this gets the first ``<li>`` element under an ``<ul>`` parent in the whole document:: +And this gets the first ``<li>`` element under an ``<ul>`` parent in the whole document: - >>> xp("(//ul/li)[1]") - ['<li>1</li>'] +>>> xp("(//ul/li)[1]") +['<li>1</li>'] Using text nodes in a condition ------------------------------- @@ -584,34 +583,34 @@ This is because the expression ``.//text()`` yields a collection of text element And when a node-set is converted to a string, which happens when it is passed as argument to a string function like ``contains()`` or ``starts-with()``, it results in the text for the first element only. -Example:: +Example: - >>> from scrapy import Selector - >>> sel = Selector(text='<a href="#">Click here to go to the <strong>Next Page</strong></a>') +>>> from scrapy import Selector +>>> sel = Selector(text='<a href="#">Click here to go to the <strong>Next Page</strong></a>') -Converting a *node-set* to string:: +Converting a *node-set* to string: - >>> sel.xpath('//a//text()').getall() # take a peek at the node-set - ['Click here to go to the ', 'Next Page'] - >>> sel.xpath("string(//a[1]//text())").getall() # convert it to string - ['Click here to go to the '] +>>> sel.xpath('//a//text()').getall() # take a peek at the node-set +['Click here to go to the ', 'Next Page'] +>>> sel.xpath("string(//a[1]//text())").getall() # convert it to string +['Click here to go to the '] -A *node* converted to a string, however, puts together the text of itself plus of all its descendants:: +A *node* converted to a string, however, puts together the text of itself plus of all its descendants: - >>> sel.xpath("//a[1]").getall() # select the first node - ['<a href="#">Click here to go to the <strong>Next Page</strong></a>'] - >>> sel.xpath("string(//a[1])").getall() # convert it to string - ['Click here to go to the Next Page'] +>>> sel.xpath("//a[1]").getall() # select the first node +['<a href="#">Click here to go to the <strong>Next Page</strong></a>'] +>>> sel.xpath("string(//a[1])").getall() # convert it to string +['Click here to go to the Next Page'] -So, using the ``.//text()`` node-set won't select anything in this case:: +So, using the ``.//text()`` node-set won't select anything in this case: - >>> sel.xpath("//a[contains(.//text(), 'Next Page')]").getall() - [] +>>> sel.xpath("//a[contains(.//text(), 'Next Page')]").getall() +[] -But using the ``.`` to mean the node, works:: +But using the ``.`` to mean the node, works: - >>> sel.xpath("//a[contains(., 'Next Page')]").getall() - ['<a href="#">Click here to go to the <strong>Next Page</strong></a>'] +>>> sel.xpath("//a[contains(., 'Next Page')]").getall() +['<a href="#">Click here to go to the <strong>Next Page</strong></a>'] .. _`XPath string function`: https://www.w3.org/TR/xpath/#section-String-Functions @@ -627,17 +626,17 @@ some arguments in your queries with placeholders like ``?``, which are then substituted with values passed with the query. Here's an example to match an element based on its "id" attribute value, -without hard-coding it (that was shown previously):: +without hard-coding it (that was shown previously): - >>> # `$val` used in the expression, a `val` argument needs to be passed - >>> response.xpath('//div[@id=$val]/a/text()', val='images').get() - 'Name: My image 1 ' +>>> # `$val` used in the expression, a `val` argument needs to be passed +>>> response.xpath('//div[@id=$val]/a/text()', val='images').get() +'Name: My image 1 ' Here's another example, to find the "id" attribute of a ``<div>`` tag containing -five ``<a>`` children (here we pass the value ``5`` as an integer):: +five ``<a>`` children (here we pass the value ``5`` as an integer): - >>> response.xpath('//div[count(a)=$cnt]/@id', cnt=5).get() - 'images' +>>> response.xpath('//div[count(a)=$cnt]/@id', cnt=5).get() +'images' All variable references must have a binding value when calling ``.xpath()`` (otherwise you'll get a ``ValueError: XPath error:`` exception). @@ -687,19 +686,19 @@ You can see several namespace declarations including a default .. highlight:: python Once in the shell we can try selecting all ``<link>`` objects and see that it -doesn't work (because the Atom XML namespace is obfuscating those nodes):: +doesn't work (because the Atom XML namespace is obfuscating those nodes): - >>> response.xpath("//link") - [] +>>> response.xpath("//link") +[] But once we call the :meth:`Selector.remove_namespaces` method, all -nodes can be accessed directly by their names:: +nodes can be accessed directly by their names: - >>> response.selector.remove_namespaces() - >>> response.xpath("//link") - [<Selector xpath='//link' data='<link rel="alternate" type="text/html" h'>, - <Selector xpath='//link' data='<link rel="next" type="application/atom+'>, - ... +>>> response.selector.remove_namespaces() +>>> response.xpath("//link") +[<Selector xpath='//link' data='<link rel="alternate" type="text/html" h'>, + <Selector xpath='//link' data='<link rel="next" type="application/atom+'>, + ... If you wonder why the namespace removal procedure isn't always called by default instead of having to call it manually, this is because of two reasons, which, in order @@ -734,26 +733,25 @@ Regular expressions The ``test()`` function, for example, can prove quite useful when XPath's ``starts-with()`` or ``contains()`` are not sufficient. -Example selecting links in list item with a "class" attribute ending with a digit:: +Example selecting links in list item with a "class" attribute ending with a digit: - >>> from scrapy import Selector - >>> doc = u""" - ... <div> - ... <ul> - ... <li class="item-0"><a href="link1.html">first item</a></li> - ... <li class="item-1"><a href="link2.html">second item</a></li> - ... <li class="item-inactive"><a href="link3.html">third item</a></li> - ... <li class="item-1"><a href="link4.html">fourth item</a></li> - ... <li class="item-0"><a href="link5.html">fifth item</a></li> - ... </ul> - ... </div> - ... """ - >>> sel = Selector(text=doc, type="html") - >>> sel.xpath('//li//@href').getall() - ['link1.html', 'link2.html', 'link3.html', 'link4.html', 'link5.html'] - >>> sel.xpath('//li[re:test(@class, "item-\d$")]//@href').getall() - ['link1.html', 'link2.html', 'link4.html', 'link5.html'] - >>> +>>> from scrapy import Selector +>>> doc = u""" +... <div> +... <ul> +... <li class="item-0"><a href="link1.html">first item</a></li> +... <li class="item-1"><a href="link2.html">second item</a></li> +... <li class="item-inactive"><a href="link3.html">third item</a></li> +... <li class="item-1"><a href="link4.html">fourth item</a></li> +... <li class="item-0"><a href="link5.html">fifth item</a></li> +... </ul> +... </div> +... """ +>>> sel = Selector(text=doc, type="html") +>>> sel.xpath('//li//@href').getall() +['link1.html', 'link2.html', 'link3.html', 'link4.html', 'link5.html'] +>>> sel.xpath('//li[re:test(@class, "item-\d$")]//@href').getall() +['link1.html', 'link2.html', 'link4.html', 'link5.html'] .. warning:: C library ``libxslt`` doesn't natively support EXSLT regular expressions so `lxml`_'s implementation uses hooks to Python's ``re`` module. @@ -849,7 +847,6 @@ with groups of itemscopes and corresponding itemprops:: current scope: ['http://schema.org/Rating'] properties: ['worstRating', 'ratingValue', 'bestRating'] - >>> Here we first iterate over ``itemscope`` elements, and for each one, we look for all ``itemprops`` elements and exclude those that are themselves @@ -877,15 +874,15 @@ For the following HTML:: .. highlight:: python -You can use it like this:: +You can use it like this: - >>> response.xpath('//p[has-class("foo")]') - [<Selector xpath='//p[has-class("foo")]' data='<p class="foo bar-baz">First</p>'>, - <Selector xpath='//p[has-class("foo")]' data='<p class="foo">Second</p>'>] - >>> response.xpath('//p[has-class("foo", "bar-baz")]') - [<Selector xpath='//p[has-class("foo", "bar-baz")]' data='<p class="foo bar-baz">First</p>'>] - >>> response.xpath('//p[has-class("foo", "bar")]') - [] +>>> response.xpath('//p[has-class("foo")]') +[<Selector xpath='//p[has-class("foo")]' data='<p class="foo bar-baz">First</p>'>, + <Selector xpath='//p[has-class("foo")]' data='<p class="foo">Second</p>'>] +>>> response.xpath('//p[has-class("foo", "bar-baz")]') +[<Selector xpath='//p[has-class("foo", "bar-baz")]' data='<p class="foo bar-baz">First</p>'>] +>>> response.xpath('//p[has-class("foo", "bar")]') +[] So XPath ``//p[has-class("foo", "bar-baz")]`` is roughly equivalent to CSS ``p.foo.bar-baz``. Please note, that it is slower in most of the cases, diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index fd46c614e..f4c3494f5 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -160,6 +160,27 @@ to any particular component. In that case the module of that component will be shown, typically an extension, middleware or pipeline. It also means that the component must be enabled in order for the setting to have any effect. +.. setting:: ASYNCIO_REACTOR + +ASYNCIO_REACTOR +--------------- + +Default: ``False`` + +Whether to install and require the Twisted reactor that uses the asyncio loop. + +When this option is set to ``True``, Scrapy will require +:class:`~twisted.internet.asyncioreactor.AsyncioSelectorReactor`. It will +install this reactor if no reactor is installed yet, such as when using the +``scrapy`` script or :class:`~scrapy.crawler.CrawlerProcess`. If you are using +:class:`~scrapy.crawler.CrawlerRunner`, you need to install the correct reactor +manually. If a different reactor is installed outside Scrapy, it will raise an +exception. + +The default value for this option is currently ``False`` to maintain backward +compatibility and avoid possible problems caused by using a different Twisted +reactor. + .. setting:: AWS_ACCESS_KEY_ID AWS_ACCESS_KEY_ID @@ -188,7 +209,6 @@ AWS_ENDPOINT_URL Default: ``None`` Endpoint URL used for S3-like storage, for example Minio or s3.scality. -Only supported with ``botocore`` library. .. setting:: AWS_USE_SSL @@ -199,7 +219,6 @@ Default: ``None`` Use this option if you want to disable SSL connection for communication with S3 or S3-like storage. By default SSL will be used. -Only supported with ``botocore`` library. .. setting:: AWS_VERIFY @@ -209,7 +228,7 @@ AWS_VERIFY Default: ``None`` Verify SSL connection between Scrapy and S3 or S3-like storage. By default -SSL verification will occur. Only supported with ``botocore`` library. +SSL verification will occur. .. setting:: AWS_REGION_NAME @@ -219,7 +238,6 @@ AWS_REGION_NAME Default: ``None`` The name of the region associated with the AWS client. -Only supported with ``botocore`` library. .. setting:: BOT_NAME @@ -229,8 +247,7 @@ BOT_NAME Default: ``'scrapybot'`` The name of the bot implemented by this Scrapy project (also known as the -project name). This will be used to construct the User-Agent by default, and -also for logging. +project name). This name will be used for the logging too. It's automatically populated with your project name when you create your project with the :command:`startproject` command. @@ -440,9 +457,29 @@ or even enable client-side authentication (and various other things). which uses the platform's certificates to validate remote endpoints. **This is only available if you use Twisted>=14.0.** -If you do use a custom ContextFactory, make sure it accepts a ``method`` -parameter at init (this is the ``OpenSSL.SSL`` method mapping -:setting:`DOWNLOADER_CLIENT_TLS_METHOD`). +If you do use a custom ContextFactory, make sure its ``__init__`` method +accepts a ``method`` parameter (this is the ``OpenSSL.SSL`` method mapping +:setting:`DOWNLOADER_CLIENT_TLS_METHOD`), a ``tls_verbose_logging`` +parameter (``bool``) and a ``tls_ciphers`` parameter (see +:setting:`DOWNLOADER_CLIENT_TLS_CIPHERS`). + +.. setting:: DOWNLOADER_CLIENT_TLS_CIPHERS + +DOWNLOADER_CLIENT_TLS_CIPHERS +----------------------------- + +Default: ``'DEFAULT'`` + +Use this setting to customize the TLS/SSL ciphers used by the default +HTTP/1.1 downloader. + +The setting should contain a string in the `OpenSSL cipher list format`_, +these ciphers will be used as client ciphers. Changing this setting may be +necessary to access certain HTTPS websites: for example, you may need to use +``'DEFAULT:!DH'`` for a website with weak DH parameters or enable a +specific cipher that is not included in ``DEFAULT`` if a website requires it. + +.. _OpenSSL cipher list format: https://www.openssl.org/docs/manmaster/man1/ciphers.html#CIPHER-LIST-FORMAT .. setting:: DOWNLOADER_CLIENT_TLS_METHOD @@ -470,6 +507,20 @@ This setting must be one of these string values: We recommend that you use PyOpenSSL>=0.13 and Twisted>=0.13 or above (Twisted>=14.0 if you can). +.. setting:: DOWNLOADER_CLIENT_TLS_VERBOSE_LOGGING + +DOWNLOADER_CLIENT_TLS_VERBOSE_LOGGING +------------------------------------- + +Default: ``False`` + +Setting this to ``True`` will enable DEBUG level messages about TLS connection +parameters after establishing HTTPS connections. The kind of information logged +depends on the versions of OpenSSL and pyOpenSSL. + +This setting is only used for the default +:setting:`DOWNLOADER_CLIENTCONTEXTFACTORY`. + .. setting:: DOWNLOADER_MIDDLEWARES DOWNLOADER_MIDDLEWARES @@ -762,6 +813,7 @@ Default: ``True`` Whether or not to use passive mode when initiating FTP transfers. +.. reqmeta:: ftp_password .. setting:: FTP_PASSWORD FTP_PASSWORD @@ -780,6 +832,7 @@ in ``Request`` meta. .. _RFC 1635: https://tools.ietf.org/html/rfc1635 +.. reqmeta:: ftp_user .. setting:: FTP_USER FTP_USER @@ -852,7 +905,7 @@ LOG_FORMAT Default: ``'%(asctime)s [%(name)s] %(levelname)s: %(message)s'`` -String for formatting log messsages. Refer to the `Python logging documentation`_ for the whole list of available +String for formatting log messages. Refer to the `Python logging documentation`_ for the whole list of available placeholders. .. _Python logging documentation: https://docs.python.org/2/library/logging.html#logrecord-attributes @@ -870,6 +923,15 @@ directives. .. _Python datetime documentation: https://docs.python.org/2/library/datetime.html#strftime-and-strptime-behavior +.. setting:: LOG_FORMATTER + +LOG_FORMATTER +------------- + +Default: :class:`scrapy.logformatter.LogFormatter` + +The class to use for :ref:`formatting log messages <custom-log-formats>` for different actions. + .. setting:: LOG_LEVEL LOG_LEVEL @@ -908,8 +970,8 @@ LOGSTATS_INTERVAL Default: ``60.0`` -The interval (in seconds) between each logging printout of the stats -by :class:`~extensions.logstats.LogStats`. +The interval (in seconds) between each logging printout of the stats +by :class:`~scrapy.extensions.logstats.LogStats`. .. setting:: MEMDEBUG_ENABLED @@ -1117,6 +1179,28 @@ If enabled, Scrapy will respect robots.txt policies. For more information see this option is enabled by default in settings.py file generated by ``scrapy startproject`` command. +.. setting:: ROBOTSTXT_PARSER + +ROBOTSTXT_PARSER +---------------- + +Default: ``'scrapy.robotstxt.ProtegoRobotParser'`` + +The parser backend to use for parsing ``robots.txt`` files. For more information see +:ref:`topics-dlmw-robots`. + +.. setting:: ROBOTSTXT_USER_AGENT + +ROBOTSTXT_USER_AGENT +^^^^^^^^^^^^^^^^^^^^ + +Default: ``None`` + +The user agent string to use for matching in the robots.txt file. If ``None``, +the User-Agent header you are sending with the request or the +:setting:`USER_AGENT` setting (in that order) will be used for determining +the user agent to use in the robots.txt file. + .. setting:: SCHEDULER SCHEDULER @@ -1178,6 +1262,17 @@ Type of priority queue used by the scheduler. Another available type is domains in parallel. But currently ``scrapy.pqueues.DownloaderAwarePriorityQueue`` does not work together with :setting:`CONCURRENT_REQUESTS_PER_IP`. +.. setting:: SCRAPER_SLOT_MAX_ACTIVE_SIZE + +SCRAPER_SLOT_MAX_ACTIVE_SIZE +---------------------------- +Default: ``5_000_000`` + +Soft limit (in bytes) for response data being processed. + +While the sum of the sizes of all responses being processed is above this value, +Scrapy does not process new requests. + .. setting:: SPIDER_CONTRACTS SPIDER_CONTRACTS @@ -1201,7 +1296,7 @@ Default:: 'scrapy.contracts.default.ScrapesContract': 3, } -A dict containing the scrapy contracts enabled by default in Scrapy. You should +A dict containing the Scrapy contracts enabled by default in Scrapy. You should never modify this setting in your project, modify :setting:`SPIDER_CONTRACTS` instead. For more info see :ref:`topics-contracts`. @@ -1232,7 +1327,7 @@ SPIDER_LOADER_WARN_ONLY Default: ``False`` -By default, when scrapy tries to import spider classes from :setting:`SPIDER_MODULES`, +By default, when Scrapy tries to import spider classes from :setting:`SPIDER_MODULES`, it will fail loudly if there is any ``ImportError`` exception. But you can choose to silence this exception and turn it into a simple warning by setting ``SPIDER_LOADER_WARN_ONLY = True``. @@ -1375,7 +1470,10 @@ USER_AGENT Default: ``"Scrapy/VERSION (+https://scrapy.org)"`` -The default User-Agent to use when crawling, unless overridden. +The default User-Agent to use when crawling, unless overridden. This user agent is +also used by :class:`~scrapy.downloadermiddlewares.robotstxt.RobotsTxtMiddleware` +if :setting:`ROBOTSTXT_USER_AGENT` setting is ``None`` and +there is no overridding User-Agent header specified for the request. Settings documented elsewhere: diff --git a/docs/topics/shell.rst b/docs/topics/shell.rst index 68a0b19b5..3cf8311a6 100644 --- a/docs/topics/shell.rst +++ b/docs/topics/shell.rst @@ -31,7 +31,7 @@ for more info. Scrapy also has support for `bpython`_, and will try to use it where `IPython`_ is unavailable. -Through scrapy's settings you can configure it to use any one of +Through Scrapy's settings you can configure it to use any one of ``ipython``, ``bpython`` or the standard ``python`` shell, regardless of which are installed. This is done by setting the ``SCRAPY_PYTHON_SHELL`` environment variable; or by defining it in your :ref:`scrapy.cfg <topics-config-settings>`:: @@ -177,47 +177,46 @@ all start with the ``[s]`` prefix):: >>> -After that, we can start playing with the objects:: +After that, we can start playing with the objects: - >>> response.xpath('//title/text()').get() - 'Scrapy | A Fast and Powerful Scraping and Web Crawling Framework' +>>> response.xpath('//title/text()').get() +'Scrapy | A Fast and Powerful Scraping and Web Crawling Framework' - >>> fetch("https://reddit.com") +>>> fetch("https://reddit.com") - >>> response.xpath('//title/text()').get() - 'reddit: the front page of the internet' +>>> response.xpath('//title/text()').get() +'reddit: the front page of the internet' - >>> request = request.replace(method="POST") +>>> request = request.replace(method="POST") - >>> fetch(request) +>>> fetch(request) - >>> response.status - 404 +>>> response.status +404 - >>> from pprint import pprint +>>> from pprint import pprint - >>> pprint(response.headers) - {'Accept-Ranges': ['bytes'], - 'Cache-Control': ['max-age=0, must-revalidate'], - 'Content-Type': ['text/html; charset=UTF-8'], - 'Date': ['Thu, 08 Dec 2016 16:21:19 GMT'], - 'Server': ['snooserv'], - 'Set-Cookie': ['loid=KqNLou0V9SKMX4qb4n; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure', - 'loidcreated=2016-12-08T16%3A21%3A19.445Z; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure', - 'loid=vi0ZVe4NkxNWdlH7r7; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure', - 'loidcreated=2016-12-08T16%3A21%3A19.459Z; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure'], - 'Vary': ['accept-encoding'], - 'Via': ['1.1 varnish'], - 'X-Cache': ['MISS'], - 'X-Cache-Hits': ['0'], - 'X-Content-Type-Options': ['nosniff'], - 'X-Frame-Options': ['SAMEORIGIN'], - 'X-Moose': ['majestic'], - 'X-Served-By': ['cache-cdg8730-CDG'], - 'X-Timer': ['S1481214079.394283,VS0,VE159'], - 'X-Ua-Compatible': ['IE=edge'], - 'X-Xss-Protection': ['1; mode=block']} - >>> +>>> pprint(response.headers) +{'Accept-Ranges': ['bytes'], + 'Cache-Control': ['max-age=0, must-revalidate'], + 'Content-Type': ['text/html; charset=UTF-8'], + 'Date': ['Thu, 08 Dec 2016 16:21:19 GMT'], + 'Server': ['snooserv'], + 'Set-Cookie': ['loid=KqNLou0V9SKMX4qb4n; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure', + 'loidcreated=2016-12-08T16%3A21%3A19.445Z; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure', + 'loid=vi0ZVe4NkxNWdlH7r7; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure', + 'loidcreated=2016-12-08T16%3A21%3A19.459Z; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure'], + 'Vary': ['accept-encoding'], + 'Via': ['1.1 varnish'], + 'X-Cache': ['MISS'], + 'X-Cache-Hits': ['0'], + 'X-Content-Type-Options': ['nosniff'], + 'X-Frame-Options': ['SAMEORIGIN'], + 'X-Moose': ['majestic'], + 'X-Served-By': ['cache-cdg8730-CDG'], + 'X-Timer': ['S1481214079.394283,VS0,VE159'], + 'X-Ua-Compatible': ['IE=edge'], + 'X-Xss-Protection': ['1; mode=block']} .. _topics-shell-inspect-response: @@ -263,16 +262,16 @@ When you run the spider, you will get something similar to this:: >>> response.url 'http://example.org' -Then, you can check if the extraction code is working:: +Then, you can check if the extraction code is working: - >>> response.xpath('//h1[@class="fn"]') - [] +>>> response.xpath('//h1[@class="fn"]') +[] Nope, it doesn't. So you can open the response in your web browser and see if -it's the response you were expecting:: +it's the response you were expecting: - >>> view(response) - True +>>> view(response) +True Finally you hit Ctrl-D (or Ctrl-Z in Windows) to exit the shell and resume the crawling:: diff --git a/docs/topics/signals.rst b/docs/topics/signals.rst index ff07b9d55..3f29aa323 100644 --- a/docs/topics/signals.rst +++ b/docs/topics/signals.rst @@ -50,10 +50,10 @@ Here is a simple example showing how you can catch signals and perform some acti Deferred signal handlers ======================== -Some signals support returning `Twisted deferreds`_ from their handlers, see -the :ref:`topics-signals-ref` below to know which ones. +Some signals support returning :class:`~twisted.internet.defer.Deferred` +objects from their handlers, see the :ref:`topics-signals-ref` below to know +which ones. -.. _Twisted deferreds: https://twistedmatrix.com/documents/current/core/howto/defer.html .. _topics-signals-ref: @@ -155,8 +155,8 @@ item_error :param spider: the spider which raised the exception :type spider: :class:`~scrapy.spiders.Spider` object - :param failure: the exception raised as a Twisted `Failure`_ object - :type failure: `Failure`_ object + :param failure: the exception raised + :type failure: twisted.python.failure.Failure spider_closed ------------- @@ -236,8 +236,8 @@ spider_error This signal does not support returning deferreds from their handlers. - :param failure: the exception raised as a Twisted `Failure`_ object - :type failure: `Failure`_ object + :param failure: the exception raised + :type failure: twisted.python.failure.Failure :param response: the response being processed when the exception was raised :type response: :class:`~scrapy.http.Response` object @@ -333,5 +333,3 @@ response_downloaded :param spider: the spider for which the response is intended :type spider: :class:`~scrapy.spiders.Spider` object - -.. _Failure: https://twistedmatrix.com/documents/current/api/twisted.python.failure.Failure.html diff --git a/docs/topics/spiders.rst b/docs/topics/spiders.rst index 869a61441..b0fb14e24 100644 --- a/docs/topics/spiders.rst +++ b/docs/topics/spiders.rst @@ -72,8 +72,6 @@ scrapy.Spider spider that crawls ``mywebsite.com`` would often be called ``mywebsite``. - .. note:: In Python 2 this must be ASCII only. - .. attribute:: allowed_domains An optional list of strings containing domains that this spider is @@ -374,12 +372,14 @@ CrawlSpider Crawling rules ~~~~~~~~~~~~~~ -.. class:: Rule(link_extractor, callback=None, cb_kwargs=None, follow=None, process_links=None, process_request=None) +.. autoclass:: Rule ``link_extractor`` is a :ref:`Link Extractor <topics-link-extractors>` object which defines how links will be extracted from each crawled page. Each produced link will be used to generate a :class:`~scrapy.http.Request` object, which will contain the link's text in its ``meta`` dictionary (under the ``link_text`` key). + If omitted, a default link extractor created with no arguments will be used, + resulting in all links being extracted. ``callback`` is a callable or a string (in which case a method from the spider object with that name will be used) to be called for each link extracted with @@ -414,6 +414,12 @@ Crawling rules from which the request originated as second argument. It must return a ``Request`` object or ``None`` (to filter out the request). + ``errback`` is a callable or a string (in which case a method from the spider + object with that name will be used) to be called if any exception is + raised while processing a request generated by the rule. + It receives a :class:`Twisted Failure <twisted.python.failure.Failure>` + instance as first parameter. + CrawlSpider example ~~~~~~~~~~~~~~~~~~~ @@ -690,7 +696,7 @@ SitemapSpider .. method:: sitemap_filter(entries) - This is a filter funtion that could be overridden to select sitemap entries + This is a filter function that could be overridden to select sitemap entries based on their attributes. For example:: diff --git a/docs/topics/stats.rst b/docs/topics/stats.rst index 38648ec55..3dd829ebe 100644 --- a/docs/topics/stats.rst +++ b/docs/topics/stats.rst @@ -57,15 +57,15 @@ Set stat value only if lower than previous:: stats.min_value('min_free_memory_percent', value) -Get stat value:: +Get stat value: - >>> stats.get_value('custom_count') - 1 +>>> stats.get_value('custom_count') +1 -Get all stats:: +Get all stats: - >>> stats.get_stats() - {'custom_count': 1, 'start_time': datetime.datetime(2009, 7, 14, 21, 47, 28, 977139)} +>>> stats.get_stats() +{'custom_count': 1, 'start_time': datetime.datetime(2009, 7, 14, 21, 47, 28, 977139)} Available Stats Collectors ========================== diff --git a/docs/topics/telnetconsole.rst b/docs/topics/telnetconsole.rst index 7db7e4f6b..47d8d393c 100644 --- a/docs/topics/telnetconsole.rst +++ b/docs/topics/telnetconsole.rst @@ -44,11 +44,11 @@ the console you need to type:: >>> By default Username is ``scrapy`` and Password is autogenerated. The -autogenerated Password can be seen on scrapy logs like the example below:: +autogenerated Password can be seen on Scrapy logs like the example below:: 2018-10-16 14:35:21 [scrapy.extensions.telnet] INFO: Telnet Password: 16f92501e8a59326 -Default Username and Password can be overriden by the settings +Default Username and Password can be overridden by the settings :setting:`TELNETCONSOLE_USERNAME` and :setting:`TELNETCONSOLE_PASSWORD`. .. warning:: diff --git a/extras/qps-bench-server.py b/extras/qps-bench-server.py index 3bef20bf3..da7a0022b 100755 --- a/extras/qps-bench-server.py +++ b/extras/qps-bench-server.py @@ -1,5 +1,4 @@ #!/usr/bin/env python -from __future__ import print_function from time import time from collections import deque from twisted.web.server import Site, NOT_DONE_YET diff --git a/extras/scrapy_zsh_completion b/extras/scrapy_zsh_completion index 564991aa8..e995947cb 100644 --- a/extras/scrapy_zsh_completion +++ b/extras/scrapy_zsh_completion @@ -1,25 +1,210 @@ #compdef scrapy - -# zsh completion for the Scrapy command-line tool - _scrapy() { - local curcontext="$curcontext" cmd spiders + local context state state_descr line typeset -A opt_args - cmd=$words[2] - - case "$cmd" in - crawl|edit|check) - spiders=$(scrapy list 2>/dev/null) || spiders="" - if [[ -n "$spiders" ]]; then - compadd `echo $spiders` - fi - ;; - *) - if [[ CURRENT -eq 2 ]]; then - _arguments '*: :(check crawl edit fetch genspider list parse runspider settings shell startproject version view)' - fi - ;; + _arguments \ + "(- 1 *)--help[Help]" \ + "1: :->command" \ + "*:: :->args" + + case $state in + command) + _scrapy_cmds + ;; + args) + case $words[1] in + bench) + _scrapy_glb_opts + ;; + fetch) + local options=( + '--headers[print response HTTP headers instead of body]' + '--no-redirect[do not handle HTTP 3xx status codes and print response as-is]' + '--spider[use this spider]:spider:_scrapy_spiders' + '1::URL:_httpie_urls' + ) + _scrapy_glb_opts $options + ;; + genspider) + local options=( + {-l,--list}'[List available templates]' + {-e,--edit}'[Edit spider after creating it]' + '--force[If the spider already exists, overwrite it with the template]' + {-d,--dump=}'[Dump template to standard output]:template:(basic crawl csvfeed xmlfeed)' + {-t,--template=}'[Uses a custom template]:template:(basic crawl csvfeed xmlfeed)' + '1:name:(NAME)' + '2:domain:_httpie_urls' + ) + _scrapy_glb_opts $options + ;; + runspider) + local options=( + {-o,--output}'[dump scraped items into FILE (use - for stdout)]:file:_files' + {-t,--output-format}'[format to use for dumping items with -o]:format:(FORMAT)' + '*-a[set spider argument (may be repeated)]:value pair:(NAME=VALUE)' + '1:spider file:_files -g \*.py' + ) + _scrapy_glb_opts $options + ;; + settings) + local options=( + '--get=[print raw setting value]:option:(SETTING)' + '--getbool=[print setting value, interpreted as a boolean]:option:(SETTING)' + '--getint=[print setting value, interpreted as an integer]:option:(SETTING)' + '--getfloat=[print setting value, interpreted as a float]:option:(SETTING)' + '--getlist=[print setting value, interpreted as a list]:option:(SETTING)' + ) + _scrapy_glb_opts $options + ;; + shell) + local options=( + '-c[evaluate the code in the shell, print the result and exit]:code:(CODE)' + '--no-redirect[do not handle HTTP 3xx status codes and print response as-is]' + '--spider[use this spider]:spider:_scrapy_spiders' + '::file:_files -g \*.html' + '::URL:_httpie_urls' + ) + _scrapy_glb_opts $options + ;; + startproject) + local options=( + '1:name:(NAME)' + '2:dir:_dir_list' + ) + _scrapy_glb_opts $options + ;; + version) + local options=( + {-v,--verbose}'[also display twisted/python/platform info (useful for bug reports)]' + ) + _scrapy_glb_opts $options + ;; + view) + local options=( + '--no-redirect[do not handle HTTP 3xx status codes and print response as-is]' + '--spider[use this spider]:spider:_scrapy_spiders' + '1:URL:_httpie_urls' + ) + _scrapy_glb_opts $options + ;; + check) + local options=( + '(- 1 *)'{-l,--list}'[only list contracts, without checking them]' + {-v,--verbose}'[print contract tests for all spiders]' + '1:spider:_scrapy_spiders' + ) + _scrapy_glb_opts $options + ;; + crawl) + local options=( + {-o,--output}'[dump scraped items into FILE (use - for stdout)]:file:_files' + {-t,--output-format}'[format to use for dumping items with -o]:format:(FORMAT)' + '*-a[set spider argument (may be repeated)]:value pair:(NAME=VALUE)' + '1:spider:_scrapy_spiders' + ) + _scrapy_glb_opts $options + ;; + edit) + local options=( + '1:spider:_scrapy_spiders' + ) + _scrapy_glb_opts $options + ;; + list) + _scrapy_glb_opts + ;; + parse) + local options=( + '*-a[set spider argument (may be repeated)]:value pair:(NAME=VALUE)' + '--spider[use this spider without looking for one]:spider:_scrapy_spiders' + '--pipelines[process items through pipelines]' + "--nolinks[don't show links to follow (extracted requests)]" + "--noitems[don't show scraped items]" + '--nocolour[avoid using pygments to colorize the output]' + {-r,--rules}'[use CrawlSpider rules to discover the callback]' + {-c,--callback=}'[use this callback for parsing, instead looking for a callback]:callback:(CALLBACK)' + {-m,--meta=}'[inject extra meta into the Request, it must be a valid raw json string]:meta:(META)' + '--cbkwargs=[inject extra callback kwargs into the Request, it must be a valid raw json string]:arguments:(CBKWARGS)' + {-d,--depth=}'[maximum depth for parsing requests (default: 1)]:depth:(DEPTH)' + {-v,--verbose}'[print each depth level one by one]' + '1:URL:_httpie_urls' + ) + _scrapy_glb_opts $options + ;; + esac + ;; esac } -_scrapy \ No newline at end of file +_scrapy_cmds() { + local -a commands project_commands + commands=( + 'bench:Run quick benchmark test' + 'fetch:Fetch a URL using the Scrapy downloader' + 'genspider:Generate new spider using pre-defined templates' + 'runspider:Run a self-contained spider (without creating a project)' + 'settings:Get settings values' + 'shell:Interactive scraping console' + 'startproject:Create new project' + 'version:Print Scrapy version' + 'view:Open URL in browser, as seen by Scrapy' + ) + project_commands=( + 'check:Check spider contracts' + 'crawl:Run a spider' + 'edit:Edit spider' + 'list:List available spiders' + 'parse:Parse URL (using its spider) and print the results' + ) + if [[ $(scrapy -h | grep -s "no active project") == "" ]]; then + commands=(${commands[@]} ${project_commands[@]}) + fi + _describe -t common-commands 'common commands' commands +} + +_scrapy_glb_opts() { + local -a options + options=( + '(- *)'{-h,--help}'[show this help message and exit]' + '(--nolog)--logfile=[log file. if omitted stderr will be used]:file:_files' + '--pidfile=[write process ID to FILE]:file:_files' + '--profile=[write python cProfile stats to FILE]:file:_files' + '(--nolog)'{-L,--loglevel=}'[log level (default: INFO)]:log level:(DEBUG INFO WARN ERROR)' + '(-L --loglevel --logfile)--nolog[disable logging completely]' + '--pdb[enable pdb on failure]' + '*'{-s,--set=}'[set/override setting (may be repeated)]:value pair:(NAME=VALUE)' + ) + options=(${options[@]} "$@") + _arguments $options +} + +_httpie_urls() { + + local ret=1 + + if ! [[ -prefix [-+.a-z0-9]#:// ]]; then + local expl + compset -S '[^:/]*' && compstate[to_end]='' + _wanted url-schemas expl 'URL schema' compadd -S '' http:// https:// && ret=0 + else + _urls && ret=0 + fi + + return $ret + +} + +_scrapy_spiders() { + + local ret=1 + + if [[ $(scrapy -h | grep -s "no active project") == "" ]]; then + compadd -S '' $(scrapy list) && ret=0 + else + compadd -S '' SPIDER && ret=0 + fi + + return $ret +} + +_scrapy $@ diff --git a/pytest.ini b/pytest.ini index 73d169601..1030d5530 100644 --- a/pytest.ini +++ b/pytest.ini @@ -2,5 +2,251 @@ usefixtures = chdir python_files=test_*.py __init__.py python_classes= -addopts = --doctest-modules --assert=plain +addopts = + --assert=plain + --doctest-modules + --ignore=docs/_ext + --ignore=docs/conf.py + --ignore=docs/news.rst + --ignore=docs/topics/dynamic-content.rst + --ignore=docs/topics/items.rst + --ignore=docs/topics/leaks.rst + --ignore=docs/topics/loaders.rst + --ignore=docs/topics/selectors.rst + --ignore=docs/topics/shell.rst + --ignore=docs/topics/stats.rst + --ignore=docs/topics/telnetconsole.rst + --ignore=docs/utils twisted = 1 +markers = + only_asyncio: marks tests as only enabled when --reactor=asyncio is passed +flake8-ignore = + # Files that are only meant to provide top-level imports are expected not + # to use any of their imports: + scrapy/core/downloader/handlers/http.py F401 + scrapy/http/__init__.py F401 + # Issues pending a review: + # extras + extras/qps-bench-server.py E501 + extras/qpsclient.py E501 E501 + # scrapy/commands + scrapy/commands/__init__.py E128 E501 + scrapy/commands/check.py E501 + scrapy/commands/crawl.py E501 + scrapy/commands/edit.py E501 + scrapy/commands/fetch.py E401 E501 E128 E731 + scrapy/commands/genspider.py E128 E501 E502 + scrapy/commands/parse.py E128 E501 E731 E226 + scrapy/commands/runspider.py E501 + scrapy/commands/settings.py E128 + scrapy/commands/shell.py E128 E501 E502 + scrapy/commands/startproject.py E127 E501 E128 + scrapy/commands/version.py E501 E128 + # scrapy/contracts + scrapy/contracts/__init__.py E501 W504 + scrapy/contracts/default.py E128 + # scrapy/core + scrapy/core/engine.py E501 E128 E127 E306 E502 + scrapy/core/scheduler.py E501 + scrapy/core/scraper.py E501 E306 E128 W504 + scrapy/core/spidermw.py E501 E731 E126 E226 + scrapy/core/downloader/__init__.py E501 + scrapy/core/downloader/contextfactory.py E501 E128 E126 + scrapy/core/downloader/middleware.py E501 E502 + scrapy/core/downloader/tls.py E501 E305 E241 + scrapy/core/downloader/webclient.py E731 E501 E128 E126 E226 + scrapy/core/downloader/handlers/__init__.py E501 + scrapy/core/downloader/handlers/ftp.py E501 E305 E128 E127 + scrapy/core/downloader/handlers/http10.py E501 + scrapy/core/downloader/handlers/http11.py E501 + scrapy/core/downloader/handlers/s3.py E501 E128 E126 + # scrapy/downloadermiddlewares + scrapy/downloadermiddlewares/ajaxcrawl.py E501 E226 + scrapy/downloadermiddlewares/decompression.py E501 + scrapy/downloadermiddlewares/defaultheaders.py E501 + scrapy/downloadermiddlewares/httpcache.py E501 E126 + scrapy/downloadermiddlewares/httpcompression.py E501 E128 + scrapy/downloadermiddlewares/httpproxy.py E501 + scrapy/downloadermiddlewares/redirect.py E501 W504 + scrapy/downloadermiddlewares/retry.py E501 E126 + scrapy/downloadermiddlewares/robotstxt.py E501 + scrapy/downloadermiddlewares/stats.py E501 + # scrapy/extensions + scrapy/extensions/closespider.py E501 E128 E123 + scrapy/extensions/corestats.py E501 + scrapy/extensions/feedexport.py E128 E501 + scrapy/extensions/httpcache.py E128 E501 E303 + scrapy/extensions/memdebug.py E501 + scrapy/extensions/spiderstate.py E501 + scrapy/extensions/telnet.py E501 W504 + scrapy/extensions/throttle.py E501 + # scrapy/http + scrapy/http/common.py E501 + scrapy/http/cookies.py E501 + scrapy/http/request/__init__.py E501 + scrapy/http/request/form.py E501 E123 + scrapy/http/request/json_request.py E501 + scrapy/http/response/__init__.py E501 E128 + scrapy/http/response/text.py E501 E128 E124 + # scrapy/linkextractors + scrapy/linkextractors/__init__.py E731 E501 E402 W504 + scrapy/linkextractors/lxmlhtml.py E501 E731 E226 + # scrapy/loader + scrapy/loader/__init__.py E501 E128 + scrapy/loader/processors.py E501 + # scrapy/pipelines + scrapy/pipelines/__init__.py E501 + scrapy/pipelines/files.py E116 E501 E266 + scrapy/pipelines/images.py E265 E501 + scrapy/pipelines/media.py E125 E501 E266 + # scrapy/selector + scrapy/selector/__init__.py F403 + scrapy/selector/unified.py E501 E111 + # scrapy/settings + scrapy/settings/__init__.py E501 + scrapy/settings/default_settings.py E501 E114 E116 E226 + scrapy/settings/deprecated.py E501 + # scrapy/spidermiddlewares + scrapy/spidermiddlewares/httperror.py E501 + scrapy/spidermiddlewares/offsite.py E501 + scrapy/spidermiddlewares/referer.py E501 E129 W503 W504 + scrapy/spidermiddlewares/urllength.py E501 + # scrapy/spiders + scrapy/spiders/__init__.py E501 E402 + scrapy/spiders/crawl.py E501 + scrapy/spiders/feed.py E501 + scrapy/spiders/sitemap.py E501 + # scrapy/utils + scrapy/utils/asyncio.py E501 + scrapy/utils/benchserver.py E501 + scrapy/utils/conf.py E402 E501 + scrapy/utils/console.py E306 E305 + scrapy/utils/datatypes.py E501 E226 + scrapy/utils/decorators.py E501 + scrapy/utils/defer.py E501 E128 + scrapy/utils/deprecate.py E128 E501 E127 E502 + scrapy/utils/gz.py E305 E501 W504 + scrapy/utils/http.py F403 E226 + scrapy/utils/httpobj.py E501 + scrapy/utils/iterators.py E501 E701 + scrapy/utils/log.py E128 W503 + scrapy/utils/markup.py F403 + scrapy/utils/misc.py E501 E226 + scrapy/utils/multipart.py F403 + scrapy/utils/project.py E501 + scrapy/utils/python.py E501 + scrapy/utils/reactor.py E226 + scrapy/utils/reqser.py E501 + scrapy/utils/request.py E127 E501 + scrapy/utils/response.py E501 E128 + scrapy/utils/signal.py E501 E128 + scrapy/utils/sitemap.py E501 + scrapy/utils/spider.py E271 E501 + scrapy/utils/ssl.py E501 + scrapy/utils/test.py E501 + scrapy/utils/url.py E501 F403 E128 F405 + # scrapy + scrapy/__init__.py E402 E501 + scrapy/cmdline.py E501 + scrapy/crawler.py E501 + scrapy/dupefilters.py E501 E202 + scrapy/exceptions.py E501 + scrapy/exporters.py E501 E226 + scrapy/interfaces.py E501 + scrapy/item.py E501 E128 + scrapy/link.py E501 + scrapy/logformatter.py E501 + scrapy/mail.py E402 E128 E501 E502 + scrapy/middleware.py E128 E501 + scrapy/pqueues.py E501 + scrapy/responsetypes.py E128 E501 E305 + scrapy/robotstxt.py E501 + scrapy/shell.py E501 + scrapy/signalmanager.py E501 + scrapy/spiderloader.py E225 F841 E501 E126 + scrapy/squeues.py E128 + scrapy/statscollectors.py E501 + # tests + tests/__init__.py E402 E501 + tests/mockserver.py E401 E501 E126 E123 + tests/pipelines.py F841 E226 + tests/spiders.py E501 E127 + tests/test_closespider.py E501 E127 + tests/test_command_fetch.py E501 + tests/test_command_parse.py E501 E128 E303 E226 + tests/test_command_shell.py E501 E128 + tests/test_commands.py E128 E501 + tests/test_contracts.py E501 E128 + tests/test_crawl.py E501 E741 E265 + tests/test_crawler.py F841 E306 E501 + tests/test_dependencies.py F841 E501 E305 + tests/test_downloader_handlers.py E124 E127 E128 E225 E265 E501 E701 E126 E226 E123 + tests/test_downloadermiddleware.py E501 + tests/test_downloadermiddleware_ajaxcrawlable.py E501 + tests/test_downloadermiddleware_cookies.py E731 E741 E501 E128 E303 E265 E126 + tests/test_downloadermiddleware_decompression.py E127 + tests/test_downloadermiddleware_defaultheaders.py E501 + tests/test_downloadermiddleware_downloadtimeout.py E501 + tests/test_downloadermiddleware_httpcache.py E501 E305 + tests/test_downloadermiddleware_httpcompression.py E501 E251 E126 E123 + tests/test_downloadermiddleware_httpproxy.py E501 E128 + tests/test_downloadermiddleware_redirect.py E501 E303 E128 E306 E127 E305 + tests/test_downloadermiddleware_retry.py E501 E128 E251 E303 E126 + tests/test_downloadermiddleware_robotstxt.py E501 + tests/test_downloadermiddleware_stats.py E501 + tests/test_dupefilters.py E221 E501 E741 E128 E124 + tests/test_engine.py E401 E501 E128 + tests/test_exporters.py E501 E731 E306 E128 E124 + tests/test_extension_telnet.py F841 + tests/test_feedexport.py E501 F841 E241 + tests/test_http_cookies.py E501 + tests/test_http_headers.py E501 + tests/test_http_request.py E402 E501 E127 E128 E128 E126 E123 + tests/test_http_response.py E501 E301 E128 E265 + tests/test_item.py E701 E128 F841 E306 + tests/test_link.py E501 + tests/test_linkextractors.py E501 E128 E124 + tests/test_loader.py E501 E731 E303 E741 E128 E117 E241 + tests/test_logformatter.py E128 E501 E122 + tests/test_mail.py E128 E501 E305 + tests/test_middleware.py E501 E128 + tests/test_pipeline_crawl.py E131 E501 E128 E126 + tests/test_pipeline_files.py E501 E303 E272 E226 + tests/test_pipeline_images.py F841 E501 E303 + tests/test_pipeline_media.py E501 E741 E731 E128 E306 E502 + tests/test_proxy_connect.py E501 E741 + tests/test_request_cb_kwargs.py E501 + tests/test_responsetypes.py E501 E305 + tests/test_robotstxt_interface.py E501 E501 + tests/test_scheduler.py E501 E126 E123 + tests/test_selector.py E501 E127 + tests/test_spider.py E501 + tests/test_spidermiddleware.py E501 E226 + tests/test_spidermiddleware_httperror.py E128 E501 E127 E121 + tests/test_spidermiddleware_offsite.py E501 E128 E111 + tests/test_spidermiddleware_output_chain.py E501 E226 + tests/test_spidermiddleware_referer.py E501 F841 E125 E201 E124 E501 E241 E121 + tests/test_squeues.py E501 E701 E741 + tests/test_utils_asyncio.py E501 + tests/test_utils_conf.py E501 E303 E128 + tests/test_utils_curl.py E501 + tests/test_utils_datatypes.py E402 E501 E305 + tests/test_utils_defer.py E306 E501 F841 E226 + tests/test_utils_deprecate.py F841 E306 E501 + tests/test_utils_http.py E501 E128 W504 + tests/test_utils_iterators.py E501 E128 E129 E303 E241 + tests/test_utils_log.py E741 E226 + tests/test_utils_python.py E501 E303 E731 E701 E305 + tests/test_utils_reqser.py E501 E128 + tests/test_utils_request.py E501 E128 E305 + tests/test_utils_response.py E501 + tests/test_utils_signal.py E741 F841 E731 E226 + tests/test_utils_sitemap.py E128 E501 E124 + tests/test_utils_spider.py E305 + tests/test_utils_template.py E305 + tests/test_utils_url.py E501 E127 E305 E211 E125 E501 E226 E241 E126 E123 + tests/test_webclient.py E501 E128 E122 E303 E402 E306 E226 E241 E123 E126 + tests/test_cmdline/__init__.py E501 + tests/test_settings/__init__.py E501 E128 + tests/test_spiderloader/__init__.py E128 E501 + tests/test_utils_misc/__init__.py E501 diff --git a/requirements-py2.txt b/requirements-py2.txt deleted file mode 100644 index 0771aae3a..000000000 --- a/requirements-py2.txt +++ /dev/null @@ -1,10 +0,0 @@ -Twisted>=13.1.0 -lxml -pyOpenSSL -cssselect>=0.9 -queuelib -w3lib>=1.17.0 -six>=1.5.2 -PyDispatcher>=2.0.5 -parsel>=1.5 -service_identity diff --git a/requirements-py3.txt b/requirements-py3.txt deleted file mode 100644 index 5a5d4c95a..000000000 --- a/requirements-py3.txt +++ /dev/null @@ -1,10 +0,0 @@ -Twisted>=17.9.0 -lxml>=3.2.4 -pyOpenSSL>=0.13.1 -cssselect>=0.9 -queuelib>=1.1.1 -w3lib>=1.17.0 -six>=1.5.2 -PyDispatcher>=2.0.5 -parsel>=1.5 -service_identity diff --git a/scrapy/VERSION b/scrapy/VERSION index bd8bf882d..27f9cd322 100644 --- a/scrapy/VERSION +++ b/scrapy/VERSION @@ -1 +1 @@ -1.7.0 +1.8.0 diff --git a/scrapy/__init__.py b/scrapy/__init__.py index 03ec6c667..fb8357f3c 100644 --- a/scrapy/__init__.py +++ b/scrapy/__init__.py @@ -14,8 +14,8 @@ del pkgutil # Check minimum required Python version import sys -if sys.version_info < (2, 7): - print("Scrapy %s requires Python 2.7" % __version__) +if sys.version_info < (3, 5): + print("Scrapy %s requires Python 3.5" % __version__) sys.exit(1) # Ignore noisy twisted deprecation warnings @@ -24,7 +24,7 @@ warnings.filterwarnings('ignore', category=DeprecationWarning, module='twisted') del warnings # Apply monkey patches to fix issues in external libraries -from . import _monkeypatches +from scrapy import _monkeypatches del _monkeypatches from twisted import version as _txv diff --git a/scrapy/_monkeypatches.py b/scrapy/_monkeypatches.py index 935c4bfa3..f74f89bda 100644 --- a/scrapy/_monkeypatches.py +++ b/scrapy/_monkeypatches.py @@ -1,19 +1,4 @@ -import six -from six.moves import copyreg - - -if six.PY2: - from urlparse import urlparse - - # workaround for https://bugs.python.org/issue7904 - Python < 2.7 - if urlparse('s3://bucket/key').netloc != 'bucket': - from urlparse import uses_netloc - uses_netloc.append('s3') - - # workaround for https://bugs.python.org/issue9374 - Python < 2.7.4 - if urlparse('s3://bucket/key?key=value').query != 'key=value': - from urlparse import uses_query - uses_query.append('s3') +import copyreg # Undo what Twisted's perspective broker adds to pickle register diff --git a/scrapy/cmdline.py b/scrapy/cmdline.py index 418dc1ac9..ec78f7c91 100644 --- a/scrapy/cmdline.py +++ b/scrapy/cmdline.py @@ -1,4 +1,3 @@ -from __future__ import print_function import sys import os import optparse @@ -68,7 +67,7 @@ def _pop_command_name(argv): def _print_header(settings, inproject): if inproject: - print("Scrapy %s - project: %s\n" % (scrapy.__version__, \ + print("Scrapy %s - project: %s\n" % (scrapy.__version__, settings['BOT_NAME'])) else: print("Scrapy %s - no active project\n" % scrapy.__version__) @@ -124,7 +123,7 @@ def execute(argv=None, settings=None): inproject = inside_project() cmds = _get_commands_dict(settings, inproject) cmdname = _pop_command_name(argv) - parser = optparse.OptionParser(formatter=optparse.TitledHelpFormatter(), \ + parser = optparse.OptionParser(formatter=optparse.TitledHelpFormatter(), conflict_handler='resolve') if not cmdname: _print_commands(settings, inproject) diff --git a/scrapy/commands/__init__.py b/scrapy/commands/__init__.py index 43b420821..0b24193c2 100644 --- a/scrapy/commands/__init__.py +++ b/scrapy/commands/__init__.py @@ -47,7 +47,7 @@ class ScrapyCommand(object): def help(self): """An extensive help for the command. It will be shown when using the - "help" command. It can contain newlines, since not post-formatting will + "help" command. It can contain newlines, since no post-formatting will be applied to its contents. """ return self.long_desc() diff --git a/scrapy/commands/bench.py b/scrapy/commands/bench.py index 90c8d56a2..7bbe362e7 100644 --- a/scrapy/commands/bench.py +++ b/scrapy/commands/bench.py @@ -1,8 +1,7 @@ import sys import time import subprocess - -from six.moves.urllib.parse import urlencode +from urllib.parse import urlencode import scrapy from scrapy.commands import ScrapyCommand diff --git a/scrapy/commands/check.py b/scrapy/commands/check.py index ab73e85e7..9d4437a47 100644 --- a/scrapy/commands/check.py +++ b/scrapy/commands/check.py @@ -1,6 +1,4 @@ -from __future__ import print_function import time -import sys from collections import defaultdict from unittest import TextTestRunner, TextTestResult as _TextTestResult @@ -96,4 +94,3 @@ class Command(ScrapyCommand): result.printErrors() result.printSummary(start, stop) self.exitcode = int(not result.wasSuccessful()) - diff --git a/scrapy/commands/fetch.py b/scrapy/commands/fetch.py index 7d4840529..0e149941d 100644 --- a/scrapy/commands/fetch.py +++ b/scrapy/commands/fetch.py @@ -1,5 +1,4 @@ -from __future__ import print_function -import sys, six +import sys from w3lib.url import is_url from scrapy.commands import ScrapyCommand @@ -8,6 +7,7 @@ from scrapy.exceptions import UsageError from scrapy.utils.datatypes import SequenceExclude from scrapy.utils.spider import spidercls_for_request, DefaultSpider + class Command(ScrapyCommand): requires_project = False @@ -24,12 +24,11 @@ class Command(ScrapyCommand): def add_options(self, parser): ScrapyCommand.add_options(self, parser) - parser.add_option("--spider", dest="spider", - help="use this spider") - parser.add_option("--headers", dest="headers", action="store_true", \ - help="print response HTTP headers instead of body") - parser.add_option("--no-redirect", dest="no_redirect", action="store_true", \ - default=False, help="do not handle HTTP 3xx status codes and print response as-is") + parser.add_option("--spider", dest="spider", help="use this spider") + parser.add_option("--headers", dest="headers", action="store_true", + help="print response HTTP headers instead of body") + parser.add_option("--no-redirect", dest="no_redirect", action="store_true", + default=False, help="do not handle HTTP 3xx status codes and print response as-is") def _print_headers(self, headers, prefix): for key, values in headers.items(): @@ -45,8 +44,7 @@ class Command(ScrapyCommand): self._print_bytes(response.body) def _print_bytes(self, bytes_): - bytes_writer = sys.stdout if six.PY2 else sys.stdout.buffer - bytes_writer.write(bytes_ + b'\n') + sys.stdout.buffer.write(bytes_ + b'\n') def run(self, args, opts): if len(args) != 1 or not is_url(args[0]): diff --git a/scrapy/commands/genspider.py b/scrapy/commands/genspider.py index d5498bb5c..adb01fa70 100644 --- a/scrapy/commands/genspider.py +++ b/scrapy/commands/genspider.py @@ -1,4 +1,3 @@ -from __future__ import print_function import os import shutil import string diff --git a/scrapy/commands/list.py b/scrapy/commands/list.py index a255b3b94..54d7bb228 100644 --- a/scrapy/commands/list.py +++ b/scrapy/commands/list.py @@ -1,6 +1,6 @@ -from __future__ import print_function from scrapy.commands import ScrapyCommand + class Command(ScrapyCommand): requires_project = True diff --git a/scrapy/commands/parse.py b/scrapy/commands/parse.py index d4f2234b0..ff6f1d8cd 100644 --- a/scrapy/commands/parse.py +++ b/scrapy/commands/parse.py @@ -1,4 +1,3 @@ -from __future__ import print_function import json import logging @@ -60,10 +59,12 @@ class Command(ScrapyCommand): @property def max_level(self): - levels = list(self.items.keys()) + list(self.requests.keys()) - if not levels: - return 0 - return max(levels) + max_items, max_requests = 0, 0 + if self.items: + max_items = max(self.items) + if self.requests: + max_requests = max(self.requests) + return max(max_items, max_requests) def add_items(self, lvl, new_items): old_items = self.items.get(lvl, []) @@ -84,9 +85,8 @@ class Command(ScrapyCommand): def print_requests(self, lvl=None, colour=True): if lvl is None: - levels = list(self.requests.keys()) - if levels: - requests = self.requests[max(levels)] + if self.requests: + requests = self.requests[max(self.requests)] else: requests = [] else: diff --git a/scrapy/commands/runspider.py b/scrapy/commands/runspider.py index 376d3c84e..57d8471ca 100644 --- a/scrapy/commands/runspider.py +++ b/scrapy/commands/runspider.py @@ -60,14 +60,13 @@ class Command(ScrapyCommand): else: self.settings.set('FEED_URI', opts.output, priority='cmdline') feed_exporters = without_none_values(self.settings.getwithbase('FEED_EXPORTERS')) - valid_output_formats = feed_exporters.keys() if not opts.output_format: opts.output_format = os.path.splitext(opts.output)[1].replace(".", "") - if opts.output_format not in valid_output_formats: + if opts.output_format not in feed_exporters: raise UsageError("Unrecognized output format '%s', set one" " using the '-t' switch or as a file extension" " from the supported list %s" % (opts.output_format, - tuple(valid_output_formats))) + tuple(feed_exporters))) self.settings.set('FEED_FORMAT', opts.output_format, priority='cmdline') def run(self, args, opts): diff --git a/scrapy/commands/settings.py b/scrapy/commands/settings.py index bee52f06a..603bafb9f 100644 --- a/scrapy/commands/settings.py +++ b/scrapy/commands/settings.py @@ -1,9 +1,9 @@ -from __future__ import print_function import json from scrapy.commands import ScrapyCommand from scrapy.settings import BaseSettings + class Command(ScrapyCommand): requires_project = False diff --git a/scrapy/commands/shell.py b/scrapy/commands/shell.py index e05084272..d44a32d5f 100644 --- a/scrapy/commands/shell.py +++ b/scrapy/commands/shell.py @@ -6,8 +6,8 @@ See documentation in docs/topics/shell.rst from threading import Thread from scrapy.commands import ScrapyCommand -from scrapy.shell import Shell from scrapy.http import Request +from scrapy.shell import Shell from scrapy.utils.spider import spidercls_for_request, DefaultSpider from scrapy.utils.url import guess_scheme diff --git a/scrapy/commands/startproject.py b/scrapy/commands/startproject.py index 67337c26e..b123e5c84 100644 --- a/scrapy/commands/startproject.py +++ b/scrapy/commands/startproject.py @@ -1,4 +1,3 @@ -from __future__ import print_function import re import os import string @@ -44,8 +43,8 @@ class Command(ScrapyCommand): return False if not re.search(r'^[_a-zA-Z]\w*$', project_name): - print('Error: Project names must begin with a letter and contain'\ - ' only\nletters, numbers and underscores') + print('Error: Project names must begin with a letter and contain' + ' only\nletters, numbers and underscores') elif _module_exists(project_name): print('Error: Module %r already exists' % project_name) else: @@ -119,4 +118,3 @@ class Command(ScrapyCommand): _templates_base_dir = self.settings['TEMPLATES_DIR'] or \ join(scrapy.__path__[0], 'templates') return join(_templates_base_dir, 'project') - diff --git a/scrapy/commands/version.py b/scrapy/commands/version.py index 577365c3b..1516c5997 100644 --- a/scrapy/commands/version.py +++ b/scrapy/commands/version.py @@ -1,5 +1,3 @@ -from __future__ import print_function - import scrapy from scrapy.commands import ScrapyCommand from scrapy.utils.versions import scrapy_components_versions @@ -30,4 +28,3 @@ class Command(ScrapyCommand): print(patt % (name, version)) else: print("Scrapy %s" % scrapy.__version__) - diff --git a/scrapy/commands/view.py b/scrapy/commands/view.py index 59e665016..41e77ba3b 100644 --- a/scrapy/commands/view.py +++ b/scrapy/commands/view.py @@ -1,6 +1,7 @@ -from scrapy.commands import fetch, ScrapyCommand +from scrapy.commands import fetch from scrapy.utils.response import open_in_browser + class Command(fetch.Command): def short_desc(self): diff --git a/scrapy/contracts/__init__.py b/scrapy/contracts/__init__.py index 536bbdafb..7b6591d86 100644 --- a/scrapy/contracts/__init__.py +++ b/scrapy/contracts/__init__.py @@ -90,9 +90,9 @@ class ContractsManager(object): cb = request.callback @wraps(cb) - def cb_wrapper(response): + def cb_wrapper(response, **cb_kwargs): try: - output = cb(response) + output = cb(response, **cb_kwargs) output = list(iterate_spider_output(output)) except Exception: case = _create_testcase(method, 'callback') @@ -121,7 +121,7 @@ class Contract(object): cb = request.callback @wraps(cb) - def wrapper(response): + def wrapper(response, **cb_kwargs): try: results.startTest(self.testcase_pre) self.pre_process(response) @@ -133,7 +133,7 @@ class Contract(object): else: results.addSuccess(self.testcase_pre) finally: - return list(iterate_spider_output(cb(response))) + return list(iterate_spider_output(cb(response, **cb_kwargs))) request.callback = wrapper @@ -144,8 +144,8 @@ class Contract(object): cb = request.callback @wraps(cb) - def wrapper(response): - output = list(iterate_spider_output(cb(response))) + def wrapper(response, **cb_kwargs): + output = list(iterate_spider_output(cb(response, **cb_kwargs))) try: results.startTest(self.testcase_post) self.post_process(output) diff --git a/scrapy/contracts/default.py b/scrapy/contracts/default.py index 20582503d..3002fc702 100644 --- a/scrapy/contracts/default.py +++ b/scrapy/contracts/default.py @@ -1,8 +1,10 @@ +import json + from scrapy.item import BaseItem from scrapy.http import Request from scrapy.exceptions import ContractFail -from . import Contract +from scrapy.contracts import Contract # contracts @@ -18,6 +20,20 @@ class UrlContract(Contract): return args +class CallbackKeywordArgumentsContract(Contract): + """ Contract to set the keyword arguments for the request. + The value should be a JSON-encoded dictionary, e.g.: + + @cb_kwargs {"arg1": "some value"} + """ + + name = 'cb_kwargs' + + def adjust_request_args(self, args): + args['cb_kwargs'] = json.loads(' '.join(self.args)) + return args + + class ReturnsContract(Contract): """ Contract to check the output of a callback @@ -70,8 +86,8 @@ class ReturnsContract(Contract): else: expected = '%s..%s' % (self.min_bound, self.max_bound) - raise ContractFail("Returned %s %s, expected %s" % \ - (occurrences, self.obj_name, expected)) + raise ContractFail("Returned %s %s, expected %s" % + (occurrences, self.obj_name, expected)) class ScrapesContract(Contract): @@ -84,6 +100,7 @@ class ScrapesContract(Contract): def post_process(self, output): for x in output: if isinstance(x, (BaseItem, dict)): - for arg in self.args: - if not arg in x: - raise ContractFail("'%s' field is missing" % arg) + missing = [arg for arg in self.args if arg not in x] + if missing: + raise ContractFail( + "Missing fields: %s" % ", ".join(missing)) diff --git a/scrapy/core/downloader/__init__.py b/scrapy/core/downloader/__init__.py index 949dacbc8..157dc3418 100644 --- a/scrapy/core/downloader/__init__.py +++ b/scrapy/core/downloader/__init__.py @@ -1,19 +1,16 @@ -from __future__ import absolute_import import random -import warnings from time import time from datetime import datetime from collections import deque -import six from twisted.internet import reactor, defer, task from scrapy.utils.defer import mustbe_deferred from scrapy.utils.httpobj import urlparse_cached from scrapy.resolver import dnscache from scrapy import signals -from .middleware import DownloaderMiddlewareManager -from .handlers import DownloadHandlers +from scrapy.core.downloader.middleware import DownloaderMiddlewareManager +from scrapy.core.downloader.handlers import DownloadHandlers class Slot(object): @@ -190,7 +187,7 @@ class Downloader(object): def close(self): self._slot_gc_loop.stop() - for slot in six.itervalues(self.slots): + for slot in self.slots.values(): slot.close() def _slot_gc(self, age=60): diff --git a/scrapy/core/downloader/contextfactory.py b/scrapy/core/downloader/contextfactory.py index 783d4c383..6e023ebcc 100644 --- a/scrapy/core/downloader/contextfactory.py +++ b/scrapy/core/downloader/contextfactory.py @@ -1,105 +1,93 @@ from OpenSSL import SSL -from twisted.internet.ssl import ClientContextFactory +from twisted.internet.ssl import optionsForClientTLS, CertificateOptions, platformTrust, AcceptableCiphers +from twisted.web.client import BrowserLikePolicyForHTTPS +from twisted.web.iweb import IPolicyForHTTPS +from zope.interface.declarations import implementer -from scrapy import twisted_version - -if twisted_version >= (14, 0, 0): - - from zope.interface.declarations import implementer - - from twisted.internet.ssl import (optionsForClientTLS, - CertificateOptions, - platformTrust) - from twisted.web.client import BrowserLikePolicyForHTTPS - from twisted.web.iweb import IPolicyForHTTPS - - from scrapy.core.downloader.tls import ScrapyClientTLSOptions, DEFAULT_CIPHERS +from scrapy.core.downloader.tls import ScrapyClientTLSOptions, DEFAULT_CIPHERS - @implementer(IPolicyForHTTPS) - class ScrapyClientContextFactory(BrowserLikePolicyForHTTPS): - """ - Non-peer-certificate verifying HTTPS context factory +@implementer(IPolicyForHTTPS) +class ScrapyClientContextFactory(BrowserLikePolicyForHTTPS): + """ + Non-peer-certificate verifying HTTPS context factory - Default OpenSSL method is TLS_METHOD (also called SSLv23_METHOD) - which allows TLS protocol negotiation + Default OpenSSL method is TLS_METHOD (also called SSLv23_METHOD) + which allows TLS protocol negotiation - 'A TLS/SSL connection established with [this method] may - understand the SSLv3, TLSv1, TLSv1.1 and TLSv1.2 protocols.' - """ + 'A TLS/SSL connection established with [this method] may + understand the SSLv3, TLSv1, TLSv1.1 and TLSv1.2 protocols.' + """ - def __init__(self, method=SSL.SSLv23_METHOD, *args, **kwargs): - super(ScrapyClientContextFactory, self).__init__(*args, **kwargs) - self._ssl_method = method + def __init__(self, method=SSL.SSLv23_METHOD, tls_verbose_logging=False, tls_ciphers=None, *args, **kwargs): + super(ScrapyClientContextFactory, self).__init__(*args, **kwargs) + self._ssl_method = method + self.tls_verbose_logging = tls_verbose_logging + if tls_ciphers: + self.tls_ciphers = AcceptableCiphers.fromOpenSSLCipherString(tls_ciphers) + else: + self.tls_ciphers = DEFAULT_CIPHERS - def getCertificateOptions(self): - # setting verify=True will require you to provide CAs - # to verify against; in other words: it's not that simple + @classmethod + def from_settings(cls, settings, method=SSL.SSLv23_METHOD, *args, **kwargs): + tls_verbose_logging = settings.getbool('DOWNLOADER_CLIENT_TLS_VERBOSE_LOGGING') + tls_ciphers = settings['DOWNLOADER_CLIENT_TLS_CIPHERS'] + return cls(method=method, tls_verbose_logging=tls_verbose_logging, tls_ciphers=tls_ciphers, *args, **kwargs) - # backward-compatible SSL/TLS method: - # - # * this will respect `method` attribute in often recommended - # `ScrapyClientContextFactory` subclass - # (https://github.com/scrapy/scrapy/issues/1429#issuecomment-131782133) - # - # * getattr() for `_ssl_method` attribute for context factories - # not calling super(..., self).__init__ - return CertificateOptions(verify=False, - method=getattr(self, 'method', - getattr(self, '_ssl_method', None)), - fixBrokenPeers=True, - acceptableCiphers=DEFAULT_CIPHERS) + def getCertificateOptions(self): + # setting verify=True will require you to provide CAs + # to verify against; in other words: it's not that simple - # kept for old-style HTTP/1.0 downloader context twisted calls, - # e.g. connectSSL() - def getContext(self, hostname=None, port=None): - return self.getCertificateOptions().getContext() + # backward-compatible SSL/TLS method: + # + # * this will respect `method` attribute in often recommended + # `ScrapyClientContextFactory` subclass + # (https://github.com/scrapy/scrapy/issues/1429#issuecomment-131782133) + # + # * getattr() for `_ssl_method` attribute for context factories + # not calling super(..., self).__init__ + return CertificateOptions(verify=False, + method=getattr(self, 'method', + getattr(self, '_ssl_method', None)), + fixBrokenPeers=True, + acceptableCiphers=self.tls_ciphers) - def creatorForNetloc(self, hostname, port): - return ScrapyClientTLSOptions(hostname.decode("ascii"), self.getContext()) + # kept for old-style HTTP/1.0 downloader context twisted calls, + # e.g. connectSSL() + def getContext(self, hostname=None, port=None): + return self.getCertificateOptions().getContext() + + def creatorForNetloc(self, hostname, port): + return ScrapyClientTLSOptions(hostname.decode("ascii"), self.getContext(), + verbose_logging=self.tls_verbose_logging) - @implementer(IPolicyForHTTPS) - class BrowserLikeContextFactory(ScrapyClientContextFactory): - """ - Twisted-recommended context factory for web clients. +@implementer(IPolicyForHTTPS) +class BrowserLikeContextFactory(ScrapyClientContextFactory): + """ + Twisted-recommended context factory for web clients. - Quoting https://twistedmatrix.com/documents/current/api/twisted.web.client.Agent.html: - "The default is to use a BrowserLikePolicyForHTTPS, - so unless you have special requirements you can leave this as-is." + Quoting the documentation of the :class:`~twisted.web.client.Agent` class: - creatorForNetloc() is the same as BrowserLikePolicyForHTTPS - except this context factory allows setting the TLS/SSL method to use. + The default is to use a + :class:`~twisted.web.client.BrowserLikePolicyForHTTPS`, so unless you + have special requirements you can leave this as-is. - Default OpenSSL method is TLS_METHOD (also called SSLv23_METHOD) - which allows TLS protocol negotiation. - """ - def creatorForNetloc(self, hostname, port): + :meth:`creatorForNetloc` is the same as + :class:`~twisted.web.client.BrowserLikePolicyForHTTPS` except this context + factory allows setting the TLS/SSL method to use. - # trustRoot set to platformTrust() will use the platform's root CAs. - # - # This means that a website like https://www.cacert.org will be rejected - # by default, since CAcert.org CA certificate is seldom shipped. - return optionsForClientTLS(hostname.decode("ascii"), - trustRoot=platformTrust(), - extraCertificateOptions={ - 'method': self._ssl_method, - }) + The default OpenSSL method is ``TLS_METHOD`` (also called + ``SSLv23_METHOD``) which allows TLS protocol negotiation. + """ + def creatorForNetloc(self, hostname, port): -else: - - class ScrapyClientContextFactory(ClientContextFactory): - "A SSL context factory which is more permissive against SSL bugs." - # see https://github.com/scrapy/scrapy/issues/82 - # and https://github.com/scrapy/scrapy/issues/26 - # and https://github.com/scrapy/scrapy/issues/981 - - def __init__(self, method=SSL.SSLv23_METHOD): - self.method = method - - def getContext(self, hostname=None, port=None): - ctx = ClientContextFactory.getContext(self) - # Enable all workarounds to SSL bugs as documented by - # https://www.openssl.org/docs/manmaster/man3/SSL_CTX_set_options.html - ctx.set_options(SSL.OP_ALL) - return ctx + # trustRoot set to platformTrust() will use the platform's root CAs. + # + # This means that a website like https://www.cacert.org will be rejected + # by default, since CAcert.org CA certificate is seldom shipped. + return optionsForClientTLS(hostname.decode("ascii"), + trustRoot=platformTrust(), + extraCertificateOptions={ + 'method': self._ssl_method, + }) diff --git a/scrapy/core/downloader/handlers/__init__.py b/scrapy/core/downloader/handlers/__init__.py index 0b55d32fa..39a0b1f51 100644 --- a/scrapy/core/downloader/handlers/__init__.py +++ b/scrapy/core/downloader/handlers/__init__.py @@ -1,8 +1,9 @@ """Download handlers for different schemes""" import logging + from twisted.internet import defer -import six + from scrapy.exceptions import NotSupported, NotConfigured from scrapy.utils.httpobj import urlparse_cached from scrapy.utils.misc import load_object @@ -22,7 +23,7 @@ class DownloadHandlers(object): self._notconfigured = {} # remembers failed handlers handlers = without_none_values( crawler.settings.getwithbase('DOWNLOAD_HANDLERS')) - for scheme, clspath in six.iteritems(handlers): + for scheme, clspath in handlers.items(): self._schemes[scheme] = clspath self._load_handler(scheme, skip_lazy=True) diff --git a/scrapy/core/downloader/handlers/datauri.py b/scrapy/core/downloader/handlers/datauri.py index ad25beb3b..9e5020753 100644 --- a/scrapy/core/downloader/handlers/datauri.py +++ b/scrapy/core/downloader/handlers/datauri.py @@ -17,8 +17,8 @@ class DataURIDownloadHandler(object): respcls = responsetypes.from_mimetype(uri.media_type) resp_kwargs = {} - if (issubclass(respcls, TextResponse) and - uri.media_type.split('/')[0] == 'text'): + if (issubclass(respcls, TextResponse) + and uri.media_type.split('/')[0] == 'text'): charset = uri.media_type_parameters.get('charset') resp_kwargs['encoding'] = charset diff --git a/scrapy/core/downloader/handlers/ftp.py b/scrapy/core/downloader/handlers/ftp.py index 806a537d4..aef231e82 100644 --- a/scrapy/core/downloader/handlers/ftp.py +++ b/scrapy/core/downloader/handlers/ftp.py @@ -30,7 +30,7 @@ In case of status 200 request, response.headers will come with two keys: import re from io import BytesIO -from six.moves.urllib.parse import unquote +from urllib.parse import unquote from twisted.internet import reactor from twisted.protocols.ftp import FTPClient, CommandFailed @@ -112,4 +112,3 @@ class FTPDownloadHandler(object): httpcode = self.CODE_MAPPING.get(ftpcode, self.CODE_MAPPING["default"]) return Response(url=request.url, status=httpcode, body=to_bytes(message)) raise result.type(result.value) - diff --git a/scrapy/core/downloader/handlers/http.py b/scrapy/core/downloader/handlers/http.py index ac4b867c3..52535bd8b 100644 --- a/scrapy/core/downloader/handlers/http.py +++ b/scrapy/core/downloader/handlers/http.py @@ -1,3 +1,4 @@ -from __future__ import absolute_import -from .http10 import HTTP10DownloadHandler -from .http11 import HTTP11DownloadHandler as HTTPDownloadHandler +from scrapy.core.downloader.handlers.http10 import HTTP10DownloadHandler +from scrapy.core.downloader.handlers.http11 import ( + HTTP11DownloadHandler as HTTPDownloadHandler, +) diff --git a/scrapy/core/downloader/handlers/http10.py b/scrapy/core/downloader/handlers/http10.py index d875fb1e4..be7298531 100644 --- a/scrapy/core/downloader/handlers/http10.py +++ b/scrapy/core/downloader/handlers/http10.py @@ -1,7 +1,7 @@ """Download handlers for http and https schemes """ from twisted.internet import reactor -from scrapy.utils.misc import load_object +from scrapy.utils.misc import load_object, create_instance from scrapy.utils.python import to_unicode @@ -11,6 +11,7 @@ class HTTP10DownloadHandler(object): def __init__(self, settings): self.HTTPClientFactory = load_object(settings['DOWNLOADER_HTTPCLIENTFACTORY']) self.ClientContextFactory = load_object(settings['DOWNLOADER_CLIENTCONTEXTFACTORY']) + self._settings = settings def download_request(self, request, spider): """Return a deferred for the HTTP download""" @@ -21,7 +22,7 @@ class HTTP10DownloadHandler(object): def _connect(self, factory): host, port = to_unicode(factory.host), factory.port if factory.scheme == b'https': - return reactor.connectSSL(host, port, factory, - self.ClientContextFactory()) + client_context_factory = create_instance(self.ClientContextFactory, settings=self._settings, crawler=None) + return reactor.connectSSL(host, port, factory, client_context_factory) else: return reactor.connectTCP(host, port, factory) diff --git a/scrapy/core/downloader/handlers/http11.py b/scrapy/core/downloader/handlers/http11.py index cbaa36b2d..1212feb79 100644 --- a/scrapy/core/downloader/handlers/http11.py +++ b/scrapy/core/downloader/handlers/http11.py @@ -2,10 +2,10 @@ import re import logging +import warnings from io import BytesIO from time import time -import warnings -from six.moves.urllib.parse import urldefrag +from urllib.parse import urldefrag from zope.interface import implementer from twisted.internet import defer, reactor, protocol @@ -13,21 +13,17 @@ from twisted.web.http_headers import Headers as TxHeaders from twisted.web.iweb import IBodyProducer, UNKNOWN_LENGTH from twisted.internet.error import TimeoutError from twisted.web.http import _DataLoss, PotentialDataLoss -from twisted.web.client import Agent, ProxyAgent, ResponseDone, \ - HTTPConnectionPool, ResponseFailed -try: - from twisted.web.client import URI -except ImportError: - from twisted.web.client import _URI as URI +from twisted.web.client import Agent, ResponseDone, HTTPConnectionPool, ResponseFailed, URI from twisted.internet.endpoints import TCP4ClientEndpoint +from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.http import Headers from scrapy.responsetypes import responsetypes from scrapy.core.downloader.webclient import _parse from scrapy.core.downloader.tls import openssl_methods -from scrapy.utils.misc import load_object +from scrapy.utils.misc import load_object, create_instance from scrapy.utils.python import to_bytes, to_unicode -from scrapy import twisted_version + logger = logging.getLogger(__name__) @@ -44,14 +40,23 @@ class HTTP11DownloadHandler(object): self._contextFactoryClass = load_object(settings['DOWNLOADER_CLIENTCONTEXTFACTORY']) # try method-aware context factory try: - self._contextFactory = self._contextFactoryClass(method=self._sslMethod) + self._contextFactory = create_instance( + self._contextFactoryClass, + settings=settings, + crawler=None, + method=self._sslMethod, + ) except TypeError: # use context factory defaults - self._contextFactory = self._contextFactoryClass() + self._contextFactory = create_instance( + self._contextFactoryClass, + settings=settings, + crawler=None, + ) msg = """ '%s' does not accept `method` argument (type OpenSSL.SSL method,\ - e.g. OpenSSL.SSL.SSLv23_METHOD).\ - Please upgrade your context factory class to handle it or ignore it.""" % ( + e.g. OpenSSL.SSL.SSLv23_METHOD) and/or `tls_verbose_logging` argument and/or `tls_ciphers` argument.\ + Please upgrade your context factory class to handle them or ignore them.""" % ( settings['DOWNLOADER_CLIENTCONTEXTFACTORY'],) warnings.warn(msg) self._default_maxsize = settings.getint('DOWNLOAD_MAXSIZE') @@ -61,10 +66,13 @@ class HTTP11DownloadHandler(object): def download_request(self, request, spider): """Return a deferred for the HTTP download""" - agent = ScrapyAgent(contextFactory=self._contextFactory, pool=self._pool, + agent = ScrapyAgent( + contextFactory=self._contextFactory, + pool=self._pool, maxsize=getattr(spider, 'download_maxsize', self._default_maxsize), warnsize=getattr(spider, 'download_warnsize', self._default_warnsize), - fail_on_dataloss=self._fail_on_dataloss) + fail_on_dataloss=self._fail_on_dataloss, + ) return agent.download_request(request) def close(self): @@ -103,11 +111,9 @@ class TunnelingTCP4ClientEndpoint(TCP4ClientEndpoint): _responseMatcher = re.compile(br'HTTP/1\.. (?P<status>\d{3})(?P<reason>.{,32})') - def __init__(self, reactor, host, port, proxyConf, contextFactory, - timeout=30, bindAddress=None): + def __init__(self, reactor, host, port, proxyConf, contextFactory, timeout=30, bindAddress=None): proxyHost, proxyPort, self._proxyAuthHeader = proxyConf - super(TunnelingTCP4ClientEndpoint, self).__init__(reactor, proxyHost, - proxyPort, timeout, bindAddress) + super(TunnelingTCP4ClientEndpoint, self).__init__(reactor, proxyHost, proxyPort, timeout, bindAddress) self._tunnelReadyDeferred = defer.Deferred() self._tunneledHost = host self._tunneledPort = port @@ -116,8 +122,7 @@ class TunnelingTCP4ClientEndpoint(TCP4ClientEndpoint): def requestTunnel(self, protocol): """Asks the proxy to open a tunnel.""" - tunnelReq = tunnel_request_data(self._tunneledHost, self._tunneledPort, - self._proxyAuthHeader) + tunnelReq = tunnel_request_data(self._tunneledHost, self._tunneledPort, self._proxyAuthHeader) protocol.transport.write(tunnelReq) self._protocolDataReceived = protocol.dataReceived protocol.dataReceived = self.processProxyResponse @@ -140,16 +145,9 @@ class TunnelingTCP4ClientEndpoint(TCP4ClientEndpoint): self._protocol.dataReceived = self._protocolDataReceived respm = TunnelingTCP4ClientEndpoint._responseMatcher.match(self._connectBuffer) if respm and int(respm.group('status')) == 200: - try: - # this sets proper Server Name Indication extension - # but is only available for Twisted>=14.0 - sslOptions = self._contextFactory.creatorForNetloc( - self._tunneledHost, self._tunneledPort) - except AttributeError: - # fall back to non-SNI SSL context factory - sslOptions = self._contextFactory - self._protocol.transport.startTLS(sslOptions, - self._protocolFactory) + # set proper Server Name Indication extension + sslOptions = self._contextFactory.creatorForNetloc(self._tunneledHost, self._tunneledPort) + self._protocol.transport.startTLS(sslOptions, self._protocolFactory) self._tunnelReadyDeferred.callback(self._protocol) else: if respm: @@ -167,8 +165,7 @@ class TunnelingTCP4ClientEndpoint(TCP4ClientEndpoint): def connect(self, protocolFactory): self._protocolFactory = protocolFactory - connectDeferred = super(TunnelingTCP4ClientEndpoint, - self).connect(protocolFactory) + connectDeferred = super(TunnelingTCP4ClientEndpoint, self).connect(protocolFactory) connectDeferred.addCallback(self.requestTunnel) connectDeferred.addErrback(self.connectFailed) return self._tunnelReadyDeferred @@ -178,7 +175,7 @@ def tunnel_request_data(host, port, proxy_auth_header=None): r""" Return binary content of a CONNECT request. - >>> from scrapy.utils.python import to_native_str as s + >>> from scrapy.utils.python import to_unicode as s >>> s(tunnel_request_data("example.com", 8080)) 'CONNECT example.com:8080 HTTP/1.1\r\nHost: example.com:8080\r\n\r\n' >>> s(tunnel_request_data("example.com", 8080, b"123")) @@ -205,42 +202,46 @@ class TunnelingAgent(Agent): def __init__(self, reactor, proxyConf, contextFactory=None, connectTimeout=None, bindAddress=None, pool=None): - super(TunnelingAgent, self).__init__(reactor, contextFactory, - connectTimeout, bindAddress, pool) + super(TunnelingAgent, self).__init__(reactor, contextFactory, connectTimeout, bindAddress, pool) self._proxyConf = proxyConf self._contextFactory = contextFactory - if twisted_version >= (15, 0, 0): - def _getEndpoint(self, uri): - return TunnelingTCP4ClientEndpoint( - self._reactor, uri.host, uri.port, self._proxyConf, - self._contextFactory, self._endpointFactory._connectTimeout, - self._endpointFactory._bindAddress) - else: - def _getEndpoint(self, scheme, host, port): - return TunnelingTCP4ClientEndpoint( - self._reactor, host, port, self._proxyConf, - self._contextFactory, self._connectTimeout, - self._bindAddress) + def _getEndpoint(self, uri): + return TunnelingTCP4ClientEndpoint( + reactor=self._reactor, + host=uri.host, + port=uri.port, + proxyConf=self._proxyConf, + contextFactory=self._contextFactory, + timeout=self._endpointFactory._connectTimeout, + bindAddress=self._endpointFactory._bindAddress, + ) - def _requestWithEndpoint(self, key, endpoint, method, parsedURI, - headers, bodyProducer, requestPath): + def _requestWithEndpoint(self, key, endpoint, method, parsedURI, headers, bodyProducer, requestPath): # proxy host and port are required for HTTP pool `key` # otherwise, same remote host connection request could reuse # a cached tunneled connection to a different proxy key = key + self._proxyConf - return super(TunnelingAgent, self)._requestWithEndpoint(key, endpoint, method, parsedURI, - headers, bodyProducer, requestPath) + return super(TunnelingAgent, self)._requestWithEndpoint( + key=key, + endpoint=endpoint, + method=method, + parsedURI=parsedURI, + headers=headers, + bodyProducer=bodyProducer, + requestPath=requestPath, + ) class ScrapyProxyAgent(Agent): - def __init__(self, reactor, proxyURI, - connectTimeout=None, bindAddress=None, pool=None): - super(ScrapyProxyAgent, self).__init__(reactor, - connectTimeout=connectTimeout, - bindAddress=bindAddress, - pool=pool) + def __init__(self, reactor, proxyURI, connectTimeout=None, bindAddress=None, pool=None): + super(ScrapyProxyAgent, self).__init__( + reactor=reactor, + connectTimeout=connectTimeout, + bindAddress=bindAddress, + pool=pool, + ) self._proxyURI = URI.fromBytes(proxyURI) def request(self, method, uri, headers=None, bodyProducer=None): @@ -249,16 +250,15 @@ class ScrapyProxyAgent(Agent): """ # Cache *all* connections under the same key, since we are only # connecting to a single destination, the proxy: - if twisted_version >= (15, 0, 0): - proxyEndpoint = self._getEndpoint(self._proxyURI) - else: - proxyEndpoint = self._getEndpoint(self._proxyURI.scheme, - self._proxyURI.host, - self._proxyURI.port) - key = ("http-proxy", self._proxyURI.host, self._proxyURI.port) - return self._requestWithEndpoint(key, proxyEndpoint, method, - URI.fromBytes(uri), headers, - bodyProducer, uri) + return self._requestWithEndpoint( + key=("http-proxy", self._proxyURI.host, self._proxyURI.port), + endpoint=self._getEndpoint(self._proxyURI), + method=method, + parsedURI=URI.fromBytes(uri), + headers=headers, + bodyProducer=bodyProducer, + requestPath=uri, + ) class ScrapyAgent(object): @@ -286,18 +286,39 @@ class ScrapyAgent(object): scheme = _parse(request.url)[0] proxyHost = to_unicode(proxyHost) omitConnectTunnel = b'noconnect' in proxyParams - if scheme == b'https' and not omitConnectTunnel: - proxyConf = (proxyHost, proxyPort, - request.headers.get(b'Proxy-Authorization', None)) - return self._TunnelingAgent(reactor, proxyConf, - contextFactory=self._contextFactory, connectTimeout=timeout, - bindAddress=bindaddress, pool=self._pool) + if omitConnectTunnel: + warnings.warn("Using HTTPS proxies in the noconnect mode is deprecated. " + "If you use Crawlera, it doesn't require this mode anymore, " + "so you should update scrapy-crawlera to 1.3.0+ " + "and remove '?noconnect' from the Crawlera URL.", + ScrapyDeprecationWarning) + if scheme == b'https' and not omitConnectTunnel: + proxyAuth = request.headers.get(b'Proxy-Authorization', None) + proxyConf = (proxyHost, proxyPort, proxyAuth) + return self._TunnelingAgent( + reactor=reactor, + proxyConf=proxyConf, + contextFactory=self._contextFactory, + connectTimeout=timeout, + bindAddress=bindaddress, + pool=self._pool, + ) else: - return self._ProxyAgent(reactor, proxyURI=to_bytes(proxy, encoding='ascii'), - connectTimeout=timeout, bindAddress=bindaddress, pool=self._pool) + return self._ProxyAgent( + reactor=reactor, + proxyURI=to_bytes(proxy, encoding='ascii'), + connectTimeout=timeout, + bindAddress=bindaddress, + pool=self._pool, + ) - return self._Agent(reactor, contextFactory=self._contextFactory, - connectTimeout=timeout, bindAddress=bindaddress, pool=self._pool) + return self._Agent( + reactor=reactor, + contextFactory=self._contextFactory, + connectTimeout=timeout, + bindAddress=bindaddress, + pool=self._pool, + ) def download_request(self, request): timeout = request.meta.get('download_timeout') or self._connectTimeout @@ -328,8 +349,7 @@ class ScrapyAgent(object): else: bodyproducer = None start_time = time() - d = agent.request( - method, to_bytes(url, encoding='ascii'), headers, bodyproducer) + d = agent.request(method, to_bytes(url, encoding='ascii'), headers, bodyproducer) # set download latency d.addCallback(self._cb_latency, request, start_time) # response body is ready to be consumed @@ -384,8 +404,9 @@ class ScrapyAgent(object): txresponse._transport._producer.abortConnection() d = defer.Deferred(_cancel) - txresponse.deliverBody(_ResponseReader( - d, txresponse, request, maxsize, warnsize, fail_on_dataloss)) + txresponse.deliverBody( + _ResponseReader(d, txresponse, request, maxsize, warnsize, fail_on_dataloss) + ) # save response for timeouts self._txresponse = txresponse @@ -420,22 +441,20 @@ class _RequestBodyProducer(object): class _ResponseReader(protocol.Protocol): - def __init__(self, finished, txresponse, request, maxsize, warnsize, - fail_on_dataloss): + def __init__(self, finished, txresponse, request, maxsize, warnsize, fail_on_dataloss): self._finished = finished self._txresponse = txresponse self._request = request self._bodybuf = BytesIO() - self._maxsize = maxsize - self._warnsize = warnsize + self._maxsize = maxsize + self._warnsize = warnsize self._fail_on_dataloss = fail_on_dataloss self._fail_on_dataloss_warned = False self._reached_warnsize = False self._bytes_received = 0 def dataReceived(self, bodyBytes): - # This maybe called several times after cancel was called with buffered - # data. + # This maybe called several times after cancel was called with buffered data. if self._finished.called: return @@ -448,8 +467,7 @@ class _ResponseReader(protocol.Protocol): {'bytes': self._bytes_received, 'maxsize': self._maxsize, 'request': self._request}) - # Clear buffer earlier to avoid keeping data in memory for a long - # time. + # Clear buffer earlier to avoid keeping data in memory for a long time. self._bodybuf.truncate(0) self._finished.cancel() diff --git a/scrapy/core/downloader/handlers/s3.py b/scrapy/core/downloader/handlers/s3.py index d8bbdd326..d6fbd54ee 100644 --- a/scrapy/core/downloader/handlers/s3.py +++ b/scrapy/core/downloader/handlers/s3.py @@ -1,9 +1,9 @@ -from six.moves.urllib.parse import unquote +from urllib.parse import unquote from scrapy.exceptions import NotConfigured from scrapy.utils.httpobj import urlparse_cached from scrapy.utils.boto import is_botocore -from .http import HTTPDownloadHandler +from scrapy.core.downloader.handlers.http import HTTPDownloadHandler def _get_boto_connection(): @@ -21,7 +21,7 @@ def _get_boto_connection(): return http_request.headers try: - import boto.auth + import boto.auth # noqa: F401 except ImportError: _S3Connection = _v19_S3Connection else: @@ -32,8 +32,8 @@ def _get_boto_connection(): class S3DownloadHandler(object): - def __init__(self, settings, aws_access_key_id=None, aws_secret_access_key=None, \ - httpdownloadhandler=HTTPDownloadHandler, **kw): + def __init__(self, settings, aws_access_key_id=None, aws_secret_access_key=None, + httpdownloadhandler=HTTPDownloadHandler, **kw): if not aws_access_key_id: aws_access_key_id = settings['AWS_ACCESS_KEY_ID'] diff --git a/scrapy/core/downloader/middleware.py b/scrapy/core/downloader/middleware.py index 7a6a4dfac..9c0014206 100644 --- a/scrapy/core/downloader/middleware.py +++ b/scrapy/core/downloader/middleware.py @@ -3,14 +3,12 @@ Downloader Middleware manager See documentation in docs/topics/downloader-middleware.rst """ -import six - from twisted.internet import defer from scrapy.exceptions import _InvalidOutput from scrapy.http import Request, Response from scrapy.middleware import MiddlewareManager -from scrapy.utils.defer import mustbe_deferred +from scrapy.utils.defer import mustbe_deferred, deferred_from_coro from scrapy.utils.conf import build_component_list @@ -35,10 +33,10 @@ class DownloaderMiddlewareManager(MiddlewareManager): @defer.inlineCallbacks def process_request(request): for method in self.methods['process_request']: - response = yield method(request=request, spider=spider) + response = yield deferred_from_coro(method(request=request, spider=spider)) if response is not None and not isinstance(response, (Response, Request)): raise _InvalidOutput('Middleware %s.process_request must return None, Response or Request, got %s' % \ - (six.get_method_self(method).__class__.__name__, response.__class__.__name__)) + (method.__self__.__class__.__name__, response.__class__.__name__)) if response: defer.returnValue(response) defer.returnValue((yield download_func(request=request, spider=spider))) @@ -50,10 +48,10 @@ class DownloaderMiddlewareManager(MiddlewareManager): defer.returnValue(response) for method in self.methods['process_response']: - response = yield method(request=request, response=response, spider=spider) + response = yield deferred_from_coro(method(request=request, response=response, spider=spider)) if not isinstance(response, (Response, Request)): raise _InvalidOutput('Middleware %s.process_response must return Response or Request, got %s' % \ - (six.get_method_self(method).__class__.__name__, type(response))) + (method.__self__.__class__.__name__, type(response))) if isinstance(response, Request): defer.returnValue(response) defer.returnValue(response) @@ -62,10 +60,10 @@ class DownloaderMiddlewareManager(MiddlewareManager): def process_exception(_failure): exception = _failure.value for method in self.methods['process_exception']: - response = yield method(request=request, exception=exception, spider=spider) + response = yield deferred_from_coro(method(request=request, exception=exception, spider=spider)) if response is not None and not isinstance(response, (Response, Request)): raise _InvalidOutput('Middleware %s.process_exception must return None, Response or Request, got %s' % \ - (six.get_method_self(method).__class__.__name__, type(response))) + (method.__self__.__class__.__name__, type(response))) if response: defer.returnValue(response) defer.returnValue(_failure) diff --git a/scrapy/core/downloader/tls.py b/scrapy/core/downloader/tls.py index df8051182..4ed482058 100644 --- a/scrapy/core/downloader/tls.py +++ b/scrapy/core/downloader/tls.py @@ -1,17 +1,24 @@ import logging + from OpenSSL import SSL +from service_identity.exceptions import CertificateError +from twisted.internet._sslverify import ClientTLSOptions, verifyHostname, VerificationError +from twisted.internet.ssl import AcceptableCiphers from scrapy import twisted_version +from scrapy.utils.ssl import x509name_to_string, get_temp_key_info logger = logging.getLogger(__name__) + METHOD_SSLv3 = 'SSLv3' METHOD_TLS = 'TLS' METHOD_TLSv10 = 'TLSv1.0' METHOD_TLSv11 = 'TLSv1.1' METHOD_TLSv12 = 'TLSv1.2' + openssl_methods = { METHOD_TLS: SSL.SSLv23_METHOD, # protocol negotiation (recommended) METHOD_SSLv3: SSL.SSLv3_METHOD, # SSL 3 (NOT recommended) @@ -20,69 +27,66 @@ openssl_methods = { METHOD_TLSv12: getattr(SSL, 'TLSv1_2_METHOD', 6), # TLS 1.2 only } -if twisted_version >= (14, 0, 0): - # ClientTLSOptions requires a recent-enough version of Twisted. - # Not having ScrapyClientTLSOptions should not matter for older - # Twisted versions because it is not used in the fallback - # ScrapyClientContextFactory. - # taken from twisted/twisted/internet/_sslverify.py - - try: - # XXX: this try-except is not needed in Twisted 17.0.0+ because - # it requires pyOpenSSL 0.16+. - from OpenSSL.SSL import SSL_CB_HANDSHAKE_DONE, SSL_CB_HANDSHAKE_START - except ImportError: - SSL_CB_HANDSHAKE_START = 0x10 - SSL_CB_HANDSHAKE_DONE = 0x20 - - from twisted.internet.ssl import AcceptableCiphers - from twisted.internet._sslverify import (ClientTLSOptions, - verifyHostname, - VerificationError) - try: - # XXX: this import would fail on Debian jessie with system installed - # service_identity library, due to lack of cryptography.x509 dependency - # See https://github.com/pyca/service_identity/issues/21 - from service_identity.exceptions import CertificateError - verification_errors = (CertificateError, VerificationError) - except ImportError: - verification_errors = VerificationError - - if twisted_version < (17, 0, 0): - from twisted.internet._sslverify import _maybeSetHostNameIndication - set_tlsext_host_name = _maybeSetHostNameIndication - else: - def set_tlsext_host_name(connection, hostNameBytes): - connection.set_tlsext_host_name(hostNameBytes) +if twisted_version < (17, 0, 0): + from twisted.internet._sslverify import _maybeSetHostNameIndication as set_tlsext_host_name +else: + def set_tlsext_host_name(connection, hostNameBytes): + connection.set_tlsext_host_name(hostNameBytes) - class ScrapyClientTLSOptions(ClientTLSOptions): - """ - SSL Client connection creator ignoring certificate verification errors - (for genuinely invalid certificates or bugs in verification code). +class ScrapyClientTLSOptions(ClientTLSOptions): + """ + SSL Client connection creator ignoring certificate verification errors + (for genuinely invalid certificates or bugs in verification code). - Same as Twisted's private _sslverify.ClientTLSOptions, - except that VerificationError, CertificateError and ValueError - exceptions are caught, so that the connection is not closed, only - logging warnings. - """ + Same as Twisted's private _sslverify.ClientTLSOptions, + except that VerificationError, CertificateError and ValueError + exceptions are caught, so that the connection is not closed, only + logging warnings. Also, HTTPS connection parameters logging is added. + """ - def _identityVerifyingInfoCallback(self, connection, where, ret): - if where & SSL_CB_HANDSHAKE_START: - set_tlsext_host_name(connection, self._hostnameBytes) - elif where & SSL_CB_HANDSHAKE_DONE: - try: - verifyHostname(connection, self._hostnameASCII) - except verification_errors as e: - logger.warning( - 'Remote certificate is not valid for hostname "{}"; {}'.format( - self._hostnameASCII, e)) + def __init__(self, hostname, ctx, verbose_logging=False): + super(ScrapyClientTLSOptions, self).__init__(hostname, ctx) + self.verbose_logging = verbose_logging - except ValueError as e: - logger.warning( - 'Ignoring error while verifying certificate ' - 'from host "{}" (exception: {})'.format( - self._hostnameASCII, repr(e))) + def _identityVerifyingInfoCallback(self, connection, where, ret): + if where & SSL.SSL_CB_HANDSHAKE_START: + set_tlsext_host_name(connection, self._hostnameBytes) + elif where & SSL.SSL_CB_HANDSHAKE_DONE: + if self.verbose_logging: + if hasattr(connection, 'get_cipher_name'): # requires pyOPenSSL 0.15 + if hasattr(connection, 'get_protocol_version_name'): # requires pyOPenSSL 16.0.0 + logger.debug('SSL connection to %s using protocol %s, cipher %s', + self._hostnameASCII, + connection.get_protocol_version_name(), + connection.get_cipher_name(), + ) + else: + logger.debug('SSL connection to %s using cipher %s', + self._hostnameASCII, + connection.get_cipher_name(), + ) + server_cert = connection.get_peer_certificate() + logger.debug('SSL connection certificate: issuer "%s", subject "%s"', + x509name_to_string(server_cert.get_issuer()), + x509name_to_string(server_cert.get_subject()), + ) + key_info = get_temp_key_info(connection._ssl) + if key_info: + logger.debug('SSL temp key: %s', key_info) - DEFAULT_CIPHERS = AcceptableCiphers.fromOpenSSLCipherString('DEFAULT') + try: + verifyHostname(connection, self._hostnameASCII) + except (CertificateError, VerificationError) as e: + logger.warning( + 'Remote certificate is not valid for hostname "{}"; {}'.format( + self._hostnameASCII, e)) + + except ValueError as e: + logger.warning( + 'Ignoring error while verifying certificate ' + 'from host "{}" (exception: {})'.format( + self._hostnameASCII, repr(e))) + +DEFAULT_CIPHERS = AcceptableCiphers.fromOpenSSLCipherString('DEFAULT') diff --git a/scrapy/core/downloader/webclient.py b/scrapy/core/downloader/webclient.py index 1c89a0f9e..fc796e8bb 100644 --- a/scrapy/core/downloader/webclient.py +++ b/scrapy/core/downloader/webclient.py @@ -1,5 +1,5 @@ from time import time -from six.moves.urllib.parse import urlparse, urlunparse, urldefrag +from urllib.parse import urlparse, urlunparse, urldefrag from twisted.web.client import HTTPClientFactory from twisted.web.http import HTTPClient @@ -42,7 +42,7 @@ class ScrapyHTTPPageGetter(HTTPClient): delimiter = b'\n' def connectionMade(self): - self.headers = Headers() # bucket for response headers + self.headers = Headers() # bucket for response headers # Method command self.sendCommand(self.factory.method, self.factory.path) @@ -88,14 +88,14 @@ class ScrapyHTTPPageGetter(HTTPClient): if self.factory.url.startswith(b'https'): self.transport.stopProducing() - self.factory.noPage(\ - defer.TimeoutError("Getting %s took longer than %s seconds." % \ - (self.factory.url, self.factory.timeout))) + self.factory.noPage( + defer.TimeoutError("Getting %s took longer than %s seconds." % + (self.factory.url, self.factory.timeout))) class ScrapyHTTPClientFactory(HTTPClientFactory): """Scrapy implementation of the HTTPClientFactory overwriting the - serUrl method to make use of our Url object that cache the parse + setUrl method to make use of our Url object that cache the parse result. """ @@ -157,4 +157,3 @@ class ScrapyHTTPClientFactory(HTTPClientFactory): def gotHeaders(self, headers): self.headers_time = time() self.response_headers = headers - diff --git a/scrapy/core/engine.py b/scrapy/core/engine.py index 37fe0a873..829e69993 100644 --- a/scrapy/core/engine.py +++ b/scrapy/core/engine.py @@ -25,7 +25,7 @@ class Slot(object): def __init__(self, start_requests, close_if_idle, nextcall, scheduler): self.closing = False - self.inprogress = set() # requests in progress + self.inprogress = set() # requests in progress self.start_requests = iter(start_requests) self.close_if_idle = close_if_idle self.nextcall = nextcall @@ -233,10 +233,11 @@ class ExecutionEngine(object): def _on_success(response): assert isinstance(response, (Response, Request)) if isinstance(response, Response): - response.request = request # tie request to response received + response.request = request # tie request to response received logkws = self.logformatter.crawled(request, response, spider) - logger.log(*logformatter_adapter(logkws), extra={'spider': spider}) - self.signals.send_catch_log(signal=signals.response_received, \ + if logkws is not None: + logger.log(*logformatter_adapter(logkws), extra={'spider': spider}) + self.signals.send_catch_log(signal=signals.response_received, response=response, request=request, spider=spider) return response diff --git a/scrapy/core/scraper.py b/scrapy/core/scraper.py index 66f5d0e05..facbd8b73 100644 --- a/scrapy/core/scraper.py +++ b/scrapy/core/scraper.py @@ -9,7 +9,7 @@ from twisted.internet import defer from scrapy.utils.defer import defer_result, defer_succeed, parallel, iter_errback from scrapy.utils.spider import iterate_spider_output -from scrapy.utils.misc import load_object +from scrapy.utils.misc import load_object, warn_on_generator_with_return_value from scrapy.utils.log import logformatter_adapter, failure_to_exc_info from scrapy.exceptions import CloseSpider, DropItem, IgnoreRequest from scrapy import signals @@ -18,6 +18,7 @@ from scrapy.item import BaseItem from scrapy.core.spidermw import SpiderMiddlewareManager from scrapy.utils.request import referer_str + logger = logging.getLogger(__name__) @@ -77,7 +78,7 @@ class Scraper(object): @defer.inlineCallbacks def open_spider(self, spider): """Open the given spider for scraping and allocate resources for it""" - self.slot = Slot() + self.slot = Slot(self.crawler.settings.getint('SCRAPER_SLOT_MAX_ACTIVE_SIZE')) yield self.itemproc.open_spider(spider) def close_spider(self, spider): @@ -99,11 +100,13 @@ class Scraper(object): def enqueue_scrape(self, response, request, spider): slot = self.slot dfd = slot.add_response_request(response, request) + def finish_scraping(_): slot.finish_response(response, request) self._check_if_closing(spider, slot) self._scrape_next(spider, slot) return _ + dfd.addBoth(finish_scraping) dfd.addErrback( lambda f: logger.error('Scraper bug processing %(request)s', @@ -123,7 +126,7 @@ class Scraper(object): callback/errback""" assert isinstance(response, (Response, Failure)) - dfd = self._scrape2(response, request, spider) # returns spiders processed output + dfd = self._scrape2(response, request, spider) # returns spider's processed output dfd.addErrback(self.handle_spider_error, request, response, spider) dfd.addCallback(self.handle_spider_output, request, response, spider) return dfd @@ -142,7 +145,10 @@ class Scraper(object): def call_spider(self, result, request, spider): result.request = request dfd = defer_result(result) - dfd.addCallbacks(callback=request.callback or spider.parse, + callback = request.callback or spider.parse + warn_on_generator_with_return_value(spider, callback) + warn_on_generator_with_return_value(spider, request.errback) + dfd.addCallbacks(callback=callback, errback=request.errback, callbackKeywords=request.cb_kwargs) return dfd.addCallback(iterate_spider_output) @@ -172,8 +178,8 @@ class Scraper(object): if not result: return defer_succeed(None) it = iter_errback(result, self.handle_spider_error, request, response, spider) - dfd = parallel(it, self.concurrent_items, - self._process_spidermw_output, request, response, spider) + dfd = parallel(it, self.concurrent_items, self._process_spidermw_output, + request, response, spider) return dfd def _process_spidermw_output(self, output, request, response, spider): @@ -200,8 +206,7 @@ class Scraper(object): """Log and silence errors that come from the engine (typically download errors that got propagated thru here) """ - if (isinstance(download_failure, Failure) and - not download_failure.check(IgnoreRequest)): + if isinstance(download_failure, Failure) and not download_failure.check(IgnoreRequest): if download_failure.frames: logger.error('Error downloading %(request)s', {'request': request}, @@ -225,21 +230,22 @@ class Scraper(object): ex = output.value if isinstance(ex, DropItem): logkws = self.logformatter.dropped(item, ex, response, spider) - logger.log(*logformatter_adapter(logkws), extra={'spider': spider}) + if logkws is not None: + logger.log(*logformatter_adapter(logkws), extra={'spider': spider}) return self.signals.send_catch_log_deferred( signal=signals.item_dropped, item=item, response=response, spider=spider, exception=output.value) else: - logger.error('Error processing %(item)s', {'item': item}, - exc_info=failure_to_exc_info(output), - extra={'spider': spider}) + logkws = self.logformatter.error(item, ex, response, spider) + logger.log(*logformatter_adapter(logkws), extra={'spider': spider}, + exc_info=failure_to_exc_info(output)) return self.signals.send_catch_log_deferred( signal=signals.item_error, item=item, response=response, spider=spider, failure=output) else: logkws = self.logformatter.scraped(output, response, spider) - logger.log(*logformatter_adapter(logkws), extra={'spider': spider}) + if logkws is not None: + logger.log(*logformatter_adapter(logkws), extra={'spider': spider}) return self.signals.send_catch_log_deferred( signal=signals.item_scraped, item=output, response=response, spider=spider) - diff --git a/scrapy/core/spidermw.py b/scrapy/core/spidermw.py index b5f9837ff..097a374bf 100644 --- a/scrapy/core/spidermw.py +++ b/scrapy/core/spidermw.py @@ -5,7 +5,6 @@ See documentation in docs/topics/spider-middleware.rst """ from itertools import chain, islice -import six from twisted.python.failure import Failure from scrapy.exceptions import _InvalidOutput from scrapy.middleware import MiddlewareManager @@ -36,16 +35,16 @@ class SpiderMiddlewareManager(MiddlewareManager): self.methods['process_spider_exception'].appendleft(getattr(mw, 'process_spider_exception', None)) def scrape_response(self, scrape_func, response, request, spider): - fname = lambda f:'%s.%s' % ( - six.get_method_self(f).__class__.__name__, - six.get_method_function(f).__name__) + fname = lambda f: '%s.%s' % ( + f.__self__.__class__.__name__, + f.__func__.__name__) def process_spider_input(response): for method in self.methods['process_spider_input']: try: result = method(response=response, spider=spider) if result is not None: - raise _InvalidOutput('Middleware {} must return None or raise an exception, got {}' \ + raise _InvalidOutput('Middleware {} must return None or raise an exception, got {}' .format(fname(method), type(result))) except _InvalidOutput: raise @@ -70,7 +69,7 @@ class SpiderMiddlewareManager(MiddlewareManager): elif result is None: continue else: - raise _InvalidOutput('Middleware {} must return None or an iterable, got {}' \ + raise _InvalidOutput('Middleware {} must return None or an iterable, got {}' .format(fname(method), type(result))) return _failure @@ -104,7 +103,7 @@ class SpiderMiddlewareManager(MiddlewareManager): if _isiterable(result): result = evaluate_iterable(result, method_index) else: - raise _InvalidOutput('Middleware {} must return an iterable, got {}' \ + raise _InvalidOutput('Middleware {} must return an iterable, got {}' .format(fname(method), type(result))) return chain(result, recovered) diff --git a/scrapy/crawler.py b/scrapy/crawler.py index ded3c082b..f87e67d93 100644 --- a/scrapy/crawler.py +++ b/scrapy/crawler.py @@ -1,10 +1,9 @@ -import six -import signal import logging +import pprint +import signal import warnings -import sys -from twisted.internet import reactor, defer +from twisted.internet import defer from zope.interface.verify import verifyClass, DoesNotImplement from scrapy import Spider @@ -15,6 +14,7 @@ from scrapy.extension import ExtensionManager from scrapy.settings import overridden_settings, Settings from scrapy.signalmanager import SignalManager from scrapy.exceptions import ScrapyDeprecationWarning +from scrapy.utils.asyncio import install_asyncio_reactor, is_asyncio_reactor_installed from scrapy.utils.ossignal import install_shutdown_handlers, signal_names from scrapy.utils.misc import load_object from scrapy.utils.log import ( @@ -22,6 +22,7 @@ from scrapy.utils.log import ( get_scrapy_root_handler, install_scrapy_root_handler) from scrapy import signals + logger = logging.getLogger(__name__) @@ -46,7 +47,8 @@ class Crawler(object): logging.root.addHandler(handler) d = dict(overridden_settings(self.settings)) - logger.info("Overridden settings: %(settings)r", {'settings': d}) + logger.info("Overridden settings:\n%(settings)s", + {'settings': pprint.pformat(d)}) if get_scrapy_root_handler() is not None: # scrapy root handler already installed: update it with new settings @@ -88,20 +90,9 @@ class Crawler(object): yield self.engine.open_spider(self.spider, start_requests) yield defer.maybeDeferred(self.engine.start) except Exception: - # In Python 2 reraising an exception after yield discards - # the original traceback (see https://bugs.python.org/issue7563), - # so sys.exc_info() workaround is used. - # This workaround also works in Python 3, but it is not needed, - # and it is slower, so in Python 3 we use native `raise`. - if six.PY2: - exc_info = sys.exc_info() - self.crawling = False if self.engine is not None: yield self.engine.close() - - if six.PY2: - six.reraise(*exc_info) raise def _create_spider(self, *args, **kwargs): @@ -122,7 +113,7 @@ class Crawler(object): class CrawlerRunner(object): """ This is a convenient helper class that keeps track of, manages and runs - crawlers inside an already setup Twisted `reactor`_. + crawlers inside an already setup :mod:`~twisted.internet.reactor`. The CrawlerRunner object must be instantiated with a :class:`~scrapy.settings.Settings` object. @@ -146,6 +137,7 @@ class CrawlerRunner(object): self._crawlers = set() self._active = set() self.bootstrap_failed = False + self._handle_asyncio_reactor() @property def spiders(self): @@ -216,7 +208,7 @@ class CrawlerRunner(object): return self._create_crawler(crawler_or_spidercls) def _create_crawler(self, spidercls): - if isinstance(spidercls, six.string_types): + if isinstance(spidercls, str): spidercls = self.spider_loader.load(spidercls) return Crawler(spidercls, self.settings) @@ -239,18 +231,24 @@ class CrawlerRunner(object): while self._active: yield defer.DeferredList(self._active) + def _handle_asyncio_reactor(self): + if self.settings.getbool('ASYNCIO_REACTOR') and not is_asyncio_reactor_installed(): + raise Exception("ASYNCIO_REACTOR is on but the Twisted asyncio " + "reactor is not installed.") + class CrawlerProcess(CrawlerRunner): """ A class to run multiple scrapy crawlers in a process simultaneously. This class extends :class:`~scrapy.crawler.CrawlerRunner` by adding support - for starting a Twisted `reactor`_ and handling shutdown signals, like the - keyboard interrupt command Ctrl-C. It also configures top-level logging. + for starting a :mod:`~twisted.internet.reactor` and handling shutdown + signals, like the keyboard interrupt command Ctrl-C. It also configures + top-level logging. This utility should be a better fit than :class:`~scrapy.crawler.CrawlerRunner` if you aren't running another - Twisted `reactor`_ within your application. + :mod:`~twisted.internet.reactor` within your application. The CrawlerProcess object must be instantiated with a :class:`~scrapy.settings.Settings` object. @@ -270,6 +268,7 @@ class CrawlerProcess(CrawlerRunner): log_scrapy_info(self.settings) def _signal_shutdown(self, signum, _): + from twisted.internet import reactor install_shutdown_handlers(self._signal_kill) signame = signal_names[signum] logger.info("Received %(signame)s, shutting down gracefully. Send again to force ", @@ -277,6 +276,7 @@ class CrawlerProcess(CrawlerRunner): reactor.callFromThread(self._graceful_stop_reactor) def _signal_kill(self, signum, _): + from twisted.internet import reactor install_shutdown_handlers(signal.SIG_IGN) signame = signal_names[signum] logger.info('Received %(signame)s twice, forcing unclean shutdown', @@ -285,9 +285,9 @@ class CrawlerProcess(CrawlerRunner): def start(self, stop_after_crawl=True): """ - This method starts a Twisted `reactor`_, adjusts its pool size to - :setting:`REACTOR_THREADPOOL_MAXSIZE`, and installs a DNS cache based - on :setting:`DNSCACHE_ENABLED` and :setting:`DNSCACHE_SIZE`. + This method starts a :mod:`~twisted.internet.reactor`, adjusts its pool + size to :setting:`REACTOR_THREADPOOL_MAXSIZE`, and installs a DNS cache + based on :setting:`DNSCACHE_ENABLED` and :setting:`DNSCACHE_SIZE`. If ``stop_after_crawl`` is True, the reactor will be stopped after all crawlers have finished, using :meth:`join`. @@ -295,6 +295,7 @@ class CrawlerProcess(CrawlerRunner): :param boolean stop_after_crawl: stop or not the reactor when all crawlers have finished """ + from twisted.internet import reactor if stop_after_crawl: d = self.join() # Don't start the reactor if the deferreds are already fired @@ -309,6 +310,7 @@ class CrawlerProcess(CrawlerRunner): reactor.run(installSignalHandlers=False) # blocking call def _get_dns_resolver(self): + from twisted.internet import reactor if self.settings.getbool('DNSCACHE_ENABLED'): cache_size = self.settings.getint('DNSCACHE_SIZE') else: @@ -325,11 +327,17 @@ class CrawlerProcess(CrawlerRunner): return d def _stop_reactor(self, _=None): + from twisted.internet import reactor try: reactor.stop() except RuntimeError: # raised if already stopped or in shutdown stage pass + def _handle_asyncio_reactor(self): + if self.settings.getbool('ASYNCIO_REACTOR'): + install_asyncio_reactor() + super()._handle_asyncio_reactor() + def _get_spider_loader(settings): """ Get SpiderLoader instance from settings """ diff --git a/scrapy/downloadermiddlewares/ajaxcrawl.py b/scrapy/downloadermiddlewares/ajaxcrawl.py index 72715dba7..7a140fcad 100644 --- a/scrapy/downloadermiddlewares/ajaxcrawl.py +++ b/scrapy/downloadermiddlewares/ajaxcrawl.py @@ -1,9 +1,7 @@ # -*- coding: utf-8 -*- -from __future__ import absolute_import import re import logging -import six from w3lib import html from scrapy.exceptions import NotConfigured @@ -67,7 +65,9 @@ class AjaxCrawlMiddleware(object): # XXX: move it to w3lib? -_ajax_crawlable_re = re.compile(six.u(r'<meta\s+name=["\']fragment["\']\s+content=["\']!["\']/?>')) +_ajax_crawlable_re = re.compile(r'<meta\s+name=["\']fragment["\']\s+content=["\']!["\']/?>') + + def _has_ajaxcrawlable_meta(text): """ >>> _has_ajaxcrawlable_meta('<html><head><meta name="fragment" content="!"/></head><body></body></html>') diff --git a/scrapy/downloadermiddlewares/cookies.py b/scrapy/downloadermiddlewares/cookies.py index 321c0171b..d8dabdf13 100644 --- a/scrapy/downloadermiddlewares/cookies.py +++ b/scrapy/downloadermiddlewares/cookies.py @@ -1,12 +1,11 @@ -import os -import six import logging from collections import defaultdict from scrapy.exceptions import NotConfigured from scrapy.http import Response from scrapy.http.cookies import CookieJar -from scrapy.utils.python import to_native_str +from scrapy.utils.python import to_unicode + logger = logging.getLogger(__name__) @@ -53,7 +52,7 @@ class CookiesMiddleware(object): def _debug_cookie(self, request, spider): if self.debug: - cl = [to_native_str(c, errors='replace') + cl = [to_unicode(c, errors='replace') for c in request.headers.getlist('Cookie')] if cl: cookies = "\n".join("Cookie: {}\n".format(c) for c in cl) @@ -62,7 +61,7 @@ class CookiesMiddleware(object): def _debug_set_cookie(self, response, spider): if self.debug: - cl = [to_native_str(c, errors='replace') + cl = [to_unicode(c, errors='replace') for c in response.headers.getlist('Set-Cookie')] if cl: cookies = "\n".join("Set-Cookie: {}\n".format(c) for c in cl) @@ -82,8 +81,10 @@ class CookiesMiddleware(object): def _get_request_cookies(self, jar, request): if isinstance(request.cookies, dict): - cookie_list = [{'name': k, 'value': v} for k, v in \ - six.iteritems(request.cookies)] + cookie_list = [ + {'name': k, 'value': v} + for k, v in request.cookies.items() + ] else: cookie_list = request.cookies diff --git a/scrapy/downloadermiddlewares/decompression.py b/scrapy/downloadermiddlewares/decompression.py index 49313cc04..fcea38ef5 100644 --- a/scrapy/downloadermiddlewares/decompression.py +++ b/scrapy/downloadermiddlewares/decompression.py @@ -4,20 +4,15 @@ and extract the potentially compressed responses that may arrive. import bz2 import gzip -import zipfile -import tarfile import logging +import tarfile +import zipfile +from io import BytesIO from tempfile import mktemp -import six - -try: - from cStringIO import StringIO as BytesIO -except ImportError: - from io import BytesIO - from scrapy.responsetypes import responsetypes + logger = logging.getLogger(__name__) @@ -79,7 +74,7 @@ class DecompressionMiddleware(object): if not response.body: return response - for fmt, func in six.iteritems(self._formats): + for fmt, func in self._formats.items(): new_response = func(response) if new_response: logger.debug('Decompressed response with format: %(responsefmt)s', diff --git a/scrapy/downloadermiddlewares/httpcache.py b/scrapy/downloadermiddlewares/httpcache.py index 495b103d1..4e06f8236 100644 --- a/scrapy/downloadermiddlewares/httpcache.py +++ b/scrapy/downloadermiddlewares/httpcache.py @@ -1,11 +1,19 @@ from email.utils import formatdate + from twisted.internet import defer -from twisted.internet.error import TimeoutError, DNSLookupError, \ - ConnectionRefusedError, ConnectionDone, ConnectError, \ - ConnectionLost, TCPTimedOutError +from twisted.internet.error import ( + ConnectError, + ConnectionDone, + ConnectionLost, + ConnectionRefusedError, + DNSLookupError, + TCPTimedOutError, + TimeoutError, +) from twisted.web.client import ResponseFailed + from scrapy import signals -from scrapy.exceptions import NotConfigured, IgnoreRequest +from scrapy.exceptions import IgnoreRequest, NotConfigured from scrapy.utils.misc import load_object diff --git a/scrapy/downloadermiddlewares/httpcompression.py b/scrapy/downloadermiddlewares/httpcompression.py index 203dee42d..65b652953 100644 --- a/scrapy/downloadermiddlewares/httpcompression.py +++ b/scrapy/downloadermiddlewares/httpcompression.py @@ -37,8 +37,9 @@ class HttpCompressionMiddleware(object): if content_encoding: encoding = content_encoding.pop() decoded_body = self._decode(response.body, encoding.lower()) - respcls = responsetypes.from_args(headers=response.headers, \ - url=response.url, body=decoded_body) + respcls = responsetypes.from_args( + headers=response.headers, url=response.url, body=decoded_body + ) kwargs = dict(cls=respcls, body=decoded_body) if issubclass(respcls, TextResponse): # force recalculating the encoding until we make sure the diff --git a/scrapy/downloadermiddlewares/httpproxy.py b/scrapy/downloadermiddlewares/httpproxy.py index 2c35d1b90..814ce78fe 100644 --- a/scrapy/downloadermiddlewares/httpproxy.py +++ b/scrapy/downloadermiddlewares/httpproxy.py @@ -1,10 +1,6 @@ import base64 -from six.moves.urllib.parse import unquote, urlunparse -from six.moves.urllib.request import getproxies, proxy_bypass -try: - from urllib2 import _parse_proxy -except ImportError: - from urllib.request import _parse_proxy +from urllib.parse import unquote, urlunparse +from urllib.request import getproxies, proxy_bypass, _parse_proxy from scrapy.exceptions import NotConfigured from scrapy.utils.httpobj import urlparse_cached diff --git a/scrapy/downloadermiddlewares/redirect.py b/scrapy/downloadermiddlewares/redirect.py index 49468a2e4..77cb5aa94 100644 --- a/scrapy/downloadermiddlewares/redirect.py +++ b/scrapy/downloadermiddlewares/redirect.py @@ -1,5 +1,5 @@ import logging -from six.moves.urllib.parse import urljoin +from urllib.parse import urljoin, urlparse from w3lib.url import safe_url_string @@ -7,6 +7,7 @@ from scrapy.http import HtmlResponse from scrapy.utils.response import get_meta_refresh from scrapy.exceptions import IgnoreRequest, NotConfigured + logger = logging.getLogger(__name__) @@ -70,7 +71,10 @@ class RedirectMiddleware(BaseRedirectMiddleware): if 'Location' not in response.headers or response.status not in allowed_status: return response - location = safe_url_string(response.headers['location']) + location = safe_url_string(response.headers['Location']) + if response.headers['Location'].startswith(b'//'): + request_scheme = urlparse(request.url).scheme + location = request_scheme + '://' + location.lstrip('/') redirected_url = urljoin(request.url, location) diff --git a/scrapy/downloadermiddlewares/robotstxt.py b/scrapy/downloadermiddlewares/robotstxt.py index 200245210..251706c50 100644 --- a/scrapy/downloadermiddlewares/robotstxt.py +++ b/scrapy/downloadermiddlewares/robotstxt.py @@ -6,14 +6,12 @@ enable this middleware and enable the ROBOTSTXT_OBEY setting. import logging -from six.moves.urllib import robotparser - from twisted.internet.defer import Deferred, maybeDeferred from scrapy.exceptions import NotConfigured, IgnoreRequest from scrapy.http import Request from scrapy.utils.httpobj import urlparse_cached from scrapy.utils.log import failure_to_exc_info -from scrapy.utils.python import to_native_str +from scrapy.utils.misc import load_object logger = logging.getLogger(__name__) @@ -24,10 +22,14 @@ class RobotsTxtMiddleware(object): def __init__(self, crawler): if not crawler.settings.getbool('ROBOTSTXT_OBEY'): raise NotConfigured - + self._default_useragent = crawler.settings.get('USER_AGENT', 'Scrapy') + self._robotstxt_useragent = crawler.settings.get('ROBOTSTXT_USER_AGENT', None) self.crawler = crawler - self._useragent = crawler.settings.get('USER_AGENT') self._parsers = {} + self._parserimpl = load_object(crawler.settings.get('ROBOTSTXT_PARSER')) + + # check if parser dependencies are met, this should throw an error otherwise. + self._parserimpl.from_crawler(self.crawler, b'') @classmethod def from_crawler(cls, crawler): @@ -43,7 +45,11 @@ class RobotsTxtMiddleware(object): def process_request_2(self, rp, request, spider): if rp is None: return - if not rp.can_fetch(to_native_str(self._useragent), request.url): + + useragent = self._robotstxt_useragent + if not useragent: + useragent = request.headers.get(b'User-Agent', self._default_useragent) + if not rp.allowed(request.url, useragent): logger.debug("Forbidden by robots.txt: %(request)s", {'request': request}, extra={'spider': spider}) self.crawler.stats.inc_value('robotstxt/forbidden') @@ -62,13 +68,14 @@ class RobotsTxtMiddleware(object): meta={'dont_obey_robotstxt': True} ) dfd = self.crawler.engine.download(robotsreq, spider) - dfd.addCallback(self._parse_robots, netloc) + dfd.addCallback(self._parse_robots, netloc, spider) dfd.addErrback(self._logerror, robotsreq, spider) dfd.addErrback(self._robots_error, netloc) self.crawler.stats.inc_value('robotstxt/request_count') if isinstance(self._parsers[netloc], Deferred): d = Deferred() + def cb(result): d.callback(result) return result @@ -85,27 +92,10 @@ class RobotsTxtMiddleware(object): extra={'spider': spider}) return failure - def _parse_robots(self, response, netloc): + def _parse_robots(self, response, netloc, spider): self.crawler.stats.inc_value('robotstxt/response_count') - self.crawler.stats.inc_value( - 'robotstxt/response_status_count/{}'.format(response.status)) - rp = robotparser.RobotFileParser(response.url) - body = '' - if hasattr(response, 'text'): - body = response.text - else: # last effort try - try: - body = response.body.decode('utf-8') - except UnicodeDecodeError: - # If we found garbage, disregard it:, - # but keep the lookup cached (in self._parsers) - # Running rp.parse() will set rp state from - # 'disallow all' to 'allow any'. - self.crawler.stats.inc_value('robotstxt/unicode_error_count') - # stdlib's robotparser expects native 'str' ; - # with unicode input, non-ASCII encoded bytes decoding fails in Python2 - rp.parse(to_native_str(body).splitlines()) - + self.crawler.stats.inc_value('robotstxt/response_status_count/{}'.format(response.status)) + rp = self._parserimpl.from_crawler(self.crawler, response.body) rp_dfd = self._parsers[netloc] self._parsers[netloc] = rp rp_dfd.callback(rp) diff --git a/scrapy/dupefilters.py b/scrapy/dupefilters.py index 0bcdd3495..ea6a4cfc3 100644 --- a/scrapy/dupefilters.py +++ b/scrapy/dupefilters.py @@ -1,10 +1,10 @@ -from __future__ import print_function import os import logging from scrapy.utils.job import job_dir from scrapy.utils.request import referer_str, request_fingerprint + class BaseDupeFilter(object): @classmethod diff --git a/scrapy/exceptions.py b/scrapy/exceptions.py index 96949bdd9..7c4bb3d00 100644 --- a/scrapy/exceptions.py +++ b/scrapy/exceptions.py @@ -7,10 +7,12 @@ new exceptions here without documenting them there. # Internal + class NotConfigured(Exception): """Indicates a missing configuration situation""" pass + class _InvalidOutput(TypeError): """ Indicates an invalid value has been returned by a middleware's processing method. @@ -18,15 +20,19 @@ class _InvalidOutput(TypeError): """ pass + # HTTP and crawling + class IgnoreRequest(Exception): """Indicates a decision was made not to process a request""" + class DontCloseSpider(Exception): """Request the spider not to be closed yet""" pass + class CloseSpider(Exception): """Raise this from callbacks to request the spider to be closed""" @@ -34,30 +40,37 @@ class CloseSpider(Exception): super(CloseSpider, self).__init__() self.reason = reason + # Items + class DropItem(Exception): """Drop item from the item pipeline""" pass + class NotSupported(Exception): """Indicates a feature or method is not supported""" pass + # Commands + class UsageError(Exception): """To indicate a command-line usage error""" def __init__(self, *a, **kw): self.print_help = kw.pop('print_help', True) super(UsageError, self).__init__(*a, **kw) + class ScrapyDeprecationWarning(Warning): """Warning category for deprecated features, since the default DeprecationWarning is silenced on Python 2.7+ """ pass + class ContractFail(AssertionError): """Error raised in case of a failing contract""" pass diff --git a/scrapy/exporters.py b/scrapy/exporters.py index 695c74fec..1a3c9345f 100644 --- a/scrapy/exporters.py +++ b/scrapy/exporters.py @@ -4,18 +4,16 @@ Item Exporters are used to export/serialize items into different formats. import csv import io -import sys import pprint import marshal -import six -from six.moves import cPickle as pickle +import warnings +import pickle from xml.sax.saxutils import XMLGenerator from scrapy.utils.serialize import ScrapyJSONEncoder -from scrapy.utils.python import to_bytes, to_unicode, to_native_str, is_listlike +from scrapy.utils.python import to_bytes, to_unicode, is_listlike from scrapy.item import BaseItem from scrapy.exceptions import ScrapyDeprecationWarning -import warnings __all__ = ['BaseItemExporter', 'PprintItemExporter', 'PickleItemExporter', @@ -25,13 +23,14 @@ __all__ = ['BaseItemExporter', 'PprintItemExporter', 'PickleItemExporter', class BaseItemExporter(object): - def __init__(self, **kwargs): - self._configure(kwargs) + def __init__(self, dont_fail=False, **kwargs): + self._kwargs = kwargs + self._configure(kwargs, dont_fail=dont_fail) def _configure(self, options, dont_fail=False): """Configure the exporter by poping options from the ``options`` dict. If dont_fail is set, it won't raise an exception on unexpected options - (useful for using with keyword arguments in subclasses constructors) + (useful for using with keyword arguments in subclasses ``__init__`` methods) """ self.encoding = options.pop('encoding', None) self.fields_to_export = options.pop('fields_to_export', None) @@ -61,9 +60,9 @@ class BaseItemExporter(object): include_empty = self.export_empty_fields if self.fields_to_export is None: if include_empty and not isinstance(item, dict): - field_iter = six.iterkeys(item.fields) + field_iter = item.fields.keys() else: - field_iter = six.iterkeys(item) + field_iter = item.keys() else: if include_empty: field_iter = self.fields_to_export @@ -83,10 +82,10 @@ class BaseItemExporter(object): class JsonLinesItemExporter(BaseItemExporter): def __init__(self, file, **kwargs): - self._configure(kwargs, dont_fail=True) + super().__init__(dont_fail=True, **kwargs) self.file = file - kwargs.setdefault('ensure_ascii', not self.encoding) - self.encoder = ScrapyJSONEncoder(**kwargs) + self._kwargs.setdefault('ensure_ascii', not self.encoding) + self.encoder = ScrapyJSONEncoder(**self._kwargs) def export_item(self, item): itemdict = dict(self._get_serialized_fields(item)) @@ -97,15 +96,15 @@ class JsonLinesItemExporter(BaseItemExporter): class JsonItemExporter(BaseItemExporter): def __init__(self, file, **kwargs): - self._configure(kwargs, dont_fail=True) + super().__init__(dont_fail=True, **kwargs) self.file = file # there is a small difference between the behaviour or JsonItemExporter.indent # and ScrapyJSONEncoder.indent. ScrapyJSONEncoder.indent=None is needed to prevent # the addition of newlines everywhere json_indent = self.indent if self.indent is not None and self.indent > 0 else None - kwargs.setdefault('indent', json_indent) - kwargs.setdefault('ensure_ascii', not self.encoding) - self.encoder = ScrapyJSONEncoder(**kwargs) + self._kwargs.setdefault('indent', json_indent) + self._kwargs.setdefault('ensure_ascii', not self.encoding) + self.encoder = ScrapyJSONEncoder(**self._kwargs) self.first_item = True def _beautify_newline(self): @@ -136,18 +135,18 @@ class XmlItemExporter(BaseItemExporter): def __init__(self, file, **kwargs): self.item_element = kwargs.pop('item_element', 'item') self.root_element = kwargs.pop('root_element', 'items') - self._configure(kwargs) + super().__init__(**kwargs) if not self.encoding: self.encoding = 'utf-8' self.xg = XMLGenerator(file, encoding=self.encoding) def _beautify_newline(self, new_item=False): if self.indent is not None and (self.indent > 0 or new_item): - self._xg_characters('\n') + self.xg.characters('\n') def _beautify_indent(self, depth=1): if self.indent: - self._xg_characters(' ' * self.indent * depth) + self.xg.characters(' ' * self.indent * depth) def start_exporting(self): self.xg.startDocument() @@ -181,32 +180,18 @@ class XmlItemExporter(BaseItemExporter): for value in serialized_value: self._export_xml_field('value', value, depth=depth+1) self._beautify_indent(depth=depth) - elif isinstance(serialized_value, six.text_type): - self._xg_characters(serialized_value) + elif isinstance(serialized_value, str): + self.xg.characters(serialized_value) else: - self._xg_characters(str(serialized_value)) + self.xg.characters(str(serialized_value)) self.xg.endElement(name) self._beautify_newline() - # Workaround for https://bugs.python.org/issue17606 - # Before Python 2.7.4 xml.sax.saxutils required bytes; - # since 2.7.4 it requires unicode. The bug is likely to be - # fixed in 2.7.6, but 2.7.6 will still support unicode, - # and Python 3.x will require unicode, so ">= 2.7.4" should be fine. - if sys.version_info[:3] >= (2, 7, 4): - def _xg_characters(self, serialized_value): - if not isinstance(serialized_value, six.text_type): - serialized_value = serialized_value.decode(self.encoding) - return self.xg.characters(serialized_value) - else: # pragma: no cover - def _xg_characters(self, serialized_value): - return self.xg.characters(serialized_value) - class CsvItemExporter(BaseItemExporter): def __init__(self, file, include_headers_line=True, join_multivalued=',', **kwargs): - self._configure(kwargs, dont_fail=True) + super().__init__(dont_fail=True, **kwargs) if not self.encoding: self.encoding = 'utf-8' self.include_headers_line = include_headers_line @@ -215,9 +200,9 @@ class CsvItemExporter(BaseItemExporter): line_buffering=False, write_through=True, encoding=self.encoding, - newline='' # Windows needs this https://github.com/scrapy/scrapy/issues/3034 - ) if six.PY3 else file - self.csv_writer = csv.writer(self.stream, **kwargs) + newline='' # Windows needs this https://github.com/scrapy/scrapy/issues/3034 + ) + self.csv_writer = csv.writer(self.stream, **self._kwargs) self._headers_not_written = True self._join_multivalued = join_multivalued @@ -246,7 +231,7 @@ class CsvItemExporter(BaseItemExporter): def _build_row(self, values): for s in values: try: - yield to_native_str(s, self.encoding) + yield to_unicode(s, self.encoding) except TypeError: yield s @@ -266,7 +251,7 @@ class CsvItemExporter(BaseItemExporter): class PickleItemExporter(BaseItemExporter): def __init__(self, file, protocol=2, **kwargs): - self._configure(kwargs) + super().__init__(**kwargs) self.file = file self.protocol = protocol @@ -276,9 +261,16 @@ class PickleItemExporter(BaseItemExporter): class MarshalItemExporter(BaseItemExporter): + """Exports items in a Python-specific binary format (see + :mod:`marshal`). + + :param file: The file-like object to use for exporting the data. Its + ``write`` method should accept :class:`bytes` (a disk file + opened in binary mode, a :class:`~io.BytesIO` object, etc) + """ def __init__(self, file, **kwargs): - self._configure(kwargs) + super().__init__(**kwargs) self.file = file def export_item(self, item): @@ -288,7 +280,7 @@ class MarshalItemExporter(BaseItemExporter): class PprintItemExporter(BaseItemExporter): def __init__(self, file, **kwargs): - self._configure(kwargs) + super().__init__(**kwargs) self.file = file def export_item(self, item): @@ -297,10 +289,13 @@ class PprintItemExporter(BaseItemExporter): class PythonItemExporter(BaseItemExporter): - """The idea behind this exporter is to have a mechanism to serialize items - to built-in python types so any serialization library (like - json, msgpack, binc, etc) can be used on top of it. Its main goal is to - seamless support what BaseItemExporter does plus nested items. + """This is a base class for item exporters that extends + :class:`BaseItemExporter` with support for nested items. + + It serializes items to built-in Python types, so that any serialization + library (e.g. :mod:`json` or msgpack_) can be used on top of it. + + .. _msgpack: https://pypi.org/project/msgpack/ """ def _configure(self, options, dont_fail=False): self.binary = options.pop('binary', True) @@ -324,12 +319,12 @@ class PythonItemExporter(BaseItemExporter): if is_listlike(value): return [self._serialize_value(v) for v in value] encode_func = to_bytes if self.binary else to_unicode - if isinstance(value, (six.text_type, bytes)): + if isinstance(value, (str, bytes)): return encode_func(value, encoding=self.encoding) return value def _serialize_dict(self, value): - for key, val in six.iteritems(value): + for key, val in value.items(): key = to_bytes(key) if self.binary else key yield key, self._serialize_value(val) diff --git a/scrapy/extension.py b/scrapy/extension.py index e39e456fa..050b87e5f 100644 --- a/scrapy/extension.py +++ b/scrapy/extension.py @@ -6,6 +6,7 @@ See documentation in docs/topics/extensions.rst from scrapy.middleware import MiddlewareManager from scrapy.utils.conf import build_component_list + class ExtensionManager(MiddlewareManager): component_name = 'extension' diff --git a/scrapy/extensions/closespider.py b/scrapy/extensions/closespider.py index 9ccf356ec..afb2ed049 100644 --- a/scrapy/extensions/closespider.py +++ b/scrapy/extensions/closespider.py @@ -54,9 +54,9 @@ class CloseSpider(object): self.crawler.engine.close_spider(spider, 'closespider_pagecount') def spider_opened(self, spider): - self.task = reactor.callLater(self.close_on['timeout'], \ - self.crawler.engine.close_spider, spider, \ - reason='closespider_timeout') + self.task = reactor.callLater(self.close_on['timeout'], + self.crawler.engine.close_spider, spider, + reason='closespider_timeout') def item_scraped(self, item, spider): self.counter['itemcount'] += 1 diff --git a/scrapy/extensions/corestats.py b/scrapy/extensions/corestats.py index 8cc5e18ac..20adfbe4b 100644 --- a/scrapy/extensions/corestats.py +++ b/scrapy/extensions/corestats.py @@ -1,14 +1,16 @@ """ Extension for collecting core stats like items scraped and start/finish times """ -import datetime +from datetime import datetime from scrapy import signals + class CoreStats(object): def __init__(self, stats): self.stats = stats + self.start_time = None @classmethod def from_crawler(cls, crawler): @@ -21,11 +23,12 @@ class CoreStats(object): return o def spider_opened(self, spider): - self.stats.set_value('start_time', datetime.datetime.utcnow(), spider=spider) + self.start_time = datetime.utcnow() + self.stats.set_value('start_time', self.start_time, spider=spider) def spider_closed(self, spider, reason): - finish_time = datetime.datetime.utcnow() - elapsed_time = finish_time - self.stats.get_value('start_time') + finish_time = datetime.utcnow() + elapsed_time = finish_time - self.start_time elapsed_time_seconds = elapsed_time.total_seconds() self.stats.set_value('elapsed_time_seconds', elapsed_time_seconds, spider=spider) self.stats.set_value('finish_time', finish_time, spider=spider) diff --git a/scrapy/extensions/feedexport.py b/scrapy/extensions/feedexport.py index 06b5a0dd9..c1ca0ca67 100644 --- a/scrapy/extensions/feedexport.py +++ b/scrapy/extensions/feedexport.py @@ -10,8 +10,7 @@ import logging import posixpath from tempfile import NamedTemporaryFile from datetime import datetime -import six -from six.moves.urllib.parse import urlparse +from urllib.parse import urlparse, unquote from ftplib import FTP from zope.interface import Interface, implementer @@ -65,7 +64,7 @@ class StdoutFeedStorage(object): def __init__(self, uri, _stdout=None): if not _stdout: - _stdout = sys.stdout if six.PY2 else sys.stdout.buffer + _stdout = sys.stdout.buffer self._stdout = _stdout def open(self, spider): @@ -162,7 +161,7 @@ class FTPFeedStorage(BlockingFeedStorage): self.host = u.hostname self.port = int(u.port or '21') self.username = u.username - self.password = u.password + self.password = unquote(u.password) self.path = u.path self.use_active_mode = use_active_mode @@ -194,9 +193,9 @@ class FeedExporter(object): def __init__(self, settings): self.settings = settings - self.urifmt = settings['FEED_URI'] - if not self.urifmt: + if not settings['FEED_URI']: raise NotConfigured + self.urifmt = str(settings['FEED_URI']) self.format = settings['FEED_FORMAT'].lower() self.export_encoding = settings['FEED_EXPORT_ENCODING'] self.storages = self._load_components('FEED_STORAGES') @@ -237,7 +236,9 @@ class FeedExporter(object): def close_spider(self, spider): slot = self.slot if not slot.itemcount and not self.store_empty: - return + # We need to call slot.storage.store nonetheless to get the file + # properly closed. + return defer.maybeDeferred(slot.storage.store, slot.file) if self._exporting: slot.exporter.finish_exporting() self._exporting = False diff --git a/scrapy/extensions/httpcache.py b/scrapy/extensions/httpcache.py index c6094643d..91850683f 100644 --- a/scrapy/extensions/httpcache.py +++ b/scrapy/extensions/httpcache.py @@ -1,19 +1,20 @@ -from __future__ import print_function -import os import gzip import logging -from six.moves import cPickle as pickle +import os +import pickle +from email.utils import mktime_tz, parsedate_tz from importlib import import_module from time import time from weakref import WeakKeyDictionary -from email.utils import mktime_tz, parsedate_tz + from w3lib.http import headers_raw_to_dict, headers_dict_to_raw + from scrapy.http import Headers, Response from scrapy.responsetypes import responsetypes -from scrapy.utils.request import request_fingerprint -from scrapy.utils.project import data_path from scrapy.utils.httpobj import urlparse_cached -from scrapy.utils.python import to_bytes, to_unicode, garbage_collect +from scrapy.utils.project import data_path +from scrapy.utils.python import to_bytes, to_unicode +from scrapy.utils.request import request_fingerprint logger = logging.getLogger(__name__) @@ -342,75 +343,6 @@ class FilesystemCacheStorage(object): return pickle.load(f) -class LeveldbCacheStorage(object): - - def __init__(self, settings): - import leveldb - self._leveldb = leveldb - self.cachedir = data_path(settings['HTTPCACHE_DIR'], createdir=True) - self.expiration_secs = settings.getint('HTTPCACHE_EXPIRATION_SECS') - self.db = None - - def open_spider(self, spider): - dbpath = os.path.join(self.cachedir, '%s.leveldb' % spider.name) - self.db = self._leveldb.LevelDB(dbpath) - - logger.debug("Using LevelDB cache storage in %(cachepath)s" % {'cachepath': dbpath}, extra={'spider': spider}) - - def close_spider(self, spider): - # Do compactation each time to save space and also recreate files to - # avoid them being removed in storages with timestamp-based autoremoval. - self.db.CompactRange() - del self.db - garbage_collect() - - def retrieve_response(self, spider, request): - data = self._read_data(spider, request) - if data is None: - return # not cached - url = data['url'] - status = data['status'] - headers = Headers(data['headers']) - body = data['body'] - respcls = responsetypes.from_args(headers=headers, url=url) - response = respcls(url=url, headers=headers, status=status, body=body) - return response - - def store_response(self, spider, request, response): - key = self._request_key(request) - data = { - 'status': response.status, - 'url': response.url, - 'headers': dict(response.headers), - 'body': response.body, - } - batch = self._leveldb.WriteBatch() - batch.Put(key + b'_data', pickle.dumps(data, protocol=2)) - batch.Put(key + b'_time', to_bytes(str(time()))) - self.db.Write(batch) - - def _read_data(self, spider, request): - key = self._request_key(request) - try: - ts = self.db.Get(key + b'_time') - except KeyError: - return # not found or invalid entry - - if 0 < self.expiration_secs < time() - float(ts): - return # expired - - try: - data = self.db.Get(key + b'_data') - except KeyError: - return # invalid entry - else: - return pickle.loads(data) - - def _request_key(self, request): - return to_bytes(request_fingerprint(request)) - - - def parse_cachecontrol(header): """Parse Cache-Control header diff --git a/scrapy/extensions/memdebug.py b/scrapy/extensions/memdebug.py index 263d8ce4c..892aa8a86 100644 --- a/scrapy/extensions/memdebug.py +++ b/scrapy/extensions/memdebug.py @@ -5,7 +5,6 @@ See documentation in docs/topics/extensions.rst """ import gc -import six from scrapy import signals from scrapy.exceptions import NotConfigured @@ -28,7 +27,7 @@ class MemoryDebugger(object): def spider_closed(self, spider, reason): gc.collect() self.stats.set_value('memdebug/gc_garbage_count', len(gc.garbage), spider=spider) - for cls, wdict in six.iteritems(live_refs): + for cls, wdict in live_refs.items(): if not wdict: continue self.stats.set_value('memdebug/live_refs/%s' % cls.__name__, len(wdict), spider=spider) diff --git a/scrapy/extensions/spiderstate.py b/scrapy/extensions/spiderstate.py index 2220cbd8f..2c8e46914 100644 --- a/scrapy/extensions/spiderstate.py +++ b/scrapy/extensions/spiderstate.py @@ -1,10 +1,11 @@ import os -from six.moves import cPickle as pickle +import pickle from scrapy import signals from scrapy.exceptions import NotConfigured from scrapy.utils.job import job_dir + class SpiderState(object): """Store and load spider state during a scraping job""" diff --git a/scrapy/http/__init__.py b/scrapy/http/__init__.py index 4b2f7b33f..e6c58e1f1 100644 --- a/scrapy/http/__init__.py +++ b/scrapy/http/__init__.py @@ -10,7 +10,7 @@ from scrapy.http.headers import Headers from scrapy.http.request import Request from scrapy.http.request.form import FormRequest from scrapy.http.request.rpc import XmlRpcRequest -from scrapy.http.request.json_request import JSONRequest +from scrapy.http.request.json_request import JsonRequest from scrapy.http.response import Response from scrapy.http.response.html import HtmlResponse diff --git a/scrapy/http/cookies.py b/scrapy/http/cookies.py index 4e8056750..0903fd4f8 100644 --- a/scrapy/http/cookies.py +++ b/scrapy/http/cookies.py @@ -1,9 +1,8 @@ import time -from six.moves.http_cookiejar import ( - CookieJar as _CookieJar, DefaultCookiePolicy, IPV4_RE -) +from http.cookiejar import CookieJar as _CookieJar, DefaultCookiePolicy, IPV4_RE + from scrapy.utils.httpobj import urlparse_cached -from scrapy.utils.python import to_native_str +from scrapy.utils.python import to_unicode class CookieJar(object): @@ -165,13 +164,13 @@ class WrappedRequest(object): return name in self.request.headers def get_header(self, name, default=None): - return to_native_str(self.request.headers.get(name, default), - errors='replace') + return to_unicode(self.request.headers.get(name, default), + errors='replace') def header_items(self): return [ - (to_native_str(k, errors='replace'), - [to_native_str(x, errors='replace') for x in v]) + (to_unicode(k, errors='replace'), + [to_unicode(x, errors='replace') for x in v]) for k, v in self.request.headers.items() ] @@ -189,7 +188,7 @@ class WrappedResponse(object): # python3 cookiejars calls get_all def get_all(self, name, default=None): - return [to_native_str(v, errors='replace') + return [to_unicode(v, errors='replace') for v in self.response.headers.getlist(name)] # python2 cookiejars calls getheaders getheaders = get_all diff --git a/scrapy/http/headers.py b/scrapy/http/headers.py index 62507eb19..dcaaeddfa 100644 --- a/scrapy/http/headers.py +++ b/scrapy/http/headers.py @@ -1,4 +1,3 @@ -import six from w3lib.http import headers_dict_to_raw from scrapy.utils.datatypes import CaselessDict from scrapy.utils.python import to_unicode @@ -19,7 +18,7 @@ class Headers(CaselessDict): """Normalize values to bytes""" if value is None: value = [] - elif isinstance(value, (six.text_type, bytes)): + elif isinstance(value, (str, bytes)): value = [value] elif not hasattr(value, '__iter__'): value = [value] @@ -29,10 +28,10 @@ class Headers(CaselessDict): def _tobytes(self, x): if isinstance(x, bytes): return x - elif isinstance(x, six.text_type): + elif isinstance(x, str): return x.encode(self.encoding) elif isinstance(x, int): - return six.text_type(x).encode(self.encoding) + return str(x).encode(self.encoding) else: raise TypeError('Unsupported value type: {}'.format(type(x))) @@ -68,9 +67,6 @@ class Headers(CaselessDict): self[key] = lst def items(self): - return list(self.iteritems()) - - def iteritems(self): return ((k, self.getlist(k)) for k in self.keys()) def values(self): @@ -91,5 +87,3 @@ class Headers(CaselessDict): def __copy__(self): return self.__class__(self) copy = __copy__ - - diff --git a/scrapy/http/request/__init__.py b/scrapy/http/request/__init__.py index f5935c4ef..6c536cb71 100644 --- a/scrapy/http/request/__init__.py +++ b/scrapy/http/request/__init__.py @@ -4,7 +4,6 @@ requests in Scrapy. See documentation in docs/topics/request-response.rst """ -import six from w3lib.url import safe_url_string from scrapy.http.headers import Headers @@ -12,6 +11,7 @@ from scrapy.utils.python import to_bytes from scrapy.utils.trackref import object_ref from scrapy.utils.url import escape_ajax from scrapy.http.common import obsolete_setter +from scrapy.utils.curl import curl_to_request_kwargs class Request(object_ref): @@ -31,7 +31,6 @@ class Request(object_ref): raise TypeError('callback must be a callable, got %s' % type(callback).__name__) if errback is not None and not callable(errback): raise TypeError('errback must be a callable, got %s' % type(errback).__name__) - assert callback or not errback, "Cannot use errback without a callback" self.callback = callback self.errback = errback @@ -59,13 +58,13 @@ class Request(object_ref): return self._url def _set_url(self, url): - if not isinstance(url, six.string_types): + if not isinstance(url, str): raise TypeError('Request url must be str or unicode, got %s:' % type(url).__name__) s = safe_url_string(url, self.encoding) self._url = escape_ajax(s) - if ':' not in self._url: + if ('://' not in self._url) and (not self._url.startswith('data:')): raise ValueError('Missing scheme in request url: %s' % self._url) url = property(_get_url, obsolete_setter(_set_url, 'url')) @@ -103,3 +102,34 @@ class Request(object_ref): kwargs.setdefault(x, getattr(self, x)) cls = kwargs.pop('cls', self.__class__) return cls(*args, **kwargs) + + @classmethod + def from_curl(cls, curl_command, ignore_unknown_options=True, **kwargs): + """Create a Request object from a string containing a `cURL + <https://curl.haxx.se/>`_ command. It populates the HTTP method, the + URL, the headers, the cookies and the body. It accepts the same + arguments as the :class:`Request` class, taking preference and + overriding the values of the same arguments contained in the cURL + command. + + Unrecognized options are ignored by default. To raise an error when + finding unknown options call this method by passing + ``ignore_unknown_options=False``. + + .. caution:: Using :meth:`from_curl` from :class:`~scrapy.http.Request` + subclasses, such as :class:`~scrapy.http.JSONRequest`, or + :class:`~scrapy.http.XmlRpcRequest`, as well as having + :ref:`downloader middlewares <topics-downloader-middleware>` + and + :ref:`spider middlewares <topics-spider-middleware>` + enabled, such as + :class:`~scrapy.downloadermiddlewares.defaultheaders.DefaultHeadersMiddleware`, + :class:`~scrapy.downloadermiddlewares.useragent.UserAgentMiddleware`, + or + :class:`~scrapy.downloadermiddlewares.httpcompression.HttpCompressionMiddleware`, + may modify the :class:`~scrapy.http.Request` object. + + """ + request_kwargs = curl_to_request_kwargs(curl_command, ignore_unknown_options) + request_kwargs.update(kwargs) + return cls(**request_kwargs) diff --git a/scrapy/http/request/form.py b/scrapy/http/request/form.py index 3ce8fc48e..af02c8484 100644 --- a/scrapy/http/request/form.py +++ b/scrapy/http/request/form.py @@ -5,8 +5,7 @@ This module implements the FormRequest class which is a more convenient class See documentation in docs/topics/request-response.rst """ -import six -from six.moves.urllib.parse import urljoin, urlencode +from urllib.parse import urljoin, urlencode import lxml.html from parsel.selector import create_root_node @@ -104,8 +103,7 @@ def _get_form(response, formname, formid, formnumber, formxpath): el = el.getparent() if el is None: break - encoded = formxpath if six.PY3 else formxpath.encode('unicode_escape') - raise ValueError('No <form> element found with %s' % encoded) + raise ValueError('No <form> element found with %s' % formxpath) # If we get here, it means that either formname was None # or invalid @@ -209,7 +207,7 @@ def _get_clickable(clickdata, form): # We didn't find it, so now we build an XPath expression out of the other # arguments, because they can be used as such xpath = u'.//*' + \ - u''.join(u'[@%s="%s"]' % c for c in six.iteritems(clickdata)) + u''.join(u'[@%s="%s"]' % c for c in clickdata.items()) el = form.xpath(xpath) if len(el) == 1: return (el[0].get('name'), el[0].get('value') or '') diff --git a/scrapy/http/request/json_request.py b/scrapy/http/request/json_request.py index 8f7a61a6d..f08b25280 100644 --- a/scrapy/http/request/json_request.py +++ b/scrapy/http/request/json_request.py @@ -1,5 +1,5 @@ """ -This module implements the JSONRequest class which is a more convenient class +This module implements the JsonRequest class which is a more convenient class (than Request) to generate JSON Requests. See documentation in docs/topics/request-response.rst @@ -10,9 +10,10 @@ import json import warnings from scrapy.http.request import Request +from scrapy.utils.deprecate import create_deprecated_class -class JSONRequest(Request): +class JsonRequest(Request): def __init__(self, *args, **kwargs): dumps_kwargs = copy.deepcopy(kwargs.pop('dumps_kwargs', {})) dumps_kwargs.setdefault('sort_keys', True) @@ -31,7 +32,7 @@ class JSONRequest(Request): if 'method' not in kwargs: kwargs['method'] = 'POST' - super(JSONRequest, self).__init__(*args, **kwargs) + super(JsonRequest, self).__init__(*args, **kwargs) self.headers.setdefault('Content-Type', 'application/json') self.headers.setdefault('Accept', 'application/json, text/javascript, */*; q=0.01') @@ -46,8 +47,11 @@ class JSONRequest(Request): elif not body_passed and data_passed: kwargs['body'] = self._dumps(data) - return super(JSONRequest, self).replace(*args, **kwargs) + return super(JsonRequest, self).replace(*args, **kwargs) def _dumps(self, data): """Convert to JSON """ return json.dumps(data, **self._dumps_kwargs) + + +JSONRequest = create_deprecated_class("JSONRequest", JsonRequest) diff --git a/scrapy/http/request/rpc.py b/scrapy/http/request/rpc.py index bd09f7534..811d3ad6b 100644 --- a/scrapy/http/request/rpc.py +++ b/scrapy/http/request/rpc.py @@ -4,7 +4,7 @@ This module implements the XmlRpcRequest class which is a more convenient class See documentation in docs/topics/request-response.rst """ -from six.moves import xmlrpc_client as xmlrpclib +import xmlrpc.client as xmlrpclib from scrapy.http.request import Request from scrapy.utils.python import get_func_args diff --git a/scrapy/http/response/__init__.py b/scrapy/http/response/__init__.py index b0a526b72..f92d0901c 100644 --- a/scrapy/http/response/__init__.py +++ b/scrapy/http/response/__init__.py @@ -4,14 +4,15 @@ responses in Scrapy. See documentation in docs/topics/request-response.rst """ -from six.moves.urllib.parse import urljoin +from typing import Generator +from urllib.parse import urljoin -from scrapy.http.request import Request +from scrapy.exceptions import NotSupported +from scrapy.http.common import obsolete_setter from scrapy.http.headers import Headers +from scrapy.http.request import Request from scrapy.link import Link from scrapy.utils.trackref import object_ref -from scrapy.http.common import obsolete_setter -from scrapy.exceptions import NotSupported class Response(object_ref): @@ -41,8 +42,8 @@ class Response(object_ref): if isinstance(url, str): self._url = url else: - raise TypeError('%s url must be str, got %s:' % (type(self).__name__, - type(url).__name__)) + raise TypeError('%s url must be str, got %s:' % + (type(self).__name__, type(url).__name__)) url = property(_get_url, obsolete_setter(_set_url, 'url')) @@ -88,7 +89,7 @@ class Response(object_ref): @property def text(self): """For subclasses of TextResponse, this will return the body - as text (unicode object in Python 2 and str in Python 3) + as str """ raise AttributeError("Response content isn't text") @@ -113,8 +114,8 @@ class Response(object_ref): It accepts the same arguments as ``Request.__init__`` method, but ``url`` can be a relative URL or a ``scrapy.link.Link`` object, not only an absolute URL. - - :class:`~.TextResponse` provides a :meth:`~.TextResponse.follow` + + :class:`~.TextResponse` provides a :meth:`~.TextResponse.follow` method which supports selectors in addition to absolute/relative URLs and Link objects. """ @@ -123,14 +124,51 @@ class Response(object_ref): elif url is None: raise ValueError("url can't be None") url = self.urljoin(url) - return Request(url, callback, - method=method, - headers=headers, - body=body, - cookies=cookies, - meta=meta, - encoding=encoding, - priority=priority, - dont_filter=dont_filter, - errback=errback, - cb_kwargs=cb_kwargs) + return Request( + url=url, + callback=callback, + method=method, + headers=headers, + body=body, + cookies=cookies, + meta=meta, + encoding=encoding, + priority=priority, + dont_filter=dont_filter, + errback=errback, + cb_kwargs=cb_kwargs, + ) + + def follow_all(self, urls, callback=None, method='GET', headers=None, body=None, + cookies=None, meta=None, encoding='utf-8', priority=0, + dont_filter=False, errback=None, cb_kwargs=None): + # type: (...) -> Generator[Request, None, None] + """ + Return an iterable of :class:`~.Request` instances to follow all links + in ``urls``. It accepts the same arguments as ``Request.__init__`` method, + but elements of ``urls`` can be relative URLs or :class:`~scrapy.link.Link` objects, + not only absolute URLs. + + :class:`~.TextResponse` provides a :meth:`~.TextResponse.follow_all` + method which supports selectors in addition to absolute/relative URLs + and Link objects. + """ + if not hasattr(urls, '__iter__'): + raise TypeError("'urls' argument must be an iterable") + return ( + self.follow( + url=url, + callback=callback, + method=method, + headers=headers, + body=body, + cookies=cookies, + meta=meta, + encoding=encoding, + priority=priority, + dont_filter=dont_filter, + errback=errback, + cb_kwargs=cb_kwargs, + ) + for url in urls + ) diff --git a/scrapy/http/response/html.py b/scrapy/http/response/html.py index bd3559fbb..7eed052c2 100644 --- a/scrapy/http/response/html.py +++ b/scrapy/http/response/html.py @@ -7,5 +7,6 @@ See documentation in docs/topics/request-response.rst from scrapy.http.response.text import TextResponse + class HtmlResponse(TextResponse): pass diff --git a/scrapy/http/response/text.py b/scrapy/http/response/text.py index 339913d4e..09049c157 100644 --- a/scrapy/http/response/text.py +++ b/scrapy/http/response/text.py @@ -5,18 +5,19 @@ discovering (through HTTP headers) to base Response class. See documentation in docs/topics/request-response.rst """ -import six -from six.moves.urllib.parse import urljoin +from contextlib import suppress +from typing import Generator +from urllib.parse import urljoin import parsel -from w3lib.encoding import html_to_unicode, resolve_encoding, \ - html_body_declared_encoding, http_content_type_encoding +from w3lib.encoding import (html_body_declared_encoding, html_to_unicode, + http_content_type_encoding, resolve_encoding) from w3lib.html import strip_html5_whitespace -from scrapy.http.request import Request +from scrapy.http import Request from scrapy.http.response import Response +from scrapy.utils.python import memoizemethod_noargs, to_unicode from scrapy.utils.response import get_base_url -from scrapy.utils.python import memoizemethod_noargs, to_native_str class TextResponse(Response): @@ -31,20 +32,17 @@ class TextResponse(Response): super(TextResponse, self).__init__(*args, **kwargs) def _set_url(self, url): - if isinstance(url, six.text_type): - if six.PY2 and self.encoding is None: - raise TypeError("Cannot convert unicode url - %s " - "has no encoding" % type(self).__name__) - self._url = to_native_str(url, self.encoding) + if isinstance(url, str): + self._url = to_unicode(url, self.encoding) else: super(TextResponse, self)._set_url(url) def _set_body(self, body): self._body = b'' # used by encoding detection - if isinstance(body, six.text_type): + if isinstance(body, str): if self._encoding is None: raise TypeError('Cannot convert unicode body - %s has no encoding' % - type(self).__name__) + type(self).__name__) self._body = body.encode(self._encoding) else: super(TextResponse, self)._set_body(body) @@ -84,14 +82,14 @@ class TextResponse(Response): @memoizemethod_noargs def _headers_encoding(self): content_type = self.headers.get(b'Content-Type', b'') - return http_content_type_encoding(to_native_str(content_type)) + return http_content_type_encoding(to_unicode(content_type)) def _body_inferred_encoding(self): if self._cached_benc is None: - content_type = to_native_str(self.headers.get(b'Content-Type', b'')) + content_type = to_unicode(self.headers.get(b'Content-Type', b'')) benc, ubody = html_to_unicode(content_type, self.body, - auto_detect_fun=self._auto_detect_fun, - default_encoding=self._DEFAULT_ENCODING) + auto_detect_fun=self._auto_detect_fun, + default_encoding=self._DEFAULT_ENCODING) self._cached_benc = benc self._cached_ubody = ubody return self._cached_benc @@ -129,15 +127,16 @@ class TextResponse(Response): Return a :class:`~.Request` instance to follow a link ``url``. It accepts the same arguments as ``Request.__init__`` method, but ``url`` can be not only an absolute URL, but also - - * a relative URL; - * a scrapy.link.Link object (e.g. a link extractor result); - * an attribute Selector (not SelectorList) - e.g. + + * a relative URL + * a :class:`~scrapy.link.Link` object, e.g. the result of + :ref:`topics-link-extractors` + * a :class:`~scrapy.selector.Selector` object for a ``<link>`` or ``<a>`` element, e.g. + ``response.css('a.my_link')[0]`` + * an attribute :class:`~scrapy.selector.Selector` (not SelectorList), e.g. ``response.css('a::attr(href)')[0]`` or - ``response.xpath('//img/@src')[0]``. - * a Selector for ``<a>`` or ``<link>`` element, e.g. - ``response.css('a.my_link')[0]``. - + ``response.xpath('//img/@src')[0]`` + See :ref:`response-follow-example` for usage examples. """ if isinstance(url, parsel.Selector): @@ -145,7 +144,66 @@ class TextResponse(Response): elif isinstance(url, parsel.SelectorList): raise ValueError("SelectorList is not supported") encoding = self.encoding if encoding is None else encoding - return super(TextResponse, self).follow(url, callback, + return super(TextResponse, self).follow( + url=url, + callback=callback, + method=method, + headers=headers, + body=body, + cookies=cookies, + meta=meta, + encoding=encoding, + priority=priority, + dont_filter=dont_filter, + errback=errback, + cb_kwargs=cb_kwargs, + ) + + def follow_all(self, urls=None, callback=None, method='GET', headers=None, body=None, + cookies=None, meta=None, encoding=None, priority=0, + dont_filter=False, errback=None, cb_kwargs=None, + css=None, xpath=None): + # type: (...) -> Generator[Request, None, None] + """ + A generator that produces :class:`~.Request` instances to follow all + links in ``urls``. It accepts the same arguments as the :class:`~.Request`'s + ``__init__`` method, except that each ``urls`` element does not need to be + an absolute URL, it can be any of the following: + + * a relative URL + * a :class:`~scrapy.link.Link` object, e.g. the result of + :ref:`topics-link-extractors` + * a :class:`~scrapy.selector.Selector` object for a ``<link>`` or ``<a>`` element, e.g. + ``response.css('a.my_link')[0]`` + * an attribute :class:`~scrapy.selector.Selector` (not SelectorList), e.g. + ``response.css('a::attr(href)')[0]`` or + ``response.xpath('//img/@src')[0]`` + + In addition, ``css`` and ``xpath`` arguments are accepted to perform the link extraction + within the ``follow_all`` method (only one of ``urls``, ``css`` and ``xpath`` is accepted). + + Note that when passing a ``SelectorList`` as argument for the ``urls`` parameter or + using the ``css`` or ``xpath`` parameters, this method will not produce requests for + selectors from which links cannot be obtained (for instance, anchor tags without an + ``href`` attribute) + """ + arg_count = len(list(filter(None, (urls, css, xpath)))) + if arg_count != 1: + raise ValueError('Please supply exactly one of the following arguments: urls, css, xpath') + if not urls: + if css: + urls = self.css(css) + if xpath: + urls = self.xpath(xpath) + if isinstance(urls, parsel.SelectorList): + selectors = urls + urls = [] + for sel in selectors: + with suppress(_InvalidSelector): + urls.append(_url_from_selector(sel)) + return super(TextResponse, self).follow_all( + urls=urls, + callback=callback, method=method, headers=headers, body=body, @@ -159,18 +217,24 @@ class TextResponse(Response): ) +class _InvalidSelector(ValueError): + """ + Raised when a URL cannot be obtained from a Selector + """ + + def _url_from_selector(sel): # type: (parsel.Selector) -> str - if isinstance(sel.root, six.string_types): + if isinstance(sel.root, str): # e.g. ::attr(href) result return strip_html5_whitespace(sel.root) if not hasattr(sel.root, 'tag'): - raise ValueError("Unsupported selector: %s" % sel) + raise _InvalidSelector("Unsupported selector: %s" % sel) if sel.root.tag not in ('a', 'link'): - raise ValueError("Only <a> and <link> elements are supported; got <%s>" % - sel.root.tag) + raise _InvalidSelector("Only <a> and <link> elements are supported; got <%s>" % + sel.root.tag) href = sel.root.get('href') if href is None: - raise ValueError("<%s> element has no href attribute: %s" % - (sel.root.tag, sel)) + raise _InvalidSelector("<%s> element has no href attribute: %s" % + (sel.root.tag, sel)) return strip_html5_whitespace(href) diff --git a/scrapy/http/response/xml.py b/scrapy/http/response/xml.py index 1df33fee5..abf474a2f 100644 --- a/scrapy/http/response/xml.py +++ b/scrapy/http/response/xml.py @@ -7,5 +7,6 @@ See documentation in docs/topics/request-response.rst from scrapy.http.response.text import TextResponse + class XmlResponse(TextResponse): pass diff --git a/scrapy/interfaces.py b/scrapy/interfaces.py index 89ad2b14f..1896ec31e 100644 --- a/scrapy/interfaces.py +++ b/scrapy/interfaces.py @@ -1,5 +1,6 @@ from zope.interface import Interface + class ISpiderLoader(Interface): def from_settings(settings): @@ -15,4 +16,3 @@ class ISpiderLoader(Interface): def find_by_request(request): """Return the list of spiders names that can handle the given request""" - diff --git a/scrapy/item.py b/scrapy/item.py index 9d4786788..1d39b48b2 100644 --- a/scrapy/item.py +++ b/scrapy/item.py @@ -5,23 +5,29 @@ See documentation in docs/topics/item.rst """ from abc import ABCMeta -from pprint import pformat +from collections.abc import MutableMapping from copy import deepcopy -import collections - -import six +from pprint import pformat +from warnings import warn +from scrapy.utils.deprecate import ScrapyDeprecationWarning from scrapy.utils.trackref import object_ref -if six.PY2: - MutableMapping = collections.MutableMapping -else: - MutableMapping = collections.abc.MutableMapping - - class BaseItem(object_ref): - """Base class for all scraped items.""" + """Base class for all scraped items. + + In Scrapy, an object is considered an *item* if it is an instance of either + :class:`BaseItem` or :class:`dict`. For example, when the output of a + spider callback is evaluated, only instances of :class:`BaseItem` or + :class:`dict` are passed to :ref:`item pipelines <topics-item-pipeline>`. + + If you need instances of a custom class to be considered items by Scrapy, + you must inherit from either :class:`BaseItem` or :class:`dict`. + + Unlike instances of :class:`dict`, instances of :class:`BaseItem` may be + :ref:`tracked <topics-leaks-trackrefs>` to debug memory leaks. + """ pass @@ -30,6 +36,10 @@ class Field(dict): class ItemMeta(ABCMeta): + """Metaclass_ of :class:`Item` that handles field definitions. + + .. _metaclass: https://realpython.com/python-metaclasses + """ def __new__(mcs, class_name, bases, attrs): classcell = attrs.pop('__classcell__', None) @@ -56,10 +66,17 @@ class DictItem(MutableMapping, BaseItem): fields = {} + def __new__(cls, *args, **kwargs): + if issubclass(cls, DictItem) and not issubclass(cls, Item): + warn('scrapy.item.DictItem is deprecated, please use ' + 'scrapy.item.Item instead', + ScrapyDeprecationWarning, stacklevel=2) + return super(DictItem, cls).__new__(cls, *args, **kwargs) + def __init__(self, *args, **kwargs): self._values = {} if args or kwargs: # avoid creating dict for most common case - for k, v in six.iteritems(dict(*args, **kwargs)): + for k, v in dict(*args, **kwargs).items(): self[k] = v def __getitem__(self, key): @@ -111,6 +128,5 @@ class DictItem(MutableMapping, BaseItem): return deepcopy(self) -@six.add_metaclass(ItemMeta) -class Item(DictItem): +class Item(DictItem, metaclass=ItemMeta): pass diff --git a/scrapy/link.py b/scrapy/link.py index 2c8301680..a809c5ca4 100644 --- a/scrapy/link.py +++ b/scrapy/link.py @@ -4,10 +4,6 @@ This module defines the Link object used in Link extractors. For actual link extractors implementation see scrapy.linkextractors, or its documentation in: docs/topics/link-extractors.rst """ -import warnings -import six - -from scrapy.utils.python import to_bytes class Link(object): @@ -17,13 +13,8 @@ class Link(object): def __init__(self, url, text='', fragment='', nofollow=False): if not isinstance(url, str): - if six.PY2: - warnings.warn("Link urls must be str objects. " - "Assuming utf-8 encoding (which could be wrong)") - url = to_bytes(url, encoding='utf8') - else: - got = url.__class__.__name__ - raise TypeError("Link urls must be str objects, got %s" % got) + got = url.__class__.__name__ + raise TypeError("Link urls must be str objects, got %s" % got) self.url = url self.text = text self.fragment = fragment @@ -39,4 +30,3 @@ class Link(object): def __repr__(self): return 'Link(url=%r, text=%r, fragment=%r, nofollow=%r)' % \ (self.url, self.text, self.fragment, self.nofollow) - diff --git a/scrapy/linkextractors/__init__.py b/scrapy/linkextractors/__init__.py index ebf3cd7d8..bdeab3a75 100644 --- a/scrapy/linkextractors/__init__.py +++ b/scrapy/linkextractors/__init__.py @@ -6,11 +6,13 @@ This package contains a collection of Link Extractors. For more info see docs/topics/link-extractors.rst """ import re +from urllib.parse import urlparse +from warnings import warn -from six.moves.urllib.parse import urlparse from parsel.csstranslator import HTMLTranslator from w3lib.url import canonicalize_url +from scrapy.utils.deprecate import ScrapyDeprecationWarning from scrapy.utils.misc import arg_to_iter from scrapy.utils.url import ( url_is_from_any_domain, url_has_any_extension, @@ -19,36 +21,47 @@ from scrapy.utils.url import ( # common file extensions that are not followed if they occur in links IGNORED_EXTENSIONS = [ + # archives + '7z', '7zip', 'bz2', 'rar', 'tar', 'tar.gz', 'xz', 'zip', + # images 'mng', 'pct', 'bmp', 'gif', 'jpg', 'jpeg', 'png', 'pst', 'psp', 'tif', - 'tiff', 'ai', 'drw', 'dxf', 'eps', 'ps', 'svg', + 'tiff', 'ai', 'drw', 'dxf', 'eps', 'ps', 'svg', 'cdr', 'ico', # audio 'mp3', 'wma', 'ogg', 'wav', 'ra', 'aac', 'mid', 'au', 'aiff', # video '3gp', 'asf', 'asx', 'avi', 'mov', 'mp4', 'mpg', 'qt', 'rm', 'swf', 'wmv', - 'm4a', 'm4v', 'flv', + 'm4a', 'm4v', 'flv', 'webm', # office suites 'xls', 'xlsx', 'ppt', 'pptx', 'pps', 'doc', 'docx', 'odt', 'ods', 'odg', 'odp', # other - 'css', 'pdf', 'exe', 'bin', 'rss', 'zip', 'rar', + 'css', 'pdf', 'exe', 'bin', 'rss', 'dmg', 'iso', 'apk' ] _re_type = type(re.compile("", 0)) _matches = lambda url, regexs: any(r.search(url) for r in regexs) -_is_valid_url = lambda url: url.split('://', 1)[0] in {'http', 'https', \ - 'file', 'ftp'} +_is_valid_url = lambda url: url.split('://', 1)[0] in {'http', 'https', 'file', 'ftp'} class FilteringLinkExtractor(object): _csstranslator = HTMLTranslator() + def __new__(cls, *args, **kwargs): + from scrapy.linkextractors.lxmlhtml import LxmlLinkExtractor + if (issubclass(cls, FilteringLinkExtractor) and + not issubclass(cls, LxmlLinkExtractor)): + warn('scrapy.linkextractors.FilteringLinkExtractor is deprecated, ' + 'please use scrapy.linkextractors.LinkExtractor instead', + ScrapyDeprecationWarning, stacklevel=2) + return super(FilteringLinkExtractor, cls).__new__(cls) + def __init__(self, link_extractor, allow, deny, allow_domains, deny_domains, restrict_xpaths, canonicalize, deny_extensions, restrict_css, restrict_text): @@ -115,4 +128,4 @@ class FilteringLinkExtractor(object): # Top-level imports -from .lxmlhtml import LxmlLinkExtractor as LinkExtractor +from scrapy.linkextractors.lxmlhtml import LxmlLinkExtractor as LinkExtractor # noqa: F401 diff --git a/scrapy/linkextractors/htmlparser.py b/scrapy/linkextractors/htmlparser.py index 27978a8a1..0425d4340 100644 --- a/scrapy/linkextractors/htmlparser.py +++ b/scrapy/linkextractors/htmlparser.py @@ -2,9 +2,8 @@ HTMLParser-based link extractor """ import warnings -import six -from six.moves.html_parser import HTMLParser -from six.moves.urllib.parse import urljoin +from html.parser import HTMLParser +from urllib.parse import urljoin from w3lib.url import safe_url_string from w3lib.html import strip_html5_whitespace @@ -42,7 +41,7 @@ class HtmlParserLinkExtractor(HTMLParser): ret = [] base_url = urljoin(response_url, self.base_url) if self.base_url else response_url for link in links: - if isinstance(link.url, six.text_type): + if isinstance(link.url, str): link.url = link.url.encode(response_encoding) try: link.url = urljoin(base_url, link.url) diff --git a/scrapy/linkextractors/lxmlhtml.py b/scrapy/linkextractors/lxmlhtml.py index 8f6f93a44..fdfa92370 100644 --- a/scrapy/linkextractors/lxmlhtml.py +++ b/scrapy/linkextractors/lxmlhtml.py @@ -1,8 +1,7 @@ """ Link extractor based on lxml.html """ -import six -from six.moves.urllib.parse import urljoin +from urllib.parse import urljoin import lxml.etree as etree from w3lib.html import strip_html5_whitespace @@ -10,7 +9,7 @@ from w3lib.url import canonicalize_url from scrapy.link import Link from scrapy.utils.misc import arg_to_iter, rel_has_nofollow -from scrapy.utils.python import unique as unique_list, to_native_str +from scrapy.utils.python import unique as unique_list, to_unicode from scrapy.utils.response import get_base_url from scrapy.linkextractors import FilteringLinkExtractor @@ -22,7 +21,7 @@ _collect_string_content = etree.XPath("string()") def _nons(tag): - if isinstance(tag, six.string_types): + if isinstance(tag, str): if tag[0] == '{' and tag[1:len(XHTML_NAMESPACE)+1] == XHTML_NAMESPACE: return tag.split('}')[-1] return tag @@ -67,7 +66,7 @@ class LxmlParserLinkExtractor(object): url = self.process_attr(attr_val) if url is None: continue - url = to_native_str(url, encoding=response_encoding) + url = to_unicode(url, encoding=response_encoding) # to fix relative links after process_value url = urljoin(response_url, url) link = Link(url, _collect_string_content(el) or u'', @@ -117,6 +116,14 @@ class LxmlLinkExtractor(FilteringLinkExtractor): restrict_text=restrict_text) def extract_links(self, response): + """Returns a list of :class:`~scrapy.link.Link` objects from the + specified :class:`response <scrapy.http.Response>`. + + Only links that match the settings passed to the ``__init__`` method of + the link extractor are returned. + + Duplicate links are omitted. + """ base_url = get_base_url(response) if self.restrict_xpaths: docs = [subdoc diff --git a/scrapy/linkextractors/regex.py b/scrapy/linkextractors/regex.py index e689b4727..3f2557248 100644 --- a/scrapy/linkextractors/regex.py +++ b/scrapy/linkextractors/regex.py @@ -1,10 +1,11 @@ import re -from six.moves.urllib.parse import urljoin +from urllib.parse import urljoin from w3lib.html import remove_tags, replace_entities, replace_escape_chars, get_base_url from scrapy.link import Link -from .sgml import SgmlLinkExtractor +from scrapy.linkextractors.sgml import SgmlLinkExtractor + linkre = re.compile( "<a\s.*?href=(\"[.#]+?\"|\'[.#]+?\'|[^\s]+?)(>|\s.*?>)(.*?)<[/ ]?a>", diff --git a/scrapy/linkextractors/sgml.py b/scrapy/linkextractors/sgml.py index 8940a4d77..2ba6bca45 100644 --- a/scrapy/linkextractors/sgml.py +++ b/scrapy/linkextractors/sgml.py @@ -1,9 +1,8 @@ """ SGMLParser-based Link extractors """ -import six -from six.moves.urllib.parse import urljoin import warnings +from urllib.parse import urljoin from sgmllib import SGMLParser from w3lib.url import safe_url_string, canonicalize_url @@ -49,7 +48,7 @@ class BaseSgmlLinkExtractor(SGMLParser): if base_url is None: base_url = urljoin(response_url, self.base_url) if self.base_url else response_url for link in self.links: - if isinstance(link.url, six.text_type): + if isinstance(link.url, str): link.url = link.url.encode(response_encoding) try: link.url = urljoin(base_url, link.url) diff --git a/scrapy/loader/__init__.py b/scrapy/loader/__init__.py index 6665eba16..7cf67e29e 100644 --- a/scrapy/loader/__init__.py +++ b/scrapy/loader/__init__.py @@ -1,18 +1,28 @@ -"""Item Loader +""" +Item Loader See documentation in docs/topics/loaders.rst - """ from collections import defaultdict -import six +from contextlib import suppress from scrapy.item import Item +from scrapy.loader.common import wrap_loader_context +from scrapy.loader.processors import Identity from scrapy.selector import Selector from scrapy.utils.misc import arg_to_iter, extract_regex from scrapy.utils.python import flatten -from .common import wrap_loader_context -from .processors import Identity + +def unbound_method(method): + """ + Allow to use single-argument functions as input or output processors + (no need to define an unused first 'self' argument) + """ + with suppress(AttributeError): + if '.' not in method.__qualname__: + return method.__func__ + return method class ItemLoader(object): @@ -33,10 +43,9 @@ class ItemLoader(object): self.parent = parent self._local_item = context['item'] = item self._local_values = defaultdict(list) - # Preprocess values if item built from dict - # Values need to be added to item._values if added them from dict (not with add_values) + # values from initial item for field_name, value in item.items(): - self._values[field_name] = self._process_input_value(field_name, value) + self._values[field_name] += arg_to_iter(value) @property def _values(self): @@ -73,7 +82,7 @@ class ItemLoader(object): if value is None: return if not field_name: - for k, v in six.iteritems(value): + for k, v in value.items(): self._add_value(k, v) else: self._add_value(field_name, value) @@ -83,7 +92,7 @@ class ItemLoader(object): if value is None: return if not field_name: - for k, v in six.iteritems(value): + for k, v in value.items(): self._replace_value(k, v) else: self._replace_value(field_name, value) @@ -132,8 +141,8 @@ class ItemLoader(object): try: return proc(self._values[field_name]) except Exception as e: - raise ValueError("Error with output processor: field=%r value=%r error='%s: %s'" % \ - (field_name, self._values[field_name], type(e).__name__, str(e))) + raise ValueError("Error with output processor: field=%r value=%r error='%s: %s'" % + (field_name, self._values[field_name], type(e).__name__, str(e))) def get_collected_values(self, field_name): return self._values[field_name] @@ -141,16 +150,16 @@ class ItemLoader(object): def get_input_processor(self, field_name): proc = getattr(self, '%s_in' % field_name, None) if not proc: - proc = self._get_item_field_attr(field_name, 'input_processor', \ - self.default_input_processor) - return proc + proc = self._get_item_field_attr(field_name, 'input_processor', + self.default_input_processor) + return unbound_method(proc) def get_output_processor(self, field_name): proc = getattr(self, '%s_out' % field_name, None) if not proc: - proc = self._get_item_field_attr(field_name, 'output_processor', \ - self.default_output_processor) - return proc + proc = self._get_item_field_attr(field_name, 'output_processor', + self.default_output_processor) + return unbound_method(proc) def _process_input_value(self, field_name, value): proc = self.get_input_processor(field_name) @@ -174,8 +183,8 @@ class ItemLoader(object): def _check_selector_method(self): if self.selector is None: raise RuntimeError("To use XPath or CSS selectors, " - "%s must be instantiated with a selector " - "or a response" % self.__class__.__name__) + "%s must be instantiated with a selector " + "or a response" % self.__class__.__name__) def add_xpath(self, field_name, xpath, *processors, **kw): values = self._get_xpathvalues(xpath, **kw) diff --git a/scrapy/loader/common.py b/scrapy/loader/common.py index 916524947..42f8de636 100644 --- a/scrapy/loader/common.py +++ b/scrapy/loader/common.py @@ -3,6 +3,7 @@ from functools import partial from scrapy.utils.python import get_func_args + def wrap_loader_context(function, context): """Wrap functions that receive loader_context to contain the context "pre-loaded" and expose a interface that receives only one argument diff --git a/scrapy/loader/processors.py b/scrapy/loader/processors.py index 2acdc8093..02c625acc 100644 --- a/scrapy/loader/processors.py +++ b/scrapy/loader/processors.py @@ -3,10 +3,7 @@ This module provides some commonly used processors for Item Loaders. See documentation in docs/topics/loaders.rst """ -try: - from collections import ChainMap -except ImportError: - from scrapy.utils.datatypes import MergeDict as ChainMap +from collections import ChainMap from scrapy.utils.misc import arg_to_iter from scrapy.loader.common import wrap_loader_context diff --git a/scrapy/logformatter.py b/scrapy/logformatter.py index 65f347dcf..4e5963e99 100644 --- a/scrapy/logformatter.py +++ b/scrapy/logformatter.py @@ -8,30 +8,49 @@ from scrapy.utils.request import referer_str SCRAPEDMSG = u"Scraped from %(src)s" + os.linesep + "%(item)s" DROPPEDMSG = u"Dropped: %(exception)s" + os.linesep + "%(item)s" CRAWLEDMSG = u"Crawled (%(status)s) %(request)s%(request_flags)s (referer: %(referer)s)%(response_flags)s" +ERRORMSG = u"'Error processing %(item)s'" class LogFormatter(object): """Class for generating log messages for different actions. - All methods must return a dictionary listing the parameters ``level``, - ``msg`` and ``args`` which are going to be used for constructing the log - message when calling logging.log. + All methods must return a dictionary listing the parameters ``level``, ``msg`` + and ``args`` which are going to be used for constructing the log message when + calling ``logging.log``. Dictionary keys for the method outputs: - * ``level`` should be the log level for that action, you can use those - from the python logging library: logging.DEBUG, logging.INFO, - logging.WARNING, logging.ERROR and logging.CRITICAL. - * ``msg`` should be a string that can contain different formatting - placeholders. This string, formatted with the provided ``args``, is - going to be the log message for that action. + * ``level`` is the log level for that action, you can use those from the + `python logging library <https://docs.python.org/3/library/logging.html>`_ : + ``logging.DEBUG``, ``logging.INFO``, ``logging.WARNING``, ``logging.ERROR`` + and ``logging.CRITICAL``. + * ``msg`` should be a string that can contain different formatting placeholders. + This string, formatted with the provided ``args``, is going to be the long message + for that action. + * ``args`` should be a tuple or dict with the formatting placeholders for ``msg``. + The final log message is computed as ``msg % args``. - * ``args`` should be a tuple or dict with the formatting placeholders - for ``msg``. The final log message is computed as output['msg'] % - output['args']. + Users can define their own ``LogFormatter`` class if they want to customize how + each action is logged or if they want to omit it entirely. In order to omit + logging an action the method must return ``None``. + + Here is an example on how to create a custom log formatter to lower the severity level of + the log message when an item is dropped from the pipeline:: + + class PoliteLogFormatter(logformatter.LogFormatter): + def dropped(self, item, exception, response, spider): + return { + 'level': logging.INFO, # lowering the level from logging.WARNING + 'msg': u"Dropped: %(exception)s" + os.linesep + "%(item)s", + 'args': { + 'exception': exception, + 'item': item, + } + } """ def crawled(self, request, response, spider): + """Logs a message when the crawler finds a webpage.""" request_flags = ' %s' % str(request.flags) if request.flags else '' response_flags = ' %s' % str(response.flags) if response.flags else '' return { @@ -40,7 +59,7 @@ class LogFormatter(object): 'args': { 'status': response.status, 'request': request, - 'request_flags' : request_flags, + 'request_flags': request_flags, 'referer': referer_str(request), 'response_flags': response_flags, # backward compatibility with Scrapy logformatter below 1.4 version @@ -49,6 +68,7 @@ class LogFormatter(object): } def scraped(self, item, response, spider): + """Logs a message when an item is scraped by a spider.""" if isinstance(response, Failure): src = response.getErrorMessage() else: @@ -63,6 +83,7 @@ class LogFormatter(object): } def dropped(self, item, exception, response, spider): + """Logs a message when an item is dropped while it is passing through the item pipeline.""" return { 'level': logging.WARNING, 'msg': DROPPEDMSG, @@ -72,6 +93,16 @@ class LogFormatter(object): } } + def error(self, item, exception, response, spider): + """Logs a message when an item causes an error while it is passing through the item pipeline.""" + return { + 'level': logging.ERROR, + 'msg': ERRORMSG, + 'args': { + 'item': item, + } + } + @classmethod def from_crawler(cls, crawler): return cls() diff --git a/scrapy/mail.py b/scrapy/mail.py index 5b944e1c4..9655b8114 100644 --- a/scrapy/mail.py +++ b/scrapy/mail.py @@ -4,29 +4,20 @@ Mail sending helpers See documentation in docs/topics/email.rst """ import logging - -try: - from cStringIO import StringIO as BytesIO -except ImportError: - from io import BytesIO -import six - +from email import encoders as Encoders +from email.mime.base import MIMEBase +from email.mime.multipart import MIMEMultipart +from email.mime.nonmultipart import MIMENonMultipart +from email.mime.text import MIMEText from email.utils import COMMASPACE, formatdate -from six.moves.email_mime_multipart import MIMEMultipart -from six.moves.email_mime_text import MIMEText -from six.moves.email_mime_base import MIMEBase -if six.PY2: - from email.MIMENonMultipart import MIMENonMultipart - from email import Encoders -else: - from email.mime.nonmultipart import MIMENonMultipart - from email import encoders as Encoders +from io import BytesIO from twisted.internet import defer, reactor, ssl from scrapy.utils.misc import arg_to_iter from scrapy.utils.python import to_bytes + logger = logging.getLogger(__name__) @@ -82,8 +73,7 @@ class MailSender(object): part = MIMEBase(*mimetype.split('/')) part.set_payload(f.read()) Encoders.encode_base64(part) - part.add_header('Content-Disposition', 'attachment; filename="%s"' \ - % attach_name) + part.add_header('Content-Disposition', 'attachment', filename=attach_name) msg.attach(part) else: msg.set_payload(body) diff --git a/scrapy/middleware.py b/scrapy/middleware.py index 1cfd8a782..53fa435bb 100644 --- a/scrapy/middleware.py +++ b/scrapy/middleware.py @@ -65,8 +65,8 @@ class MiddlewareManager(object): return process_chain(self.methods[methodname], obj, *args) def _process_chain_both(self, cb_methodname, eb_methodname, obj, *args): - return process_chain_both(self.methods[cb_methodname], \ - self.methods[eb_methodname], obj, *args) + return process_chain_both(self.methods[cb_methodname], + self.methods[eb_methodname], obj, *args) def open_spider(self, spider): return self._process_parallel('open_spider', spider) diff --git a/scrapy/pipelines/__init__.py b/scrapy/pipelines/__init__.py index 2ef8786d0..b5725a8ee 100644 --- a/scrapy/pipelines/__init__.py +++ b/scrapy/pipelines/__init__.py @@ -6,6 +6,8 @@ See documentation in docs/item-pipeline.rst from scrapy.middleware import MiddlewareManager from scrapy.utils.conf import build_component_list +from scrapy.utils.defer import deferred_f_from_coro_f + class ItemPipelineManager(MiddlewareManager): @@ -18,7 +20,7 @@ class ItemPipelineManager(MiddlewareManager): def _add_middleware(self, pipe): super(ItemPipelineManager, self)._add_middleware(pipe) if hasattr(pipe, 'process_item'): - self.methods['process_item'].append(pipe.process_item) + self.methods['process_item'].append(deferred_f_from_coro_f(pipe.process_item)) def process_item(self, item, spider): return self._process_chain('process_item', item, spider) diff --git a/scrapy/pipelines/files.py b/scrapy/pipelines/files.py index 5383b05fe..2d286eed8 100644 --- a/scrapy/pipelines/files.py +++ b/scrapy/pipelines/files.py @@ -5,20 +5,16 @@ See documentation in topics/media-pipeline.rst """ import functools import hashlib -import os -import os.path -import time import logging +import mimetypes +import os +import time +from collections import defaultdict from email.utils import parsedate_tz, mktime_tz from ftplib import FTP +from io import BytesIO from six.moves.urllib.parse import urlparse -from collections import defaultdict -import six - -try: - from cStringIO import StringIO as BytesIO -except ImportError: - from io import BytesIO +from urllib.parse import urlparse from twisted.internet import defer, threads @@ -34,6 +30,7 @@ from scrapy.utils.boto import is_botocore from scrapy.utils.datatypes import CaselessDict from scrapy.utils.ftp import ftp_makedirs_cwd, ftp_store_file + logger = logging.getLogger(__name__) @@ -158,14 +155,14 @@ class S3FilesStore(object): Bucket=self.bucket, Key=key_name, Body=buf, - Metadata={k: str(v) for k, v in six.iteritems(meta or {})}, + Metadata={k: str(v) for k, v in (meta or {}).items()}, ACL=self.POLICY, **extra) else: b = self._get_boto_bucket() k = b.new_key(key_name) if meta: - for metakey, metavalue in six.iteritems(meta): + for metakey, metavalue in meta.items(): k.set_metadata(metakey, str(metavalue)) h = self.HEADERS.copy() if headers: @@ -191,9 +188,22 @@ class S3FilesStore(object): 'X-Amz-Grant-Read': 'GrantRead', 'X-Amz-Grant-Read-ACP': 'GrantReadACP', 'X-Amz-Grant-Write-ACP': 'GrantWriteACP', + 'X-Amz-Object-Lock-Legal-Hold': 'ObjectLockLegalHoldStatus', + 'X-Amz-Object-Lock-Mode': 'ObjectLockMode', + 'X-Amz-Object-Lock-Retain-Until-Date': 'ObjectLockRetainUntilDate', + 'X-Amz-Request-Payer': 'RequestPayer', + 'X-Amz-Server-Side-Encryption': 'ServerSideEncryption', + 'X-Amz-Server-Side-Encryption-Aws-Kms-Key-Id': 'SSEKMSKeyId', + 'X-Amz-Server-Side-Encryption-Context': 'SSEKMSEncryptionContext', + 'X-Amz-Server-Side-Encryption-Customer-Algorithm': 'SSECustomerAlgorithm', + 'X-Amz-Server-Side-Encryption-Customer-Key': 'SSECustomerKey', + 'X-Amz-Server-Side-Encryption-Customer-Key-Md5': 'SSECustomerKeyMD5', + 'X-Amz-Storage-Class': 'StorageClass', + 'X-Amz-Tagging': 'Tagging', + 'X-Amz-Website-Redirect-Location': 'WebsiteRedirectLocation', }) extra = {} - for key, value in six.iteritems(headers): + for key, value in headers.items(): try: kwarg = mapping[key] except KeyError: @@ -241,7 +251,7 @@ class GCSFilesStore(object): def persist_file(self, path, buf, info, meta=None, headers=None): blob = self.bucket.blob(self.prefix + path) blob.cache_control = self.CACHE_CONTROL - blob.metadata = {k: str(v) for k, v in six.iteritems(meta or {})} + blob.metadata = {k: str(v) for k, v in (meta or {}).items()} return threads.deferToThread( blob.upload_from_string, data=buf.getvalue(), @@ -511,4 +521,11 @@ class FilesPipeline(MediaPipeline): def file_path(self, request, response=None, info=None): media_guid = hashlib.sha1(to_bytes(request.url)).hexdigest() media_ext = os.path.splitext(request.url)[1] + # Handles empty and wild extensions by trying to guess the + # mime type then extension or default to empty string otherwise + if media_ext not in mimetypes.types_map: + media_ext = '' + media_type = mimetypes.guess_type(request.url)[0] + if media_type: + media_ext = mimetypes.guess_extension(media_type) return 'full/%s%s' % (media_guid, media_ext) diff --git a/scrapy/pipelines/images.py b/scrapy/pipelines/images.py index 872342fc0..2e646379c 100644 --- a/scrapy/pipelines/images.py +++ b/scrapy/pipelines/images.py @@ -5,12 +5,7 @@ See documentation in topics/media-pipeline.rst """ import functools import hashlib -import six - -try: - from cStringIO import StringIO as BytesIO -except ImportError: - from io import BytesIO +from io import BytesIO from PIL import Image @@ -135,7 +130,7 @@ class ImagesPipeline(FilesPipeline): image, buf = self.convert_image(orig_image) yield path, image, buf - for thumb_id, size in six.iteritems(self.thumbs): + for thumb_id, size in self.thumbs.items(): thumb_path = self.thumb_path(request, thumb_id, response=response, info=info) thumb_image, thumb_buf = self.convert_image(image, size) yield thumb_path, thumb_image, thumb_buf diff --git a/scrapy/pipelines/media.py b/scrapy/pipelines/media.py index 95dca9a3f..c174addf9 100644 --- a/scrapy/pipelines/media.py +++ b/scrapy/pipelines/media.py @@ -1,5 +1,3 @@ -from __future__ import print_function - import functools import logging from collections import defaultdict diff --git a/scrapy/pqueues.py b/scrapy/pqueues.py index 6ecd1b51a..717ed4d27 100644 --- a/scrapy/pqueues.py +++ b/scrapy/pqueues.py @@ -86,9 +86,6 @@ class _SlotPriorityQueues(object): def __len__(self): return sum(len(x) for x in self.pqueues.values()) if self.pqueues else 0 - def __contains__(self, slot): - return slot in self.pqueues - class ScrapyPriorityQueue(PriorityQueue): """ diff --git a/scrapy/resolver.py b/scrapy/resolver.py index 0aaced7e4..4df949015 100644 --- a/scrapy/resolver.py +++ b/scrapy/resolver.py @@ -7,6 +7,7 @@ from scrapy.utils.datatypes import LocalCache dnscache = LocalCache(10000) + class CachingThreadedResolver(ThreadedResolver): def __init__(self, reactor, cache_size, timeout): super(CachingThreadedResolver, self).__init__(reactor) diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 4a2d5bf52..91d309147 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -2,15 +2,13 @@ This module implements a class which returns the appropriate Response class based on different criteria. """ -from __future__ import absolute_import from mimetypes import MimeTypes from pkgutil import get_data from io import StringIO -import six from scrapy.http import Response from scrapy.utils.misc import load_object -from scrapy.utils.python import binary_is_text, to_bytes, to_native_str +from scrapy.utils.python import binary_is_text, to_bytes, to_unicode class ResponseTypes(object): @@ -37,7 +35,7 @@ class ResponseTypes(object): self.mimetypes = MimeTypes() mimedata = get_data('scrapy', 'mime.types').decode('utf8') self.mimetypes.readfp(StringIO(mimedata)) - for mimetype, cls in six.iteritems(self.CLASSES): + for mimetype, cls in self.CLASSES.items(): self.classes[mimetype] = load_object(cls) def from_mimetype(self, mimetype): @@ -55,12 +53,12 @@ class ResponseTypes(object): header """ if content_encoding: return Response - mimetype = to_native_str(content_type).split(';')[0].strip().lower() + mimetype = to_unicode(content_type).split(';')[0].strip().lower() return self.from_mimetype(mimetype) def from_content_disposition(self, content_disposition): try: - filename = to_native_str(content_disposition, + filename = to_unicode(content_disposition, encoding='latin-1', errors='replace').split(';')[1].split('=')[1] filename = filename.strip('"\'') return self.from_filename(filename) diff --git a/scrapy/robotstxt.py b/scrapy/robotstxt.py new file mode 100644 index 000000000..0a9af3a62 --- /dev/null +++ b/scrapy/robotstxt.py @@ -0,0 +1,128 @@ +import sys +import logging +from abc import ABCMeta, abstractmethod + +from scrapy.utils.python import to_unicode + + +logger = logging.getLogger(__name__) + + +def decode_robotstxt(robotstxt_body, spider, to_native_str_type=False): + try: + if to_native_str_type: + robotstxt_body = to_unicode(robotstxt_body) + else: + robotstxt_body = robotstxt_body.decode('utf-8') + except UnicodeDecodeError: + # If we found garbage or robots.txt in an encoding other than UTF-8, disregard it. + # Switch to 'allow all' state. + logger.warning("Failure while parsing robots.txt. " + "File either contains garbage or is in an encoding other than UTF-8, treating it as an empty file.", + exc_info=sys.exc_info(), + extra={'spider': spider}) + robotstxt_body = '' + return robotstxt_body + + +class RobotParser(metaclass=ABCMeta): + @classmethod + @abstractmethod + def from_crawler(cls, crawler, robotstxt_body): + """Parse the content of a robots.txt_ file as bytes. This must be a class method. + It must return a new instance of the parser backend. + + :param crawler: crawler which made the request + :type crawler: :class:`~scrapy.crawler.Crawler` instance + + :param robotstxt_body: content of a robots.txt_ file. + :type robotstxt_body: bytes + """ + pass + + @abstractmethod + def allowed(self, url, user_agent): + """Return ``True`` if ``user_agent`` is allowed to crawl ``url``, otherwise return ``False``. + + :param url: Absolute URL + :type url: string + + :param user_agent: User agent + :type user_agent: string + """ + pass + + +class PythonRobotParser(RobotParser): + def __init__(self, robotstxt_body, spider): + from urllib.robotparser import RobotFileParser + self.spider = spider + robotstxt_body = decode_robotstxt(robotstxt_body, spider, to_native_str_type=True) + self.rp = RobotFileParser() + self.rp.parse(robotstxt_body.splitlines()) + + @classmethod + def from_crawler(cls, crawler, robotstxt_body): + spider = None if not crawler else crawler.spider + o = cls(robotstxt_body, spider) + return o + + def allowed(self, url, user_agent): + user_agent = to_unicode(user_agent) + url = to_unicode(url) + return self.rp.can_fetch(user_agent, url) + + +class ReppyRobotParser(RobotParser): + def __init__(self, robotstxt_body, spider): + from reppy.robots import Robots + self.spider = spider + self.rp = Robots.parse('', robotstxt_body) + + @classmethod + def from_crawler(cls, crawler, robotstxt_body): + spider = None if not crawler else crawler.spider + o = cls(robotstxt_body, spider) + return o + + def allowed(self, url, user_agent): + return self.rp.allowed(url, user_agent) + + +class RerpRobotParser(RobotParser): + def __init__(self, robotstxt_body, spider): + from robotexclusionrulesparser import RobotExclusionRulesParser + self.spider = spider + self.rp = RobotExclusionRulesParser() + robotstxt_body = decode_robotstxt(robotstxt_body, spider) + self.rp.parse(robotstxt_body) + + @classmethod + def from_crawler(cls, crawler, robotstxt_body): + spider = None if not crawler else crawler.spider + o = cls(robotstxt_body, spider) + return o + + def allowed(self, url, user_agent): + user_agent = to_unicode(user_agent) + url = to_unicode(url) + return self.rp.is_allowed(user_agent, url) + + +class ProtegoRobotParser(RobotParser): + def __init__(self, robotstxt_body, spider): + from protego import Protego + self.spider = spider + robotstxt_body = decode_robotstxt(robotstxt_body, spider) + self.rp = Protego.parse(robotstxt_body) + + @classmethod + def from_crawler(cls, crawler, robotstxt_body): + spider = None if not crawler else crawler.spider + o = cls(robotstxt_body, spider) + return o + + def allowed(self, url, user_agent): + user_agent = to_unicode(user_agent) + url = to_unicode(url) + return self.rp.can_fetch(url, user_agent) diff --git a/scrapy/selector/__init__.py b/scrapy/selector/__init__.py index 90e96ee92..a9240c1f6 100644 --- a/scrapy/selector/__init__.py +++ b/scrapy/selector/__init__.py @@ -1,4 +1,4 @@ """ Selectors """ -from scrapy.selector.unified import * +from scrapy.selector.unified import * # noqa: F401 diff --git a/scrapy/selector/unified.py b/scrapy/selector/unified.py index 62fda40b0..a08955dc9 100644 --- a/scrapy/selector/unified.py +++ b/scrapy/selector/unified.py @@ -2,12 +2,10 @@ XPath selectors based on lxml """ -import warnings from parsel import Selector as _ParselSelector from scrapy.utils.trackref import object_ref from scrapy.utils.python import to_bytes from scrapy.http import HtmlResponse, XmlResponse -from scrapy.utils.decorators import deprecated __all__ = ['Selector', 'SelectorList'] diff --git a/scrapy/settings/__init__.py b/scrapy/settings/__init__.py index f28c7940d..b6133619c 100644 --- a/scrapy/settings/__init__.py +++ b/scrapy/settings/__init__.py @@ -1,19 +1,12 @@ -import six import json import copy -import collections +from collections.abc import MutableMapping from importlib import import_module from pprint import pformat from scrapy.settings import default_settings -if six.PY2: - MutableMapping = collections.MutableMapping -else: - MutableMapping = collections.abc.MutableMapping - - SETTINGS_PRIORITIES = { 'default': 0, 'command': 10, @@ -29,7 +22,7 @@ def get_settings_priority(priority): :attr:`~scrapy.settings.SETTINGS_PRIORITIES` dictionary and returns its numerical value, or directly returns a given numerical priority. """ - if isinstance(priority, six.string_types): + if isinstance(priority, str): return SETTINGS_PRIORITIES[priority] else: return priority @@ -179,7 +172,7 @@ class BaseSettings(MutableMapping): :type default: any """ value = self.get(name, default or []) - if isinstance(value, six.string_types): + if isinstance(value, str): value = value.split(',') return list(value) @@ -200,7 +193,7 @@ class BaseSettings(MutableMapping): :type default: any """ value = self.get(name, default or {}) - if isinstance(value, six.string_types): + if isinstance(value, str): value = json.loads(value) return dict(value) @@ -290,7 +283,7 @@ class BaseSettings(MutableMapping): :type priority: string or int """ self._assert_mutability() - if isinstance(module, six.string_types): + if isinstance(module, str): module = import_module(module) for key in dir(module): if key.isupper(): @@ -319,14 +312,14 @@ class BaseSettings(MutableMapping): :type priority: string or int """ self._assert_mutability() - if isinstance(values, six.string_types): + if isinstance(values, str): values = json.loads(values) if values is not None: if isinstance(values, BaseSettings): - for name, value in six.iteritems(values): + for name, value in values.items(): self.set(name, value, values.getpriority(name)) else: - for name, value in six.iteritems(values): + for name, value in values.items(): self.set(name, value, priority) def delete(self, name, priority='project'): @@ -383,7 +376,7 @@ class BaseSettings(MutableMapping): def _to_dict(self): return {k: (v._to_dict() if isinstance(v, BaseSettings) else v) - for k, v in six.iteritems(self)} + for k, v in self.items()} def copy_to_dict(self): """ @@ -451,7 +444,7 @@ class Settings(BaseSettings): self.setmodule(default_settings, 'default') # Promote default dictionaries to BaseSettings instances for per-key # priorities - for name, val in six.iteritems(self): + for name, val in self.items(): if isinstance(val, dict): self.set(name, BaseSettings(val, 'default'), 'default') self.update(values, priority) diff --git a/scrapy/settings/default_settings.py b/scrapy/settings/default_settings.py index d17eb3125..ddd48d327 100644 --- a/scrapy/settings/default_settings.py +++ b/scrapy/settings/default_settings.py @@ -17,10 +17,10 @@ import sys from importlib import import_module from os.path import join, abspath, dirname -import six - AJAXCRAWL_ENABLED = False +ASYNCIO_REACTOR = False + AUTOTHROTTLE_ENABLED = False AUTOTHROTTLE_DEBUG = False AUTOTHROTTLE_MAX_DELAY = 60.0 @@ -85,8 +85,10 @@ DOWNLOADER = 'scrapy.core.downloader.Downloader' DOWNLOADER_HTTPCLIENTFACTORY = 'scrapy.core.downloader.webclient.ScrapyHTTPClientFactory' DOWNLOADER_CLIENTCONTEXTFACTORY = 'scrapy.core.downloader.contextfactory.ScrapyClientContextFactory' -DOWNLOADER_CLIENT_TLS_METHOD = 'TLS' # Use highest TLS/SSL protocol version supported by the platform, - # also allowing negotiation +DOWNLOADER_CLIENT_TLS_CIPHERS = 'DEFAULT' +# Use highest TLS/SSL protocol version supported by the platform, also allowing negotiation: +DOWNLOADER_CLIENT_TLS_METHOD = 'TLS' +DOWNLOADER_CLIENT_TLS_VERBOSE_LOGGING = False DOWNLOADER_MIDDLEWARES = {} @@ -177,7 +179,7 @@ HTTPCACHE_ALWAYS_STORE = False HTTPCACHE_IGNORE_HTTP_CODES = [] HTTPCACHE_IGNORE_SCHEMES = ['file'] HTTPCACHE_IGNORE_RESPONSE_CACHE_CONTROLS = [] -HTTPCACHE_DBM_MODULE = 'anydbm' if six.PY2 else 'dbm' +HTTPCACHE_DBM_MODULE = 'dbm' HTTPCACHE_POLICY = 'scrapy.extensions.httpcache.DummyPolicy' HTTPCACHE_GZIP = False @@ -244,12 +246,16 @@ RETRY_HTTP_CODES = [500, 502, 503, 504, 522, 524, 408, 429] RETRY_PRIORITY_ADJUST = -1 ROBOTSTXT_OBEY = False +ROBOTSTXT_PARSER = 'scrapy.robotstxt.ProtegoRobotParser' +ROBOTSTXT_USER_AGENT = None SCHEDULER = 'scrapy.core.scheduler.Scheduler' SCHEDULER_DISK_QUEUE = 'scrapy.squeues.PickleLifoDiskQueue' SCHEDULER_MEMORY_QUEUE = 'scrapy.squeues.LifoMemoryQueue' SCHEDULER_PRIORITY_QUEUE = 'scrapy.pqueues.ScrapyPriorityQueue' +SCRAPER_SLOT_MAX_ACTIVE_SIZE = 5000000 + SPIDER_LOADER_CLASS = 'scrapy.spiderloader.SpiderLoader' SPIDER_LOADER_WARN_ONLY = False @@ -287,6 +293,7 @@ TELNETCONSOLE_PASSWORD = None SPIDER_CONTRACTS = {} SPIDER_CONTRACTS_BASE = { 'scrapy.contracts.default.UrlContract': 1, + 'scrapy.contracts.default.CallbackKeywordArgumentsContract': 1, 'scrapy.contracts.default.ReturnsContract': 2, 'scrapy.contracts.default.ScrapesContract': 3, } diff --git a/scrapy/shell.py b/scrapy/shell.py index 80b625633..a23b04df9 100644 --- a/scrapy/shell.py +++ b/scrapy/shell.py @@ -3,13 +3,11 @@ See documentation in docs/topics/shell.rst """ -from __future__ import print_function - import os import signal import warnings -from twisted.internet import reactor, threads, defer +from twisted.internet import threads, defer from twisted.python import threadable from w3lib.url import any_to_uri @@ -100,6 +98,7 @@ class Shell(object): return spider def fetch(self, request_or_url, spider=None, redirect=True, **kwargs): + from twisted.internet import reactor if isinstance(request_or_url, Request): request = request_or_url else: diff --git a/scrapy/signalmanager.py b/scrapy/signalmanager.py index 296d27ed8..481d97e9a 100644 --- a/scrapy/signalmanager.py +++ b/scrapy/signalmanager.py @@ -1,4 +1,3 @@ -from __future__ import absolute_import from pydispatch import dispatcher from scrapy.utils import signal as _signal @@ -46,16 +45,14 @@ class SignalManager(object): def send_catch_log_deferred(self, signal, **kwargs): """ - Like :meth:`send_catch_log` but supports returning `deferreds`_ from - signal handlers. + Like :meth:`send_catch_log` but supports returning + :class:`~twisted.internet.defer.Deferred` objects from signal handlers. Returns a Deferred that gets fired once all signal handlers deferreds were fired. Send a signal, catch exceptions and log them. The keyword arguments are passed to the signal handlers (connected through the :meth:`connect` method). - - .. _deferreds: https://twistedmatrix.com/documents/current/core/howto/defer.html """ kwargs.setdefault('sender', self.sender) return _signal.send_catch_log_deferred(signal, **kwargs) diff --git a/scrapy/spiderloader.py b/scrapy/spiderloader.py index 7478faa78..3beca4060 100644 --- a/scrapy/spiderloader.py +++ b/scrapy/spiderloader.py @@ -1,5 +1,4 @@ # -*- coding: utf-8 -*- -from __future__ import absolute_import from collections import defaultdict import traceback import warnings diff --git a/scrapy/spidermiddlewares/referer.py b/scrapy/spidermiddlewares/referer.py index 1ddfb37f4..dce2b3598 100644 --- a/scrapy/spidermiddlewares/referer.py +++ b/scrapy/spidermiddlewares/referer.py @@ -2,16 +2,15 @@ RefererMiddleware: populates Request referer field, based on the Response which originated it. """ -from six.moves.urllib.parse import urlparse import warnings +from urllib.parse import urlparse from w3lib.url import safe_url_string from scrapy.http import Request, Response from scrapy.exceptions import NotConfigured from scrapy import signals -from scrapy.utils.python import to_native_str -from scrapy.utils.httpobj import urlparse_cached +from scrapy.utils.python import to_unicode from scrapy.utils.misc import load_object from scrapy.utils.url import strip_url @@ -322,7 +321,7 @@ class RefererMiddleware(object): if isinstance(resp_or_url, Response): policy_header = resp_or_url.headers.get('Referrer-Policy') if policy_header is not None: - policy_name = to_native_str(policy_header.decode('latin1')) + policy_name = to_unicode(policy_header.decode('latin1')) if policy_name is None: return self.default_policy() diff --git a/scrapy/spiders/__init__.py b/scrapy/spiders/__init__.py index 94095bc27..9429f6cb2 100644 --- a/scrapy/spiders/__init__.py +++ b/scrapy/spiders/__init__.py @@ -10,7 +10,6 @@ from scrapy import signals from scrapy.http import Request from scrapy.utils.trackref import object_ref from scrapy.utils.url import url_is_from_spider -from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.utils.deprecate import method_is_overridden @@ -58,6 +57,11 @@ class Spider(object_ref): def start_requests(self): cls = self.__class__ + if not self.start_urls and hasattr(self, 'start_url'): + raise AttributeError( + "Crawling could not start: 'start_urls' not found " + "or empty (but found 'start_url' attribute instead, " + "did you miss an 's'?)") if method_is_overridden(cls, Spider, 'make_requests_from_url'): warnings.warn( "Spider.make_requests_from_url method is deprecated; it " @@ -100,6 +104,6 @@ class Spider(object_ref): # Top-level imports -from scrapy.spiders.crawl import CrawlSpider, Rule -from scrapy.spiders.feed import XMLFeedSpider, CSVFeedSpider -from scrapy.spiders.sitemap import SitemapSpider +from scrapy.spiders.crawl import CrawlSpider, Rule # noqa: F401 +from scrapy.spiders.feed import XMLFeedSpider, CSVFeedSpider # noqa: F401 +from scrapy.spiders.sitemap import SitemapSpider # noqa: F401 diff --git a/scrapy/spiders/crawl.py b/scrapy/spiders/crawl.py index 90a6eb806..a2c364c0e 100644 --- a/scrapy/spiders/crawl.py +++ b/scrapy/spiders/crawl.py @@ -8,39 +8,48 @@ See documentation in docs/topics/spiders.rst import copy import warnings -import six - from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.http import Request, HtmlResponse -from scrapy.utils.spider import iterate_spider_output -from scrapy.utils.python import get_func_args +from scrapy.linkextractors import LinkExtractor from scrapy.spiders import Spider +from scrapy.utils.python import get_func_args +from scrapy.utils.spider import iterate_spider_output -def _identity(request, response): +def _identity(x): + return x + + +def _identity_process_request(request, response): return request def _get_method(method, spider): if callable(method): return method - elif isinstance(method, six.string_types): + elif isinstance(method, str): return getattr(spider, method, None) +_default_link_extractor = LinkExtractor() + + class Rule(object): - def __init__(self, link_extractor, callback=None, cb_kwargs=None, follow=None, process_links=None, process_request=None): - self.link_extractor = link_extractor + def __init__(self, link_extractor=None, callback=None, cb_kwargs=None, follow=None, + process_links=None, process_request=None, errback=None): + self.link_extractor = link_extractor or _default_link_extractor self.callback = callback + self.errback = errback self.cb_kwargs = cb_kwargs or {} - self.process_links = process_links - self.process_request = process_request or _identity + self.process_links = process_links or _identity + self.process_request = process_request or _identity_process_request self.process_request_argcount = None self.follow = follow if follow is not None else not callback def _compile(self, spider): self.callback = _get_method(self.callback, spider) + self.errback = _get_method(self.errback, spider) self.process_links = _get_method(self.process_links, spider) self.process_request = _get_method(self.process_request, spider) self.process_request_argcount = len(get_func_args(self.process_request)) @@ -74,48 +83,59 @@ class CrawlSpider(Spider): def process_results(self, response, results): return results - def _build_request(self, rule, link): - r = Request(url=link.url, callback=self._response_downloaded) - r.meta.update(rule=rule, link_text=link.text) - return r + def _build_request(self, rule_index, link): + return Request( + url=link.url, + callback=self._callback, + errback=self._errback, + meta=dict(rule=rule_index, link_text=link.text), + ) def _requests_to_follow(self, response): if not isinstance(response, HtmlResponse): return seen = set() - for n, rule in enumerate(self._rules): + for rule_index, rule in enumerate(self._rules): links = [lnk for lnk in rule.link_extractor.extract_links(response) if lnk not in seen] - if links and rule.process_links: - links = rule.process_links(links) - for link in links: + for link in rule.process_links(links): seen.add(link) - request = self._build_request(n, link) + request = self._build_request(rule_index, link) yield rule._process_request(request, response) - def _response_downloaded(self, response): + def _callback(self, response): rule = self._rules[response.meta['rule']] return self._parse_response(response, rule.callback, rule.cb_kwargs, rule.follow) + def _errback(self, failure): + rule = self._rules[failure.request.meta['rule']] + return self._handle_failure(failure, rule.errback) + def _parse_response(self, response, callback, cb_kwargs, follow=True): if callback: cb_res = callback(response, **cb_kwargs) or () cb_res = self.process_results(response, cb_res) - for requests_or_item in iterate_spider_output(cb_res): - yield requests_or_item + for request_or_item in iterate_spider_output(cb_res): + yield request_or_item if follow and self._follow_links: for request_or_item in self._requests_to_follow(response): yield request_or_item + def _handle_failure(self, failure, errback): + if errback: + results = errback(failure) or () + for request_or_item in iterate_spider_output(results): + yield request_or_item + def _compile_rules(self): - self._rules = [copy.copy(r) for r in self.rules] - for rule in self._rules: - rule._compile(self) + self._rules = [] + for rule in self.rules: + self._rules.append(copy.copy(rule)) + self._rules[-1]._compile(self) @classmethod def from_crawler(cls, crawler, *args, **kwargs): spider = super(CrawlSpider, cls).from_crawler(crawler, *args, **kwargs) - spider._follow_links = crawler.settings.getbool( - 'CRAWLSPIDER_FOLLOW_LINKS', True) + spider._follow_links = crawler.settings.getbool('CRAWLSPIDER_FOLLOW_LINKS', True) return spider diff --git a/scrapy/spiders/feed.py b/scrapy/spiders/feed.py index 06e212e1c..c566f0236 100644 --- a/scrapy/spiders/feed.py +++ b/scrapy/spiders/feed.py @@ -100,8 +100,8 @@ class CSVFeedSpider(Spider): and the file's headers. """ - delimiter = None # When this is None, python's csv module's default delimiter is used - quotechar = None # When this is None, python's csv module's default quotechar is used + delimiter = None # When this is None, python's csv module's default delimiter is used + quotechar = None # When this is None, python's csv module's default quotechar is used headers = None def process_results(self, response, results): @@ -133,4 +133,3 @@ class CSVFeedSpider(Spider): raise NotConfigured('You must define parse_row method in order to scrape this CSV feed') response = self.adapt_response(response) return self.parse_rows(response) - diff --git a/scrapy/spiders/init.py b/scrapy/spiders/init.py index 2efb1a869..fd41133ea 100644 --- a/scrapy/spiders/init.py +++ b/scrapy/spiders/init.py @@ -29,4 +29,3 @@ class InitSpider(Spider): spider """ return self.initialized() - diff --git a/scrapy/spiders/sitemap.py b/scrapy/spiders/sitemap.py index 534c45c70..d368c7108 100644 --- a/scrapy/spiders/sitemap.py +++ b/scrapy/spiders/sitemap.py @@ -1,6 +1,5 @@ import re import logging -import six from scrapy.spiders import Spider from scrapy.http import Request, XmlResponse @@ -22,7 +21,7 @@ class SitemapSpider(Spider): super(SitemapSpider, self).__init__(*a, **kw) self._cbs = [] for r, c in self.sitemap_rules: - if isinstance(c, six.string_types): + if isinstance(c, str): c = getattr(self, c) self._cbs.append((regex(r), c)) self._follow = [regex(x) for x in self.sitemap_follow] @@ -86,7 +85,7 @@ class SitemapSpider(Spider): def regex(x): - if isinstance(x, six.string_types): + if isinstance(x, str): return re.compile(x) return x diff --git a/scrapy/squeues.py b/scrapy/squeues.py index 30cc926e5..d5d3be67e 100644 --- a/scrapy/squeues.py +++ b/scrapy/squeues.py @@ -3,7 +3,7 @@ Scheduler queues """ import marshal -from six.moves import cPickle as pickle +import pickle from queuelib import queue diff --git a/scrapy/statscollectors.py b/scrapy/statscollectors.py index 6da9ddcd2..f0bfaed34 100644 --- a/scrapy/statscollectors.py +++ b/scrapy/statscollectors.py @@ -80,5 +80,3 @@ class DummyStatsCollector(StatsCollector): def min_value(self, key, value, spider=None): pass - - diff --git a/scrapy/utils/asyncio.py b/scrapy/utils/asyncio.py new file mode 100644 index 000000000..917973de2 --- /dev/null +++ b/scrapy/utils/asyncio.py @@ -0,0 +1,17 @@ +import asyncio +from contextlib import suppress + +from twisted.internet import asyncioreactor +from twisted.internet.error import ReactorAlreadyInstalledError + + +def install_asyncio_reactor(): + """ Tries to install AsyncioSelectorReactor + """ + with suppress(ReactorAlreadyInstalledError): + asyncioreactor.install(asyncio.get_event_loop()) + + +def is_asyncio_reactor_installed(): + from twisted.internet import reactor + return isinstance(reactor, asyncioreactor.AsyncioSelectorReactor) diff --git a/scrapy/utils/benchserver.py b/scrapy/utils/benchserver.py index 5bbda6e27..cdbe21942 100644 --- a/scrapy/utils/benchserver.py +++ b/scrapy/utils/benchserver.py @@ -1,5 +1,6 @@ import random -from six.moves.urllib.parse import urlencode +from urllib.parse import urlencode + from twisted.web.server import Site from twisted.web.resource import Resource from twisted.internet import reactor diff --git a/scrapy/utils/boto.py b/scrapy/utils/boto.py index 421ab2f7e..12321caa5 100644 --- a/scrapy/utils/boto.py +++ b/scrapy/utils/boto.py @@ -1,21 +1,11 @@ """Boto/botocore helpers""" -from __future__ import absolute_import -import six - from scrapy.exceptions import NotConfigured def is_botocore(): try: - import botocore + import botocore # noqa: F401 return True except ImportError: - if six.PY2: - try: - import boto - return False - except ImportError: - raise NotConfigured('missing botocore or boto library') - else: - raise NotConfigured('missing botocore library') + raise NotConfigured('missing botocore library') diff --git a/scrapy/utils/conf.py b/scrapy/utils/conf.py index 26d66eaf8..23306ca28 100644 --- a/scrapy/utils/conf.py +++ b/scrapy/utils/conf.py @@ -1,11 +1,9 @@ import os import sys import numbers -import configparser +from configparser import ConfigParser from operator import itemgetter -import six - from scrapy.settings import BaseSettings from scrapy.utils.deprecate import update_classpath from scrapy.utils.python import without_none_values @@ -22,7 +20,7 @@ def build_component_list(compdict, custom=None, convert=update_classpath): def _map_keys(compdict): if isinstance(compdict, BaseSettings): compbs = BaseSettings() - for k, v in six.iteritems(compdict): + for k, v in compdict.items(): prio = compdict.getpriority(k) if compbs.getpriority(convert(k)) == prio: raise ValueError('Some paths in {!r} convert to the same ' @@ -33,13 +31,13 @@ def build_component_list(compdict, custom=None, convert=update_classpath): return compbs else: _check_components(compdict) - return {convert(k): v for k, v in six.iteritems(compdict)} + return {convert(k): v for k, v in compdict.items()} def _validate_values(compdict): """Fail if a value in the components dict is not a real number or None.""" - for name, value in six.iteritems(compdict): + for name, value in compdict.items(): if value is not None and not isinstance(value, numbers.Real): - raise ValueError('Invalid value {} for component {}, please provide ' \ + raise ValueError('Invalid value {} for component {}, please provide ' 'a real number or None instead'.format(value, name)) # BEGIN Backward compatibility for old (base, custom) call signature @@ -53,7 +51,7 @@ def build_component_list(compdict, custom=None, convert=update_classpath): _validate_values(compdict) compdict = without_none_values(_map_keys(compdict)) - return [k for k, v in sorted(six.iteritems(compdict), key=itemgetter(1))] + return [k for k, v in sorted(compdict.items(), key=itemgetter(1))] def arglist_to_dict(arglist): @@ -94,7 +92,7 @@ def init_env(project='default', set_syspath=True): def get_config(use_closest=True): """Get Scrapy config file as a ConfigParser""" sources = get_sources(use_closest) - cfg = configparser.ConfigParser() + cfg = ConfigParser() cfg.read(sources) return cfg diff --git a/scrapy/utils/console.py b/scrapy/utils/console.py index 2e9981556..7eb40f0ce 100644 --- a/scrapy/utils/console.py +++ b/scrapy/utils/console.py @@ -1,6 +1,7 @@ from functools import wraps from collections import OrderedDict + def _embed_ipython_shell(namespace={}, banner=''): """Start an IPython Shell""" try: @@ -23,6 +24,7 @@ def _embed_ipython_shell(namespace={}, banner=''): shell() return wrapper + def _embed_bpython_shell(namespace={}, banner=''): """Start a bpython shell""" import bpython @@ -31,6 +33,7 @@ def _embed_bpython_shell(namespace={}, banner=''): bpython.embed(locals_=namespace, banner=banner) return wrapper + def _embed_ptpython_shell(namespace={}, banner=''): """Start a ptpython shell""" import ptpython.repl @@ -40,21 +43,23 @@ def _embed_ptpython_shell(namespace={}, banner=''): ptpython.repl.embed(locals=namespace) return wrapper + def _embed_standard_shell(namespace={}, banner=''): """Start a standard python shell""" import code - try: # readline module is only available on unix systems + try: # readline module is only available on unix systems import readline except ImportError: pass else: - import rlcompleter + import rlcompleter # noqa: F401 readline.parse_and_bind("tab:complete") @wraps(_embed_standard_shell) def wrapper(namespace=namespace, banner=''): code.interact(banner=banner, local=namespace) return wrapper + DEFAULT_PYTHON_SHELLS = OrderedDict([ ('ptpython', _embed_ptpython_shell), ('ipython', _embed_ipython_shell), @@ -62,13 +67,14 @@ DEFAULT_PYTHON_SHELLS = OrderedDict([ ('python', _embed_standard_shell), ]) + def get_shell_embed_func(shells=None, known_shells=None): """Return the first acceptable shell-embed function from a given list of shell names. """ - if shells is None: # list, preference order of shells + if shells is None: # list, preference order of shells shells = DEFAULT_PYTHON_SHELLS.keys() - if known_shells is None: # available embeddable shells + if known_shells is None: # available embeddable shells known_shells = DEFAULT_PYTHON_SHELLS.copy() for shell in shells: if shell in known_shells: @@ -79,6 +85,7 @@ def get_shell_embed_func(shells=None, known_shells=None): except ImportError: continue + def start_python_console(namespace=None, banner='', shells=None): """Start Python console bound to the given namespace. Readline support and tab completion will be used on Unix, if available. @@ -90,5 +97,5 @@ def start_python_console(namespace=None, banner='', shells=None): shell = get_shell_embed_func(shells) if shell is not None: shell(namespace=namespace, banner=banner) - except SystemExit: # raised when using exit() in python code.interact + except SystemExit: # raised when using exit() in python code.interact pass diff --git a/scrapy/utils/curl.py b/scrapy/utils/curl.py new file mode 100644 index 000000000..16639356e --- /dev/null +++ b/scrapy/utils/curl.py @@ -0,0 +1,94 @@ +import argparse +import warnings +from shlex import split +from http.cookies import SimpleCookie +from urllib.parse import urlparse + +from w3lib.http import basic_auth_header + + +class CurlParser(argparse.ArgumentParser): + def error(self, message): + error_msg = \ + 'There was an error parsing the curl command: {}'.format(message) + raise ValueError(error_msg) + + +curl_parser = CurlParser() +curl_parser.add_argument('url') +curl_parser.add_argument('-H', '--header', dest='headers', action='append') +curl_parser.add_argument('-X', '--request', dest='method', default='get') +curl_parser.add_argument('-d', '--data', dest='data') +curl_parser.add_argument('-u', '--user', dest='auth') + + +safe_to_ignore_arguments = [ + ['--compressed'], + # `--compressed` argument is not safe to ignore, but it's included here + # because the `HttpCompressionMiddleware` is enabled by default + ['-s', '--silent'], + ['-v', '--verbose'], + ['-#', '--progress-bar'] +] + +for argument in safe_to_ignore_arguments: + curl_parser.add_argument(*argument, action='store_true') + + +def curl_to_request_kwargs(curl_command, ignore_unknown_options=True): + """Convert a cURL command syntax to Request kwargs. + + :param str curl_command: string containing the curl command + :param bool ignore_unknown_options: If true, only a warning is emitted when + cURL options are unknown. Otherwise raises an error. (default: True) + :return: dictionary of Request kwargs + """ + + curl_args = split(curl_command) + + if curl_args[0] != 'curl': + raise ValueError('A curl command must start with "curl"') + + parsed_args, argv = curl_parser.parse_known_args(curl_args[1:]) + + if argv: + msg = 'Unrecognized options: {}'.format(', '.join(argv)) + if ignore_unknown_options: + warnings.warn(msg) + else: + raise ValueError(msg) + + url = parsed_args.url + + # curl automatically prepends 'http' if the scheme is missing, but Request + # needs the scheme to work + parsed_url = urlparse(url) + if not parsed_url.scheme: + url = 'http://' + url + + result = {'method': parsed_args.method.upper(), 'url': url} + + headers = [] + cookies = {} + for header in parsed_args.headers or (): + name, val = header.split(':', 1) + name = name.strip() + val = val.strip() + if name.title() == 'Cookie': + for name, morsel in SimpleCookie(val).items(): + cookies[name] = morsel.value + else: + headers.append((name, val)) + + if parsed_args.auth: + user, password = parsed_args.auth.split(':', 1) + headers.append(('Authorization', basic_auth_header(user, password))) + + if headers: + result['headers'] = headers + if cookies: + result['cookies'] = cookies + if parsed_args.data: + result['body'] = parsed_args.data + + return result diff --git a/scrapy/utils/datatypes.py b/scrapy/utils/datatypes.py index df2b99c28..a52bbc70e 100644 --- a/scrapy/utils/datatypes.py +++ b/scrapy/utils/datatypes.py @@ -5,21 +5,15 @@ Python Standard Library. This module must not depend on any module outside the Standard Library. """ -import copy import collections +import copy import warnings - -import six +import weakref +from collections.abc import Mapping from scrapy.exceptions import ScrapyDeprecationWarning -if six.PY2: - Mapping = collections.Mapping -else: - Mapping = collections.abc.Mapping - - class MultiValueDictKeyError(KeyError): def __init__(self, *args, **kwargs): warnings.warn( @@ -156,7 +150,7 @@ class MultiValueDict(dict): self.setlistdefault(key, []).append(value) except TypeError: raise ValueError("MultiValueDict.update() takes either a MultiValueDict or dictionary") - for key, value in six.iteritems(kwargs): + for key, value in kwargs.items(): self.setlistdefault(key, []).append(value) @@ -243,71 +237,10 @@ class CaselessDict(dict): return dict.pop(self, self.normkey(key), *args) -class MergeDict(object): - """ - A simple class for creating new "virtual" dictionaries that actually look - up values in more than one dictionary, passed in the constructor. - - If a key appears in more than one of the given dictionaries, only the - first occurrence will be used. - """ - def __init__(self, *dicts): - if not six.PY2: - warnings.warn( - "scrapy.utils.datatypes.MergeDict is deprecated in favor " - "of collections.ChainMap (introduced in Python 3.3)", - category=ScrapyDeprecationWarning, - stacklevel=2, - ) - self.dicts = dicts - - def __getitem__(self, key): - for dict_ in self.dicts: - try: - return dict_[key] - except KeyError: - pass - raise KeyError - - def __copy__(self): - return self.__class__(*self.dicts) - - def get(self, key, default=None): - try: - return self[key] - except KeyError: - return default - - def getlist(self, key): - for dict_ in self.dicts: - if key in dict_.keys(): - return dict_.getlist(key) - return [] - - def items(self): - item_list = [] - for dict_ in self.dicts: - item_list.extend(dict_.items()) - return item_list - - def has_key(self, key): - for dict_ in self.dicts: - if key in dict_: - return True - return False - - __contains__ = has_key - - def copy(self): - """Returns a copy of this object.""" - return self.__copy__() - - class LocalCache(collections.OrderedDict): """Dictionary with a finite number of keys. Older items expires first. - """ def __init__(self, limit=None): @@ -315,11 +248,41 @@ class LocalCache(collections.OrderedDict): self.limit = limit def __setitem__(self, key, value): - while len(self) >= self.limit: - self.popitem(last=False) + if self.limit: + while len(self) >= self.limit: + self.popitem(last=False) super(LocalCache, self).__setitem__(key, value) +class LocalWeakReferencedCache(weakref.WeakKeyDictionary): + """ + A weakref.WeakKeyDictionary implementation that uses LocalCache as its + underlying data structure, making it ordered and capable of being size-limited. + + Useful for memoization, while avoiding keeping received + arguments in memory only because of the cached references. + + Note: like LocalCache and unlike weakref.WeakKeyDictionary, + it cannot be instantiated with an initial dictionary. + """ + + def __init__(self, limit=None): + super(LocalWeakReferencedCache, self).__init__() + self.data = LocalCache(limit=limit) + + def __setitem__(self, key, value): + try: + super(LocalWeakReferencedCache, self).__setitem__(key, value) + except TypeError: + pass # key is not weak-referenceable, skip caching + + def __getitem__(self, key): + try: + return super(LocalWeakReferencedCache, self).__getitem__(key) + except TypeError: + return None # key is not weak-referenceable, it's not cached + + class SequenceExclude(object): """Object to test if an item is NOT within some sequence.""" diff --git a/scrapy/utils/decorators.py b/scrapy/utils/decorators.py index 38bee1a6c..2e2c7adc1 100644 --- a/scrapy/utils/decorators.py +++ b/scrapy/utils/decorators.py @@ -34,6 +34,7 @@ def defers(func): return defer.maybeDeferred(func, *a, **kw) return wrapped + def inthread(func): """Decorator to call a function in a thread and return a deferred with the result diff --git a/scrapy/utils/defer.py b/scrapy/utils/defer.py index 69d621830..62b43a96c 100644 --- a/scrapy/utils/defer.py +++ b/scrapy/utils/defer.py @@ -1,11 +1,16 @@ """ Helper functions for dealing with Twisted deferreds """ +import asyncio +from functools import wraps +import inspect -from twisted.internet import defer, reactor, task +from twisted.internet import defer, task from twisted.python import failure from scrapy.exceptions import IgnoreRequest +from scrapy.utils.asyncio import is_asyncio_reactor_installed + def defer_fail(_failure): """Same as twisted.internet.defer.fail but delay calling errback until @@ -14,10 +19,12 @@ def defer_fail(_failure): It delays by 100ms so reactor has a chance to go through readers and writers before attending pending delayed calls, so do not set delay to zero. """ + from twisted.internet import reactor d = defer.Deferred() reactor.callLater(0.1, d.errback, _failure) return d + def defer_succeed(result): """Same as twisted.internet.defer.succeed but delay calling callback until next reactor loop @@ -25,10 +32,12 @@ def defer_succeed(result): It delays by 100ms so reactor has a chance to go trough readers and writers before attending pending delayed calls, so do not set delay to zero. """ + from twisted.internet import reactor d = defer.Deferred() reactor.callLater(0.1, d.callback, result) return d + def defer_result(result): if isinstance(result, defer.Deferred): return result @@ -37,6 +46,7 @@ def defer_result(result): else: return defer_succeed(result) + def mustbe_deferred(f, *args, **kw): """Same as twisted.internet.defer.maybeDeferred, but delay calling callback/errback to next reactor loop @@ -53,6 +63,7 @@ def mustbe_deferred(f, *args, **kw): else: return defer_result(result) + def parallel(iterable, count, callable, *args, **named): """Execute a callable over the objects in the given iterable, in parallel, using no more than ``count`` concurrent calls. @@ -63,6 +74,7 @@ def parallel(iterable, count, callable, *args, **named): work = (callable(elem, *args, **named) for elem in iterable) return defer.DeferredList([coop.coiterate(work) for _ in range(count)]) + def process_chain(callbacks, input, *a, **kw): """Return a Deferred built by chaining the given callbacks""" d = defer.Deferred() @@ -71,6 +83,7 @@ def process_chain(callbacks, input, *a, **kw): d.callback(input) return d + def process_chain_both(callbacks, errbacks, input, *a, **kw): """Return a Deferred built by chaining the given callbacks and errbacks""" d = defer.Deferred() @@ -83,6 +96,7 @@ def process_chain_both(callbacks, errbacks, input, *a, **kw): d.callback(input) return d + def process_parallel(callbacks, input, *a, **kw): """Return a Deferred with the output of all successful calls to the given callbacks @@ -92,6 +106,7 @@ def process_parallel(callbacks, input, *a, **kw): d.addCallbacks(lambda r: [x[1] for x in r], lambda f: f.value.subFailure) return d + def iter_errback(iterable, errback, *a, **kw): """Wraps an iterable calling an errback if an error is caught while iterating it. @@ -104,3 +119,37 @@ def iter_errback(iterable, errback, *a, **kw): break except Exception: errback(failure.Failure(), *a, **kw) + + +def _isfuture(o): + # workaround for Python before 3.5.3 not having asyncio.isfuture + if hasattr(asyncio, 'isfuture'): + return asyncio.isfuture(o) + return isinstance(o, asyncio.Future) + + +def deferred_from_coro(o): + """Converts a coroutine into a Deferred, or returns the object as is if it isn't a coroutine""" + if isinstance(o, defer.Deferred): + return o + if _isfuture(o) or inspect.isawaitable(o): + if not is_asyncio_reactor_installed(): + # wrapping the coroutine directly into a Deferred, this doesn't work correctly with coroutines + # that use asyncio, e.g. "await asyncio.sleep(1)" + return defer.ensureDeferred(o) + else: + # wrapping the coroutine into a Future and then into a Deferred, this requires AsyncioSelectorReactor + return defer.Deferred.fromFuture(asyncio.ensure_future(o)) + return o + + +def deferred_f_from_coro_f(coro_f): + """ Converts a coroutine function into a function that returns a Deferred. + + The coroutine function will be called at the time when the wrapper is called. Wrapper args will be passed to it. + This is useful for callback chains, as callback functions are called with the previous callback result. + """ + @wraps(coro_f) + def f(*coro_args, **coro_kwargs): + return deferred_from_coro(coro_f(*coro_args, **coro_kwargs)) + return f diff --git a/scrapy/utils/display.py b/scrapy/utils/display.py index f6a6c4645..9735220ef 100644 --- a/scrapy/utils/display.py +++ b/scrapy/utils/display.py @@ -2,10 +2,10 @@ pprint and pformat wrappers with colorization support """ -from __future__ import print_function import sys from pprint import pformat as pformat_ + def _colorize(text, colorize=True): if not colorize or not sys.stdout.isatty(): return text @@ -17,8 +17,10 @@ def _colorize(text, colorize=True): except ImportError: return text + def pformat(obj, *args, **kwargs): return _colorize(pformat_(obj), kwargs.pop('colorize', True)) + def pprint(obj, *args, **kwargs): print(pformat(obj, *args, **kwargs)) diff --git a/scrapy/utils/engine.py b/scrapy/utils/engine.py index 11dd36d91..267c7ecd1 100644 --- a/scrapy/utils/engine.py +++ b/scrapy/utils/engine.py @@ -1,7 +1,8 @@ """Some debugging functions for working with the Scrapy engine""" -from __future__ import print_function -from time import time # used in global tests code +# used in global tests code +from time import time # noqa: F401 + def get_engine_status(engine): """Return a report of the current engine status""" @@ -32,6 +33,7 @@ def get_engine_status(engine): return checks + def format_engine_status(engine=None): checks = get_engine_status(engine) s = "Execution engine status\n\n" @@ -41,5 +43,6 @@ def format_engine_status(engine=None): return s + def print_engine_status(engine): print(format_engine_status(engine)) diff --git a/scrapy/utils/ftp.py b/scrapy/utils/ftp.py index 9992a916e..1bb754a69 100644 --- a/scrapy/utils/ftp.py +++ b/scrapy/utils/ftp.py @@ -3,6 +3,7 @@ import posixpath from ftplib import error_perm, FTP from posixpath import dirname + def ftp_makedirs_cwd(ftp, path, first_call=True): """Set the current directory of the FTP connection given in the ``ftp`` argument (as a ftplib.FTP object), creating all parent directories if they diff --git a/scrapy/utils/gz.py b/scrapy/utils/gz.py index b3fb16b1e..9672e28da 100644 --- a/scrapy/utils/gz.py +++ b/scrapy/utils/gz.py @@ -1,13 +1,7 @@ -import struct - -try: - from cStringIO import StringIO as BytesIO -except ImportError: - from io import BytesIO from gzip import GzipFile - -import six +from io import BytesIO import re +import struct from scrapy.utils.decorators import deprecated @@ -17,14 +11,9 @@ from scrapy.utils.decorators import deprecated # (regression or bug-fix compared to Python 3.4) # - read1(), which fetches data before raising EOFError on next call # works here but is only available from Python>=3.3 -# - scrapy does not support Python 3.2 -# - Python 2.7 GzipFile works fine with standard read() + extrabuf -if six.PY2: - def read1(gzf, size=-1): - return gzf.read(size) -else: - def read1(gzf, size=-1): - return gzf.read1(size) +@deprecated('GzipFile.read1') +def read1(gzf, size=-1): + return gzf.read1(size) def gunzip(data): @@ -37,7 +26,7 @@ def gunzip(data): chunk = b'.' while chunk: try: - chunk = read1(f, 8196) + chunk = f.read1(8196) output_list.append(chunk) except (IOError, EOFError, struct.error): # complete only if there is some data, otherwise re-raise @@ -56,6 +45,7 @@ def gunzip(data): _is_gzipped = re.compile(br'^application/(x-)?gzip\b', re.I).search _is_octetstream = re.compile(br'^(application|binary)/octet-stream\b', re.I).search + @deprecated def is_gzipped(response): """Return True if the response is gzipped, or False otherwise""" diff --git a/scrapy/utils/http.py b/scrapy/utils/http.py index b6e05c862..bab262393 100644 --- a/scrapy/utils/http.py +++ b/scrapy/utils/http.py @@ -8,7 +8,7 @@ import warnings from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.utils.decorators import deprecated -from w3lib.http import * +from w3lib.http import * # noqa: F401 warnings.warn("Module `scrapy.utils.http` is deprecated, " @@ -34,4 +34,3 @@ def decode_chunked_transfer(chunked_body): body += t[:size] t = t[size+2:] return body - diff --git a/scrapy/utils/httpobj.py b/scrapy/utils/httpobj.py index b4c929b0e..c8d4391b1 100644 --- a/scrapy/utils/httpobj.py +++ b/scrapy/utils/httpobj.py @@ -1,10 +1,12 @@ """Helper functions for scrapy.http objects (Request, Response)""" import weakref +from urllib.parse import urlparse -from six.moves.urllib.parse import urlparse _urlparse_cache = weakref.WeakKeyDictionary() + + def urlparse_cached(request_or_response): """Return urlparse.urlparse caching the result, where the argument can be a Request or Response object diff --git a/scrapy/utils/iterators.py b/scrapy/utils/iterators.py index a12e14005..3c0cb68c3 100644 --- a/scrapy/utils/iterators.py +++ b/scrapy/utils/iterators.py @@ -1,17 +1,13 @@ -import re import csv import logging -try: - from cStringIO import StringIO as BytesIO -except ImportError: - from io import BytesIO +import re from io import StringIO -import six from scrapy.http import TextResponse, Response from scrapy.selector import Selector from scrapy.utils.python import re_rsearch, to_unicode + logger = logging.getLogger(__name__) @@ -64,7 +60,7 @@ class _StreamReader(object): self._text, self.encoding = obj.body, obj.encoding else: self._text, self.encoding = obj, 'utf-8' - self._is_unicode = isinstance(self._text, six.text_type) + self._is_unicode = isinstance(self._text, str) def read(self, n=65535): self.read = self._read_unicode if self._is_unicode else self._read_string @@ -102,11 +98,7 @@ def csviter(obj, delimiter=None, headers=None, encoding=None, quotechar=None): def row_to_unicode(row_): return [to_unicode(field, encoding) for field in row_] - # Python 3 csv reader input object needs to return strings - if six.PY3: - lines = StringIO(_body_or_str(obj, unicode=True)) - else: - lines = BytesIO(_body_or_str(obj, unicode=False)) + lines = StringIO(_body_or_str(obj, unicode=True)) kwargs = {} if delimiter: kwargs["delimiter"] = delimiter @@ -133,7 +125,7 @@ def csviter(obj, delimiter=None, headers=None, encoding=None, quotechar=None): def _body_or_str(obj, unicode=True): - expected_types = (Response, six.text_type, six.binary_type) + expected_types = (Response, str, bytes) assert isinstance(obj, expected_types), \ "obj must be %s, not %s" % ( " or ".join(t.__name__ for t in expected_types), @@ -145,7 +137,7 @@ def _body_or_str(obj, unicode=True): return obj.text else: return obj.body.decode('utf-8') - elif isinstance(obj, six.text_type): + elif isinstance(obj, str): return obj if unicode else obj.encode('utf-8') else: return obj.decode('utf-8') if unicode else obj diff --git a/scrapy/utils/job.py b/scrapy/utils/job.py index 389fde73a..4f1e601fc 100644 --- a/scrapy/utils/job.py +++ b/scrapy/utils/job.py @@ -1,5 +1,6 @@ import os + def job_dir(settings): path = settings['JOBDIR'] if path and not os.path.exists(path): diff --git a/scrapy/utils/log.py b/scrapy/utils/log.py index e07fb8698..e4cf0196b 100644 --- a/scrapy/utils/log.py +++ b/scrapy/utils/log.py @@ -11,6 +11,7 @@ from twisted.python import log as twisted_log import scrapy from scrapy.settings import Settings from scrapy.exceptions import ScrapyDeprecationWarning +from scrapy.utils.asyncio import is_asyncio_reactor_installed from scrapy.utils.versions import scrapy_components_versions @@ -148,6 +149,8 @@ def log_scrapy_info(settings): {'versions': ", ".join("%s %s" % (name, version) for name, version in scrapy_components_versions() if name != "Scrapy")}) + if is_asyncio_reactor_installed(): + logger.debug("Asyncio reactor is installed") class StreamLogger(object): diff --git a/scrapy/utils/markup.py b/scrapy/utils/markup.py index a18f308a3..9728c542a 100644 --- a/scrapy/utils/markup.py +++ b/scrapy/utils/markup.py @@ -6,9 +6,9 @@ For new code, always import from w3lib.html instead of this module import warnings from scrapy.exceptions import ScrapyDeprecationWarning -from w3lib.html import * +from w3lib.html import * # noqa: F401 warnings.warn("Module `scrapy.utils.markup` is deprecated. " "Please import from `w3lib.html` instead.", - ScrapyDeprecationWarning, stacklevel=2) \ No newline at end of file + ScrapyDeprecationWarning, stacklevel=2) diff --git a/scrapy/utils/misc.py b/scrapy/utils/misc.py index f638adb25..cb0ee5af3 100644 --- a/scrapy/utils/misc.py +++ b/scrapy/utils/misc.py @@ -1,19 +1,23 @@ """Helper functions which don't fit anywhere else""" +import ast +import inspect import os import re import hashlib +import warnings from contextlib import contextmanager from importlib import import_module from pkgutil import iter_modules +from textwrap import dedent -import six from w3lib.html import replace_entities +from scrapy.utils.datatypes import LocalWeakReferencedCache from scrapy.utils.python import flatten, to_unicode from scrapy.item import BaseItem -_ITERABLE_SINGLE_VALUES = dict, BaseItem, six.text_type, bytes +_ITERABLE_SINGLE_VALUES = dict, BaseItem, str, bytes def arg_to_iter(arg): @@ -83,7 +87,7 @@ def extract_regex(regex, text, encoding='utf-8'): * if the regex doesn't contain any group the entire regex matching is returned """ - if isinstance(regex, six.string_types): + if isinstance(regex, str): regex = re.compile(regex, re.UNICODE) try: @@ -92,7 +96,7 @@ def extract_regex(regex, text, encoding='utf-8'): strings = regex.findall(text) # full regex or numbered groups strings = flatten(strings) - if isinstance(text, six.text_type): + if isinstance(text, str): return [replace_entities(s, keep=['lt', 'amp']) for s in strings] else: return [replace_entities(to_unicode(s, encoding), keep=['lt', 'amp']) @@ -136,7 +140,7 @@ def create_instance(objcls, settings, crawler, *args, **kwargs): """ if settings is None: if crawler is None: - raise ValueError("Specifiy at least one of settings and crawler.") + raise ValueError("Specify at least one of settings and crawler.") settings = crawler.settings if crawler and hasattr(objcls, 'from_crawler'): return objcls.from_crawler(crawler, *args, **kwargs) @@ -162,3 +166,44 @@ def set_environ(**kwargs): del os.environ[k] else: os.environ[k] = v + + +_generator_callbacks_cache = LocalWeakReferencedCache(limit=128) + + +def is_generator_with_return_value(callable): + """ + Returns True if a callable is a generator function which includes a + 'return' statement with a value different than None, False otherwise + """ + if callable in _generator_callbacks_cache: + return _generator_callbacks_cache[callable] + + def returns_none(return_node): + value = return_node.value + return value is None or isinstance(value, ast.NameConstant) and value.value is None + + if inspect.isgeneratorfunction(callable): + tree = ast.parse(dedent(inspect.getsource(callable))) + for node in ast.walk(tree): + if isinstance(node, ast.Return) and not returns_none(node): + _generator_callbacks_cache[callable] = True + return _generator_callbacks_cache[callable] + + _generator_callbacks_cache[callable] = False + return _generator_callbacks_cache[callable] + + +def warn_on_generator_with_return_value(spider, callable): + """ + Logs a warning if a callable is a generator function and includes + a 'return' statement with a value different than None + """ + if is_generator_with_return_value(callable): + warnings.warn( + 'The "{}.{}" method is a generator and includes a "return" statement with a ' + 'value different than None. This could lead to unexpected behaviour. Please see ' + 'https://docs.python.org/3/reference/simple_stmts.html#the-return-statement ' + 'for details about the semantics of the "return" statement within generators' + .format(spider.__class__.__name__, callable.__name__), stacklevel=2, + ) diff --git a/scrapy/utils/multipart.py b/scrapy/utils/multipart.py index c2d8afd07..5dcf791b8 100644 --- a/scrapy/utils/multipart.py +++ b/scrapy/utils/multipart.py @@ -6,10 +6,10 @@ For new code, always import from w3lib.form instead of this module import warnings from scrapy.exceptions import ScrapyDeprecationWarning -from w3lib.form import * +from w3lib.form import * # noqa: F401 warnings.warn("Module `scrapy.utils.multipart` is deprecated. " "If you're using `encode_multipart` function, please use " "`urllib3.filepost.encode_multipart_formdata` instead", - ScrapyDeprecationWarning, stacklevel=2) \ No newline at end of file + ScrapyDeprecationWarning, stacklevel=2) diff --git a/scrapy/utils/ossignal.py b/scrapy/utils/ossignal.py index f87d5a803..45c9cef0c 100644 --- a/scrapy/utils/ossignal.py +++ b/scrapy/utils/ossignal.py @@ -1,9 +1,5 @@ - -from __future__ import absolute_import import signal -from twisted.internet import reactor - signal_names = {} for signame in dir(signal): @@ -19,6 +15,7 @@ def install_shutdown_handlers(function, override_sigint=True): SIGINT handler won't be install if there is already a handler in place (e.g. Pdb) """ + from twisted.internet import reactor reactor._handleSignals() signal.signal(signal.SIGTERM, function) if signal.getsignal(signal.SIGINT) == signal.default_int_handler or \ diff --git a/scrapy/utils/project.py b/scrapy/utils/project.py index 95c6a8035..f28c2eaa1 100644 --- a/scrapy/utils/project.py +++ b/scrapy/utils/project.py @@ -1,5 +1,5 @@ import os -from six.moves import cPickle as pickle +import pickle import warnings from importlib import import_module @@ -7,7 +7,8 @@ from os.path import join, dirname, abspath, isabs, exists from scrapy.utils.conf import closest_scrapy_cfg, get_config, init_env from scrapy.settings import Settings -from scrapy.exceptions import NotConfigured +from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning + ENVVAR = 'SCRAPY_SETTINGS_MODULE' DATADIR_CFG_SECTION = 'datadir' @@ -70,6 +71,9 @@ def get_project_settings(): # XXX: remove this hack pickled_settings = os.environ.get("SCRAPY_PICKLED_SETTINGS_TO_OVERRIDE") if pickled_settings: + warnings.warn("Use of environment variable " + "'SCRAPY_PICKLED_SETTINGS_TO_OVERRIDE' " + "is deprecated.", ScrapyDeprecationWarning) settings.setdict(pickle.loads(pickled_settings), priority='project') # XXX: deprecate and remove this functionality diff --git a/scrapy/utils/python.py b/scrapy/utils/python.py index c6140f885..8d829c5a5 100644 --- a/scrapy/utils/python.py +++ b/scrapy/utils/python.py @@ -7,7 +7,6 @@ import re import inspect import weakref import errno -import six from functools import partial, wraps from itertools import chain import sys @@ -65,10 +64,10 @@ def is_listlike(x): True >>> is_listlike((x for x in range(3))) True - >>> is_listlike(six.moves.xrange(5)) + >>> is_listlike(range(5)) True """ - return hasattr(x, "__iter__") and not isinstance(x, (six.text_type, bytes)) + return hasattr(x, "__iter__") and not isinstance(x, (str, bytes)) def unique(list_, key=lambda x: x): @@ -87,10 +86,10 @@ def unique(list_, key=lambda x: x): def to_unicode(text, encoding=None, errors='strict'): """Return the unicode representation of a bytes object ``text``. If ``text`` is already an unicode object, return it as-is.""" - if isinstance(text, six.text_type): + if isinstance(text, str): return text - if not isinstance(text, (bytes, six.text_type)): - raise TypeError('to_unicode must receive a bytes, str or unicode ' + if not isinstance(text, (bytes, str)): + raise TypeError('to_unicode must receive a bytes or str ' 'object, got %s' % type(text).__name__) if encoding is None: encoding = 'utf-8' @@ -102,21 +101,18 @@ def to_bytes(text, encoding=None, errors='strict'): is already a bytes object, return it as-is.""" if isinstance(text, bytes): return text - if not isinstance(text, six.string_types): - raise TypeError('to_bytes must receive a unicode, str or bytes ' + if not isinstance(text, str): + raise TypeError('to_bytes must receive a str or bytes ' 'object, got %s' % type(text).__name__) if encoding is None: encoding = 'utf-8' return text.encode(encoding, errors) +@deprecated('to_unicode') def to_native_str(text, encoding=None, errors='strict'): - """ Return str representation of ``text`` - (bytes in Python 2.x and unicode in Python 3.x). """ - if six.PY2: - return to_bytes(text, encoding, errors) - else: - return to_unicode(text, encoding, errors) + """ Return str representation of ``text``. """ + return to_unicode(text, encoding, errors) def re_rsearch(pattern, text, chunk_size=1024): @@ -141,7 +137,7 @@ def re_rsearch(pattern, text, chunk_size=1024): yield (text[offset:], offset) yield (text, 0) - if isinstance(pattern, six.string_types): + if isinstance(pattern, str): pattern = re.compile(pattern) for chunk, offset in _chunk_iter(): @@ -165,9 +161,10 @@ def memoizemethod_noargs(method): return new_method -_BINARYCHARS = {six.b(chr(i)) for i in range(32)} - {b"\0", b"\t", b"\n", b"\r"} +_BINARYCHARS = {to_bytes(chr(i)) for i in range(32)} - {b"\0", b"\t", b"\n", b"\r"} _BINARYCHARS |= {ord(ch) for ch in _BINARYCHARS} + @deprecated("scrapy.utils.python.binary_is_text") def isbinarytext(text): """ This function is deprecated. @@ -189,7 +186,7 @@ def _getargspec_py23(func): """_getargspec_py23(function) -> named tuple ArgSpec(args, varargs, keywords, defaults) - Identical to inspect.getargspec() in python2, but uses + Was identical to inspect.getargspec() in python2, but uses inspect.getfullargspec() for python3 behind the scenes to avoid DeprecationWarning. @@ -199,9 +196,6 @@ def _getargspec_py23(func): >>> _getargspec_py23(f) ArgSpec(args=['a', 'b'], varargs='ar', keywords='kw', defaults=(2,)) """ - if six.PY2: - return inspect.getargspec(func) - return inspect.ArgSpec(*inspect.getfullargspec(func)[:4]) @@ -303,13 +297,13 @@ class WeakKeyCache(object): def stringify_dict(dct_or_tuples, encoding='utf-8', keys_only=True): """Return a (new) dict with unicode keys (and values when "keys_only" is False) of the given dict converted to strings. ``dct_or_tuples`` can be a - dict or a list of tuples, like any dict constructor supports. + dict or a list of tuples, like any dict ``__init__`` method supports. """ d = {} - for k, v in six.iteritems(dict(dct_or_tuples)): - k = k.encode(encoding) if isinstance(k, six.text_type) else k + for k, v in dict(dct_or_tuples).items(): + k = k.encode(encoding) if isinstance(k, str) else k if not keys_only: - v = v.encode(encoding) if isinstance(v, six.text_type) else v + v = v.encode(encoding) if isinstance(v, str) else v d[k] = v return d @@ -351,7 +345,7 @@ def without_none_values(iterable): value ``None`` have been removed. """ try: - return {k: v for k, v in six.iteritems(iterable) if v is not None} + return {k: v for k, v in iterable.items() if v is not None} except AttributeError: return type(iterable)((v for v in iterable if v is not None)) @@ -388,9 +382,11 @@ class MutableChain(object): self.data = chain(self.data, *iterables) def __iter__(self): - return self.data.__iter__() + return self def __next__(self): return next(self.data) - next = __next__ + @deprecated("scrapy.utils.python.MutableChain.__next__") + def next(self): + return self.__next__() diff --git a/scrapy/utils/reactor.py b/scrapy/utils/reactor.py index 83186a372..b98fff6ec 100644 --- a/scrapy/utils/reactor.py +++ b/scrapy/utils/reactor.py @@ -1,12 +1,14 @@ -from twisted.internet import reactor, error +from twisted.internet import error + def listen_tcp(portrange, host, factory): """Like reactor.listenTCP but tries different ports in a range.""" + from twisted.internet import reactor assert len(portrange) <= 2, "invalid portrange: %s" % portrange - if not hasattr(portrange, '__iter__'): - return reactor.listenTCP(portrange, factory, interface=host) if not portrange: return reactor.listenTCP(0, factory, interface=host) + if not hasattr(portrange, '__iter__'): + return reactor.listenTCP(portrange, factory, interface=host) if len(portrange) == 1: return reactor.listenTCP(portrange[0], factory, interface=host) for x in range(portrange[0], portrange[1]+1): @@ -29,6 +31,7 @@ class CallLaterOnce(object): self._call = None def schedule(self, delay=0): + from twisted.internet import reactor if self._call is None: self._call = reactor.callLater(delay, self) diff --git a/scrapy/utils/reqser.py b/scrapy/utils/reqser.py index c7ea7b425..749bbc387 100644 --- a/scrapy/utils/reqser.py +++ b/scrapy/utils/reqser.py @@ -1,10 +1,8 @@ """ Helper functions for serializing (and deserializing) requests. """ -import six - from scrapy.http import Request -from scrapy.utils.python import to_unicode, to_native_str +from scrapy.utils.python import to_unicode from scrapy.utils.misc import load_object @@ -54,7 +52,7 @@ def request_from_dict(d, spider=None): eb = _get_method(spider, eb) request_cls = load_object(d['_class']) if '_class' in d else Request return request_cls( - url=to_native_str(d['url']), + url=to_unicode(d['url']), callback=cb, errback=eb, method=d['method'], @@ -87,12 +85,12 @@ def _mangle_private_name(obj, func, name): def _find_method(obj, func): if obj: try: - func_self = six.get_method_self(func) + func_self = func.__self__ except AttributeError: # func has no __self__ pass else: if func_self is obj: - name = six.get_method_function(func).__name__ + name = func.__func__.__name__ if _is_private_method(name): return _mangle_private_name(obj, func, name) return name diff --git a/scrapy/utils/request.py b/scrapy/utils/request.py index 50bc3cb1e..356753ab5 100644 --- a/scrapy/utils/request.py +++ b/scrapy/utils/request.py @@ -3,20 +3,21 @@ This module provides some useful functions for working with scrapy.http.Request objects """ -from __future__ import print_function import hashlib import weakref -from six.moves.urllib.parse import urlunparse +from urllib.parse import urlunparse from w3lib.http import basic_auth_header -from scrapy.utils.python import to_bytes, to_native_str - from w3lib.url import canonicalize_url + from scrapy.utils.httpobj import urlparse_cached +from scrapy.utils.python import to_bytes, to_unicode _fingerprint_cache = weakref.WeakKeyDictionary() -def request_fingerprint(request, include_headers=None): + + +def request_fingerprint(request, include_headers=None, keep_fragments=False): """ Return the request fingerprint. @@ -30,7 +31,7 @@ def request_fingerprint(request, include_headers=None): and are equivalent (ie. they should return the same response). Another example are cookies used to store session ids. Suppose the - following page is only accesible to authenticated users: + following page is only accessible to authenticated users: http://www.example.com/members/offers.html @@ -42,15 +43,21 @@ def request_fingerprint(request, include_headers=None): the fingeprint. If you want to include specific headers use the include_headers argument, which is a list of Request headers to include. + Also, servers usually ignore fragments in urls when handling requests, + so they are also ignored by default when calculating the fingerprint. + If you want to include them, set the keep_fragments argument to True + (for instance when handling requests with a headless browser). + """ if include_headers: include_headers = tuple(to_bytes(h.lower()) for h in sorted(include_headers)) cache = _fingerprint_cache.setdefault(request, {}) - if include_headers not in cache: + cache_key = (include_headers, keep_fragments) + if cache_key not in cache: fp = hashlib.sha1() fp.update(to_bytes(request.method)) - fp.update(to_bytes(canonicalize_url(request.url))) + fp.update(to_bytes(canonicalize_url(request.url, keep_fragments=keep_fragments))) fp.update(request.body or b'') if include_headers: for hdr in include_headers: @@ -58,8 +65,8 @@ def request_fingerprint(request, include_headers=None): fp.update(hdr) for v in request.headers.getlist(hdr): fp.update(v) - cache[include_headers] = fp.hexdigest() - return cache[include_headers] + cache[cache_key] = fp.hexdigest() + return cache[cache_key] def request_authenticate(request, username, password): @@ -91,4 +98,4 @@ def referer_str(request): referrer = request.headers.get('Referer') if referrer is None: return referrer - return to_native_str(referrer, errors='replace') + return to_unicode(referrer, errors='replace') diff --git a/scrapy/utils/response.py b/scrapy/utils/response.py index c3236afd4..29fdaaf2c 100644 --- a/scrapy/utils/response.py +++ b/scrapy/utils/response.py @@ -8,11 +8,13 @@ import webbrowser import tempfile from twisted.web import http -from scrapy.utils.python import to_bytes, to_native_str +from scrapy.utils.python import to_bytes, to_unicode from w3lib import html _baseurl_cache = weakref.WeakKeyDictionary() + + def get_base_url(response): """Return the base url of the given response, joined with the response url""" if response not in _baseurl_cache: @@ -23,6 +25,8 @@ def get_base_url(response): _metaref_cache = weakref.WeakKeyDictionary() + + def get_meta_refresh(response, ignore_tags=('script', 'noscript')): """Parse the http-equiv refrsh parameter from the given response""" if response not in _metaref_cache: @@ -36,7 +40,7 @@ def response_status_message(status): """Return status code plus status text descriptive message """ message = http.RESPONSES.get(int(status), "Unknown Status") - return '%s %s' % (status, to_native_str(message)) + return '%s %s' % (status, to_unicode(message)) def response_httprepr(response): diff --git a/scrapy/utils/sitemap.py b/scrapy/utils/sitemap.py index 4742b3e13..2f10cf4de 100644 --- a/scrapy/utils/sitemap.py +++ b/scrapy/utils/sitemap.py @@ -5,8 +5,9 @@ Note: The main purpose of this module is to provide support for the SitemapSpider, its API is subject to change without notice. """ +from urllib.parse import urljoin + import lxml.etree -from six.moves.urllib.parse import urljoin class Sitemap(object): diff --git a/scrapy/utils/spider.py b/scrapy/utils/spider.py index 94b24f67e..4061d1ea3 100644 --- a/scrapy/utils/spider.py +++ b/scrapy/utils/spider.py @@ -1,11 +1,10 @@ import logging import inspect -import six - from scrapy.spiders import Spider from scrapy.utils.misc import arg_to_iter + logger = logging.getLogger(__name__) @@ -21,13 +20,14 @@ def iter_spider_classes(module): # singleton in scrapy.spider.spiders from scrapy.spiders import Spider - for obj in six.itervalues(vars(module)): + for obj in vars(module).values(): if inspect.isclass(obj) and \ issubclass(obj, Spider) and \ obj.__module__ == module.__name__ and \ getattr(obj, 'name', None): yield obj + def spidercls_for_request(spider_loader, request, default_spidercls=None, log_none=False, log_multiple=False): """Return a spider class that handles the given Request. diff --git a/scrapy/utils/ssl.py b/scrapy/utils/ssl.py new file mode 100644 index 000000000..6e81b33ff --- /dev/null +++ b/scrapy/utils/ssl.py @@ -0,0 +1,63 @@ +# -*- coding: utf-8 -*- + +import OpenSSL +import OpenSSL._util as pyOpenSSLutil + +from scrapy.utils.python import to_unicode + + +# The OpenSSL symbol is present since 1.1.1 but it's not currently supported in any version of pyOpenSSL. +# Using the binding directly, as this code does, requires cryptography 2.4. +SSL_OP_NO_TLSv1_3 = getattr(pyOpenSSLutil.lib, 'SSL_OP_NO_TLSv1_3', 0) + + +def ffi_buf_to_string(buf): + return to_unicode(pyOpenSSLutil.ffi.string(buf)) + + +def x509name_to_string(x509name): + # from OpenSSL.crypto.X509Name.__repr__ + result_buffer = pyOpenSSLutil.ffi.new("char[]", 512) + pyOpenSSLutil.lib.X509_NAME_oneline(x509name._name, result_buffer, len(result_buffer)) + + return ffi_buf_to_string(result_buffer) + + +def get_temp_key_info(ssl_object): + if not hasattr(pyOpenSSLutil.lib, 'SSL_get_server_tmp_key'): # requires OpenSSL 1.0.2 + return None + + # adapted from OpenSSL apps/s_cb.c::ssl_print_tmp_key() + temp_key_p = pyOpenSSLutil.ffi.new("EVP_PKEY **") + if not pyOpenSSLutil.lib.SSL_get_server_tmp_key(ssl_object, temp_key_p): + return None + temp_key = temp_key_p[0] + if temp_key == pyOpenSSLutil.ffi.NULL: + return None + temp_key = pyOpenSSLutil.ffi.gc(temp_key, pyOpenSSLutil.lib.EVP_PKEY_free) + key_info = [] + key_type = pyOpenSSLutil.lib.EVP_PKEY_id(temp_key) + if key_type == pyOpenSSLutil.lib.EVP_PKEY_RSA: + key_info.append('RSA') + elif key_type == pyOpenSSLutil.lib.EVP_PKEY_DH: + key_info.append('DH') + elif key_type == pyOpenSSLutil.lib.EVP_PKEY_EC: + key_info.append('ECDH') + ec_key = pyOpenSSLutil.lib.EVP_PKEY_get1_EC_KEY(temp_key) + ec_key = pyOpenSSLutil.ffi.gc(ec_key, pyOpenSSLutil.lib.EC_KEY_free) + nid = pyOpenSSLutil.lib.EC_GROUP_get_curve_name(pyOpenSSLutil.lib.EC_KEY_get0_group(ec_key)) + cname = pyOpenSSLutil.lib.EC_curve_nid2nist(nid) + if cname == pyOpenSSLutil.ffi.NULL: + cname = pyOpenSSLutil.lib.OBJ_nid2sn(nid) + key_info.append(ffi_buf_to_string(cname)) + else: + key_info.append(ffi_buf_to_string(pyOpenSSLutil.lib.OBJ_nid2sn(key_type))) + key_info.append('%s bits' % pyOpenSSLutil.lib.EVP_PKEY_bits(temp_key)) + return ', '.join(key_info) + + +def get_openssl_version(): + system_openssl = OpenSSL.SSL.SSLeay_version( + OpenSSL.SSL.SSLEAY_VERSION + ).decode('ascii', errors='replace') + return '{} ({})'.format(OpenSSL.version.__version__, system_openssl) diff --git a/scrapy/utils/template.py b/scrapy/utils/template.py index 615372fc8..96ff4b09b 100644 --- a/scrapy/utils/template.py +++ b/scrapy/utils/template.py @@ -19,6 +19,8 @@ def render_templatefile(path, **kwargs): CAMELCASE_INVALID_CHARS = re.compile(r'[^a-zA-Z\d]') + + def string_camelcase(string): """ Convert a word to its CamelCase version and remove invalid chars diff --git a/scrapy/utils/test.py b/scrapy/utils/test.py index 59467f105..fd8f41137 100644 --- a/scrapy/utils/test.py +++ b/scrapy/utils/test.py @@ -2,9 +2,10 @@ This module contains some assorted functions used in tests """ +import asyncio +import os from __future__ import absolute_import from posixpath import split -import os from importlib import import_module from twisted.trial.unittest import SkipTest @@ -33,6 +34,7 @@ def skip_if_no_boto(): except NotConfigured as e: raise SkipTest(e) + def get_s3_content_and_delete(bucket, path, with_key=False): """ Get content from s3 key, and delete key afterwards. """ @@ -52,6 +54,7 @@ def get_s3_content_and_delete(bucket, path, with_key=False): bucket.delete_key(path) return (content, key) if with_key else content + def get_gcs_content_and_delete(bucket, path): from google.cloud import storage client = storage.Client(project=os.environ.get('GCS_PROJECT_ID')) @@ -62,6 +65,7 @@ def get_gcs_content_and_delete(bucket, path): bucket.delete_blob(path) return content, acl, blob + def get_ftp_content_and_delete(path, host ,port, username, password, use_active_mode=False): from ftplib import FTP @@ -79,6 +83,7 @@ def get_ftp_content_and_delete(path, host ,port, ftp.delete(filename) return "".join(ftp_data) + def get_crawler(spidercls=None, settings_dict=None): """Return an unconfigured Crawler object. If settings_dict is given, it will be used to populate the crawler settings with a project level @@ -90,12 +95,14 @@ def get_crawler(spidercls=None, settings_dict=None): runner = CrawlerRunner(settings_dict) return runner.create_crawler(spidercls or Spider) + def get_pythonpath(): """Return a PYTHONPATH suitable to use in processes so that they find this installation of Scrapy""" scrapy_path = import_module('scrapy').__path__[0] return os.path.dirname(scrapy_path) + os.pathsep + os.environ.get('PYTHONPATH', '') + def get_testenv(): """Return a OS environment dict suitable to fork processes that need to import this installation of Scrapy, instead of a system installed one. @@ -104,8 +111,16 @@ def get_testenv(): env['PYTHONPATH'] = get_pythonpath() return env + def assert_samelines(testcase, text1, text2, msg=None): """Asserts text1 and text2 have the same lines, ignoring differences in line endings between platforms """ testcase.assertEqual(text1.splitlines(), text2.splitlines(), msg) + + +def get_from_asyncio_queue(value): + q = asyncio.Queue() + getter = q.get() + q.put_nowait(value) + return getter diff --git a/scrapy/utils/testproc.py b/scrapy/utils/testproc.py index f268e91ff..0f15cf60a 100644 --- a/scrapy/utils/testproc.py +++ b/scrapy/utils/testproc.py @@ -1,4 +1,3 @@ -from __future__ import absolute_import import sys import os diff --git a/scrapy/utils/testsite.py b/scrapy/utils/testsite.py index e50a989b3..6f5c21624 100644 --- a/scrapy/utils/testsite.py +++ b/scrapy/utils/testsite.py @@ -1,5 +1,4 @@ -from __future__ import print_function -from six.moves.urllib.parse import urljoin +from urllib.parse import urljoin from twisted.internet import reactor from twisted.web import server, resource, static, util diff --git a/scrapy/utils/trackref.py b/scrapy/utils/trackref.py index eed14c5a1..4842b95df 100644 --- a/scrapy/utils/trackref.py +++ b/scrapy/utils/trackref.py @@ -9,12 +9,10 @@ and no performance penalty at all when disabled (as object_ref becomes just an alias to object in that case). """ -from __future__ import print_function import weakref from time import time from operator import itemgetter from collections import defaultdict -import six NoneType = type(None) @@ -37,13 +35,13 @@ def format_live_refs(ignore=NoneType): """Return a tabular representation of tracked objects""" s = "Live References\n\n" now = time() - for cls, wdict in sorted(six.iteritems(live_refs), + for cls, wdict in sorted(live_refs.items(), key=lambda x: x[0].__name__): if not wdict: continue if issubclass(cls, ignore): continue - oldest = min(six.itervalues(wdict)) + oldest = min(wdict.values()) s += "%-30s %6d oldest: %ds ago\n" % ( cls.__name__, len(wdict), now - oldest ) @@ -57,15 +55,15 @@ def print_live_refs(*a, **kw): def get_oldest(class_name): """Get the oldest object for a specific class name""" - for cls, wdict in six.iteritems(live_refs): + for cls, wdict in live_refs.items(): if cls.__name__ == class_name: if not wdict: break - return min(six.iteritems(wdict), key=itemgetter(1))[0] + return min(wdict.items(), key=itemgetter(1))[0] def iter_all(class_name): """Iterate over all objects of the same class by its class name""" - for cls, wdict in six.iteritems(live_refs): + for cls, wdict in live_refs.items(): if cls.__name__ == class_name: - return six.iterkeys(wdict) + return wdict.keys() diff --git a/scrapy/utils/url.py b/scrapy/utils/url.py index b3a4be007..c9abb12d5 100644 --- a/scrapy/utils/url.py +++ b/scrapy/utils/url.py @@ -7,12 +7,12 @@ to the w3lib.url module. Always import those from there instead. """ import posixpath import re -from six.moves.urllib.parse import (ParseResult, urldefrag, urlparse, urlunparse) +from urllib.parse import ParseResult, urldefrag, urlparse, urlunparse # scrapy.utils.url was moved to w3lib.url and import * ensures this # move doesn't break old code from w3lib.url import * -from w3lib.url import _safe_chars, _unquotepath +from w3lib.url import _safe_chars, _unquotepath # noqa: F401 from scrapy.utils.python import to_unicode diff --git a/scrapy/utils/versions.py b/scrapy/utils/versions.py index 58c7aef85..b0737d3d5 100644 --- a/scrapy/utils/versions.py +++ b/scrapy/utils/versions.py @@ -1,6 +1,7 @@ import platform import sys +import cryptography import cssselect import lxml.etree import parsel @@ -8,20 +9,12 @@ import twisted import w3lib import scrapy +from scrapy.utils.ssl import get_openssl_version def scrapy_components_versions(): lxml_version = ".".join(map(str, lxml.etree.LXML_VERSION)) libxml2_version = ".".join(map(str, lxml.etree.LIBXML_VERSION)) - try: - w3lib_version = w3lib.__version__ - except AttributeError: - w3lib_version = "<1.14.3" - try: - import cryptography - cryptography_version = cryptography.__version__ - except ImportError: - cryptography_version = "unknown" return [ ("Scrapy", scrapy.__version__), @@ -29,22 +22,10 @@ def scrapy_components_versions(): ("libxml2", libxml2_version), ("cssselect", cssselect.__version__), ("parsel", parsel.__version__), - ("w3lib", w3lib_version), + ("w3lib", w3lib.__version__), ("Twisted", twisted.version.short()), ("Python", sys.version.replace("\n", "- ")), - ("pyOpenSSL", _get_openssl_version()), - ("cryptography", cryptography_version), - ("Platform", platform.platform()), + ("pyOpenSSL", get_openssl_version()), + ("cryptography", cryptography.__version__), + ("Platform", platform.platform()), ] - - -def _get_openssl_version(): - try: - import OpenSSL - openssl = OpenSSL.SSL.SSLeay_version(OpenSSL.SSL.SSLEAY_VERSION)\ - .decode('ascii', errors='replace') - # pyOpenSSL 0.12 does not expose openssl version - except AttributeError: - openssl = 'Unknown OpenSSL version' - - return '{} ({})'.format(OpenSSL.version.__version__, openssl) diff --git a/scrapy/xlib/__init__.py b/scrapy/xlib/__init__.py deleted file mode 100644 index 11f022087..000000000 --- a/scrapy/xlib/__init__.py +++ /dev/null @@ -1,2 +0,0 @@ -"""This package contains some third party modules that are distributed along -with Scrapy""" diff --git a/scrapy/xlib/pydispatch.py b/scrapy/xlib/pydispatch.py deleted file mode 100644 index 5ffeaf579..000000000 --- a/scrapy/xlib/pydispatch.py +++ /dev/null @@ -1,19 +0,0 @@ -from __future__ import absolute_import - -import warnings -from scrapy.exceptions import ScrapyDeprecationWarning - -from pydispatch import ( - dispatcher, - errors, - robust, - robustapply, - saferef, -) - -warnings.warn("Importing from scrapy.xlib.pydispatch is deprecated and will" - " no longer be supported in future Scrapy versions." - " If you just want to connect signals use the from_crawler class method," - " otherwise import pydispatch directly if needed." - " See: https://github.com/scrapy/scrapy/issues/1762", - ScrapyDeprecationWarning, stacklevel=2) diff --git a/scrapy/xlib/tx.py b/scrapy/xlib/tx.py deleted file mode 100644 index 0d94307b7..000000000 --- a/scrapy/xlib/tx.py +++ /dev/null @@ -1,19 +0,0 @@ -from __future__ import absolute_import - -import warnings -from scrapy.exceptions import ScrapyDeprecationWarning - -from twisted.web import client -from twisted.internet import endpoints - -Agent = client.Agent # since < 11.1 -ProxyAgent = client.ProxyAgent # since 11.1 -ResponseDone = client.ResponseDone # since 11.1 -ResponseFailed = client.ResponseFailed # since 11.1 -HTTPConnectionPool = client.HTTPConnectionPool # since 12.1 -TCP4ClientEndpoint = endpoints.TCP4ClientEndpoint # since 10.1 - -warnings.warn("Importing from scrapy.xlib.tx is deprecated and will" - " no longer be supported in future Scrapy versions." - " Update your code to import from twisted proper.", - ScrapyDeprecationWarning, stacklevel=2) diff --git a/sep/sep-001.rst b/sep/sep-001.rst index 2a66f9802..00226283f 100644 --- a/sep/sep-001.rst +++ b/sep/sep-001.rst @@ -254,8 +254,8 @@ ItemForm #!python class MySiteForm(ItemForm): - witdth = adaptor(ItemForm.witdh, default_unit='cm') - volume = adaptor(ItemForm.witdh, default_unit='lt') + width = adaptor(ItemForm.width, default_unit='cm') + volume = adaptor(ItemForm.width, default_unit='lt') ia['width'] = x.x('//p[@class="width"]') ia['volume'] = x.x('//p[@class="volume"]') diff --git a/sep/sep-004.rst b/sep/sep-004.rst index 69edfa136..05b0eb99c 100644 --- a/sep/sep-004.rst +++ b/sep/sep-004.rst @@ -53,7 +53,7 @@ Here's a simple proof-of-concept code of such script: # ... do something more interesting with scraped_items ... The behaviour of the Scrapy crawler would be controller by the Scrapy settings, -naturally, just like any typical scrapy project. But the default settings +naturally, just like any typical Scrapy project. But the default settings should be sufficient so as to not require adding any specific setting. But, at the same time, you could do it if you need to, say, for specifying a custom middleware. diff --git a/sep/sep-009.rst b/sep/sep-009.rst index 232a536a8..da87fa9aa 100644 --- a/sep/sep-009.rst +++ b/sep/sep-009.rst @@ -38,7 +38,7 @@ singletons members of that object, as explained below: ``scrapy.core.manager.ExecutionManager``) - instantiated with a ``Settings`` object - - **crawler.settings**: ``scrapy.conf.Settings`` instance (passed in the constructor) + - **crawler.settings**: ``scrapy.conf.Settings`` instance (passed in the ``__init__`` method) - **crawler.extensions**: ``scrapy.extension.ExtensionManager`` instance - **crawler.engine**: ``scrapy.core.engine.ExecutionEngine`` instance - ``crawler.engine.scheduler`` @@ -55,7 +55,7 @@ singletons members of that object, as explained below: ``STATS_CLASS`` setting) - **crawler.log**: Logger class with methods replacing the current ``scrapy.log`` functions. Logging would be started (if enabled) on - ``Crawler`` constructor, so no log starting functions are required. + ``Crawler`` instantiation, so no log starting functions are required. - ``crawler.log.msg`` - **crawler.signals**: signal handling @@ -69,12 +69,12 @@ Required code changes after singletons removal ============================================== All components (extensions, middlewares, etc) will receive this ``Crawler`` -object in their constructors, and this will be the only mechanism for accessing +object in their ``__init__`` methods, and this will be the only mechanism for accessing any other components (as opposed to importing each singleton from their respective module). This will also serve to stabilize the core API, something which we haven't documented so far (partly because of this). -So, for a typical middleware constructor code, instead of this: +So, for a typical middleware ``__init__`` method code, instead of this: :: @@ -125,13 +125,13 @@ Open issues to resolve - Should we pass ``Settings`` object to ``ScrapyCommand.add_options()``? - How should spiders access settings? - - Option 1. Pass ``Crawler`` object to spider constructors too + - Option 1. Pass ``Crawler`` object to spider ``__init__`` methods too - pro: one way to access all components (settings and signals being the most relevant to spiders) - con?: spider code can access (and control) any crawler component - since we don't want to support spiders messing with the crawler (write an extension or spider middleware if you need that) - - Option 2. Pass ``Settings`` object to spider constructors, which would + - Option 2. Pass ``Settings`` object to spider ``__init__`` methods, which would then be accessed through ``self.settings``, like logging which is accessed through ``self.log`` diff --git a/sep/sep-019.rst b/sep/sep-019.rst index 9fbf6a223..84f3a96c3 100644 --- a/sep/sep-019.rst +++ b/sep/sep-019.rst @@ -185,7 +185,7 @@ These ideas translate to the following changes on the ``SpiderManager`` class: will return a spider class, not an instance. It's basically a ``__get__`` to ``self._spiders``. -- All remaining functions should be deprecated or remove accordantly, since a +- All remaining functions should be deprecated or remove accordingly, since a crawler reference is no longer needed. - New helper ``get_spider_manager_class_from_scrapycfg`` in diff --git a/setup.py b/setup.py index 4dc6d18c1..85d797f88 100644 --- a/setup.py +++ b/setup.py @@ -50,32 +50,31 @@ setup( 'License :: OSI Approved :: BSD License', 'Operating System :: OS Independent', 'Programming Language :: Python', - 'Programming Language :: Python :: 2', - 'Programming Language :: Python :: 2.7', 'Programming Language :: Python :: 3', - 'Programming Language :: Python :: 3.4', 'Programming Language :: Python :: 3.5', 'Programming Language :: Python :: 3.6', 'Programming Language :: Python :: 3.7', + 'Programming Language :: Python :: 3.8', 'Programming Language :: Python :: Implementation :: CPython', 'Programming Language :: Python :: Implementation :: PyPy', 'Topic :: Internet :: WWW/HTTP', 'Topic :: Software Development :: Libraries :: Application Frameworks', 'Topic :: Software Development :: Libraries :: Python Modules', ], - python_requires='>=2.7, !=3.0.*, !=3.1.*, !=3.2.*, !=3.3.*', + python_requires='>=3.5', install_requires=[ - 'Twisted>=13.1.0;python_version!="3.4"', - 'Twisted>=13.1.0,<=19.2.0;python_version=="3.4"', - 'w3lib>=1.17.0', - 'queuelib', - 'lxml', - 'pyOpenSSL', - 'cssselect>=0.9', - 'six>=1.5.2', - 'parsel>=1.5', + 'Twisted>=17.9.0', + 'cryptography>=2.0', + 'cssselect>=0.9.1', + 'lxml>=3.5.0', + 'parsel>=1.5.0', 'PyDispatcher>=2.0.5', - 'service_identity', + 'pyOpenSSL>=16.2.0', + 'queuelib>=1.4.2', + 'service_identity>=16.0.0', + 'w3lib>=1.17.0', + 'zope.interface>=4.1.3', + 'protego>=0.1.15', ], extras_require=extras_require, ) diff --git a/tests/CrawlerProcess/asyncio_enabled_no_reactor.py b/tests/CrawlerProcess/asyncio_enabled_no_reactor.py new file mode 100644 index 000000000..db1b75931 --- /dev/null +++ b/tests/CrawlerProcess/asyncio_enabled_no_reactor.py @@ -0,0 +1,17 @@ +import scrapy +from scrapy.crawler import CrawlerProcess + + +class NoRequestsSpider(scrapy.Spider): + name = 'no_request' + + def start_requests(self): + return [] + + +process = CrawlerProcess(settings={ + 'ASYNCIO_REACTOR': True, +}) + +process.crawl(NoRequestsSpider) +process.start() diff --git a/tests/CrawlerProcess/asyncio_enabled_reactor.py b/tests/CrawlerProcess/asyncio_enabled_reactor.py new file mode 100644 index 000000000..cec3c9c25 --- /dev/null +++ b/tests/CrawlerProcess/asyncio_enabled_reactor.py @@ -0,0 +1,22 @@ +import asyncio + +from twisted.internet import asyncioreactor +asyncioreactor.install(asyncio.get_event_loop()) + +import scrapy +from scrapy.crawler import CrawlerProcess + + +class NoRequestsSpider(scrapy.Spider): + name = 'no_request' + + def start_requests(self): + return [] + + +process = CrawlerProcess(settings={ + 'ASYNCIO_REACTOR': True, +}) + +process.crawl(NoRequestsSpider) +process.start() diff --git a/tests/CrawlerProcess/simple.py b/tests/CrawlerProcess/simple.py new file mode 100644 index 000000000..5f6f1ae30 --- /dev/null +++ b/tests/CrawlerProcess/simple.py @@ -0,0 +1,15 @@ +import scrapy +from scrapy.crawler import CrawlerProcess + + +class NoRequestsSpider(scrapy.Spider): + name = 'no_request' + + def start_requests(self): + return [] + + +process = CrawlerProcess(settings={}) + +process.crawl(NoRequestsSpider) +process.start() diff --git a/tests/__init__.py b/tests/__init__.py index 9c9e35c35..12ce79fa9 100644 --- a/tests/__init__.py +++ b/tests/__init__.py @@ -21,11 +21,6 @@ if 'COV_CORE_CONFIG' in os.environ: os.environ['COV_CORE_CONFIG'] = os.path.join(_sourceroot, os.environ['COV_CORE_CONFIG']) -try: - import unittest.mock as mock -except ImportError: - import mock - tests_datadir = os.path.join(os.path.abspath(os.path.dirname(__file__)), 'sample_data') @@ -35,18 +30,3 @@ def get_testdata(*paths): path = os.path.join(tests_datadir, *paths) with open(path, 'rb') as f: return f.read() - - -# FIXME: delete after dropping py2 support -# Monkey patch the unittest module to prevent the -# DeprecationWarning about assertRaisesRegexp -> assertRaisesRegex -import six -if six.PY2: - import unittest - import twisted.trial.unittest - if not getattr(unittest.TestCase, 'assertRegex', None): - unittest.TestCase.assertRegex = unittest.TestCase.assertRegexpMatches - if not getattr(unittest.TestCase, 'assertRaisesRegex', None): - unittest.TestCase.assertRaisesRegex = unittest.TestCase.assertRaisesRegexp - if not getattr(twisted.trial.unittest.TestCase, 'assertRaisesRegex', None): - twisted.trial.unittest.TestCase.assertRaisesRegex = twisted.trial.unittest.TestCase.assertRaisesRegexp diff --git a/tests/constraints.txt b/tests/constraints.txt index e59e68b3f..5655ac2d3 100644 --- a/tests/constraints.txt +++ b/tests/constraints.txt @@ -1,2 +1 @@ -Twisted!=18.4.0 -lxml!=4.2.2 \ No newline at end of file +Twisted!=18.4.0 \ No newline at end of file diff --git a/tests/py3-ignores.txt b/tests/ignores.txt similarity index 74% rename from tests/py3-ignores.txt rename to tests/ignores.txt index 313e74ec9..45cf6fb92 100644 --- a/tests/py3-ignores.txt +++ b/tests/ignores.txt @@ -1,6 +1,3 @@ -tests/test_linkextractors_deprecated.py -tests/test_proxy_connect.py - scrapy/linkextractors/sgml.py scrapy/linkextractors/regex.py scrapy/linkextractors/htmlparser.py diff --git a/tests/mocks/dummydbm.py b/tests/mocks/dummydbm.py index 431428331..75c74daf5 100644 --- a/tests/mocks/dummydbm.py +++ b/tests/mocks/dummydbm.py @@ -13,6 +13,7 @@ error = KeyError _DATABASES = collections.defaultdict(DummyDB) + def open(file, flag='r', mode=0o666): """Open or create a dummy database compatible. diff --git a/tests/mockserver.py b/tests/mockserver.py index 3fa4bc0f0..a45277db9 100644 --- a/tests/mockserver.py +++ b/tests/mockserver.py @@ -1,8 +1,11 @@ -from __future__ import print_function -import sys, time, random, os, json -from six.moves.urllib.parse import urlencode +import json +import os +import random +import sys from subprocess import Popen, PIPE +from urllib.parse import urlencode +from OpenSSL import SSL from twisted.web.server import Site, NOT_DONE_YET from twisted.web.resource import Resource from twisted.web.static import File @@ -13,8 +16,8 @@ from twisted.web.util import redirectTo from twisted.internet import reactor, ssl from twisted.internet.task import deferLater - from scrapy.utils.python import to_bytes, to_unicode +from scrapy.utils.ssl import SSL_OP_NO_TLSv1_3 def getarg(request, name, default=None, type=None): @@ -161,6 +164,12 @@ class Drop(Partial): request.finish() +class ArbitraryLengthPayloadResource(LeafResource): + + def render(self, request): + return request.content.read() + + class Root(Resource): def __init__(self): @@ -174,6 +183,7 @@ class Root(Resource): self.putChild(b"echo", Echo()) self.putChild(b"payload", PayloadResource()) self.putChild(b"xpayload", EncodingResourceWrapper(PayloadResource(), [GzipEncoderFactory()])) + self.putChild(b"alpayload", ArbitraryLengthPayloadResource()) try: from tests import tests_datadir self.putChild(b"files", File(os.path.join(tests_datadir, 'test_site/files/'))) @@ -205,8 +215,7 @@ class MockServer(): def __exit__(self, exc_type, exc_value, traceback): self.proc.kill() - self.proc.wait() - time.sleep(0.2) + self.proc.communicate() def url(self, path, is_secure=False): host = self.http_address.replace('0.0.0.0', '127.0.0.1') @@ -215,11 +224,17 @@ class MockServer(): return host + path -def ssl_context_factory(keyfile='keys/localhost.key', certfile='keys/localhost.crt'): - return ssl.DefaultOpenSSLContextFactory( +def ssl_context_factory(keyfile='keys/localhost.key', certfile='keys/localhost.crt', cipher_string=None): + factory = ssl.DefaultOpenSSLContextFactory( os.path.join(os.path.dirname(__file__), keyfile), os.path.join(os.path.dirname(__file__), certfile), ) + if cipher_string: + ctx = factory.getContext() + # disabling TLS1.2+ because it unconditionally enables some strong ciphers + ctx.set_options(SSL.OP_CIPHER_SERVER_PREFERENCE | SSL.OP_NO_TLSv1_2 | SSL_OP_NO_TLSv1_3) + ctx.set_cipher_list(to_bytes(cipher_string)) + return factory if __name__ == "__main__": diff --git a/tests/pipelines.py b/tests/pipelines.py index 7e2895a5c..d7d3b5259 100644 --- a/tests/pipelines.py +++ b/tests/pipelines.py @@ -2,6 +2,7 @@ Some pipelines used for testing """ + class ZeroDivisionErrorPipeline(object): def open_spider(self, spider): diff --git a/tests/requirements-py2.txt b/tests/requirements-py2.txt deleted file mode 100644 index be809b151..000000000 --- a/tests/requirements-py2.txt +++ /dev/null @@ -1,14 +0,0 @@ -# Tests requirements -mock -mitmproxy==0.10.1 -netlib==0.10.1 -pytest -pytest-cov -pytest-twisted -pytest-xdist -jmespath -brotlipy -testfixtures -# optional for shell wrapper tests -bpython -ipython<6.0 diff --git a/tests/requirements-py3.txt b/tests/requirements-py3.txt index ed7bf0be0..d97c4b8ee 100644 --- a/tests/requirements-py3.txt +++ b/tests/requirements-py3.txt @@ -1,13 +1,16 @@ +# Tests requirements +jmespath +mitmproxy; python_version >= '3.6' +mitmproxy<4.0.0; python_version < '3.6' pytest pytest-cov -pytest-twisted +pytest-twisted >= 1.11 pytest-xdist +sybil testfixtures -jmespath -leveldb; sys_platform != "win32" -botocore + # optional for shell wrapper tests bpython -ipython brotlipy +ipython pywin32; sys_platform == "win32" diff --git a/tests/sample_data/link_extractor/sgml_linkextractor.html b/tests/sample_data/link_extractor/linkextractor.html similarity index 100% rename from tests/sample_data/link_extractor/sgml_linkextractor.html rename to tests/sample_data/link_extractor/linkextractor.html diff --git a/tests/sample_data/link_extractor/linkextractor_no_href.html b/tests/sample_data/link_extractor/linkextractor_no_href.html new file mode 100644 index 000000000..0b01cede8 --- /dev/null +++ b/tests/sample_data/link_extractor/linkextractor_no_href.html @@ -0,0 +1,25 @@ +<html> + <head> + <base href='http://example.com' /> + <title>Sample page with anchor tags containing no href attribute, to test the TextResponse.follow_all method + + + +
+ + + \ No newline at end of file diff --git a/tests/spiders.py b/tests/spiders.py index 7816bf7c7..39c8da0b6 100644 --- a/tests/spiders.py +++ b/tests/spiders.py @@ -1,14 +1,14 @@ """ Some spiders used for testing and benchmarking """ - import time -from six.moves.urllib.parse import urlencode +from urllib.parse import urlencode -from scrapy.spiders import Spider from scrapy.http import Request from scrapy.item import Item from scrapy.linkextractors import LinkExtractor +from scrapy.spiders import Spider +from scrapy.spiders.crawl import CrawlSpider, Rule class MockServerSpider(Spider): @@ -16,6 +16,7 @@ class MockServerSpider(Spider): super(MockServerSpider, self).__init__(*args, **kwargs) self.mockserver = mockserver + class MetaSpider(MockServerSpider): name = 'meta' @@ -183,3 +184,35 @@ class DuplicateStartRequestsSpider(MockServerSpider): def parse(self, response): self.visited += 1 + + +class CrawlSpiderWithErrback(MockServerSpider, CrawlSpider): + name = 'crawl_spider_with_errback' + custom_settings = { + 'RETRY_HTTP_CODES': [], # no need to retry + } + rules = ( + Rule(LinkExtractor(), callback='callback', errback='errback', follow=True), + ) + + def start_requests(self): + test_body = b""" + + Page title<title></head> + <body> + <p><a href="/status?n=200">Item 200</a></p> <!-- callback --> + <p><a href="/status?n=201">Item 201</a></p> <!-- callback --> + <p><a href="/status?n=404">Item 404</a></p> <!-- errback --> + <p><a href="/status?n=500">Item 500</a></p> <!-- errback --> + <p><a href="/status?n=501">Item 501</a></p> <!-- errback --> + </body> + </html> + """ + url = self.mockserver.url("/alpayload") + yield Request(url, method="POST", body=test_body) + + def callback(self, response): + self.logger.info('[callback] status %i', response.status) + + def errback(self, failure): + self.logger.info('[errback] status %i', failure.value.response.status) diff --git a/tests/test_cmdline/__init__.py b/tests/test_cmdline/__init__.py index 68dfb1cca..da99a6be8 100644 --- a/tests/test_cmdline/__init__.py +++ b/tests/test_cmdline/__init__.py @@ -2,15 +2,11 @@ import json import os import pstats import shutil -import six -from subprocess import Popen, PIPE import sys import tempfile import unittest -try: - from cStringIO import StringIO -except ImportError: - from io import StringIO +from io import StringIO +from subprocess import Popen, PIPE from scrapy.utils.test import get_testenv @@ -29,17 +25,15 @@ class CmdlineTest(unittest.TestCase): return comm.decode(encoding) def test_default_settings(self): - self.assertEqual(self._execute('settings', '--get', 'TEST1'), \ - 'default') + self.assertEqual(self._execute('settings', '--get', 'TEST1'), 'default') def test_override_settings_using_set_arg(self): - self.assertEqual(self._execute('settings', '--get', 'TEST1', '-s', 'TEST1=override'), \ - 'override') + self.assertEqual(self._execute('settings', '--get', 'TEST1', '-s', + 'TEST1=override'), 'override') def test_override_settings_using_envvar(self): self.env['SCRAPY_TEST1'] = 'override' - self.assertEqual(self._execute('settings', '--get', 'TEST1'), \ - 'override') + self.assertEqual(self._execute('settings', '--get', 'TEST1'), 'override') def test_profiling(self): path = tempfile.mkdtemp() @@ -68,5 +62,5 @@ class CmdlineTest(unittest.TestCase): for char in ("'", "<", ">", 'u"'): settingsstr = settingsstr.replace(char, '"') settingsdict = json.loads(settingsstr) - six.assertCountEqual(self, settingsdict.keys(), EXTENSIONS.keys()) + self.assertCountEqual(settingsdict.keys(), EXTENSIONS.keys()) self.assertEqual(200, settingsdict[EXT_PATH]) diff --git a/tests/test_cmdline/extensions.py b/tests/test_cmdline/extensions.py index 72867eb56..c64e87d81 100644 --- a/tests/test_cmdline/extensions.py +++ b/tests/test_cmdline/extensions.py @@ -1,5 +1,6 @@ """A test extension used to check the settings loading order""" + class TestExtension(object): def __init__(self, settings): @@ -12,4 +13,3 @@ class TestExtension(object): class DummyExtension(object): pass - diff --git a/tests/test_command_fetch.py b/tests/test_command_fetch.py index 3fa3ed930..9d3c8fe73 100644 --- a/tests/test_command_fetch.py +++ b/tests/test_command_fetch.py @@ -29,6 +29,6 @@ class FetchTest(ProcessTest, SiteTest, unittest.TestCase): @defer.inlineCallbacks def test_headers(self): _, out, _ = yield self.execute([self.url('/text'), '--headers']) - out = out.replace(b'\r', b'') # required on win32 + out = out.replace(b'\r', b'') # required on win32 assert b'Server: TwistedWeb' in out, out assert b'Content-Type: text/plain' in out diff --git a/tests/test_command_parse.py b/tests/test_command_parse.py index 98e415ad3..b7035fdff 100644 --- a/tests/test_command_parse.py +++ b/tests/test_command_parse.py @@ -1,17 +1,17 @@ import os from os.path import join, abspath -from twisted.trial import unittest from twisted.internet import defer from scrapy.utils.testsite import SiteTest from scrapy.utils.testproc import ProcessTest -from scrapy.utils.python import to_native_str +from scrapy.utils.python import to_unicode from tests.test_commands import CommandTest def _textmode(bstr): """Normalize input the same as writing to a file and reading from it in text mode""" - return to_native_str(bstr).replace(os.linesep, '\n') + return to_unicode(bstr).replace(os.linesep, '\n') + class ParseCommandTest(ProcessTest, SiteTest, CommandTest): command = 'parse' @@ -108,6 +108,7 @@ ITEM_PIPELINES = {'%s.pipelines.MyPipeline': 1} _, _, stderr = yield self.execute(['--spider', self.spider_name, '-a', 'test_arg=1', '-c', 'parse', + '--verbose', self.url('/html')]) self.assertIn("DEBUG: It Works!", _textmode(stderr)) @@ -117,12 +118,14 @@ ITEM_PIPELINES = {'%s.pipelines.MyPipeline': 1} _, _, stderr = yield self.execute(['--spider', self.spider_name, '--meta', raw_json_string, '-c', 'parse_request_with_meta', + '--verbose', self.url('/html')]) self.assertIn("DEBUG: It Works!", _textmode(stderr)) _, _, stderr = yield self.execute(['--spider', self.spider_name, '-m', raw_json_string, '-c', 'parse_request_with_meta', + '--verbose', self.url('/html')]) self.assertIn("DEBUG: It Works!", _textmode(stderr)) @@ -132,6 +135,7 @@ ITEM_PIPELINES = {'%s.pipelines.MyPipeline': 1} _, _, stderr = yield self.execute(['--spider', self.spider_name, '--cbkwargs', raw_json_string, '-c', 'parse_request_with_cb_kwargs', + '--verbose', self.url('/html')]) self.assertIn("DEBUG: It Works!", _textmode(stderr)) @@ -139,6 +143,7 @@ ITEM_PIPELINES = {'%s.pipelines.MyPipeline': 1} def test_request_without_meta(self): _, _, stderr = yield self.execute(['--spider', self.spider_name, '-c', 'parse_request_without_meta', + '--nolinks', self.url('/html')]) self.assertIn("DEBUG: It Works!", _textmode(stderr)) @@ -148,6 +153,7 @@ ITEM_PIPELINES = {'%s.pipelines.MyPipeline': 1} _, _, stderr = yield self.execute(['--spider', self.spider_name, '--pipelines', '-c', 'parse', + '--verbose', self.url('/html')]) self.assertIn("INFO: It Works!", _textmode(stderr)) diff --git a/tests/test_commands.py b/tests/test_commands.py index b8445ae6c..6024af71c 100644 --- a/tests/test_commands.py +++ b/tests/test_commands.py @@ -10,13 +10,10 @@ from contextlib import contextmanager from threading import Timer from twisted.trial import unittest -from twisted.internet import defer import scrapy -from scrapy.utils.python import to_native_str +from scrapy.utils.python import to_unicode from scrapy.utils.test import get_testenv -from scrapy.utils.testsite import SiteTest -from scrapy.utils.testproc import ProcessTest from tests.test_crawler import ExceptionSpider, NoRequestsSpider @@ -56,7 +53,7 @@ class ProjectTest(unittest.TestCase): finally: timer.cancel() - return p, to_native_str(stdout), to_native_str(stderr) + return p, to_unicode(stdout), to_unicode(stderr) class StartprojectTest(ProjectTest): @@ -298,6 +295,14 @@ class BadSpider(scrapy.Spider): self.assertIn("start_requests", log) self.assertIn("badspider.py", log) + def test_asyncio_enabled_true(self): + log = self.get_log(self.debug_log_spider, args=['-s', 'ASYNCIO_REACTOR=True']) + self.assertIn("DEBUG: Asyncio reactor is installed", log) + + def test_asyncio_enabled_false(self): + log = self.get_log(self.debug_log_spider, args=['-s', 'ASYNCIO_REACTOR=False']) + self.assertNotIn("DEBUG: Asyncio reactor is installed", log) + class BenchCommandTest(CommandTest): diff --git a/tests/test_contracts.py b/tests/test_contracts.py index a06bb2cc3..11d41c1fe 100644 --- a/tests/test_contracts.py +++ b/tests/test_contracts.py @@ -1,6 +1,5 @@ from unittest import TextTestResult -from six import get_unbound_function from twisted.internet import defer from twisted.python import failure from twisted.trial import unittest @@ -14,6 +13,7 @@ from scrapy.item import Item, Field from scrapy.contracts import ContractsManager, Contract from scrapy.contracts.default import ( UrlContract, + CallbackKeywordArgumentsContract, ReturnsContract, ScrapesContract, ) @@ -70,6 +70,37 @@ class TestSpider(Spider): """ return TestItem(url=response.url) + def returns_request_cb_kwargs(self, response, url): + """ method which returns request + @url https://example.org + @cb_kwargs {"url": "http://scrapy.org"} + @returns requests 1 + """ + return Request(url, callback=self.returns_item_cb_kwargs) + + def returns_item_cb_kwargs(self, response, name): + """ method which returns item + @url http://scrapy.org + @cb_kwargs {"name": "Scrapy"} + @returns items 1 1 + """ + return TestItem(name=name, url=response.url) + + def returns_item_cb_kwargs_error_unexpected_keyword(self, response): + """ method which returns item + @url http://scrapy.org + @cb_kwargs {"arg": "value"} + @returns items 1 1 + """ + return TestItem(url=response.url) + + def returns_item_cb_kwargs_error_missing_argument(self, response, arg): + """ method which returns item + @url http://scrapy.org + @returns items 1 1 + """ + return TestItem(url=response.url) + def returns_dict_item(self, response): """ method which returns item @url http://scrapy.org @@ -123,6 +154,14 @@ class TestSpider(Spider): """ return {'url': response.url} + def scrapes_multiple_missing_fields(self, response): + """ returns item with no name + @url http://scrapy.org + @returns items 1 1 + @scrapes name url + """ + return {} + def parse_no_url(self, response): """ method with no url @returns items 1 1 @@ -164,6 +203,7 @@ class InheritsTestSpider(TestSpider): class ContractsManagerTest(unittest.TestCase): contracts = [ UrlContract, + CallbackKeywordArgumentsContract, ReturnsContract, ScrapesContract, CustomFormContract, @@ -203,6 +243,51 @@ class ContractsManagerTest(unittest.TestCase): request = self.conman.from_method(spider.parse_no_url, self.results) self.assertEqual(request, None) + def test_cb_kwargs(self): + spider = TestSpider() + response = ResponseMock() + + # extract contracts correctly + contracts = self.conman.extract_contracts(spider.returns_request_cb_kwargs) + self.assertEqual(len(contracts), 3) + self.assertEqual(frozenset(type(x) for x in contracts), + frozenset([UrlContract, CallbackKeywordArgumentsContract, ReturnsContract])) + + contracts = self.conman.extract_contracts(spider.returns_item_cb_kwargs) + self.assertEqual(len(contracts), 3) + self.assertEqual(frozenset(type(x) for x in contracts), + frozenset([UrlContract, CallbackKeywordArgumentsContract, ReturnsContract])) + + contracts = self.conman.extract_contracts(spider.returns_item_cb_kwargs_error_unexpected_keyword) + self.assertEqual(len(contracts), 3) + self.assertEqual(frozenset(type(x) for x in contracts), + frozenset([UrlContract, CallbackKeywordArgumentsContract, ReturnsContract])) + + contracts = self.conman.extract_contracts(spider.returns_item_cb_kwargs_error_missing_argument) + self.assertEqual(len(contracts), 2) + self.assertEqual(frozenset(type(x) for x in contracts), + frozenset([UrlContract, ReturnsContract])) + + # returns_request + request = self.conman.from_method(spider.returns_request_cb_kwargs, self.results) + request.callback(response, **request.cb_kwargs) + self.should_succeed() + + # returns_item + request = self.conman.from_method(spider.returns_item_cb_kwargs, self.results) + request.callback(response, **request.cb_kwargs) + self.should_succeed() + + # returns_item (error, callback doesn't take keyword arguments) + request = self.conman.from_method(spider.returns_item_cb_kwargs_error_unexpected_keyword, self.results) + request.callback(response, **request.cb_kwargs) + self.should_error() + + # returns_item (error, contract doesn't provide keyword arguments) + request = self.conman.from_method(spider.returns_item_cb_kwargs_error_missing_argument, self.results) + request.callback(response, **request.cb_kwargs) + self.should_error() + def test_returns(self): spider = TestSpider() response = ResponseMock() @@ -256,6 +341,13 @@ class ContractsManagerTest(unittest.TestCase): request.callback(response) self.should_fail() + # scrapes_multiple_missing_fields + request = self.conman.from_method(spider.scrapes_multiple_missing_fields, self.results) + request.callback(response) + self.should_fail() + message = 'ContractFail: Missing fields: name, url' + assert message in self.results.failures[-1][-1] + def test_custom_contracts(self): self.conman.from_spider(CustomContractSuccessSpider(), self.results) self.should_succeed() @@ -302,8 +394,8 @@ class ContractsManagerTest(unittest.TestCase): with MockServer() as mockserver: contract_doc = '@url {}'.format(mockserver.url('/status?n=200')) - get_unbound_function(TestSameUrlSpider.parse_first).__doc__ = contract_doc - get_unbound_function(TestSameUrlSpider.parse_second).__doc__ = contract_doc + TestSameUrlSpider.parse_first.__doc__ = contract_doc + TestSameUrlSpider.parse_second.__doc__ = contract_doc crawler = CrawlerRunner().create_crawler(TestSameUrlSpider) yield crawler.crawl() diff --git a/tests/test_crawl.py b/tests/test_crawl.py index 3fc13eeb7..f433fcea6 100644 --- a/tests/test_crawl.py +++ b/tests/test_crawl.py @@ -5,12 +5,12 @@ from testfixtures import LogCapture from twisted.internet import defer from twisted.trial.unittest import TestCase -from scrapy.http import Request from scrapy.crawler import CrawlerRunner +from scrapy.http import Request from scrapy.utils.python import to_unicode -from tests.spiders import FollowAllSpider, DelaySpider, SimpleSpider, \ - BrokenStartRequestsSpider, SingleRequestSpider, DuplicateStartRequestsSpider from tests.mockserver import MockServer +from tests.spiders import (FollowAllSpider, DelaySpider, SimpleSpider, BrokenStartRequestsSpider, + SingleRequestSpider, DuplicateStartRequestsSpider, CrawlSpiderWithErrback) class CrawlTestCase(TestCase): @@ -30,25 +30,44 @@ class CrawlTestCase(TestCase): self.assertEqual(len(crawler.spider.urls_visited), 11) # 10 + start_url @defer.inlineCallbacks - def test_delay(self): - # short to long delays - yield self._test_delay(0.2, False) - yield self._test_delay(1, False) - # randoms - yield self._test_delay(0.2, True) - yield self._test_delay(1, True) + def test_fixed_delay(self): + yield self._test_delay(total=3, delay=0.1) @defer.inlineCallbacks - def _test_delay(self, delay, randomize): - settings = {"DOWNLOAD_DELAY": delay, 'RANDOMIZE_DOWNLOAD_DELAY': randomize} + def test_randomized_delay(self): + yield self._test_delay(total=3, delay=0.1, randomize=True) + + @defer.inlineCallbacks + def _test_delay(self, total, delay, randomize=False): + crawl_kwargs = dict( + maxlatency=delay * 2, + mockserver=self.mockserver, + total=total, + ) + tolerance = (1 - (0.6 if randomize else 0.2)) + + settings = {"DOWNLOAD_DELAY": delay, + 'RANDOMIZE_DOWNLOAD_DELAY': randomize} crawler = CrawlerRunner(settings).create_crawler(FollowAllSpider) - yield crawler.crawl(maxlatency=delay * 2, mockserver=self.mockserver) - t = crawler.spider.times - totaltime = t[-1] - t[0] - avgd = totaltime / (len(t) - 1) - tolerance = 0.6 if randomize else 0.2 - self.assertTrue(avgd > delay * (1 - tolerance), - "download delay too small: %s" % avgd) + yield crawler.crawl(**crawl_kwargs) + times = crawler.spider.times + total_time = times[-1] - times[0] + average = total_time / (len(times) - 1) + self.assertTrue(average > delay * tolerance, + "download delay too small: %s" % average) + + # Ensure that the same test parameters would cause a failure if no + # download delay is set. Otherwise, it means we are using a combination + # of ``total`` and ``delay`` values that are too small for the test + # code above to have any meaning. + settings["DOWNLOAD_DELAY"] = 0 + crawler = CrawlerRunner(settings).create_crawler(FollowAllSpider) + yield crawler.crawl(**crawl_kwargs) + times = crawler.spider.times + total_time = times[-1] - times[0] + average = total_time / (len(times) - 1) + self.assertFalse(average > delay / tolerance, + "test total or delay values are too small") @defer.inlineCallbacks def test_timeout_success(self): @@ -140,7 +159,7 @@ class CrawlTestCase(TestCase): def test_unbounded_response(self): # Completeness of responses without Content-Length or Transfer-Encoding # can not be determined, we treat them as valid but flagged as "partial" - from six.moves.urllib.parse import urlencode + from urllib.parse import urlencode query = urlencode({'raw': '''\ HTTP/1.1 200 OK Server: Apache-Coyote/1.1 @@ -277,3 +296,15 @@ with multiples lines self._assert_retried(log) self.assertIn("Got response 200", str(log)) + + @defer.inlineCallbacks + def test_crawlspider_with_errback(self): + self.runner.crawl(CrawlSpiderWithErrback, mockserver=self.mockserver) + + with LogCapture() as log: + yield self.runner.join() + + self.assertIn("[callback] status 200", str(log)) + self.assertIn("[callback] status 201", str(log)) + self.assertIn("[errback] status 404", str(log)) + self.assertIn("[errback] status 500", str(log)) diff --git a/tests/test_crawler.py b/tests/test_crawler.py index 8eb2389e2..fce60ca37 100644 --- a/tests/test_crawler.py +++ b/tests/test_crawler.py @@ -1,9 +1,13 @@ import logging +import os +import subprocess +import sys import warnings +from pytest import raises, mark +from testfixtures import LogCapture from twisted.internet import defer from twisted.trial import unittest -from pytest import raises import scrapy from scrapy.crawler import Crawler, CrawlerRunner, CrawlerProcess @@ -14,6 +18,7 @@ from scrapy.utils.spider import DefaultSpider from scrapy.utils.misc import load_object from scrapy.extensions.throttle import AutoThrottle from scrapy.extensions import telnet +from scrapy.utils.test import get_testenv class BaseCrawlerTest(unittest.TestCase): @@ -203,6 +208,7 @@ class NoRequestsSpider(scrapy.Spider): return [] +@mark.usefixtures('reactor_pytest') class CrawlerRunnerHasSpider(unittest.TestCase): @defer.inlineCallbacks @@ -245,3 +251,57 @@ class CrawlerRunnerHasSpider(unittest.TestCase): yield runner.crawl(NoRequestsSpider) self.assertEqual(runner.bootstrap_failed, True) + + def test_crawler_runner_asyncio_enabled_true(self): + if self.reactor_pytest == 'asyncio': + runner = CrawlerRunner(settings={'ASYNCIO_REACTOR': True}) + else: + msg = "ASYNCIO_REACTOR is on but the Twisted asyncio reactor is not installed" + with self.assertRaisesRegex(Exception, msg): + runner = CrawlerRunner(settings={'ASYNCIO_REACTOR': True}) + + @defer.inlineCallbacks + def test_crawler_process_asyncio_enabled_true(self): + with LogCapture(level=logging.DEBUG) as log: + if self.reactor_pytest == 'asyncio': + runner = CrawlerProcess(settings={'ASYNCIO_REACTOR': True}) + yield runner.crawl(NoRequestsSpider) + self.assertIn("Asyncio reactor is installed", str(log)) + else: + msg = "ASYNCIO_REACTOR is on but the Twisted asyncio reactor is not installed" + with self.assertRaisesRegex(Exception, msg): + runner = CrawlerProcess(settings={'ASYNCIO_REACTOR': True}) + + @defer.inlineCallbacks + def test_crawler_process_asyncio_enabled_false(self): + runner = CrawlerProcess(settings={'ASYNCIO_REACTOR': False}) + with LogCapture(level=logging.DEBUG) as log: + yield runner.crawl(NoRequestsSpider) + self.assertNotIn("Asyncio reactor is installed", str(log)) + + +class CrawlerProcessSubprocess(unittest.TestCase): + script_dir = os.path.join(os.path.abspath(os.path.dirname(__file__)), 'CrawlerProcess') + + def run_script(self, script_name): + script_path = os.path.join(self.script_dir, script_name) + args = (sys.executable, script_path) + p = subprocess.Popen(args, env=get_testenv(), + stdout=subprocess.PIPE, stderr=subprocess.PIPE) + stdout, stderr = p.communicate() + return stderr.decode('utf-8') + + def test_simple(self): + log = self.run_script('simple.py') + self.assertIn('Spider closed (finished)', log) + self.assertNotIn("DEBUG: Asyncio reactor is installed", log) + + def test_asyncio_enabled_no_reactor(self): + log = self.run_script('asyncio_enabled_no_reactor.py') + self.assertIn('Spider closed (finished)', log) + self.assertIn("DEBUG: Asyncio reactor is installed", log) + + def test_asyncio_enabled_reactor(self): + log = self.run_script('asyncio_enabled_reactor.py') + self.assertIn('Spider closed (finished)', log) + self.assertIn("DEBUG: Asyncio reactor is installed", log) diff --git a/tests/test_dependencies.py b/tests/test_dependencies.py index 03bf2ffcf..e31ccd9b5 100644 --- a/tests/test_dependencies.py +++ b/tests/test_dependencies.py @@ -1,6 +1,7 @@ from importlib import import_module from twisted.trial import unittest + class ScrapyUtilsTest(unittest.TestCase): def test_required_openssl_version(self): try: diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py index d2151e10e..ce39f8545 100644 --- a/tests/test_downloader_handlers.py +++ b/tests/test_downloader_handlers.py @@ -1,13 +1,10 @@ import os -import six import shutil import tempfile +from unittest import mock import contextlib -try: - from unittest import mock -except ImportError: - import mock +from testfixtures import LogCapture from twisted.trial import unittest from twisted.protocols.policies import WrappingFactory from twisted.python.filepath import FilePath @@ -36,7 +33,7 @@ from scrapy.responsetypes import responsetypes from scrapy.settings import Settings from scrapy.utils.test import get_crawler, skip_if_no_boto from scrapy.utils.python import to_bytes -from scrapy.exceptions import NotConfigured +from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning from tests.mockserver import MockServer, ssl_context_factory, Echo from tests.spiders import SingleRequestSpider @@ -352,7 +349,7 @@ class HttpTestCase(unittest.TestCase): return self.download_request(request, Spider('foo')).addCallback(_test) def test_payload(self): - body = b'1'*100 # PayloadResource requires body length to be 100 + body = b'1'*100 # PayloadResource requires body length to be 100 request = Request(self.getURL('payload'), method='POST', body=body) d = self.download_request(request, Spider('foo')) d.addCallback(lambda r: r.body) @@ -498,6 +495,24 @@ class Http11TestCase(HttpTestCase): class Https11TestCase(Http11TestCase): scheme = 'https' + tls_log_message = 'SSL connection certificate: issuer "/C=IE/O=Scrapy/CN=localhost", subject "/C=IE/O=Scrapy/CN=localhost"' + + @defer.inlineCallbacks + def test_tls_logging(self): + download_handler = self.download_handler_cls(Settings({ + 'DOWNLOADER_CLIENT_TLS_VERBOSE_LOGGING': True, + })) + try: + with LogCapture() as log_capture: + request = Request(self.getURL('file')) + d = download_handler.download_request(request, Spider('foo')) + d.addCallback(lambda r: r.body) + d.addCallback(self.assertEqual, b"0123456789") + yield d + log_capture.check_present(('scrapy.core.downloader.tls', 'DEBUG', self.tls_log_message)) + finally: + yield download_handler.close() + class Https11WrongHostnameTestCase(Http11TestCase): scheme = 'https' @@ -518,6 +533,7 @@ class Https11InvalidDNSId(Https11TestCase): super(Https11InvalidDNSId, self).setUp() self.host = '127.0.0.1' + class Https11InvalidDNSPattern(Https11TestCase): """Connect to HTTPS hosts where the certificate are issued to an ip instead of a domain.""" @@ -526,12 +542,54 @@ class Https11InvalidDNSPattern(Https11TestCase): def setUp(self): try: - from service_identity.exceptions import CertificateError + from service_identity.exceptions import CertificateError # noqa: F401 except ImportError: raise unittest.SkipTest("cryptography lib is too old") + self.tls_log_message = 'SSL connection certificate: issuer "/C=IE/O=Scrapy/CN=127.0.0.1", subject "/C=IE/O=Scrapy/CN=127.0.0.1"' super(Https11InvalidDNSPattern, self).setUp() +class Https11CustomCiphers(unittest.TestCase): + scheme = 'https' + download_handler_cls = HTTP11DownloadHandler + + keyfile = 'keys/localhost.key' + certfile = 'keys/localhost.crt' + + def setUp(self): + self.tmpname = self.mktemp() + os.mkdir(self.tmpname) + FilePath(self.tmpname).child("file").setContent(b"0123456789") + r = static.File(self.tmpname) + self.site = server.Site(r, timeout=None) + self.wrapper = WrappingFactory(self.site) + self.host = 'localhost' + self.port = reactor.listenSSL( + 0, self.wrapper, ssl_context_factory(self.keyfile, self.certfile, cipher_string='CAMELLIA256-SHA'), + interface=self.host) + self.portno = self.port.getHost().port + self.download_handler = self.download_handler_cls( + Settings({'DOWNLOADER_CLIENT_TLS_CIPHERS': 'CAMELLIA256-SHA'})) + self.download_request = self.download_handler.download_request + + @defer.inlineCallbacks + def tearDown(self): + yield self.port.stopListening() + if hasattr(self.download_handler, 'close'): + yield self.download_handler.close() + shutil.rmtree(self.tmpname) + + def getURL(self, path): + return "%s://%s:%d/%s" % (self.scheme, self.host, self.portno, path) + + def test_download(self): + request = Request(self.getURL('file')) + d = self.download_request(request, Spider('foo')) + d.addCallback(lambda r: r.body) + d.addCallback(self.assertEqual, b"0123456789") + return d + + class Http11MockServerTestCase(unittest.TestCase): """HTTP 1.1 test case with MockServer""" @@ -556,7 +614,7 @@ class Http11MockServerTestCase(unittest.TestCase): crawler = get_crawler(SingleRequestSpider) yield crawler.crawl(seed=Request(url=self.mockserver.url(''))) failure = crawler.spider.meta.get('failure') - self.assertTrue(failure == None) + self.assertTrue(failure is None) reason = crawler.spider.meta['close_reason'] self.assertTrue(reason, 'finished') @@ -571,18 +629,16 @@ class Http11MockServerTestCase(unittest.TestCase): # download_maxsize < 100, hence the CancelledError self.assertIsInstance(failure.value, defer.CancelledError) - if six.PY2: - request.headers.setdefault(b'Accept-Encoding', b'gzip,deflate') - request = request.replace(url=self.mockserver.url('/xpayload')) - yield crawler.crawl(seed=request) - # download_maxsize = 50 is enough for the gzipped response - failure = crawler.spider.meta.get('failure') - self.assertTrue(failure == None) - reason = crawler.spider.meta['close_reason'] - self.assertTrue(reason, 'finished') - else: - # See issue https://twistedmatrix.com/trac/ticket/8175 - raise unittest.SkipTest("xpayload only enabled for PY2") + # See issue https://twistedmatrix.com/trac/ticket/8175 + raise unittest.SkipTest("xpayload fails on PY3") + request.headers.setdefault(b'Accept-Encoding', b'gzip,deflate') + request = request.replace(url=self.mockserver.url('/xpayload')) + yield crawler.crawl(seed=request) + # download_maxsize = 50 is enough for the gzipped response + failure = crawler.spider.meta.get('failure') + self.assertTrue(failure is None) + reason = crawler.spider.meta['close_reason'] + self.assertTrue(reason, 'finished') class UriResource(resource.Resource): @@ -639,7 +695,9 @@ class HttpProxyTestCase(unittest.TestCase): http_proxy = '%s?noconnect' % self.getURL('') request = Request('https://example.com', meta={'proxy': http_proxy}) - return self.download_request(request, Spider('foo')).addCallback(_test) + with self.assertWarnsRegex(ScrapyDeprecationWarning, + r'Using HTTPS proxies in the noconnect mode is deprecated'): + return self.download_request(request, Spider('foo')).addCallback(_test) def test_download_without_proxy(self): def _test(response): @@ -654,6 +712,9 @@ class HttpProxyTestCase(unittest.TestCase): class Http10ProxyTestCase(HttpProxyTestCase): download_handler_cls = HTTP10DownloadHandler + def test_download_with_proxy_https_noconnect(self): + raise unittest.SkipTest('noconnect is not supported in HTTP10DownloadHandler') + class Http11ProxyTestCase(HttpProxyTestCase): download_handler_cls = HTTP11DownloadHandler @@ -719,7 +780,7 @@ class S3TestCase(unittest.TestCase): @contextlib.contextmanager def _mocked_date(self, date): try: - import botocore.auth + import botocore.auth # noqa: F401 except ImportError: yield else: @@ -744,8 +805,8 @@ class S3TestCase(unittest.TestCase): req = Request('s3://johnsmith/photos/puppy.jpg', headers={'Date': date}) with self._mocked_date(date): httpreq = self.download_request(req, self.spider) - self.assertEqual(httpreq.headers['Authorization'], \ - b'AWS 0PN5J17HBGZHT7JJ3X82:xXjDGYUmKxnwqr5KXNPGldn5LbA=') + self.assertEqual(httpreq.headers['Authorization'], + b'AWS 0PN5J17HBGZHT7JJ3X82:xXjDGYUmKxnwqr5KXNPGldn5LbA=') def test_request_signing2(self): # puts an object into the johnsmith bucket. @@ -757,21 +818,22 @@ class S3TestCase(unittest.TestCase): }) with self._mocked_date(date): httpreq = self.download_request(req, self.spider) - self.assertEqual(httpreq.headers['Authorization'], \ - b'AWS 0PN5J17HBGZHT7JJ3X82:hcicpDDvL9SsO6AkvxqmIWkmOuQ=') + self.assertEqual(httpreq.headers['Authorization'], + b'AWS 0PN5J17HBGZHT7JJ3X82:hcicpDDvL9SsO6AkvxqmIWkmOuQ=') def test_request_signing3(self): # lists the content of the johnsmith bucket. date = 'Tue, 27 Mar 2007 19:42:41 +0000' - req = Request('s3://johnsmith/?prefix=photos&max-keys=50&marker=puppy', \ - method='GET', headers={ - 'User-Agent': 'Mozilla/5.0', - 'Date': date, - }) + req = Request( + 's3://johnsmith/?prefix=photos&max-keys=50&marker=puppy', + method='GET', headers={ + 'User-Agent': 'Mozilla/5.0', + 'Date': date, + }) with self._mocked_date(date): httpreq = self.download_request(req, self.spider) - self.assertEqual(httpreq.headers['Authorization'], \ - b'AWS 0PN5J17HBGZHT7JJ3X82:jsRt/rhG+Vtp88HrYL706QhE4w4=') + self.assertEqual(httpreq.headers['Authorization'], + b'AWS 0PN5J17HBGZHT7JJ3X82:jsRt/rhG+Vtp88HrYL706QhE4w4=') def test_request_signing4(self): # fetches the access control policy sub-resource for the 'johnsmith' bucket. @@ -780,23 +842,25 @@ class S3TestCase(unittest.TestCase): method='GET', headers={'Date': date}) with self._mocked_date(date): httpreq = self.download_request(req, self.spider) - self.assertEqual(httpreq.headers['Authorization'], \ - b'AWS 0PN5J17HBGZHT7JJ3X82:thdUi9VAkzhkniLj96JIrOPGi0g=') + self.assertEqual(httpreq.headers['Authorization'], + b'AWS 0PN5J17HBGZHT7JJ3X82:thdUi9VAkzhkniLj96JIrOPGi0g=') def test_request_signing5(self): - try: import botocore - except ImportError: pass + try: + import botocore # noqa: F401 + except ImportError: + pass else: raise unittest.SkipTest( 'botocore does not support overriding date with x-amz-date') # deletes an object from the 'johnsmith' bucket using the # path-style and Date alternative. date = 'Tue, 27 Mar 2007 21:20:27 +0000' - req = Request('s3://johnsmith/photos/puppy.jpg', \ - method='DELETE', headers={ - 'Date': date, - 'x-amz-date': 'Tue, 27 Mar 2007 21:20:26 +0000', - }) + req = Request( + 's3://johnsmith/photos/puppy.jpg', method='DELETE', headers={ + 'Date': date, + 'x-amz-date': 'Tue, 27 Mar 2007 21:20:26 +0000', + }) with self._mocked_date(date): httpreq = self.download_request(req, self.spider) # botocore does not override Date with x-amz-date @@ -806,25 +870,26 @@ class S3TestCase(unittest.TestCase): def test_request_signing6(self): # uploads an object to a CNAME style virtual hosted bucket with metadata. date = 'Tue, 27 Mar 2007 21:06:08 +0000' - req = Request('s3://static.johnsmith.net:8080/db-backup.dat.gz', \ - method='PUT', headers={ - 'User-Agent': 'curl/7.15.5', - 'Host': 'static.johnsmith.net:8080', - 'Date': date, - 'x-amz-acl': 'public-read', - 'content-type': 'application/x-download', - 'Content-MD5': '4gJE4saaMU4BqNR0kLY+lw==', - 'X-Amz-Meta-ReviewedBy': 'joe@johnsmith.net,jane@johnsmith.net', - 'X-Amz-Meta-FileChecksum': '0x02661779', - 'X-Amz-Meta-ChecksumAlgorithm': 'crc32', - 'Content-Disposition': 'attachment; filename=database.dat', - 'Content-Encoding': 'gzip', - 'Content-Length': '5913339', - }) + req = Request( + 's3://static.johnsmith.net:8080/db-backup.dat.gz', + method='PUT', headers={ + 'User-Agent': 'curl/7.15.5', + 'Host': 'static.johnsmith.net:8080', + 'Date': date, + 'x-amz-acl': 'public-read', + 'content-type': 'application/x-download', + 'Content-MD5': '4gJE4saaMU4BqNR0kLY+lw==', + 'X-Amz-Meta-ReviewedBy': 'joe@johnsmith.net,jane@johnsmith.net', + 'X-Amz-Meta-FileChecksum': '0x02661779', + 'X-Amz-Meta-ChecksumAlgorithm': 'crc32', + 'Content-Disposition': 'attachment; filename=database.dat', + 'Content-Encoding': 'gzip', + 'Content-Length': '5913339', + }) with self._mocked_date(date): httpreq = self.download_request(req, self.spider) - self.assertEqual(httpreq.headers['Authorization'], \ - b'AWS 0PN5J17HBGZHT7JJ3X82:C0FlOtU8Ylb9KDTpZqYkZPX91iI=') + self.assertEqual(httpreq.headers['Authorization'], + b'AWS 0PN5J17HBGZHT7JJ3X82:C0FlOtU8Ylb9KDTpZqYkZPX91iI=') def test_request_signing7(self): # ensure that spaces are quoted properly before signing diff --git a/tests/test_downloadermiddleware.py b/tests/test_downloadermiddleware.py index 03564e748..3dd4f2351 100644 --- a/tests/test_downloadermiddleware.py +++ b/tests/test_downloadermiddleware.py @@ -1,3 +1,9 @@ +import asyncio +from unittest import mock + +from pytest import mark +from twisted.internet import defer +from twisted.internet.defer import Deferred from twisted.trial.unittest import TestCase from twisted.python.failure import Failure @@ -5,9 +11,8 @@ from scrapy.http import Request, Response from scrapy.spiders import Spider from scrapy.exceptions import _InvalidOutput from scrapy.core.downloader.middleware import DownloaderMiddlewareManager -from scrapy.utils.test import get_crawler +from scrapy.utils.test import get_crawler, get_from_asyncio_queue from scrapy.utils.python import to_bytes -from tests import mock class ManagerTestCase(TestCase): @@ -176,3 +181,75 @@ class ProcessExceptionInvalidOutput(ManagerTestCase): dfd.addBoth(results.append) self.assertIsInstance(results[0], Failure) self.assertIsInstance(results[0].value, _InvalidOutput) + + +class MiddlewareUsingDeferreds(ManagerTestCase): + """Middlewares using Deferreds should work""" + + def test_deferred(self): + resp = Response('http://example.com/index.html') + + class DeferredMiddleware: + def cb(self, result): + return result + + def process_request(self, request, spider): + d = Deferred() + d.addCallback(self.cb) + d.callback(resp) + return d + + self.mwman._add_middleware(DeferredMiddleware()) + req = Request('http://example.com/index.html') + download_func = mock.MagicMock() + dfd = self.mwman.download(download_func, req, self.spider) + results = [] + dfd.addBoth(results.append) + self._wait(dfd) + + self.assertIs(results[0], resp) + self.assertFalse(download_func.called) + + +class MiddlewareUsingCoro(ManagerTestCase): + """Middlewares using asyncio coroutines should work""" + + def test_asyncdef(self): + resp = Response('http://example.com/index.html') + + class CoroMiddleware: + async def process_request(self, request, spider): + await defer.succeed(42) + return resp + + self.mwman._add_middleware(CoroMiddleware()) + req = Request('http://example.com/index.html') + download_func = mock.MagicMock() + dfd = self.mwman.download(download_func, req, self.spider) + results = [] + dfd.addBoth(results.append) + self._wait(dfd) + + self.assertIs(results[0], resp) + self.assertFalse(download_func.called) + + @mark.only_asyncio() + def test_asyncdef_asyncio(self): + resp = Response('http://example.com/index.html') + + class CoroMiddleware: + async def process_request(self, request, spider): + await asyncio.sleep(0.1) + result = await get_from_asyncio_queue(resp) + return result + + self.mwman._add_middleware(CoroMiddleware()) + req = Request('http://example.com/index.html') + download_func = mock.MagicMock() + dfd = self.mwman.download(download_func, req, self.spider) + results = [] + dfd.addBoth(results.append) + self._wait(dfd) + + self.assertIs(results[0], resp) + self.assertFalse(download_func.called) diff --git a/tests/test_downloadermiddleware_ajaxcrawlable.py b/tests/test_downloadermiddleware_ajaxcrawlable.py index 493691ea4..5a56c9db2 100644 --- a/tests/test_downloadermiddleware_ajaxcrawlable.py +++ b/tests/test_downloadermiddleware_ajaxcrawlable.py @@ -5,8 +5,10 @@ from scrapy.spiders import Spider from scrapy.http import Request, HtmlResponse, Response from scrapy.utils.test import get_crawler + __doctests__ = ['scrapy.downloadermiddlewares.ajaxcrawl'] + class AjaxCrawlMiddlewareTest(unittest.TestCase): def setUp(self): crawler = get_crawler(Spider, {'AJAXCRAWL_ENABLED': True}) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index 22946b98c..9401dd66d 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -1,11 +1,9 @@ -from __future__ import print_function import time import tempfile import shutil import unittest import email.utils from contextlib import contextmanager -import pytest from scrapy.http import Response, HtmlResponse, Request from scrapy.spiders import Spider @@ -84,8 +82,8 @@ class _BaseTest(unittest.TestCase): def assertEqualRequestButWithCacheValidators(self, request1, request2): self.assertEqual(request1.url, request2.url) - assert not b'If-None-Match' in request1.headers - assert not b'If-Modified-Since' in request1.headers + assert b'If-None-Match' not in request1.headers + assert b'If-Modified-Since' not in request1.headers assert any(h in request2.headers for h in (b'If-None-Match', b'If-Modified-Since')) self.assertEqual(request1.body, request2.body) @@ -148,17 +146,13 @@ class FilesystemStorageTest(DefaultStorageTest): storage_class = 'scrapy.extensions.httpcache.FilesystemCacheStorage' + class FilesystemStorageGzipTest(FilesystemStorageTest): def _get_settings(self, **new_settings): new_settings.setdefault('HTTPCACHE_GZIP', True) return super(FilesystemStorageTest, self)._get_settings(**new_settings) -class LeveldbStorageTest(DefaultStorageTest): - - pytest.importorskip('leveldb') - storage_class = 'scrapy.extensions.httpcache.LeveldbCacheStorage' - class DummyPolicyTest(_BaseTest): diff --git a/tests/test_downloadermiddleware_httpcompression.py b/tests/test_downloadermiddleware_httpcompression.py index 0745c8dd3..c6a823b53 100644 --- a/tests/test_downloadermiddleware_httpcompression.py +++ b/tests/test_downloadermiddleware_httpcompression.py @@ -70,7 +70,7 @@ class HttpCompressionTest(TestCase): def test_process_response_br(self): try: - import brotli + import brotli # noqa: F401 except ImportError: raise SkipTest("no brotli") response = self._getresponse('br') diff --git a/tests/test_downloadermiddleware_httpproxy.py b/tests/test_downloadermiddleware_httpproxy.py index 30920b2da..36743b1de 100644 --- a/tests/test_downloadermiddleware_httpproxy.py +++ b/tests/test_downloadermiddleware_httpproxy.py @@ -1,11 +1,10 @@ import os -import sys from functools import partial -from twisted.trial.unittest import TestCase, SkipTest +from twisted.trial.unittest import TestCase from scrapy.downloadermiddlewares.httpproxy import HttpProxyMiddleware from scrapy.exceptions import NotConfigured -from scrapy.http import Response, Request +from scrapy.http import Request from scrapy.spiders import Spider from scrapy.crawler import Crawler from scrapy.settings import Settings diff --git a/tests/test_downloadermiddleware_redirect.py b/tests/test_downloadermiddleware_redirect.py index 0e841489d..e7faf14a7 100644 --- a/tests/test_downloadermiddleware_redirect.py +++ b/tests/test_downloadermiddleware_redirect.py @@ -106,6 +106,22 @@ class RedirectMiddlewareTest(unittest.TestCase): del rsp.headers['Location'] assert self.mw.process_response(req, rsp, self.spider) is rsp + def test_redirect_302_relative(self): + url = 'http://www.example.com/302' + url2 = '///i8n.example2.com/302' + url3 = 'http://i8n.example2.com/302' + req = Request(url, method='HEAD') + rsp = Response(url, headers={'Location': url2}, status=302) + + req2 = self.mw.process_response(req, rsp, self.spider) + assert isinstance(req2, Request) + self.assertEqual(req2.url, url3) + self.assertEqual(req2.method, 'HEAD') + + # response without Location header but with status code is 3XX should be ignored + del rsp.headers['Location'] + assert self.mw.process_response(req, rsp, self.spider) is rsp + def test_max_redirect_times(self): self.mw.max_redirect_times = 1 diff --git a/tests/test_downloadermiddleware_retry.py b/tests/test_downloadermiddleware_retry.py index 51b79b6c3..9c989977e 100644 --- a/tests/test_downloadermiddleware_retry.py +++ b/tests/test_downloadermiddleware_retry.py @@ -124,7 +124,7 @@ class MaxRetryTimesTest(unittest.TestCase): # SETTINGS: meta(max_retry_times) = 0 meta_max_retry_times = 0 - + req = Request(self.invalid_url, meta={'max_retry_times': meta_max_retry_times}) self._test_retry(req, DNSLookupError('foo'), meta_max_retry_times) @@ -137,7 +137,7 @@ class MaxRetryTimesTest(unittest.TestCase): self._test_retry(req, DNSLookupError('foo'), self.mw.max_retry_times) def test_with_metakey_greater(self): - + # SETINGS: RETRY_TIMES < meta(max_retry_times) self.mw.max_retry_times = 2 meta_max_retry_times = 3 @@ -149,7 +149,7 @@ class MaxRetryTimesTest(unittest.TestCase): self._test_retry(req2, DNSLookupError('foo'), self.mw.max_retry_times) def test_with_metakey_lesser(self): - + # SETINGS: RETRY_TIMES > meta(max_retry_times) self.mw.max_retry_times = 5 meta_max_retry_times = 4 @@ -165,14 +165,14 @@ class MaxRetryTimesTest(unittest.TestCase): # SETTINGS: meta(max_retry_times) = 4 meta_max_retry_times = 4 - req = Request(self.invalid_url, meta= \ - {'max_retry_times': meta_max_retry_times, 'dont_retry': True}) + req = Request(self.invalid_url, meta={ + 'max_retry_times': meta_max_retry_times, 'dont_retry': True + }) self._test_retry(req, DNSLookupError('foo'), 0) - def _test_retry(self, req, exception, max_retry_times): - + for i in range(0, max_retry_times): req = self.mw.process_exception(req, exception, self.spider) assert isinstance(req, Request) diff --git a/tests/test_downloadermiddleware_robotstxt.py b/tests/test_downloadermiddleware_robotstxt.py index 2b3548bdd..a1645ed96 100644 --- a/tests/test_downloadermiddleware_robotstxt.py +++ b/tests/test_downloadermiddleware_robotstxt.py @@ -1,5 +1,6 @@ # -*- coding: utf-8 -*- -from __future__ import absolute_import +from unittest import mock + from twisted.internet import reactor, error from twisted.internet.defer import Deferred, DeferredList, maybeDeferred from twisted.python import failure @@ -9,7 +10,7 @@ from scrapy.downloadermiddlewares.robotstxt import (RobotsTxtMiddleware, from scrapy.exceptions import IgnoreRequest, NotConfigured from scrapy.http import Request, Response, TextResponse from scrapy.settings import Settings -from tests import mock +from tests.test_robotstxt_interface import rerp_available, reppy_available class RobotsTxtMiddlewareTest(unittest.TestCase): @@ -41,6 +42,7 @@ User-Agent: UnicödeBöt Disallow: /some/randome/page.html """.encode('utf-8') response = TextResponse('http://site.local/robots.txt', body=ROBOTS) + def return_response(request, spider): deferred = Deferred() reactor.callFromThread(deferred.callback, response) @@ -77,6 +79,7 @@ Disallow: /some/randome/page.html crawler = self.crawler crawler.settings.set('ROBOTSTXT_OBEY', True) response = Response('http://site.local/robots.txt', body=b'GIF89a\xd3\x00\xfe\x00\xa2') + def return_response(request, spider): deferred = Deferred() reactor.callFromThread(deferred.callback, response) @@ -99,6 +102,7 @@ Disallow: /some/randome/page.html crawler = self.crawler crawler.settings.set('ROBOTSTXT_OBEY', True) response = Response('http://site.local/robots.txt') + def return_response(request, spider): deferred = Deferred() reactor.callFromThread(deferred.callback, response) @@ -118,6 +122,7 @@ Disallow: /some/randome/page.html def test_robotstxt_error(self): self.crawler.settings.set('ROBOTSTXT_OBEY', True) err = error.DNSLookupError('Robotstxt address not found') + def return_failure(request, spider): deferred = Deferred() reactor.callFromThread(deferred.errback, failure.Failure(err)) @@ -133,6 +138,7 @@ Disallow: /some/randome/page.html def test_robotstxt_immediate_error(self): self.crawler.settings.set('ROBOTSTXT_OBEY', True) err = error.DNSLookupError('Robotstxt address not found') + def immediate_failure(request, spider): deferred = Deferred() deferred.errback(failure.Failure(err)) @@ -144,6 +150,7 @@ Disallow: /some/randome/page.html def test_ignore_robotstxt_request(self): self.crawler.settings.set('ROBOTSTXT_OBEY', True) + def ignore_request(request, spider): deferred = Deferred() reactor.callFromThread(deferred.errback, failure.Failure(IgnoreRequest())) @@ -157,6 +164,15 @@ Disallow: /some/randome/page.html d.addCallback(lambda _: self.assertFalse(mw_module_logger.error.called)) return d + def test_robotstxt_user_agent_setting(self): + crawler = self._get_successful_crawler() + crawler.settings.set('ROBOTSTXT_USER_AGENT', 'Examplebot') + crawler.settings.set('USER_AGENT', 'Mozilla/5.0 (X11; Linux x86_64)') + middleware = RobotsTxtMiddleware(crawler) + rp = mock.MagicMock(return_value=True) + middleware.process_request_2(rp, Request('http://site.local/allowed'), None) + rp.allowed.assert_called_once_with('http://site.local/allowed', 'Examplebot') + def assertNotIgnored(self, request, middleware): spider = None # not actually used dfd = maybeDeferred(middleware.process_request, request, spider) @@ -167,3 +183,21 @@ Disallow: /some/randome/page.html spider = None # not actually used return self.assertFailure(maybeDeferred(middleware.process_request, request, spider), IgnoreRequest) + + +class RobotsTxtMiddlewareWithRerpTest(RobotsTxtMiddlewareTest): + if not rerp_available(): + skip = "Rerp parser is not installed" + + def setUp(self): + super(RobotsTxtMiddlewareWithRerpTest, self).setUp() + self.crawler.settings.set('ROBOTSTXT_PARSER', 'scrapy.robotstxt.RerpRobotParser') + + +class RobotsTxtMiddlewareWithReppyTest(RobotsTxtMiddlewareTest): + if not reppy_available(): + skip = "Reppy parser is not installed" + + def setUp(self): + super(RobotsTxtMiddlewareWithReppyTest, self).setUp() + self.crawler.settings.set('ROBOTSTXT_PARSER', 'scrapy.robotstxt.ReppyRobotParser') diff --git a/tests/test_dupefilters.py b/tests/test_dupefilters.py index d7eb98c97..0546558bc 100644 --- a/tests/test_dupefilters.py +++ b/tests/test_dupefilters.py @@ -12,6 +12,7 @@ from scrapy.utils.job import job_dir from scrapy.utils.test import get_crawler from tests.spiders import SimpleSpider + class FromCrawlerRFPDupeFilter(RFPDupeFilter): @classmethod @@ -141,12 +142,12 @@ class RFPDupeFilterTest(unittest.TestCase): r1 = Request('http://scrapytest.org/index.html') r2 = Request('http://scrapytest.org/index.html') - + dupefilter.log(r1, spider) dupefilter.log(r2, spider) assert crawler.stats.get_value('dupefilter/filtered') == 2 - l.check_present(('scrapy.dupefilters', 'DEBUG', + l.check_present(('scrapy.dupefilters', 'DEBUG', ('Filtered duplicate request: <GET http://scrapytest.org/index.html>' ' - no more duplicates will be shown' ' (see DUPEFILTER_DEBUG to show all duplicates)'))) @@ -168,7 +169,7 @@ class RFPDupeFilterTest(unittest.TestCase): r2 = Request('http://scrapytest.org/index.html', headers={'Referer': 'http://scrapytest.org/INDEX.html'} ) - + dupefilter.log(r1, spider) dupefilter.log(r2, spider) diff --git a/tests/test_engine.py b/tests/test_engine.py index 30150391a..25dee7c1f 100644 --- a/tests/test_engine.py +++ b/tests/test_engine.py @@ -10,9 +10,10 @@ module with the ``runserver`` argument:: python test_engine.py runserver """ -from __future__ import print_function -import sys, os, re -from six.moves.urllib.parse import urlparse +import os +import re +import sys +from urllib.parse import urlparse from twisted.internet import reactor, defer from twisted.web import server, static, util @@ -90,8 +91,8 @@ def start_test_site(debug=False): port = reactor.listenTCP(0, server.Site(r), interface="127.0.0.1") if debug: - print("Test server running at http://localhost:%d/ - hit Ctrl-C to finish." \ - % port.getHost().port) + print("Test server running at http://localhost:%d/ - hit Ctrl-C to finish." + % port.getHost().port) return port @@ -178,7 +179,6 @@ class EngineTest(unittest.TestCase): @defer.inlineCallbacks def test_crawler(self): - for spider in TestSpider, DictItemsSpider: self.run = CrawlerRun(spider) yield self.run.run() @@ -188,11 +188,15 @@ class EngineTest(unittest.TestCase): self._assert_scraped_items() self._assert_signals_catched() + @defer.inlineCallbacks + def test_crawler_dupefilter(self): self.run = CrawlerRun(TestDupeFilterSpider) yield self.run.run() self._assert_scheduled_requests(urls_to_visit=7) self._assert_dropped_requests() + @defer.inlineCallbacks + def test_crawler_itemerror(self): self.run = CrawlerRun(ItemZeroDivisionErrorSpider) yield self.run.run() self._assert_items_error() @@ -270,7 +274,6 @@ class EngineTest(unittest.TestCase): self.run.signals_catched[signals.spider_opened]) self.assertEqual({'spider': self.run.spider}, self.run.signals_catched[signals.spider_idle]) - self.run.signals_catched[signals.spider_closed].pop('spider_stats', None) # XXX: remove for scrapy 0.17 self.assertEqual({'spider': self.run.spider, 'reason': 'finished'}, self.run.signals_catched[signals.spider_closed]) diff --git a/tests/test_exporters.py b/tests/test_exporters.py index 0046c5666..5d1f5c182 100644 --- a/tests/test_exporters.py +++ b/tests/test_exporters.py @@ -1,15 +1,13 @@ -from __future__ import absolute_import import re import json import marshal +import pickle import tempfile import unittest from io import BytesIO from datetime import datetime -from six.moves import cPickle as pickle import lxml.etree -import six from scrapy.item import Item, Field from scrapy.utils.python import to_unicode @@ -80,7 +78,7 @@ class BaseItemExporterTest(unittest.TestCase): ie = self._get_exporter(fields_to_export=['name'], encoding='latin-1') _, name = list(ie._get_serialized_fields(self.i))[0] - assert isinstance(name, six.text_type) + assert isinstance(name, str) self.assertEqual(name, u'John\xa3') def test_field_custom_serializer(self): diff --git a/tests/test_extension_telnet.py b/tests/test_extension_telnet.py index 4f389e5cb..873a97248 100644 --- a/tests/test_extension_telnet.py +++ b/tests/test_extension_telnet.py @@ -1,14 +1,9 @@ -try: - import unittest.mock as mock -except ImportError: - import mock - from twisted.trial import unittest from twisted.conch.telnet import ITelnetProtocol from twisted.cred import credentials from twisted.internet import defer -from scrapy.extensions.telnet import TelnetConsole, logger +from scrapy.extensions.telnet import TelnetConsole from scrapy.utils.test import get_crawler diff --git a/tests/test_feedexport.py b/tests/test_feedexport.py index c5063253a..2ca57c19d 100644 --- a/tests/test_feedexport.py +++ b/tests/test_feedexport.py @@ -1,20 +1,21 @@ -from __future__ import absolute_import import os import csv import json import warnings -from io import BytesIO import tempfile import shutil -from six.moves.urllib.parse import urljoin, urlparse -from six.moves.urllib.request import pathname2url +import string +from io import BytesIO +from pathlib import Path +from unittest import mock +from urllib.parse import urljoin, urlparse, quote +from urllib.request import pathname2url from zope.interface.verify import verifyObject from twisted.trial import unittest from twisted.internet import defer from scrapy.crawler import CrawlerRunner from scrapy.settings import Settings -from tests import mock from tests.mockserver import MockServer from w3lib.url import path_to_file_uri @@ -25,8 +26,7 @@ from scrapy.extensions.feedexport import ( S3FeedStorage, StdoutFeedStorage, BlockingFeedStorage) from scrapy.utils.test import assert_aws_environ, get_s3_content_and_delete, get_crawler -from scrapy.utils.python import to_native_str -from scrapy.utils.project import get_project_settings +from scrapy.utils.python import to_unicode class FileFeedStorageTest(unittest.TestCase): @@ -98,6 +98,12 @@ class FTPFeedStorageTest(unittest.TestCase): verifyObject(IFeedStorage, st) return self._assert_stores(st, path) + def test_uri_auth_quote(self): + # RFC3986: 3.2.1. User Information + pw_quoted = quote(string.punctuation, safe='') + st = FTPFeedStorage('ftp://foo:%s@example.com/some_path' % pw_quoted) + self.assertEqual(st.password, string.punctuation) + @defer.inlineCallbacks def _assert_stores(self, storage, path): spider = self.get_test_spider() @@ -159,7 +165,7 @@ class S3FeedStorageTest(unittest.TestCase): create=True) def test_parse_credentials(self): try: - import boto + import boto # noqa: F401 except ImportError: raise unittest.SkipTest("S3FeedStorage requires boto") aws_credentials = {'AWS_ACCESS_KEY_ID': 'settings_key', @@ -260,7 +266,7 @@ class S3FeedStorageTest(unittest.TestCase): @defer.inlineCallbacks def test_store_botocore_without_acl(self): try: - import botocore + import botocore # noqa: F401 except ImportError: raise unittest.SkipTest('botocore is required') @@ -280,7 +286,7 @@ class S3FeedStorageTest(unittest.TestCase): @defer.inlineCallbacks def test_store_botocore_with_acl(self): try: - import botocore + import botocore # noqa: F401 except ImportError: raise unittest.SkipTest('botocore is required') @@ -398,6 +404,7 @@ class FeedExportTest(unittest.TestCase): defaults = { 'FEED_URI': res_uri, 'FEED_FORMAT': 'csv', + 'FEED_PATH': res_path } defaults.update(settings or {}) try: @@ -406,11 +413,11 @@ class FeedExportTest(unittest.TestCase): spider_cls.start_urls = [s.url('/')] yield runner.crawl(spider_cls) - with open(res_path, 'rb') as f: + with open(str(defaults['FEED_PATH']), 'rb') as f: content = f.read() finally: - shutil.rmtree(tmpdir, ignore_errors=True) + shutil.rmtree(tmpdir) defer.returnValue(content) @@ -449,7 +456,7 @@ class FeedExportTest(unittest.TestCase): settings.update({'FEED_FORMAT': 'csv'}) data = yield self.exported_data(items, settings) - reader = csv.DictReader(to_native_str(data).splitlines()) + reader = csv.DictReader(to_unicode(data).splitlines()) got_rows = list(reader) if ordered: self.assertEqual(reader.fieldnames, header) @@ -463,7 +470,7 @@ class FeedExportTest(unittest.TestCase): settings = settings or {} settings.update({'FEED_FORMAT': 'jl'}) data = yield self.exported_data(items, settings) - parsed = [json.loads(to_native_str(line)) for line in data.splitlines()] + parsed = [json.loads(to_unicode(line)) for line in data.splitlines()] rows = [{k: v for k, v in row.items() if v} for row in rows] self.assertEqual(rows, parsed) @@ -836,3 +843,17 @@ class FeedExportTest(unittest.TestCase): yield self.exported_data({}, settings) self.assertTrue(FromCrawlerCsvItemExporter.init_with_crawler) self.assertTrue(FromCrawlerFileFeedStorage.init_with_crawler) + + @defer.inlineCallbacks + def test_pathlib_uri(self): + tmpdir = tempfile.mkdtemp() + feed_uri = Path(tmpdir) / 'res' + settings = { + 'FEED_FORMAT': 'csv', + 'FEED_STORE_EMPTY': True, + 'FEED_URI': feed_uri, + 'FEED_PATH': feed_uri + } + data = yield self.exported_no_data(settings) + self.assertEqual(data, b'') + shutil.rmtree(tmpdir, ignore_errors=True) diff --git a/tests/test_http_cookies.py b/tests/test_http_cookies.py index 0a9ed500a..45ddb42ba 100644 --- a/tests/test_http_cookies.py +++ b/tests/test_http_cookies.py @@ -1,4 +1,4 @@ -from six.moves.urllib.parse import urlparse +from urllib.parse import urlparse from unittest import TestCase from scrapy.http import Request, Response diff --git a/tests/test_http_headers.py b/tests/test_http_headers.py index 69d906fbf..cf3fc8496 100644 --- a/tests/test_http_headers.py +++ b/tests/test_http_headers.py @@ -3,6 +3,7 @@ import copy from scrapy.http import Headers + class HeadersTest(unittest.TestCase): def assertSortedEqual(self, first, second, msg=None): @@ -85,9 +86,6 @@ class HeadersTest(unittest.TestCase): self.assertSortedEqual(h.items(), [(b'X-Forwarded-For', [b'ip1', b'ip2']), (b'Content-Type', [b'text/html'])]) - self.assertSortedEqual(h.iteritems(), - [(b'X-Forwarded-For', [b'ip1', b'ip2']), - (b'Content-Type', [b'text/html'])]) self.assertSortedEqual(h.values(), [b'ip2', b'text/html']) def test_update(self): diff --git a/tests/test_http_request.py b/tests/test_http_request.py index 952e208de..cc2cddda4 100644 --- a/tests/test_http_request.py +++ b/tests/test_http_request.py @@ -1,20 +1,13 @@ -# -*- coding: utf-8 -*- -import cgi import unittest import re import json +import xmlrpc.client import warnings +from unittest import mock +from urllib.parse import parse_qs, unquote_to_bytes, urlparse -import six -from six.moves import xmlrpc_client as xmlrpclib -from six.moves.urllib.parse import urlparse, parse_qs, unquote -if six.PY3: - from urllib.parse import unquote_to_bytes - -from scrapy.http import Request, FormRequest, XmlRpcRequest, JSONRequest, Headers, HtmlResponse -from scrapy.utils.python import to_bytes, to_native_str - -from tests import mock +from scrapy.http import Request, FormRequest, XmlRpcRequest, JsonRequest, Headers, HtmlResponse +from scrapy.utils.python import to_bytes, to_unicode class RequestTest(unittest.TestCase): @@ -25,7 +18,7 @@ class RequestTest(unittest.TestCase): default_meta = {} def test_init(self): - # Request requires url in the constructor + # Request requires url in the __init__ method self.assertRaises(Exception, self.request_class) # url argument must be basestring @@ -52,11 +45,13 @@ class RequestTest(unittest.TestCase): def test_url_no_scheme(self): self.assertRaises(ValueError, self.request_class, 'foo') + self.assertRaises(ValueError, self.request_class, '/foo/') + self.assertRaises(ValueError, self.request_class, '/foo:bar') def test_headers(self): # Different ways of setting headers attribute url = 'http://www.scrapy.org' - headers = {b'Accept':'gzip', b'Custom-Header':'nothing to tell you'} + headers = {b'Accept': 'gzip', b'Custom-Header': 'nothing to tell you'} r = self.request_class(url=url, headers=headers) p = self.request_class(url=url, headers=r.headers) @@ -67,7 +62,7 @@ class RequestTest(unittest.TestCase): # headers must not be unicode h = Headers({'key1': u'val1', u'key2': 'val2'}) h[u'newkey'] = u'newval' - for k, v in h.iteritems(): + for k, v in h.items(): self.assertIsInstance(k, bytes) for s in v: self.assertIsInstance(s, bytes) @@ -154,7 +149,7 @@ class RequestTest(unittest.TestCase): r2 = self.request_class(url="http://www.example.com/", body=b"") assert isinstance(r2.body, bytes) - self.assertEqual(r2.encoding, 'utf-8') # default encoding + self.assertEqual(r2.encoding, 'utf-8') # default encoding r3 = self.request_class(url="http://www.example.com/", body=u"Price: \xa3100", encoding='utf-8') assert isinstance(r3.body, bytes) @@ -249,25 +244,117 @@ class RequestTest(unittest.TestCase): self.assertRaises(AttributeError, setattr, r, 'url', 'http://example2.com') self.assertRaises(AttributeError, setattr, r, 'body', 'xxx') - def test_callback_is_callable(self): + def test_callback_and_errback(self): def a_function(): pass - r = self.request_class('http://example.com') - self.assertIsNone(r.callback) - r = self.request_class('http://example.com', a_function) - self.assertIs(r.callback, a_function) - with self.assertRaises(TypeError): - self.request_class('http://example.com', 'a_function') - def test_errback_is_callable(self): - def a_function(): - pass - r = self.request_class('http://example.com') - self.assertIsNone(r.errback) - r = self.request_class('http://example.com', a_function, errback=a_function) - self.assertIs(r.errback, a_function) + r1 = self.request_class('http://example.com') + self.assertIsNone(r1.callback) + self.assertIsNone(r1.errback) + + r2 = self.request_class('http://example.com', callback=a_function) + self.assertIs(r2.callback, a_function) + self.assertIsNone(r2.errback) + + r3 = self.request_class('http://example.com', errback=a_function) + self.assertIsNone(r3.callback) + self.assertIs(r3.errback, a_function) + + r4 = self.request_class( + url='http://example.com', + callback=a_function, + errback=a_function, + ) + self.assertIs(r4.callback, a_function) + self.assertIs(r4.errback, a_function) + + def test_callback_and_errback_type(self): with self.assertRaises(TypeError): - self.request_class('http://example.com', a_function, errback='a_function') + self.request_class('http://example.com', callback='a_function') + with self.assertRaises(TypeError): + self.request_class('http://example.com', errback='a_function') + with self.assertRaises(TypeError): + self.request_class( + url='http://example.com', + callback='a_function', + errback='a_function', + ) + + def test_from_curl(self): + # Note: more curated tests regarding curl conversion are in + # `test_utils_curl.py` + curl_command = ( + "curl 'http://httpbin.org/post' -X POST -H 'Cookie: _gauges_unique" + "_year=1; _gauges_unique=1; _gauges_unique_month=1; _gauges_unique" + "_hour=1; _gauges_unique_day=1' -H 'Origin: http://httpbin.org' -H" + " 'Accept-Encoding: gzip, deflate' -H 'Accept-Language: en-US,en;q" + "=0.9,ru;q=0.8,es;q=0.7' -H 'Upgrade-Insecure-Requests: 1' -H 'Use" + "r-Agent: Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTM" + "L, like Gecko) Ubuntu Chromium/62.0.3202.75 Chrome/62.0.3202.75 S" + "afari/537.36' -H 'Content-Type: application /x-www-form-urlencode" + "d' -H 'Accept: text/html,application/xhtml+xml,application/xml;q=" + "0.9,image/webp,image/apng,*/*;q=0.8' -H 'Cache-Control: max-age=0" + "' -H 'Referer: http://httpbin.org/forms/post' -H 'Connection: kee" + "p-alive' --data 'custname=John+Smith&custtel=500&custemail=jsmith" + "%40example.org&size=small&topping=cheese&topping=onion&delivery=1" + "2%3A15&comments=' --compressed" + ) + r = self.request_class.from_curl(curl_command) + self.assertEqual(r.method, "POST") + self.assertEqual(r.url, "http://httpbin.org/post") + self.assertEqual(r.body, + b"custname=John+Smith&custtel=500&custemail=jsmith%40" + b"example.org&size=small&topping=cheese&topping=onion" + b"&delivery=12%3A15&comments=") + self.assertEqual(r.cookies, { + '_gauges_unique_year': '1', + '_gauges_unique': '1', + '_gauges_unique_month': '1', + '_gauges_unique_hour': '1', + '_gauges_unique_day': '1' + }) + self.assertEqual(r.headers, { + b'Origin': [b'http://httpbin.org'], + b'Accept-Encoding': [b'gzip, deflate'], + b'Accept-Language': [b'en-US,en;q=0.9,ru;q=0.8,es;q=0.7'], + b'Upgrade-Insecure-Requests': [b'1'], + b'User-Agent': [b'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.' + b'36 (KHTML, like Gecko) Ubuntu Chromium/62.0.3202' + b'.75 Chrome/62.0.3202.75 Safari/537.36'], + b'Content-Type': [b'application /x-www-form-urlencoded'], + b'Accept': [b'text/html,application/xhtml+xml,application/xml;q=0.' + b'9,image/webp,image/apng,*/*;q=0.8'], + b'Cache-Control': [b'max-age=0'], + b'Referer': [b'http://httpbin.org/forms/post'], + b'Connection': [b'keep-alive']}) + + def test_from_curl_with_kwargs(self): + r = self.request_class.from_curl( + 'curl -X PATCH "http://example.org"', + method="POST", + meta={'key': 'value'} + ) + self.assertEqual(r.method, "POST") + self.assertEqual(r.meta, {"key": "value"}) + + def test_from_curl_ignore_unknown_options(self): + # By default: it works and ignores the unknown options: --foo and -z + with warnings.catch_warnings(): # avoid warning when executing tests + warnings.simplefilter('ignore') + r = self.request_class.from_curl( + 'curl -X DELETE "http://example.org" --foo -z', + ) + self.assertEqual(r.method, "DELETE") + + # If `ignore_unknon_options` is set to `False` it raises an error with + # the unknown options: --foo and -z + self.assertRaises( + ValueError, + lambda: self.request_class.from_curl( + 'curl -X PATCH "http://example.org" --foo -z', + ignore_unknown_options=False, + ), + ) class FormRequestTest(RequestTest): @@ -275,8 +362,8 @@ class FormRequestTest(RequestTest): request_class = FormRequest def assertQueryEqual(self, first, second, msg=None): - first = to_native_str(first).split("&") - second = to_native_str(second).split("&") + first = to_unicode(first).split("&") + second = to_unicode(second).split("&") return self.assertEqual(sorted(first), sorted(second), msg) def test_empty_formdata(self): @@ -424,7 +511,7 @@ class FormRequestTest(RequestTest): formdata=(('foo', 'bar'), ('foo', 'baz'))) self.assertEqual(urlparse(req.url).hostname, 'www.example.com') self.assertEqual(urlparse(req.url).query, 'foo=bar&foo=baz') - + def test_from_response_override_duplicate_form_key(self): response = _buildresponse( """<form action="get.php" method="POST"> @@ -551,8 +638,9 @@ class FormRequestTest(RequestTest): <input type="hidden" name="two" value="3"> <input type="submit" name="clickable2" value="clicked2"> </form>""") - req = self.request_class.from_response(response, formdata={'two': '2'}, \ - clickdata={'name': 'clickable2'}) + req = self.request_class.from_response( + response, formdata={'two': '2'}, clickdata={'name': 'clickable2'} + ) fs = _qs(req) self.assertEqual(fs[b'clickable2'], [b'clicked2']) self.assertFalse(b'clickable1' in fs, fs) @@ -581,7 +669,7 @@ class FormRequestTest(RequestTest): req = self.request_class.from_response(response, dont_click=True) fs = _qs(req) self.assertEqual(fs, {b'i1': [b'i1v'], b'i2': [b'i2v']}) - + def test_from_response_clickdata_does_not_ignore_image(self): response = _buildresponse( """<form> @@ -600,8 +688,9 @@ class FormRequestTest(RequestTest): <input type="hidden" name="one" value="clicked1"> <input type="hidden" name="two" value="clicked2"> </form>""") - req = self.request_class.from_response(response, \ - clickdata={u'name': u'clickable', u'value': u'clicked2'}) + req = self.request_class.from_response( + response, clickdata={u'name': u'clickable', u'value': u'clicked2'} + ) fs = _qs(req) self.assertEqual(fs[b'clickable'], [b'clicked2']) self.assertEqual(fs[b'one'], [b'clicked1']) @@ -615,8 +704,9 @@ class FormRequestTest(RequestTest): <input type="hidden" name="poundsign" value="\u00a3"> <input type="hidden" name="eurosign" value="\u20ac"> </form>""") - req = self.request_class.from_response(response, \ - clickdata={u'name': u'price in \u00a3'}) + req = self.request_class.from_response( + response, clickdata={u'name': u'price in \u00a3'} + ) fs = _qs(req, to_unicode=True) self.assertTrue(fs[u'price in \u00a3']) @@ -629,8 +719,9 @@ class FormRequestTest(RequestTest): <input type="hidden" name="yensign" value="\u00a5"> </form>""", encoding='latin1') - req = self.request_class.from_response(response, \ - clickdata={u'name': u'price in \u00a5'}) + req = self.request_class.from_response( + response, clickdata={u'name': u'price in \u00a5'} + ) fs = _qs(req, to_unicode=True, encoding='latin1') self.assertTrue(fs[u'price in \u00a5']) @@ -645,8 +736,9 @@ class FormRequestTest(RequestTest): <input type="hidden" name="field2" value="value2"> </form> """) - req = self.request_class.from_response(response, formname='form2', \ - clickdata={u'name': u'clickable'}) + req = self.request_class.from_response( + response, formname='form2', clickdata={u'name': u'clickable'} + ) fs = _qs(req) self.assertEqual(fs[b'clickable'], [b'clicked2']) self.assertEqual(fs[b'field2'], [b'value2']) @@ -654,8 +746,9 @@ class FormRequestTest(RequestTest): def test_from_response_override_clickable(self): response = _buildresponse('''<form><input type="submit" name="clickme" value="one"> </form>''') - req = self.request_class.from_response(response, \ - formdata={'clickme': 'two'}, clickdata={'name': 'clickme'}) + req = self.request_class.from_response( + response, formdata={'clickme': 'two'}, clickdata={'name': 'clickme'} + ) fs = _qs(req) self.assertEqual(fs[b'clickme'], [b'two']) @@ -740,7 +833,7 @@ class FormRequestTest(RequestTest): <input type="hidden" name="one" value="1"> <input type="hidden" name="two" value="2"> </form>""") - r1 = self.request_class.from_response(response, formdata={'two':'3'}) + r1 = self.request_class.from_response(response, formdata={'two': '3'}) self.assertEqual(r1.method, 'POST') self.assertEqual(r1.headers['Content-type'], b'application/x-www-form-urlencoded') fs = _qs(r1) @@ -782,7 +875,7 @@ class FormRequestTest(RequestTest): <form name="form2" action="post.php" method="POST"> <input type="hidden" name="two" value="2"> </form>""") - self.assertRaises(IndexError, self.request_class.from_response, \ + self.assertRaises(IndexError, self.request_class.from_response, response, formname="form3", formnumber=2) def test_from_response_formid_exists(self): @@ -836,7 +929,7 @@ class FormRequestTest(RequestTest): <form id="form2" name="form2" action="post.php" method="POST"> <input type="hidden" name="two" value="2"> </form>""") - self.assertRaises(IndexError, self.request_class.from_response, \ + self.assertRaises(IndexError, self.request_class.from_response, response, formid="form3", formnumber=2) def test_from_response_select(self): @@ -988,8 +1081,7 @@ class FormRequestTest(RequestTest): self.assertEqual(fs, {}) xpath = u"//form[@name='\u03b1']" - encoded = xpath if six.PY3 else xpath.encode('unicode_escape') - self.assertRaisesRegex(ValueError, re.escape(encoded), + self.assertRaisesRegex(ValueError, re.escape(xpath), self.request_class.from_response, response, formxpath=xpath) @@ -1132,10 +1224,7 @@ def _qs(req, encoding='utf-8', to_unicode=False): qs = req.body else: qs = req.url.partition('?')[2] - if six.PY2: - uqs = unquote(to_native_str(qs, encoding)) - elif six.PY3: - uqs = unquote_to_bytes(qs) + uqs = unquote_to_bytes(qs) if to_unicode: uqs = uqs.decode(encoding) return parse_qs(uqs, True) @@ -1151,7 +1240,7 @@ class XmlRpcRequestTest(RequestTest): r = self.request_class('http://scrapytest.org/rpc2', **kwargs) self.assertEqual(r.headers[b'Content-Type'], b'text/xml') self.assertEqual(r.body, - to_bytes(xmlrpclib.dumps(**kwargs), + to_bytes(xmlrpc.client.dumps(**kwargs), encoding=kwargs.get('encoding', 'utf-8'))) self.assertEqual(r.method, 'POST') self.assertEqual(r.encoding, kwargs.get('encoding', 'utf-8')) @@ -1170,14 +1259,14 @@ class XmlRpcRequestTest(RequestTest): self._test_request(params=(u'pas£',), encoding='latin1') -class JSONRequestTest(RequestTest): - request_class = JSONRequest +class JsonRequestTest(RequestTest): + request_class = JsonRequest default_method = 'GET' default_headers = {b'Content-Type': [b'application/json'], b'Accept': [b'application/json, text/javascript, */*; q=0.01']} def setUp(self): warnings.simplefilter("always") - super(JSONRequestTest, self).setUp() + super(JsonRequestTest, self).setUp() def test_data(self): r1 = self.request_class(url="http://www.example.com/") @@ -1331,7 +1420,7 @@ class JSONRequestTest(RequestTest): def tearDown(self): warnings.resetwarnings() - super(JSONRequestTest, self).tearDown() + super(JsonRequestTest, self).tearDown() if __name__ == "__main__": diff --git a/tests/test_http_response.py b/tests/test_http_response.py index cd5c3486e..4c1b2afc3 100644 --- a/tests/test_http_response.py +++ b/tests/test_http_response.py @@ -1,13 +1,12 @@ # -*- coding: utf-8 -*- import unittest -import six from w3lib.encoding import resolve_encoding from scrapy.http import (Request, Response, TextResponse, HtmlResponse, XmlResponse, Headers) from scrapy.selector import Selector -from scrapy.utils.python import to_native_str +from scrapy.utils.python import to_unicode from scrapy.exceptions import NotSupported from scrapy.link import Link from tests import get_testdata @@ -21,8 +20,7 @@ class BaseResponseTest(unittest.TestCase): # Response requires url in the consturctor self.assertRaises(Exception, self.response_class) self.assertTrue(isinstance(self.response_class('http://example.com/'), self.response_class)) - if not six.PY2: - self.assertRaises(TypeError, self.response_class, b"http://example.com") + self.assertRaises(TypeError, self.response_class, b"http://example.com") # body can be str or None self.assertTrue(isinstance(self.response_class('http://example.com/', body=b''), self.response_class)) self.assertTrue(isinstance(self.response_class('http://example.com/', body=b'body'), self.response_class)) @@ -103,7 +101,7 @@ class BaseResponseTest(unittest.TestCase): self.assertEqual(r4.flags, []) def _assert_response_values(self, response, encoding, body): - if isinstance(body, six.text_type): + if isinstance(body, str): body_unicode = body body_bytes = body.encode(encoding) else: @@ -111,7 +109,7 @@ class BaseResponseTest(unittest.TestCase): body_bytes = body assert isinstance(response.body, bytes) - assert isinstance(response.text, six.text_type) + assert isinstance(response.text, str) self._assert_response_encoding(response, encoding) self.assertEqual(response.body, body_bytes) self.assertEqual(response.body_as_unicode(), body_unicode) @@ -143,6 +141,8 @@ class BaseResponseTest(unittest.TestCase): r.css('body') r.xpath('//body') + # Response.follow + def test_follow_url_absolute(self): self._assert_followed_url('http://foo.example.com', 'http://foo.example.com') @@ -166,6 +166,72 @@ class BaseResponseTest(unittest.TestCase): def test_follow_whitespace_link(self): self._assert_followed_url(Link('http://example.com/foo '), 'http://example.com/foo%20') + + # Response.follow_all + + def test_follow_all_absolute(self): + url_list = ['http://example.org', 'http://www.example.org', + 'http://example.com', 'http://www.example.com'] + self._assert_followed_all_urls(url_list, url_list) + + def test_follow_all_relative(self): + relative = ['foo', 'bar', 'foo/bar', 'bar/foo'] + absolute = [ + 'http://example.com/foo', + 'http://example.com/bar', + 'http://example.com/foo/bar', + 'http://example.com/bar/foo', + ] + self._assert_followed_all_urls(relative, absolute) + + def test_follow_all_links(self): + absolute = [ + 'http://example.com/foo', + 'http://example.com/bar', + 'http://example.com/foo/bar', + 'http://example.com/bar/foo', + ] + links = map(Link, absolute) + self._assert_followed_all_urls(links, absolute) + + def test_follow_all_invalid(self): + r = self.response_class("http://example.com") + if self.response_class == Response: + with self.assertRaises(TypeError): + list(r.follow_all(urls=None)) + with self.assertRaises(TypeError): + list(r.follow_all(urls=12345)) + with self.assertRaises(ValueError): + list(r.follow_all(urls=[None])) + else: + with self.assertRaises(ValueError): + list(r.follow_all(urls=None)) + with self.assertRaises(TypeError): + list(r.follow_all(urls=12345)) + with self.assertRaises(ValueError): + list(r.follow_all(urls=[None])) + + def test_follow_all_whitespace(self): + relative = ['foo ', 'bar ', 'foo/bar ', 'bar/foo '] + absolute = [ + 'http://example.com/foo%20', + 'http://example.com/bar%20', + 'http://example.com/foo/bar%20', + 'http://example.com/bar/foo%20', + ] + self._assert_followed_all_urls(relative, absolute) + + def test_follow_all_whitespace_links(self): + absolute = [ + 'http://example.com/foo ', + 'http://example.com/bar ', + 'http://example.com/foo/bar ', + 'http://example.com/bar/foo ', + ] + links = map(Link, absolute) + expected = [u.replace(' ', '%20') for u in absolute] + self._assert_followed_all_urls(links, expected) + def _assert_followed_url(self, follow_obj, target_url, response=None): if response is None: response = self._links_response() @@ -173,8 +239,21 @@ class BaseResponseTest(unittest.TestCase): self.assertEqual(req.url, target_url) return req + def _assert_followed_all_urls(self, follow_obj, target_urls, response=None): + if response is None: + response = self._links_response() + followed = response.follow_all(follow_obj) + for req, target in zip(followed, target_urls): + self.assertEqual(req.url, target) + yield req + def _links_response(self): - body = get_testdata('link_extractor', 'sgml_linkextractor.html') + body = get_testdata('link_extractor', 'linkextractor.html') + resp = self.response_class('http://example.com/index', body=body) + return resp + + def _links_response_no_href(self): + body = get_testdata('link_extractor', 'linkextractor_no_href.html') resp = self.response_class('http://example.com/index', body=body) return resp @@ -205,11 +284,11 @@ class TextResponseTest(BaseResponseTest): assert isinstance(resp.url, str) resp = self.response_class(url=u"http://www.example.com/price/\xa3", encoding='utf-8') - self.assertEqual(resp.url, to_native_str(b'http://www.example.com/price/\xc2\xa3')) + self.assertEqual(resp.url, to_unicode(b'http://www.example.com/price/\xc2\xa3')) resp = self.response_class(url=u"http://www.example.com/price/\xa3", encoding='latin-1') self.assertEqual(resp.url, 'http://www.example.com/price/\xa3') resp = self.response_class(u"http://www.example.com/price/\xa3", headers={"Content-type": ["text/html; charset=utf-8"]}) - self.assertEqual(resp.url, to_native_str(b'http://www.example.com/price/\xc2\xa3')) + self.assertEqual(resp.url, to_unicode(b'http://www.example.com/price/\xc2\xa3')) resp = self.response_class(u"http://www.example.com/price/\xa3", headers={"Content-type": ["text/html; charset=iso-8859-1"]}) self.assertEqual(resp.url, 'http://www.example.com/price/\xa3') @@ -221,11 +300,11 @@ class TextResponseTest(BaseResponseTest): r1 = self.response_class('http://www.example.com', body=original_string, encoding='cp1251') # check body_as_unicode - self.assertTrue(isinstance(r1.body_as_unicode(), six.text_type)) + self.assertTrue(isinstance(r1.body_as_unicode(), str)) self.assertEqual(r1.body_as_unicode(), unicode_string) # check response.text - self.assertTrue(isinstance(r1.text, six.text_type)) + self.assertTrue(isinstance(r1.text, str)) self.assertEqual(r1.text, unicode_string) def test_encoding(self): @@ -318,8 +397,8 @@ class TextResponseTest(BaseResponseTest): assert u'SUFFIX' in r.text, repr(r.text) # Do not destroy html tags due to encoding bugs - r = self.response_class("http://example.com", encoding='utf-8', \ - body=b'\xf0<span>value</span>') + r = self.response_class("http://example.com", encoding='utf-8', + body=b'\xf0<span>value</span>') assert u'<span>value</span>' in r.text, repr(r.text) # FIXME: This test should pass once we stop using BeautifulSoup's UnicodeDammit in TextResponse @@ -483,6 +562,53 @@ class TextResponseTest(BaseResponseTest): ) self.assertEqual(req.encoding, 'cp1251') + def test_follow_all_css(self): + expected = [ + 'http://example.com/sample3.html', + 'http://example.com/innertag.html', + ] + response = self._links_response() + extracted = [r.url for r in response.follow_all(css='a[href*="example.com"]')] + self.assertEqual(expected, extracted) + + def test_follow_all_css_skip_invalid(self): + expected = [ + 'http://example.com/page/1/', + 'http://example.com/page/3/', + 'http://example.com/page/4/', + ] + response = self._links_response_no_href() + extracted1 = [r.url for r in response.follow_all(css='.pagination a')] + self.assertEqual(expected, extracted1) + extracted2 = [r.url for r in response.follow_all(response.css('.pagination a'))] + self.assertEqual(expected, extracted2) + + def test_follow_all_xpath(self): + expected = [ + 'http://example.com/sample3.html', + 'http://example.com/innertag.html', + ] + response = self._links_response() + extracted = response.follow_all(xpath='//a[contains(@href, "example.com")]') + self.assertEqual(expected, [r.url for r in extracted]) + + def test_follow_all_xpath_skip_invalid(self): + expected = [ + 'http://example.com/page/1/', + 'http://example.com/page/3/', + 'http://example.com/page/4/', + ] + response = self._links_response_no_href() + extracted1 = [r.url for r in response.follow_all(xpath='//div[@id="pagination"]/a')] + self.assertEqual(expected, extracted1) + extracted2 = [r.url for r in response.follow_all(response.xpath('//div[@id="pagination"]/a'))] + self.assertEqual(expected, extracted2) + + def test_follow_all_too_many_arguments(self): + response = self._links_response() + with self.assertRaises(ValueError): + response.follow_all(css='a[href*="example.com"]', xpath='//a[contains(@href, "example.com")]') + class HtmlResponseTest(TextResponseTest): @@ -534,7 +660,7 @@ class XmlResponseTest(TextResponseTest): r2 = self.response_class("http://www.example.com", body=body) self._assert_response_values(r2, 'iso-8859-1', body) - # make sure replace() preserves the explicit encoding passed in the constructor + # make sure replace() preserves the explicit encoding passed in the __init__ method body = b"""<?xml version="1.0" encoding="iso-8859-1"?><xml></xml>""" r3 = self.response_class("http://www.example.com", body=body, encoding='utf-8') body2 = b"New body" diff --git a/tests/test_item.py b/tests/test_item.py index 010d3b141..30463a0f5 100644 --- a/tests/test_item.py +++ b/tests/test_item.py @@ -1,10 +1,10 @@ import sys import unittest +from unittest import mock +from warnings import catch_warnings -import six - -from scrapy.item import ABCMeta, Item, ItemMeta, Field -from tests import mock +from scrapy.exceptions import ScrapyDeprecationWarning +from scrapy.item import ABCMeta, DictItem, Field, Item, ItemMeta PY36_PLUS = (sys.version_info.major >= 3) and (sys.version_info.minor >= 6) @@ -60,12 +60,8 @@ class ItemTest(unittest.TestCase): i['number'] = 123 itemrepr = repr(i) - if six.PY2: - self.assertEqual(itemrepr, - "{'name': u'John Doe', 'number': 123}") - else: - self.assertEqual(itemrepr, - "{'name': 'John Doe', 'number': 123}") + self.assertEqual(itemrepr, + "{'name': 'John Doe', 'number': 123}") i2 = eval(itemrepr) self.assertEqual(i2['name'], 'John Doe') @@ -243,7 +239,7 @@ class ItemTest(unittest.TestCase): def test_copy(self): class TestItem(Item): name = Field() - item = TestItem({'name':'lower'}) + item = TestItem({'name': 'lower'}) copied_item = item.copy() self.assertNotEqual(id(item), id(copied_item)) copied_item['name'] = copied_item['name'].upper() @@ -257,6 +253,17 @@ class ItemTest(unittest.TestCase): item['tags'].append('tag2') assert item['tags'] != copied_item['tags'] + def test_dictitem_deprecation_warning(self): + """Make sure the DictItem deprecation warning is not issued for + Item""" + with catch_warnings(record=True) as warnings: + item = Item() + self.assertEqual(len(warnings), 0) + class SubclassedItem(Item): + pass + subclassed_item = SubclassedItem() + self.assertEqual(len(warnings), 0) + class ItemMetaTest(unittest.TestCase): @@ -293,7 +300,7 @@ class ItemMetaTest(unittest.TestCase): class ItemMetaClassCellRegression(unittest.TestCase): def test_item_meta_classcell_regression(self): - class MyItem(six.with_metaclass(ItemMeta, Item)): + class MyItem(Item, metaclass=ItemMeta): def __init__(self, *args, **kwargs): # This call to super() trigger the __classcell__ propagation # requirement. When not done properly raises an error: @@ -302,5 +309,20 @@ class ItemMetaClassCellRegression(unittest.TestCase): super(MyItem, self).__init__(*args, **kwargs) +class DictItemTest(unittest.TestCase): + + def test_deprecation_warning(self): + with catch_warnings(record=True) as warnings: + dict_item = DictItem() + self.assertEqual(len(warnings), 1) + self.assertEqual(warnings[0].category, ScrapyDeprecationWarning) + with catch_warnings(record=True) as warnings: + class SubclassedDictItem(DictItem): + pass + subclassed_dict_item = SubclassedDictItem() + self.assertEqual(len(warnings), 1) + self.assertEqual(warnings[0].category, ScrapyDeprecationWarning) + + if __name__ == "__main__": unittest.main() diff --git a/tests/test_link.py b/tests/test_link.py index 955430b37..e0f1efffa 100644 --- a/tests/test_link.py +++ b/tests/test_link.py @@ -1,6 +1,4 @@ import unittest -import warnings -import six from scrapy.link import Link @@ -45,13 +43,6 @@ class LinkTest(unittest.TestCase): l2 = eval(repr(l1)) self._assert_same_links(l1, l2) - def test_non_str_url_py2(self): - if six.PY2: - with warnings.catch_warnings(record=True) as w: - link = Link(u"http://www.example.com/\xa3") - self.assertIsInstance(link.url, str) - self.assertEqual(link.url, b'http://www.example.com/\xc2\xa3') - assert len(w) == 1, "warning not issued" - else: - with self.assertRaises(TypeError): - Link(b"http://www.example.com/\xc2\xa3") + def test_bytes_url(self): + with self.assertRaises(TypeError): + Link(b"http://www.example.com/\xc2\xa3") diff --git a/tests/test_linkextractors.py b/tests/test_linkextractors.py index d96e259f6..38fb8fb4a 100644 --- a/tests/test_linkextractors.py +++ b/tests/test_linkextractors.py @@ -1,10 +1,13 @@ import re import unittest +from warnings import catch_warnings import pytest +from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.http import HtmlResponse, XmlResponse from scrapy.link import Link +from scrapy.linkextractors import FilteringLinkExtractor from scrapy.linkextractors.lxmlhtml import LxmlLinkExtractor from tests import get_testdata @@ -16,7 +19,7 @@ class Base: escapes_whitespace = False def setUp(self): - body = get_testdata('link_extractor', 'sgml_linkextractor.html') + body = get_testdata('link_extractor', 'linkextractor.html') self.response = HtmlResponse(url='http://example.com/index', body=body) def test_urls_type(self): @@ -322,7 +325,7 @@ class Base: Link(url=page4_url, text=u'href with whitespaces'), ]) - lx = self.extractor_cls(attrs=("href","src"), tags=("a","area","img"), deny_extensions=()) + lx = self.extractor_cls(attrs=("href", "src"), tags=("a", "area", "img"), deny_extensions=()) self.assertEqual(lx.extract_links(self.response), [ Link(url='http://example.com/sample1.html', text=u''), Link(url='http://example.com/sample2.html', text=u'sample 2'), @@ -360,7 +363,7 @@ class Base: Link(url='http://example.com/sample2.html', text=u'sample 2'), ]) - lx = self.extractor_cls(tags=("a","img"), attrs=("href", "src"), deny_extensions=()) + lx = self.extractor_cls(tags=("a", "img"), attrs=("href", "src"), deny_extensions=()) self.assertEqual(lx.extract_links(response), [ Link(url='http://example.com/sample2.html', text=u'sample 2'), Link(url='http://example.com/sample2.jpg', text=u''), @@ -506,3 +509,32 @@ class LxmlLinkExtractorTestCase(Base.LinkExtractorTestCase): @pytest.mark.xfail def test_restrict_xpaths_with_html_entities(self): super(LxmlLinkExtractorTestCase, self).test_restrict_xpaths_with_html_entities() + + def test_filteringlinkextractor_deprecation_warning(self): + """Make sure the FilteringLinkExtractor deprecation warning is not + issued for LxmlLinkExtractor""" + with catch_warnings(record=True) as warnings: + LxmlLinkExtractor() + self.assertEqual(len(warnings), 0) + + class SubclassedLxmlLinkExtractor(LxmlLinkExtractor): + pass + + SubclassedLxmlLinkExtractor() + self.assertEqual(len(warnings), 0) + + +class FilteringLinkExtractorTest(unittest.TestCase): + + def test_deprecation_warning(self): + args = [None] * 10 + with catch_warnings(record=True) as warnings: + FilteringLinkExtractor(*args) + self.assertEqual(len(warnings), 1) + self.assertEqual(warnings[0].category, ScrapyDeprecationWarning) + with catch_warnings(record=True) as warnings: + class SubclassedFilteringLinkExtractor(FilteringLinkExtractor): + pass + SubclassedFilteringLinkExtractor(*args) + self.assertEqual(len(warnings), 1) + self.assertEqual(warnings[0].category, ScrapyDeprecationWarning) diff --git a/tests/test_linkextractors_deprecated.py b/tests/test_linkextractors_deprecated.py deleted file mode 100644 index 1366971be..000000000 --- a/tests/test_linkextractors_deprecated.py +++ /dev/null @@ -1,233 +0,0 @@ -# -*- coding: utf-8 -*- -import unittest -from scrapy.linkextractors.regex import RegexLinkExtractor -from scrapy.http import HtmlResponse -from scrapy.link import Link -from scrapy.linkextractors.htmlparser import HtmlParserLinkExtractor -from scrapy.linkextractors.sgml import SgmlLinkExtractor, BaseSgmlLinkExtractor -from tests import get_testdata - -from tests.test_linkextractors import Base - - -class BaseSgmlLinkExtractorTestCase(unittest.TestCase): - # XXX: should we move some of these tests to base link extractor tests? - - def test_basic(self): - html = """<html><head><title>Page title<title> - <body><p><a href="item/12.html">Item 12</a></p> - <p><a href="/about.html">About us</a></p> - <img src="/logo.png" alt="Company logo (not a link)" /> - <p><a href="../othercat.html">Other category</a></p> - <p><a href="/">>></a></p> - <p><a href="/" /></p> - </body></html>""" - response = HtmlResponse("http://example.org/somepage/index.html", body=html) - - lx = BaseSgmlLinkExtractor() # default: tag=a, attr=href - self.assertEqual(lx.extract_links(response), - [Link(url='http://example.org/somepage/item/12.html', text='Item 12'), - Link(url='http://example.org/about.html', text='About us'), - Link(url='http://example.org/othercat.html', text='Other category'), - Link(url='http://example.org/', text='>>'), - Link(url='http://example.org/', text='')]) - - def test_base_url(self): - html = """<html><head><title>Page title<title><base href="http://otherdomain.com/base/" /> - <body><p><a href="item/12.html">Item 12</a></p> - </body></html>""" - response = HtmlResponse("http://example.org/somepage/index.html", body=html) - - lx = BaseSgmlLinkExtractor() # default: tag=a, attr=href - self.assertEqual(lx.extract_links(response), - [Link(url='http://otherdomain.com/base/item/12.html', text='Item 12')]) - - # base url is an absolute path and relative to host - html = """<html><head><title>Page title<title><base href="/" /> - <body><p><a href="item/12.html">Item 12</a></p></body></html>""" - response = HtmlResponse("https://example.org/somepage/index.html", body=html) - self.assertEqual(lx.extract_links(response), - [Link(url='https://example.org/item/12.html', text='Item 12')]) - - # base url has no scheme - html = """<html><head><title>Page title<title><base href="//noschemedomain.com/path/to/" /> - <body><p><a href="item/12.html">Item 12</a></p></body></html>""" - response = HtmlResponse("https://example.org/somepage/index.html", body=html) - self.assertEqual(lx.extract_links(response), - [Link(url='https://noschemedomain.com/path/to/item/12.html', text='Item 12')]) - - def test_link_text_wrong_encoding(self): - html = """<body><p><a href="item/12.html">Wrong: \xed</a></p></body></html>""" - response = HtmlResponse("http://www.example.com", body=html, encoding='utf-8') - lx = BaseSgmlLinkExtractor() - self.assertEqual(lx.extract_links(response), [ - Link(url='http://www.example.com/item/12.html', text=u'Wrong: \ufffd'), - ]) - - def test_extraction_encoding(self): - body = get_testdata('link_extractor', 'linkextractor_noenc.html') - response_utf8 = HtmlResponse(url='http://example.com/utf8', body=body, headers={'Content-Type': ['text/html; charset=utf-8']}) - response_noenc = HtmlResponse(url='http://example.com/noenc', body=body) - body = get_testdata('link_extractor', 'linkextractor_latin1.html') - response_latin1 = HtmlResponse(url='http://example.com/latin1', body=body) - - lx = BaseSgmlLinkExtractor() - self.assertEqual(lx.extract_links(response_utf8), [ - Link(url='http://example.com/sample_%C3%B1.html', text=''), - Link(url='http://example.com/sample_%E2%82%AC.html', text='sample \xe2\x82\xac text'.decode('utf-8')), - ]) - - self.assertEqual(lx.extract_links(response_noenc), [ - Link(url='http://example.com/sample_%C3%B1.html', text=''), - Link(url='http://example.com/sample_%E2%82%AC.html', text='sample \xe2\x82\xac text'.decode('utf-8')), - ]) - - # document encoding does not affect URL path component, only query part - # >>> u'sample_ñ.html'.encode('utf8') - # b'sample_\xc3\xb1.html' - # >>> u"sample_á.html".encode('utf8') - # b'sample_\xc3\xa1.html' - # >>> u"sample_ö.html".encode('utf8') - # b'sample_\xc3\xb6.html' - # >>> u"£32".encode('latin1') - # b'\xa332' - # >>> u"µ".encode('latin1') - # b'\xb5' - self.assertEqual(lx.extract_links(response_latin1), [ - Link(url='http://example.com/sample_%C3%B1.html', text=''), - Link(url='http://example.com/sample_%C3%A1.html', text='sample \xe1 text'.decode('latin1')), - Link(url='http://example.com/sample_%C3%B6.html?price=%A332&%B5=unit', text=''), - ]) - - def test_matches(self): - url1 = 'http://lotsofstuff.com/stuff1/index' - url2 = 'http://evenmorestuff.com/uglystuff/index' - - lx = BaseSgmlLinkExtractor() - self.assertEqual(lx.matches(url1), True) - self.assertEqual(lx.matches(url2), True) - - -class HtmlParserLinkExtractorTestCase(unittest.TestCase): - - def setUp(self): - body = get_testdata('link_extractor', 'sgml_linkextractor.html') - self.response = HtmlResponse(url='http://example.com/index', body=body) - - def test_extraction(self): - # Default arguments - lx = HtmlParserLinkExtractor() - self.assertEqual(lx.extract_links(self.response), [ - Link(url='http://example.com/sample2.html', text=u'sample 2'), - Link(url='http://example.com/sample3.html', text=u'sample 3 text'), - Link(url='http://example.com/sample3.html', text=u'sample 3 repetition'), - Link(url='http://example.com/sample3.html#foo', text=u'sample 3 repetition with fragment'), - Link(url='http://www.google.com/something', text=u''), - Link(url='http://example.com/innertag.html', text=u'inner tag'), - Link(url='http://example.com/page%204.html', text=u'href with whitespaces'), - ]) - - def test_link_wrong_href(self): - html = """ - <a href="http://example.org/item1.html">Item 1</a> - <a href="http://[example.org/item2.html">Item 2</a> - <a href="http://example.org/item3.html">Item 3</a> - """ - response = HtmlResponse("http://example.org/index.html", body=html) - lx = HtmlParserLinkExtractor() - self.assertEqual([link for link in lx.extract_links(response)], [ - Link(url='http://example.org/item1.html', text=u'Item 1', nofollow=False), - Link(url='http://example.org/item3.html', text=u'Item 3', nofollow=False), - ]) - - -class SgmlLinkExtractorTestCase(Base.LinkExtractorTestCase): - extractor_cls = SgmlLinkExtractor - escapes_whitespace = True - - def test_deny_extensions(self): - html = """<a href="page.html">asd</a> and <a href="photo.jpg">""" - response = HtmlResponse("http://example.org/", body=html) - lx = SgmlLinkExtractor(deny_extensions="jpg") - self.assertEqual(lx.extract_links(response), [ - Link(url='http://example.org/page.html', text=u'asd'), - ]) - - def test_attrs_sgml(self): - html = """<html><area href="sample1.html"></area> - <a ref="sample2.html">sample text 2</a></html>""" - response = HtmlResponse("http://example.com/index.html", body=html) - lx = SgmlLinkExtractor(attrs="href") - self.assertEqual(lx.extract_links(response), [ - Link(url='http://example.com/sample1.html', text=u''), - ]) - - def test_link_nofollow(self): - html = """ - <a href="page.html?action=print" rel="nofollow">Printer-friendly page</a> - <a href="about.html">About us</a> - <a href="http://google.com/something" rel="external nofollow">Something</a> - """ - response = HtmlResponse("http://example.org/page.html", body=html) - lx = SgmlLinkExtractor() - self.assertEqual([link for link in lx.extract_links(response)], [ - Link(url='http://example.org/page.html?action=print', text=u'Printer-friendly page', nofollow=True), - Link(url='http://example.org/about.html', text=u'About us', nofollow=False), - Link(url='http://google.com/something', text=u'Something', nofollow=True), - ]) - - -class RegexLinkExtractorTestCase(unittest.TestCase): - # XXX: RegexLinkExtractor is not deprecated yet, but it must be rewritten - # not to depend on SgmlLinkExractor. Its speed is also much worse - # than it should be. - - def setUp(self): - body = get_testdata('link_extractor', 'sgml_linkextractor.html') - self.response = HtmlResponse(url='http://example.com/index', body=body) - - def test_extraction(self): - # Default arguments - lx = RegexLinkExtractor() - self.assertEqual(lx.extract_links(self.response), - [Link(url='http://example.com/sample2.html', text=u'sample 2'), - Link(url='http://example.com/sample3.html', text=u'sample 3 text'), - Link(url='http://example.com/sample3.html#foo', text=u'sample 3 repetition with fragment'), - Link(url='http://www.google.com/something', text=u''), - Link(url='http://example.com/innertag.html', text=u'inner tag'),]) - - def test_link_wrong_href(self): - html = """ - <a href="http://example.org/item1.html">Item 1</a> - <a href="http://[example.org/item2.html">Item 2</a> - <a href="http://example.org/item3.html">Item 3</a> - """ - response = HtmlResponse("http://example.org/index.html", body=html) - lx = RegexLinkExtractor() - self.assertEqual([link for link in lx.extract_links(response)], [ - Link(url='http://example.org/item1.html', text=u'Item 1', nofollow=False), - Link(url='http://example.org/item3.html', text=u'Item 3', nofollow=False), - ]) - - def test_html_base_href(self): - html = """ - <html> - <head> - <base href="http://b.com/"> - </head> - <body> - <a href="test.html"></a> - </body> - </html> - """ - response = HtmlResponse("http://a.com/", body=html) - lx = RegexLinkExtractor() - self.assertEqual([link for link in lx.extract_links(response)], [ - Link(url='http://b.com/test.html', text=u'', nofollow=False), - ]) - - @unittest.expectedFailure - def test_extraction(self): - # RegexLinkExtractor doesn't parse URLs with leading/trailing - # whitespaces correctly. - super(RegexLinkExtractorTestCase, self).test_extraction() diff --git a/tests/test_loader.py b/tests/test_loader.py index 2725b001a..579a85ff6 100644 --- a/tests/test_loader.py +++ b/tests/test_loader.py @@ -1,13 +1,13 @@ -import unittest -import six from functools import partial +import unittest -from scrapy.loader import ItemLoader -from scrapy.loader.processors import Join, Identity, TakeFirst, \ - Compose, MapCompose, SelectJmes -from scrapy.item import Item, Field -from scrapy.selector import Selector from scrapy.http import HtmlResponse +from scrapy.item import Item, Field +from scrapy.loader import ItemLoader +from scrapy.loader.processors import (Compose, Identity, Join, + MapCompose, SelectJmes, TakeFirst) +from scrapy.selector import Selector + # test items class NameItem(Item): @@ -61,7 +61,7 @@ class BasicItemLoaderTest(unittest.TestCase): il.add_value('name', u'marta') item = il.load_item() assert item is i - self.assertEqual(item['summary'], u'lala') + self.assertEqual(item['summary'], [u'lala']) self.assertEqual(item['name'], [u'marta']) def test_load_item_using_custom_loader(self): @@ -155,7 +155,7 @@ class BasicItemLoaderTest(unittest.TestCase): def test_get_value(self): il = NameItemLoader() - self.assertEqual(u'FOO', il.get_value([u'foo', u'bar'], TakeFirst(), six.text_type.upper)) + self.assertEqual(u'FOO', il.get_value([u'foo', u'bar'], TakeFirst(), str.upper)) self.assertEqual([u'foo', u'bar'], il.get_value([u'name:foo', u'name:bar'], re=u'name:(.*)$')) self.assertEqual(u'foo', il.get_value([u'name:foo', u'name:bar'], TakeFirst(), re=u'name:(.*)$')) @@ -256,7 +256,7 @@ class BasicItemLoaderTest(unittest.TestCase): def test_extend_custom_input_processors(self): class ChildItemLoader(TestItemLoader): - name_in = MapCompose(TestItemLoader.name_in, six.text_type.swapcase) + name_in = MapCompose(TestItemLoader.name_in, str.swapcase) il = ChildItemLoader() il.add_value('name', u'marta') @@ -264,7 +264,7 @@ class BasicItemLoaderTest(unittest.TestCase): def test_extend_default_input_processors(self): class ChildDefaultedItemLoader(DefaultedItemLoader): - name_in = MapCompose(DefaultedItemLoader.default_input_processor, six.text_type.swapcase) + name_in = MapCompose(DefaultedItemLoader.default_input_processor, str.swapcase) il = ChildDefaultedItemLoader() il.add_value('name', u'marta') @@ -419,43 +419,6 @@ class BasicItemLoaderTest(unittest.TestCase): self.assertEqual(item['url'], u'rabbit.hole') self.assertEqual(item['summary'], u'rabbithole') - def test_create_item_from_dict(self): - class TestItem(Item): - title = Field() - - class TestItemLoader(ItemLoader): - default_item_class = TestItem - - input_item = {'title': 'Test item title 1'} - il = TestItemLoader(item=input_item) - # Getting output value mustn't remove value from item - self.assertEqual(il.load_item(), { - 'title': 'Test item title 1', - }) - self.assertEqual(il.get_output_value('title'), 'Test item title 1') - self.assertEqual(il.load_item(), { - 'title': 'Test item title 1', - }) - - input_item = {'title': 'Test item title 2'} - il = TestItemLoader(item=input_item) - # Values from dict must be added to item _values - self.assertEqual(il._values.get('title'), 'Test item title 2') - - input_item = {'title': [u'Test item title 3', u'Test item 4']} - il = TestItemLoader(item=input_item) - # Same rules must work for lists - self.assertEqual(il._values.get('title'), - [u'Test item title 3', u'Test item 4']) - self.assertEqual(il.load_item(), { - 'title': [u'Test item title 3', u'Test item 4'], - }) - self.assertEqual(il.get_output_value('title'), - [u'Test item title 3', u'Test item 4']) - self.assertEqual(il.load_item(), { - 'title': [u'Test item title 3', u'Test item 4'], - }) - def test_error_input_processor(self): class TestItem(Item): name = Field() @@ -493,6 +456,220 @@ class BasicItemLoaderTest(unittest.TestCase): [u'marta', u'other'], Compose(float)) +class InitializationTestMixin(object): + + item_class = None + + def test_keep_single_value(self): + """Loaded item should contain values from the initial item""" + input_item = self.item_class(name='foo') + il = ItemLoader(item=input_item) + loaded_item = il.load_item() + self.assertIsInstance(loaded_item, self.item_class) + self.assertEqual(dict(loaded_item), {'name': ['foo']}) + + def test_keep_list(self): + """Loaded item should contain values from the initial item""" + input_item = self.item_class(name=['foo', 'bar']) + il = ItemLoader(item=input_item) + loaded_item = il.load_item() + self.assertIsInstance(loaded_item, self.item_class) + self.assertEqual(dict(loaded_item), {'name': ['foo', 'bar']}) + + def test_add_value_singlevalue_singlevalue(self): + """Values added after initialization should be appended""" + input_item = self.item_class(name='foo') + il = ItemLoader(item=input_item) + il.add_value('name', 'bar') + loaded_item = il.load_item() + self.assertIsInstance(loaded_item, self.item_class) + self.assertEqual(dict(loaded_item), {'name': ['foo', 'bar']}) + + def test_add_value_singlevalue_list(self): + """Values added after initialization should be appended""" + input_item = self.item_class(name='foo') + il = ItemLoader(item=input_item) + il.add_value('name', ['item', 'loader']) + loaded_item = il.load_item() + self.assertIsInstance(loaded_item, self.item_class) + self.assertEqual(dict(loaded_item), {'name': ['foo', 'item', 'loader']}) + + def test_add_value_list_singlevalue(self): + """Values added after initialization should be appended""" + input_item = self.item_class(name=['foo', 'bar']) + il = ItemLoader(item=input_item) + il.add_value('name', 'qwerty') + loaded_item = il.load_item() + self.assertIsInstance(loaded_item, self.item_class) + self.assertEqual(dict(loaded_item), {'name': ['foo', 'bar', 'qwerty']}) + + def test_add_value_list_list(self): + """Values added after initialization should be appended""" + input_item = self.item_class(name=['foo', 'bar']) + il = ItemLoader(item=input_item) + il.add_value('name', ['item', 'loader']) + loaded_item = il.load_item() + self.assertIsInstance(loaded_item, self.item_class) + self.assertEqual(dict(loaded_item), {'name': ['foo', 'bar', 'item', 'loader']}) + + def test_get_output_value_singlevalue(self): + """Getting output value must not remove value from item""" + input_item = self.item_class(name='foo') + il = ItemLoader(item=input_item) + self.assertEqual(il.get_output_value('name'), ['foo']) + loaded_item = il.load_item() + self.assertIsInstance(loaded_item, self.item_class) + self.assertEqual(loaded_item, dict({'name': ['foo']})) + + def test_get_output_value_list(self): + """Getting output value must not remove value from item""" + input_item = self.item_class(name=['foo', 'bar']) + il = ItemLoader(item=input_item) + self.assertEqual(il.get_output_value('name'), ['foo', 'bar']) + loaded_item = il.load_item() + self.assertIsInstance(loaded_item, self.item_class) + self.assertEqual(loaded_item, dict({'name': ['foo', 'bar']})) + + def test_values_single(self): + """Values from initial item must be added to loader._values""" + input_item = self.item_class(name='foo') + il = ItemLoader(item=input_item) + self.assertEqual(il._values.get('name'), ['foo']) + + def test_values_list(self): + """Values from initial item must be added to loader._values""" + input_item = self.item_class(name=['foo', 'bar']) + il = ItemLoader(item=input_item) + self.assertEqual(il._values.get('name'), ['foo', 'bar']) + + +class InitializationFromDictTest(InitializationTestMixin, unittest.TestCase): + item_class = dict + + +class InitializationFromItemTest(InitializationTestMixin, unittest.TestCase): + item_class = NameItem + + +class BaseNoInputReprocessingLoader(ItemLoader): + title_in = MapCompose(str.upper) + title_out = TakeFirst() + + +class NoInputReprocessingDictLoader(BaseNoInputReprocessingLoader): + default_item_class = dict + + +class NoInputReprocessingFromDictTest(unittest.TestCase): + """ + Loaders initialized from loaded items must not reprocess fields (dict instances) + """ + def test_avoid_reprocessing_with_initial_values_single(self): + il = NoInputReprocessingDictLoader(item=dict(title='foo')) + il_loaded = il.load_item() + self.assertEqual(il_loaded, dict(title='foo')) + self.assertEqual(NoInputReprocessingDictLoader(item=il_loaded).load_item(), dict(title='foo')) + + def test_avoid_reprocessing_with_initial_values_list(self): + il = NoInputReprocessingDictLoader(item=dict(title=['foo', 'bar'])) + il_loaded = il.load_item() + self.assertEqual(il_loaded, dict(title='foo')) + self.assertEqual(NoInputReprocessingDictLoader(item=il_loaded).load_item(), dict(title='foo')) + + def test_avoid_reprocessing_without_initial_values_single(self): + il = NoInputReprocessingDictLoader() + il.add_value('title', 'foo') + il_loaded = il.load_item() + self.assertEqual(il_loaded, dict(title='FOO')) + self.assertEqual(NoInputReprocessingDictLoader(item=il_loaded).load_item(), dict(title='FOO')) + + def test_avoid_reprocessing_without_initial_values_list(self): + il = NoInputReprocessingDictLoader() + il.add_value('title', ['foo', 'bar']) + il_loaded = il.load_item() + self.assertEqual(il_loaded, dict(title='FOO')) + self.assertEqual(NoInputReprocessingDictLoader(item=il_loaded).load_item(), dict(title='FOO')) + + +class NoInputReprocessingItem(Item): + title = Field() + + +class NoInputReprocessingItemLoader(BaseNoInputReprocessingLoader): + default_item_class = NoInputReprocessingItem + + +class NoInputReprocessingFromItemTest(unittest.TestCase): + """ + Loaders initialized from loaded items must not reprocess fields (BaseItem instances) + """ + def test_avoid_reprocessing_with_initial_values_single(self): + il = NoInputReprocessingItemLoader(item=NoInputReprocessingItem(title='foo')) + il_loaded = il.load_item() + self.assertEqual(il_loaded, {'title': 'foo'}) + self.assertEqual(NoInputReprocessingItemLoader(item=il_loaded).load_item(), {'title': 'foo'}) + + def test_avoid_reprocessing_with_initial_values_list(self): + il = NoInputReprocessingItemLoader(item=NoInputReprocessingItem(title=['foo', 'bar'])) + il_loaded = il.load_item() + self.assertEqual(il_loaded, {'title': 'foo'}) + self.assertEqual(NoInputReprocessingItemLoader(item=il_loaded).load_item(), {'title': 'foo'}) + + def test_avoid_reprocessing_without_initial_values_single(self): + il = NoInputReprocessingItemLoader() + il.add_value('title', 'FOO') + il_loaded = il.load_item() + self.assertEqual(il_loaded, {'title': 'FOO'}) + self.assertEqual(NoInputReprocessingItemLoader(item=il_loaded).load_item(), {'title': 'FOO'}) + + def test_avoid_reprocessing_without_initial_values_list(self): + il = NoInputReprocessingItemLoader() + il.add_value('title', ['foo', 'bar']) + il_loaded = il.load_item() + self.assertEqual(il_loaded, {'title': 'FOO'}) + self.assertEqual(NoInputReprocessingItemLoader(item=il_loaded).load_item(), {'title': 'FOO'}) + + +class TestOutputProcessorDict(unittest.TestCase): + def test_output_processor(self): + + class TempDict(dict): + def __init__(self, *args, **kwargs): + super(TempDict, self).__init__(self, *args, **kwargs) + self.setdefault('temp', 0.3) + + class TempLoader(ItemLoader): + default_item_class = TempDict + default_input_processor = Identity() + default_output_processor = Compose(TakeFirst()) + + loader = TempLoader() + item = loader.load_item() + self.assertIsInstance(item, TempDict) + self.assertEqual(dict(item), {'temp': 0.3}) + + +class TestOutputProcessorItem(unittest.TestCase): + def test_output_processor(self): + + class TempItem(Item): + temp = Field() + + def __init__(self, *args, **kwargs): + super(TempItem, self).__init__(self, *args, **kwargs) + self.setdefault('temp', 0.3) + + class TempLoader(ItemLoader): + default_item_class = TempItem + default_input_processor = Identity() + default_output_processor = Compose(TakeFirst()) + + loader = TempLoader() + item = loader.load_item() + self.assertIsInstance(item, TempItem) + self.assertEqual(dict(item), {'temp': 0.3}) + + class ProcessorsTest(unittest.TestCase): def test_take_first(self): @@ -510,7 +687,7 @@ class ProcessorsTest(unittest.TestCase): self.assertRaises(TypeError, proc, [None, '', 'hello', 'world']) self.assertEqual(proc(['', 'hello', 'world']), u' hello world') self.assertEqual(proc(['hello', 'world']), u'hello world') - self.assertIsInstance(proc(['hello', 'world']), six.text_type) + self.assertIsInstance(proc(['hello', 'world']), str) def test_compose(self): proc = Compose(lambda v: v[0], str.upper) @@ -523,19 +700,19 @@ class ProcessorsTest(unittest.TestCase): self.assertRaises(ValueError, proc, 'hello') def test_mapcompose(self): - filter_world = lambda x: None if x == 'world' else x - proc = MapCompose(filter_world, six.text_type.upper) + def filter_world(x): + return None if x == 'world' else x + proc = MapCompose(filter_world, str.upper) self.assertEqual(proc([u'hello', u'world', u'this', u'is', u'scrapy']), [u'HELLO', u'THIS', u'IS', u'SCRAPY']) - proc = MapCompose(filter_world, six.text_type.upper) + proc = MapCompose(filter_world, str.upper) self.assertEqual(proc(None), []) - proc = MapCompose(filter_world, six.text_type.upper) + proc = MapCompose(filter_world, str.upper) self.assertRaises(ValueError, proc, [1]) proc = MapCompose(filter_world, lambda x: x + 1) self.assertRaises(ValueError, proc, 'hello') - class SelectortemLoaderTest(unittest.TestCase): response = HtmlResponse(url="", encoding='utf-8', body=b""" <html> @@ -548,11 +725,11 @@ class SelectortemLoaderTest(unittest.TestCase): </html> """) - def test_constructor(self): + def test_init_method(self): l = TestItemLoader() self.assertEqual(l.selector, None) - def test_constructor_errors(self): + def test_init_method_errors(self): l = TestItemLoader() self.assertRaises(RuntimeError, l.add_xpath, 'url', '//a/@href') self.assertRaises(RuntimeError, l.replace_xpath, 'url', '//a/@href') @@ -561,7 +738,7 @@ class SelectortemLoaderTest(unittest.TestCase): self.assertRaises(RuntimeError, l.replace_css, 'name', '#name::text') self.assertRaises(RuntimeError, l.get_css, '#name::text') - def test_constructor_with_selector(self): + def test_init_method_with_selector(self): sel = Selector(text=u"<html><body><div>marta</div></body></html>") l = TestItemLoader(selector=sel) self.assertIs(l.selector, sel) @@ -569,7 +746,7 @@ class SelectortemLoaderTest(unittest.TestCase): l.add_xpath('name', '//div/text()') self.assertEqual(l.get_output_value('name'), [u'Marta']) - def test_constructor_with_selector_css(self): + def test_init_method_with_selector_css(self): sel = Selector(text=u"<html><body><div>marta</div></body></html>") l = TestItemLoader(selector=sel) self.assertIs(l.selector, sel) @@ -577,14 +754,14 @@ class SelectortemLoaderTest(unittest.TestCase): l.add_css('name', 'div::text') self.assertEqual(l.get_output_value('name'), [u'Marta']) - def test_constructor_with_response(self): + def test_init_method_with_response(self): l = TestItemLoader(response=self.response) self.assertTrue(l.selector) l.add_xpath('name', '//div/text()') self.assertEqual(l.get_output_value('name'), [u'Marta']) - def test_constructor_with_response_css(self): + def test_init_method_with_response_css(self): l = TestItemLoader(response=self.response) self.assertTrue(l.selector) @@ -672,7 +849,7 @@ class SelectortemLoaderTest(unittest.TestCase): self.assertEqual(l.get_css(['p::text', 'div::text']), [u'paragraph', 'marta']) self.assertEqual(l.get_css(['a::attr(href)', 'img::attr(src)']), - [u'http://www.scrapy.org', u'/images/logo.png']) + [u'http://www.scrapy.org', u'/images/logo.png']) def test_replace_css_multi_fields(self): l = TestItemLoader(response=self.response) @@ -720,7 +897,7 @@ class SubselectorLoaderTest(unittest.TestCase): self.assertEqual(l.get_output_value('name'), [u'marta']) self.assertEqual(l.get_output_value('name_div'), [u'<div id="id">marta</div>']) - self.assertEqual(l.get_output_value('name_value'), [u'marta']) + self.assertEqual(l.get_output_value('name_value'), [u'marta']) self.assertEqual(l.get_output_value('name'), nl.get_output_value('name')) self.assertEqual(l.get_output_value('name_div'), nl.get_output_value('name_div')) @@ -735,7 +912,7 @@ class SubselectorLoaderTest(unittest.TestCase): self.assertEqual(l.get_output_value('name'), [u'marta']) self.assertEqual(l.get_output_value('name_div'), [u'<div id="id">marta</div>']) - self.assertEqual(l.get_output_value('name_value'), [u'marta']) + self.assertEqual(l.get_output_value('name_value'), [u'marta']) self.assertEqual(l.get_output_value('name'), nl.get_output_value('name')) self.assertEqual(l.get_output_value('name_div'), nl.get_output_value('name_div')) @@ -791,28 +968,76 @@ class SubselectorLoaderTest(unittest.TestCase): class SelectJmesTestCase(unittest.TestCase): - test_list_equals = { - 'simple': ('foo.bar', {"foo": {"bar": "baz"}}, "baz"), - 'invalid': ('foo.bar.baz', {"foo": {"bar": "baz"}}, None), - 'top_level': ('foo', {"foo": {"bar": "baz"}}, {"bar": "baz"}), - 'double_vs_single_quote_string': ('foo.bar', {"foo": {"bar": "baz"}}, "baz"), - 'dict': ( - 'foo.bar[*].name', - {"foo": {"bar": [{"name": "one"}, {"name": "two"}]}}, - ['one', 'two'] - ), - 'list': ('[1]', [1, 2], 2) - } + test_list_equals = { + 'simple': ('foo.bar', {"foo": {"bar": "baz"}}, "baz"), + 'invalid': ('foo.bar.baz', {"foo": {"bar": "baz"}}, None), + 'top_level': ('foo', {"foo": {"bar": "baz"}}, {"bar": "baz"}), + 'double_vs_single_quote_string': ('foo.bar', {"foo": {"bar": "baz"}}, "baz"), + 'dict': ( + 'foo.bar[*].name', + {"foo": {"bar": [{"name": "one"}, {"name": "two"}]}}, + ['one', 'two'] + ), + 'list': ('[1]', [1, 2], 2) + } - def test_output(self): - for l in self.test_list_equals: - expr, test_list, expected = self.test_list_equals[l] - test = SelectJmes(expr)(test_list) - self.assertEqual( - test, - expected, - msg='test "{}" got {} expected {}'.format(l, test, expected) - ) + def test_output(self): + for l in self.test_list_equals: + expr, test_list, expected = self.test_list_equals[l] + test = SelectJmes(expr)(test_list) + self.assertEqual( + test, + expected, + msg='test "{}" got {} expected {}'.format(l, test, expected) + ) + + +# Functions as processors + +def function_processor_strip(iterable): + return [x.strip() for x in iterable] + + +def function_processor_upper(iterable): + return [x.upper() for x in iterable] + + +class FunctionProcessorItem(Item): + foo = Field( + input_processor=function_processor_strip, + output_processor=function_processor_upper, + ) + + +class FunctionProcessorItemLoader(ItemLoader): + default_item_class = FunctionProcessorItem + + +class FunctionProcessorDictLoader(ItemLoader): + default_item_class = dict + foo_in = function_processor_strip + foo_out = function_processor_upper + + +class FunctionProcessorTestCase(unittest.TestCase): + + def test_processor_defined_in_item(self): + lo = FunctionProcessorItemLoader() + lo.add_value('foo', ' bar ') + lo.add_value('foo', [' asdf ', ' qwerty ']) + self.assertEqual( + dict(lo.load_item()), + {'foo': ['BAR', 'ASDF', 'QWERTY']} + ) + + def test_processor_defined_in_item_loader(self): + lo = FunctionProcessorDictLoader() + lo.add_value('foo', ' bar ') + lo.add_value('foo', [' asdf ', ' qwerty ']) + self.assertEqual( + dict(lo.load_item()), + {'foo': ['BAR', 'ASDF', 'QWERTY']} + ) if __name__ == "__main__": diff --git a/tests/test_logformatter.py b/tests/test_logformatter.py index 94e6c9fde..7d8c6ec7f 100644 --- a/tests/test_logformatter.py +++ b/tests/test_logformatter.py @@ -1,10 +1,17 @@ import unittest -import six -from scrapy.spiders import Spider +from testfixtures import LogCapture +from twisted.internet import defer +from twisted.trial.unittest import TestCase as TwistedTestCase + +from scrapy.crawler import CrawlerRunner +from scrapy.exceptions import DropItem from scrapy.http import Request, Response from scrapy.item import Item, Field from scrapy.logformatter import LogFormatter +from scrapy.spiders import Spider +from tests.mockserver import MockServer +from tests.spiders import ItemSpider class CustomItem(Item): @@ -15,13 +22,13 @@ class CustomItem(Item): return "name: %s" % self['name'] -class LoggingContribTest(unittest.TestCase): +class LogFormatterTestCase(unittest.TestCase): def setUp(self): self.formatter = LogFormatter() self.spider = Spider('default') - def test_crawled(self): + def test_crawled_with_referer(self): req = Request("http://www.example.com") res = Response("http://www.example.com") logkws = self.formatter.crawled(req, res, self.spider) @@ -29,6 +36,7 @@ class LoggingContribTest(unittest.TestCase): self.assertEqual(logline, "Crawled (200) <GET http://www.example.com> (referer: None)") + def test_crawled_without_referer(self): req = Request("http://www.example.com", headers={'referer': 'http://example.com'}) res = Response("http://www.example.com", flags=['cached']) logkws = self.formatter.crawled(req, res, self.spider) @@ -37,7 +45,7 @@ class LoggingContribTest(unittest.TestCase): "Crawled (200) <GET http://www.example.com> (referer: http://example.com) ['cached']") def test_flags_in_request(self): - req = Request("http://www.example.com", flags=['test','flag']) + req = Request("http://www.example.com", flags=['test', 'flag']) res = Response("http://www.example.com") logkws = self.formatter.crawled(req, res, self.spider) logline = logkws['msg'] % logkws['args'] @@ -51,9 +59,19 @@ class LoggingContribTest(unittest.TestCase): logkws = self.formatter.dropped(item, exception, response, self.spider) logline = logkws['msg'] % logkws['args'] lines = logline.splitlines() - assert all(isinstance(x, six.text_type) for x in lines) + assert all(isinstance(x, str) for x in lines) self.assertEqual(lines, [u"Dropped: \u2018", '{}']) + def test_error(self): + # In practice, the complete traceback is shown by passing the + # 'exc_info' argument to the logging function + item = {'key': 'value'} + exception = Exception() + response = Response("http://www.example.com") + logkws = self.formatter.error(item, exception, response, self.spider) + logline = logkws['msg'] % logkws['args'] + self.assertEqual(logline, u"'Error processing {'key': 'value'}'") + def test_scraped(self): item = CustomItem() item['name'] = u'\xa3' @@ -61,32 +79,109 @@ class LoggingContribTest(unittest.TestCase): logkws = self.formatter.scraped(item, response, self.spider) logline = logkws['msg'] % logkws['args'] lines = logline.splitlines() - assert all(isinstance(x, six.text_type) for x in lines) + assert all(isinstance(x, str) for x in lines) self.assertEqual(lines, [u"Scraped from <200 http://www.example.com>", u'name: \xa3']) class LogFormatterSubclass(LogFormatter): def crawled(self, request, response, spider): - kwargs = super(LogFormatterSubclass, self).crawled( - request, response, spider) + kwargs = super(LogFormatterSubclass, self).crawled(request, response, spider) CRAWLEDMSG = ( - u"Crawled (%(status)s) %(request)s (referer: " - u"%(referer)s)%(flags)s" + u"Crawled (%(status)s) %(request)s (referer: %(referer)s) %(flags)s" ) + log_args = kwargs['args'] + log_args['flags'] = str(request.flags) return { 'level': kwargs['level'], 'msg': CRAWLEDMSG, - 'args': kwargs['args'] + 'args': log_args, } -class LogformatterSubclassTest(LoggingContribTest): +class LogformatterSubclassTest(LogFormatterTestCase): def setUp(self): self.formatter = LogFormatterSubclass() self.spider = Spider('default') + def test_crawled_with_referer(self): + req = Request("http://www.example.com") + res = Response("http://www.example.com") + logkws = self.formatter.crawled(req, res, self.spider) + logline = logkws['msg'] % logkws['args'] + self.assertEqual(logline, + "Crawled (200) <GET http://www.example.com> (referer: None) []") + + def test_crawled_without_referer(self): + req = Request("http://www.example.com", headers={'referer': 'http://example.com'}, flags=['cached']) + res = Response("http://www.example.com") + logkws = self.formatter.crawled(req, res, self.spider) + logline = logkws['msg'] % logkws['args'] + self.assertEqual(logline, + "Crawled (200) <GET http://www.example.com> (referer: http://example.com) ['cached']") + def test_flags_in_request(self): - pass + req = Request("http://www.example.com", flags=['test', 'flag']) + res = Response("http://www.example.com") + logkws = self.formatter.crawled(req, res, self.spider) + logline = logkws['msg'] % logkws['args'] + self.assertEqual(logline, "Crawled (200) <GET http://www.example.com> (referer: None) ['test', 'flag']") + + +class SkipMessagesLogFormatter(LogFormatter): + def crawled(self, *args, **kwargs): + return None + + def scraped(self, *args, **kwargs): + return None + + def dropped(self, *args, **kwargs): + return None + + +class DropSomeItemsPipeline(object): + drop = True + + def process_item(self, item, spider): + if self.drop: + self.drop = False + raise DropItem("Ignoring item") + else: + self.drop = True + + +class ShowOrSkipMessagesTestCase(TwistedTestCase): + def setUp(self): + self.mockserver = MockServer() + self.mockserver.__enter__() + self.base_settings = { + 'LOG_LEVEL': 'DEBUG', + 'ITEM_PIPELINES': { + __name__ + '.DropSomeItemsPipeline': 300, + }, + } + + def tearDown(self): + self.mockserver.__exit__(None, None, None) + + @defer.inlineCallbacks + def test_show_messages(self): + crawler = CrawlerRunner(self.base_settings).create_crawler(ItemSpider) + with LogCapture() as lc: + yield crawler.crawl(mockserver=self.mockserver) + self.assertIn("Scraped from <200 http://127.0.0.1:", str(lc)) + self.assertIn("Crawled (200) <GET http://127.0.0.1:", str(lc)) + self.assertIn("Dropped: Ignoring item", str(lc)) + + @defer.inlineCallbacks + def test_skip_messages(self): + settings = self.base_settings.copy() + settings['LOG_FORMATTER'] = __name__ + '.SkipMessagesLogFormatter' + crawler = CrawlerRunner(settings).create_crawler(ItemSpider) + with LogCapture() as lc: + yield crawler.crawl(mockserver=self.mockserver) + self.assertNotIn("Scraped from <200 http://127.0.0.1:", str(lc)) + self.assertNotIn("Crawled (200) <GET http://127.0.0.1:", str(lc)) + self.assertNotIn("Dropped: Ignoring item", str(lc)) if __name__ == "__main__": diff --git a/tests/test_mail.py b/tests/test_mail.py index b139e98d8..ddb0f1e70 100644 --- a/tests/test_mail.py +++ b/tests/test_mail.py @@ -6,6 +6,7 @@ from email.charset import Charset from scrapy.mail import MailSender + class MailSenderTest(unittest.TestCase): def test_send(self): diff --git a/tests/test_middleware.py b/tests/test_middleware.py index aea0be825..ebf817c7e 100644 --- a/tests/test_middleware.py +++ b/tests/test_middleware.py @@ -3,7 +3,7 @@ from twisted.trial import unittest from scrapy.settings import Settings from scrapy.exceptions import NotConfigured from scrapy.middleware import MiddlewareManager -import six + class M1(object): @@ -16,6 +16,7 @@ class M1(object): def process(self, response, request, spider): pass + class M2(object): def open_spider(self, spider): @@ -26,6 +27,7 @@ class M2(object): pass + class M3(object): def process(self, response, request, spider): @@ -55,6 +57,7 @@ class TestMiddlewareManager(MiddlewareManager): if hasattr(mw, 'process'): self.methods['process'].append(mw.process) + class MiddlewareManagerTest(unittest.TestCase): def test_init(self): @@ -66,20 +69,12 @@ class MiddlewareManagerTest(unittest.TestCase): def test_methods(self): mwman = TestMiddlewareManager(M1(), M2(), M3()) - if six.PY2: - self.assertEqual([x.im_class for x in mwman.methods['open_spider']], - [M1, M2]) - self.assertEqual([x.im_class for x in mwman.methods['close_spider']], - [M2, M1]) - self.assertEqual([x.im_class for x in mwman.methods['process']], - [M1, M3]) - else: - self.assertEqual([x.__self__.__class__ for x in mwman.methods['open_spider']], - [M1, M2]) - self.assertEqual([x.__self__.__class__ for x in mwman.methods['close_spider']], - [M2, M1]) - self.assertEqual([x.__self__.__class__ for x in mwman.methods['process']], - [M1, M3]) + self.assertEqual([x.__self__.__class__ for x in mwman.methods['open_spider']], + [M1, M2]) + self.assertEqual([x.__self__.__class__ for x in mwman.methods['close_spider']], + [M2, M1]) + self.assertEqual([x.__self__.__class__ for x in mwman.methods['process']], + [M1, M3]) def test_enabled(self): m1, m2, m3 = M1(), M2(), M3() diff --git a/tests/test_pipeline_files.py b/tests/test_pipeline_files.py index 000c1e2e2..fc0453a97 100644 --- a/tests/test_pipeline_files.py +++ b/tests/test_pipeline_files.py @@ -1,12 +1,11 @@ import os import random import time -import hashlib -import warnings +from io import BytesIO from tempfile import mkdtemp from shutil import rmtree -from six.moves.urllib.parse import urlparse -from six import BytesIO +from unittest import mock +from urllib.parse import urlparse from twisted.trial import unittest from twisted.internet import defer @@ -15,14 +14,11 @@ from scrapy.pipelines.files import FilesPipeline, FSFilesStore, S3FilesStore, GC from scrapy.item import Item, Field from scrapy.http import Request, Response from scrapy.settings import Settings -from scrapy.utils.python import to_bytes from scrapy.utils.test import assert_aws_environ, get_s3_content_and_delete from scrapy.utils.test import assert_gcs_environ, get_gcs_content_and_delete from scrapy.utils.test import get_ftp_content_and_delete from scrapy.utils.boto import is_botocore -from tests import mock - def _mocked_download_func(request, info): response = request.meta.get('response') @@ -58,6 +54,11 @@ class FilesPipelineTestCase(unittest.TestCase): response=Response("http://www.dorma.co.uk/images/product_details/2532"), info=object()), 'full/244e0dd7d96a3b7b01f54eded250c9e272577aa1') + self.assertEqual(file_path(Request("http://www.dfsonline.co.uk/get_prod_image.php?img=status_0907_mdm.jpg.bohaha")), + 'full/76c00cef2ef669ae65052661f68d451162829507') + self.assertEqual(file_path(Request("data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAR0AAACxCAMAAADOHZloAAACClBMVEX/\ + //+F0tzCwMK76ZKQ21AMqr7oAAC96JvD5aWM2kvZ78J0N7fmAAC46Y4Ap7y")), + 'full/178059cbeba2e34120a67f2dc1afc3ecc09b61cb.png') def test_fs_store(self): assert isinstance(self.pipeline.store, FSFilesStore) diff --git a/tests/test_pipeline_images.py b/tests/test_pipeline_images.py index 4f7265763..7f1cb4a11 100644 --- a/tests/test_pipeline_images.py +++ b/tests/test_pipeline_images.py @@ -1,7 +1,6 @@ import io import hashlib import random -import warnings from tempfile import mkdtemp from shutil import rmtree diff --git a/tests/test_pipeline_media.py b/tests/test_pipeline_media.py index 28e39cefa..1fcc5799e 100644 --- a/tests/test_pipeline_media.py +++ b/tests/test_pipeline_media.py @@ -1,7 +1,3 @@ -from __future__ import print_function - -import sys - from testfixtures import LogCapture from twisted.trial import unittest from twisted.python.failure import Failure @@ -144,10 +140,8 @@ class BaseMediaPipelineTestCase(unittest.TestCase): # The Failure should encapsulate a FileException ... self.assertEqual(failure.value, file_exc) - # ... and if we're running on Python 3 ... - if sys.version_info.major >= 3: - # ... it should have the returnValue exception set as its context - self.assertEqual(failure.value.__context__, def_gen_return_exc) + # ... and it should have the returnValue exception set as its context + self.assertEqual(failure.value.__context__, def_gen_return_exc) # Let's calculate the request fingerprint and fake some runtime data... fp = request_fingerprint(request) @@ -244,10 +238,10 @@ class MediaPipelineTestCase(BaseMediaPipelineTestCase): self.assertEqual(new_item['results'], [(True, rsp1), (False, fail)]) m = self.pipe._mockcalled # only once - self.assertEqual(m[0], 'get_media_requests') # first hook called + self.assertEqual(m[0], 'get_media_requests') # first hook called self.assertEqual(m.count('get_media_requests'), 1) self.assertEqual(m.count('item_completed'), 1) - self.assertEqual(m[-1], 'item_completed') # last hook called + self.assertEqual(m[-1], 'item_completed') # last hook called # twice, one per request self.assertEqual(m.count('media_to_download'), 2) # one to handle success and other for failure @@ -258,7 +252,7 @@ class MediaPipelineTestCase(BaseMediaPipelineTestCase): def test_get_media_requests(self): # returns single Request (without callback) req = Request('http://url') - item = dict(requests=req) # pass a single item + item = dict(requests=req) # pass a single item new_item = yield self.pipe.process_item(item, self.spider) assert new_item is item assert request_fingerprint(req) in self.info.downloaded diff --git a/tests/test_pipelines.py b/tests/test_pipelines.py new file mode 100644 index 000000000..c72f1a338 --- /dev/null +++ b/tests/test_pipelines.py @@ -0,0 +1,101 @@ +import asyncio + +from pytest import mark +from twisted.internet import defer +from twisted.internet.defer import Deferred +from twisted.trial import unittest + +from scrapy import Spider, signals, Request +from scrapy.utils.test import get_crawler, get_from_asyncio_queue + +from tests.mockserver import MockServer + + +class SimplePipeline: + def process_item(self, item, spider): + item['pipeline_passed'] = True + return item + + +class DeferredPipeline: + def cb(self, item): + item['pipeline_passed'] = True + return item + + def process_item(self, item, spider): + d = Deferred() + d.addCallback(self.cb) + d.callback(item) + return d + + +class AsyncDefPipeline: + async def process_item(self, item, spider): + await defer.succeed(42) + item['pipeline_passed'] = True + return item + + +class AsyncDefAsyncioPipeline: + async def process_item(self, item, spider): + await asyncio.sleep(0.2) + item['pipeline_passed'] = await get_from_asyncio_queue(True) + return item + + +class ItemSpider(Spider): + name = 'itemspider' + + def start_requests(self): + yield Request(self.mockserver.url('/status?n=200')) + + def parse(self, response): + return {'field': 42} + + +class PipelineTestCase(unittest.TestCase): + def setUp(self): + self.mockserver = MockServer() + self.mockserver.__enter__() + + def tearDown(self): + self.mockserver.__exit__(None, None, None) + + def _on_item_scraped(self, item): + self.assertIsInstance(item, dict) + self.assertTrue(item.get('pipeline_passed')) + self.items.append(item) + + def _create_crawler(self, pipeline_class): + settings = { + 'ITEM_PIPELINES': {__name__ + '.' + pipeline_class.__name__: 1}, + } + crawler = get_crawler(ItemSpider, settings) + crawler.signals.connect(self._on_item_scraped, signals.item_scraped) + self.items = [] + return crawler + + @defer.inlineCallbacks + def test_simple_pipeline(self): + crawler = self._create_crawler(SimplePipeline) + yield crawler.crawl(mockserver=self.mockserver) + self.assertEqual(len(self.items), 1) + + @defer.inlineCallbacks + def test_deferred_pipeline(self): + crawler = self._create_crawler(DeferredPipeline) + yield crawler.crawl(mockserver=self.mockserver) + self.assertEqual(len(self.items), 1) + + @defer.inlineCallbacks + def test_asyncdef_pipeline(self): + crawler = self._create_crawler(AsyncDefPipeline) + yield crawler.crawl(mockserver=self.mockserver) + self.assertEqual(len(self.items), 1) + + @mark.only_asyncio() + @defer.inlineCallbacks + def test_asyncdef_asyncio_pipeline(self): + crawler = self._create_crawler(AsyncDefAsyncioPipeline) + yield crawler.crawl(mockserver=self.mockserver) + self.assertEqual(len(self.items), 1) diff --git a/tests/test_proxy_connect.py b/tests/test_proxy_connect.py index ae1236bcb..188ec68dd 100644 --- a/tests/test_proxy_connect.py +++ b/tests/test_proxy_connect.py @@ -1,38 +1,53 @@ import json import os -import time +import re +import sys +from subprocess import Popen, PIPE +from urllib.parse import urlsplit, urlunsplit -from six.moves.urllib.parse import urlsplit, urlunsplit -from threading import Thread -from libmproxy import controller, proxy -from netlib import http_auth +import pytest from testfixtures import LogCapture - from twisted.internet import defer from twisted.trial.unittest import TestCase -from scrapy.utils.test import get_crawler + from scrapy.http import Request -from tests.spiders import SimpleSpider, SingleRequestSpider +from scrapy.utils.test import get_crawler + from tests.mockserver import MockServer +from tests.spiders import SimpleSpider, SingleRequestSpider -class HTTPSProxy(controller.Master, Thread): +class MitmProxy: + auth_user = 'scrapy' + auth_pass = 'scrapy' - def __init__(self): - password_manager = http_auth.PassManSingleUser('scrapy', 'scrapy') - authenticator = http_auth.BasicProxyAuth(password_manager, "mitmproxy") + def start(self): + from scrapy.utils.test import get_testenv + script = """ +import sys +from mitmproxy.tools.main import mitmdump +sys.argv[0] = "mitmdump" +sys.exit(mitmdump()) + """ cert_path = os.path.join(os.path.abspath(os.path.dirname(__file__)), - 'keys', 'mitmproxy-ca.pem') - server = proxy.ProxyServer(proxy.ProxyConfig( - authenticator = authenticator, - cacert = cert_path), - 0) - self.server = server - Thread.__init__(self) - controller.Master.__init__(self, server) + 'keys', 'mitmproxy-ca.pem') + self.proc = Popen([sys.executable, + '-c', script, + '--listen-host', '127.0.0.1', + '--listen-port', '0', + '--proxyauth', '%s:%s' % (self.auth_user, self.auth_pass), + '--certs', cert_path, + '--ssl-insecure', + ], + stdout=PIPE, env=get_testenv()) + line = self.proc.stdout.readline().decode('utf-8') + host_port = re.search(r'listening at http://([^:]+:\d+)', line).group(1) + address = 'http://%s:%s@%s' % (self.auth_user, self.auth_pass, host_port) + return address - def http_address(self): - return 'http://scrapy:scrapy@%s:%d' % self.server.socket.getsockname() + def stop(self): + self.proc.kill() + self.proc.communicate() def _wrong_credentials(proxy_url): @@ -40,6 +55,7 @@ def _wrong_credentials(proxy_url): bad_auth_proxy[1] = bad_auth_proxy[1].replace('scrapy:scrapy@', 'wrong:wronger@') return urlunsplit(bad_auth_proxy) + class ProxyConnectTestCase(TestCase): def setUp(self): @@ -47,17 +63,14 @@ class ProxyConnectTestCase(TestCase): self.mockserver.__enter__() self._oldenv = os.environ.copy() - self._proxy = HTTPSProxy() - self._proxy.start() - - # Wait for the proxy to start. - time.sleep(1.0) - os.environ['https_proxy'] = self._proxy.http_address() - os.environ['http_proxy'] = self._proxy.http_address() + self._proxy = MitmProxy() + proxy_url = self._proxy.start() + os.environ['https_proxy'] = proxy_url + os.environ['http_proxy'] = proxy_url def tearDown(self): self.mockserver.__exit__(None, None, None) - self._proxy.shutdown() + self._proxy.stop() os.environ = self._oldenv @defer.inlineCallbacks @@ -67,15 +80,7 @@ class ProxyConnectTestCase(TestCase): yield crawler.crawl(self.mockserver.url("/status?n=200", is_secure=True)) self._assert_got_response_code(200, l) - @defer.inlineCallbacks - def test_https_noconnect(self): - proxy = os.environ['https_proxy'] - os.environ['https_proxy'] = proxy + '?noconnect' - crawler = get_crawler(SimpleSpider) - with LogCapture() as l: - yield crawler.crawl(self.mockserver.url("/status?n=200", is_secure=True)) - self._assert_got_response_code(200, l) - + @pytest.mark.xfail(reason='Python 3.6+ fails this earlier', condition=sys.version_info.minor >= 6) @defer.inlineCallbacks def test_https_connect_tunnel_error(self): crawler = get_crawler(SimpleSpider) @@ -100,17 +105,9 @@ class ProxyConnectTestCase(TestCase): with LogCapture() as l: yield crawler.crawl(seed=request) self._assert_got_response_code(200, l) - echo = json.loads(crawler.spider.meta['responses'][0].body) + echo = json.loads(crawler.spider.meta['responses'][0].text) self.assertTrue('Proxy-Authorization' not in echo['headers']) - @defer.inlineCallbacks - def test_https_noconnect_auth_error(self): - os.environ['https_proxy'] = _wrong_credentials(os.environ['https_proxy']) + '?noconnect' - crawler = get_crawler(SimpleSpider) - with LogCapture() as l: - yield crawler.crawl(self.mockserver.url("/status?n=200", is_secure=True)) - self._assert_got_response_code(407, l) - def _assert_got_response_code(self, code, log): print(log) self.assertEqual(str(log).count('Crawled (%d)' % code), 1) diff --git a/tests/test_pydispatch_deprecated.py b/tests/test_pydispatch_deprecated.py deleted file mode 100644 index 6d3237fe1..000000000 --- a/tests/test_pydispatch_deprecated.py +++ /dev/null @@ -1,12 +0,0 @@ -import unittest -import warnings -from six.moves import reload_module - - -class DeprecatedPydispatchTest(unittest.TestCase): - def test_import_xlib_pydispatch_show_warning(self): - with warnings.catch_warnings(record=True) as w: - from scrapy.xlib import pydispatch - reload_module(pydispatch) - self.assertIn('Importing from scrapy.xlib.pydispatch is deprecated', - str(w[0].message)) diff --git a/tests/test_request_cb_kwargs.py b/tests/test_request_cb_kwargs.py index c9943faa8..a5cdc0de0 100644 --- a/tests/test_request_cb_kwargs.py +++ b/tests/test_request_cb_kwargs.py @@ -1,7 +1,6 @@ from testfixtures import LogCapture from twisted.internet import defer from twisted.trial.unittest import TestCase -import six from scrapy.http import Request from scrapy.crawler import CrawlerRunner @@ -161,9 +160,4 @@ class CallbackKeywordArgumentsTestCase(TestCase): self.assertEqual(exceptions['takes_less'].exc_info[0], TypeError) self.assertEqual(str(exceptions['takes_less'].exc_info[1]), "parse_takes_less() got an unexpected keyword argument 'number'") self.assertEqual(exceptions['takes_more'].exc_info[0], TypeError) - # py2 and py3 messages are different - exc_message = str(exceptions['takes_more'].exc_info[1]) - if six.PY2: - self.assertEqual(exc_message, "parse_takes_more() takes exactly 5 arguments (4 given)") - elif six.PY3: - self.assertEqual(exc_message, "parse_takes_more() missing 1 required positional argument: 'other'") + self.assertEqual(str(exceptions['takes_more'].exc_info[1]), "parse_takes_more() missing 1 required positional argument: 'other'") diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index f89042b3d..d5a3371ab 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -4,6 +4,7 @@ from scrapy.responsetypes import responsetypes from scrapy.http import Response, TextResponse, XmlResponse, HtmlResponse, Headers + class ResponseTypesTest(unittest.TestCase): def test_from_filename(self): diff --git a/tests/test_robotstxt_interface.py b/tests/test_robotstxt_interface.py new file mode 100644 index 000000000..24aaaf7ec --- /dev/null +++ b/tests/test_robotstxt_interface.py @@ -0,0 +1,162 @@ +# coding=utf-8 +from twisted.trial import unittest + + +def reppy_available(): + # check if reppy parser is installed + try: + from reppy.robots import Robots # noqa: F401 + except ImportError: + return False + return True + + +def rerp_available(): + # check if robotexclusionrulesparser is installed + try: + from robotexclusionrulesparser import RobotExclusionRulesParser # noqa: F401 + except ImportError: + return False + return True + + +def protego_available(): + # check if protego parser is installed + try: + from protego import Protego # noqa: F401 + except ImportError: + return False + return True + + +class BaseRobotParserTest: + def _setUp(self, parser_cls): + self.parser_cls = parser_cls + + def test_allowed(self): + robotstxt_robotstxt_body = ("User-agent: * \n" + "Disallow: /disallowed \n" + "Allow: /allowed \n" + "Crawl-delay: 10".encode('utf-8')) + rp = self.parser_cls.from_crawler(crawler=None, robotstxt_body=robotstxt_robotstxt_body) + self.assertTrue(rp.allowed("https://www.site.local/allowed", "*")) + self.assertFalse(rp.allowed("https://www.site.local/disallowed", "*")) + + def test_allowed_wildcards(self): + robotstxt_robotstxt_body = """User-agent: first + Disallow: /disallowed/*/end$ + + User-agent: second + Allow: /*allowed + Disallow: / + """.encode('utf-8') + rp = self.parser_cls.from_crawler(crawler=None, robotstxt_body=robotstxt_robotstxt_body) + + self.assertTrue(rp.allowed("https://www.site.local/disallowed", "first")) + self.assertFalse(rp.allowed("https://www.site.local/disallowed/xyz/end", "first")) + self.assertFalse(rp.allowed("https://www.site.local/disallowed/abc/end", "first")) + self.assertTrue(rp.allowed("https://www.site.local/disallowed/xyz/endinglater", "first")) + + self.assertTrue(rp.allowed("https://www.site.local/allowed", "second")) + self.assertTrue(rp.allowed("https://www.site.local/is_still_allowed", "second")) + self.assertTrue(rp.allowed("https://www.site.local/is_allowed_too", "second")) + + def test_length_based_precedence(self): + robotstxt_robotstxt_body = ("User-agent: * \n" + "Disallow: / \n" + "Allow: /page".encode('utf-8')) + rp = self.parser_cls.from_crawler(crawler=None, robotstxt_body=robotstxt_robotstxt_body) + self.assertTrue(rp.allowed("https://www.site.local/page", "*")) + + def test_order_based_precedence(self): + robotstxt_robotstxt_body = ("User-agent: * \n" + "Disallow: / \n" + "Allow: /page".encode('utf-8')) + rp = self.parser_cls.from_crawler(crawler=None, robotstxt_body=robotstxt_robotstxt_body) + self.assertFalse(rp.allowed("https://www.site.local/page", "*")) + + def test_empty_response(self): + """empty response should equal 'allow all'""" + rp = self.parser_cls.from_crawler(crawler=None, robotstxt_body=b'') + self.assertTrue(rp.allowed("https://site.local/", "*")) + self.assertTrue(rp.allowed("https://site.local/", "chrome")) + self.assertTrue(rp.allowed("https://site.local/index.html", "*")) + self.assertTrue(rp.allowed("https://site.local/disallowed", "*")) + + def test_garbage_response(self): + """garbage response should be discarded, equal 'allow all'""" + robotstxt_robotstxt_body = b'GIF89a\xd3\x00\xfe\x00\xa2' + rp = self.parser_cls.from_crawler(crawler=None, robotstxt_body=robotstxt_robotstxt_body) + self.assertTrue(rp.allowed("https://site.local/", "*")) + self.assertTrue(rp.allowed("https://site.local/", "chrome")) + self.assertTrue(rp.allowed("https://site.local/index.html", "*")) + self.assertTrue(rp.allowed("https://site.local/disallowed", "*")) + + def test_unicode_url_and_useragent(self): + robotstxt_robotstxt_body = u""" + User-Agent: * + Disallow: /admin/ + Disallow: /static/ + # taken from https://en.wikipedia.org/robots.txt + Disallow: /wiki/K%C3%A4ytt%C3%A4j%C3%A4: + Disallow: /wiki/Käyttäjä: + + User-Agent: UnicödeBöt + Disallow: /some/randome/page.html""".encode('utf-8') + rp = self.parser_cls.from_crawler(crawler=None, robotstxt_body=robotstxt_robotstxt_body) + self.assertTrue(rp.allowed("https://site.local/", "*")) + self.assertFalse(rp.allowed("https://site.local/admin/", "*")) + self.assertFalse(rp.allowed("https://site.local/static/", "*")) + self.assertTrue(rp.allowed("https://site.local/admin/", u"UnicödeBöt")) + self.assertFalse(rp.allowed("https://site.local/wiki/K%C3%A4ytt%C3%A4j%C3%A4:", "*")) + self.assertFalse(rp.allowed(u"https://site.local/wiki/Käyttäjä:", "*")) + self.assertTrue(rp.allowed("https://site.local/some/randome/page.html", "*")) + self.assertFalse(rp.allowed("https://site.local/some/randome/page.html", u"UnicödeBöt")) + + +class PythonRobotParserTest(BaseRobotParserTest, unittest.TestCase): + def setUp(self): + from scrapy.robotstxt import PythonRobotParser + super(PythonRobotParserTest, self)._setUp(PythonRobotParser) + + def test_length_based_precedence(self): + raise unittest.SkipTest("RobotFileParser does not support length based directives precedence.") + + def test_allowed_wildcards(self): + raise unittest.SkipTest("RobotFileParser does not support wildcards.") + + +class ReppyRobotParserTest(BaseRobotParserTest, unittest.TestCase): + if not reppy_available(): + skip = "Reppy parser is not installed" + + def setUp(self): + from scrapy.robotstxt import ReppyRobotParser + super(ReppyRobotParserTest, self)._setUp(ReppyRobotParser) + + def test_order_based_precedence(self): + raise unittest.SkipTest("Reppy does not support order based directives precedence.") + + +class RerpRobotParserTest(BaseRobotParserTest, unittest.TestCase): + if not rerp_available(): + skip = "Rerp parser is not installed" + + def setUp(self): + from scrapy.robotstxt import RerpRobotParser + super(RerpRobotParserTest, self)._setUp(RerpRobotParser) + + def test_length_based_precedence(self): + raise unittest.SkipTest("Rerp does not support length based directives precedence.") + + +class ProtegoRobotParserTest(BaseRobotParserTest, unittest.TestCase): + if not protego_available(): + skip = "Protego parser is not installed" + + def setUp(self): + from scrapy.robotstxt import ProtegoRobotParser + super(ProtegoRobotParserTest, self)._setUp(ProtegoRobotParser) + + def test_order_based_precedence(self): + raise unittest.SkipTest("Protego does not support order based directives precedence.") diff --git a/tests/test_selector.py b/tests/test_selector.py index b2565dd78..09c2546fb 100644 --- a/tests/test_selector.py +++ b/tests/test_selector.py @@ -1,9 +1,9 @@ -import warnings import weakref + from twisted.trial import unittest + from scrapy.http import TextResponse, HtmlResponse, XmlResponse from scrapy.selector import Selector -from lxml import etree class SelectorTestCase(unittest.TestCase): diff --git a/tests/test_settings/__init__.py b/tests/test_settings/__init__.py index 1dbacbea3..fda44653a 100644 --- a/tests/test_settings/__init__.py +++ b/tests/test_settings/__init__.py @@ -1,17 +1,15 @@ -import six import unittest -import warnings +from unittest import mock from scrapy.settings import (BaseSettings, Settings, SettingsAttribute, SETTINGS_PRIORITIES, get_settings_priority) -from tests import mock from . import default_settings class SettingsGlobalFuncsTest(unittest.TestCase): def test_get_settings_priority(self): - for prio_str, prio_num in six.iteritems(SETTINGS_PRIORITIES): + for prio_str, prio_num in SETTINGS_PRIORITIES.items(): self.assertEqual(get_settings_priority(prio_str), prio_num) self.assertEqual(get_settings_priority(99), 99) @@ -44,14 +42,14 @@ class SettingsAttributeTest(unittest.TestCase): new_dict = {'three': 11, 'four': 21} attribute.set(new_dict, 10) self.assertIsInstance(attribute.value, BaseSettings) - six.assertCountEqual(self, attribute.value, new_dict) - six.assertCountEqual(self, original_settings, original_dict) + self.assertCountEqual(attribute.value, new_dict) + self.assertCountEqual(original_settings, original_dict) new_settings = BaseSettings({'five': 12}, 0) attribute.set(new_settings, 0) # Insufficient priority - six.assertCountEqual(self, attribute.value, new_dict) + self.assertCountEqual(attribute.value, new_dict) attribute.set(new_settings, 10) - six.assertCountEqual(self, attribute.value, new_settings) + self.assertCountEqual(attribute.value, new_settings) def test_repr(self): self.assertEqual(repr(self.attribute), @@ -60,9 +58,6 @@ class SettingsAttributeTest(unittest.TestCase): class BaseSettingsTest(unittest.TestCase): - if six.PY3: - assertItemsEqual = unittest.TestCase.assertCountEqual - def setUp(self): self.settings = BaseSettings() @@ -152,10 +147,10 @@ class BaseSettingsTest(unittest.TestCase): self.settings.setmodule( 'tests.test_settings.default_settings', 10) - self.assertItemsEqual(six.iterkeys(self.settings.attributes), - six.iterkeys(ctrl_attributes)) + self.assertCountEqual(self.settings.attributes.keys(), + ctrl_attributes.keys()) - for key in six.iterkeys(ctrl_attributes): + for key in ctrl_attributes.keys(): attr = self.settings.attributes[key] ctrl_attr = ctrl_attributes[key] self.assertEqual(attr.value, ctrl_attr.value) @@ -231,7 +226,7 @@ class BaseSettingsTest(unittest.TestCase): } settings = self.settings settings.attributes = {key: SettingsAttribute(value, 0) for key, value - in six.iteritems(test_configuration)} + in test_configuration.items()} self.assertTrue(settings.getbool('TEST_ENABLED1')) self.assertTrue(settings.getbool('TEST_ENABLED2')) @@ -280,9 +275,8 @@ class BaseSettingsTest(unittest.TestCase): 'TEST': BaseSettings({1: 10, 3: 30}, 'default'), 'HASNOBASE': BaseSettings({3: 3000}, 'default')}) s['TEST'].set(2, 200, 'cmdline') - six.assertCountEqual(self, s.getwithbase('TEST'), - {1: 1, 2: 200, 3: 30}) - six.assertCountEqual(self, s.getwithbase('HASNOBASE'), s['HASNOBASE']) + self.assertCountEqual(s.getwithbase('TEST'), {1: 1, 2: 200, 3: 30}) + self.assertCountEqual(s.getwithbase('HASNOBASE'), s['HASNOBASE']) self.assertEqual(s.getwithbase('NONEXISTENT'), {}) def test_maxpriority(self): @@ -343,9 +337,6 @@ class BaseSettingsTest(unittest.TestCase): class SettingsTest(unittest.TestCase): - if six.PY3: - assertItemsEqual = unittest.TestCase.assertCountEqual - def setUp(self): self.settings = Settings() diff --git a/tests/test_settings/default_settings.py b/tests/test_settings/default_settings.py index c24b5a9b9..26a555275 100644 --- a/tests/test_settings/default_settings.py +++ b/tests/test_settings/default_settings.py @@ -2,4 +2,3 @@ TEST_DEFAULT = 'defvalue' TEST_DICT = {'key': 'val'} - diff --git a/tests/test_spider.py b/tests/test_spider.py index e81e6d5f9..317a27076 100644 --- a/tests/test_spider.py +++ b/tests/test_spider.py @@ -1,5 +1,6 @@ import gzip import inspect +from unittest import mock import warnings from io import BytesIO @@ -14,11 +15,8 @@ from scrapy.spiders import Spider, CrawlSpider, Rule, XMLFeedSpider, \ CSVFeedSpider, SitemapSpider from scrapy.linkextractors import LinkExtractor from scrapy.exceptions import ScrapyDeprecationWarning -from scrapy.utils.trackref import object_ref from scrapy.utils.test import get_crawler -from tests import mock - class SpiderTest(unittest.TestCase): @@ -42,12 +40,12 @@ class SpiderTest(unittest.TestCase): self.assertEqual(list(start_requests), []) def test_spider_args(self): - """Constructor arguments are assigned to spider attributes""" + """``__init__`` method arguments are assigned to spider attributes""" spider = self.spider_class('example.com', foo='bar') self.assertEqual(spider.foo, 'bar') def test_spider_without_name(self): - """Constructor arguments are assigned to spider attributes""" + """``__init__`` method arguments are assigned to spider attributes""" self.assertRaises(ValueError, self.spider_class) self.assertRaises(ValueError, self.spider_class, somearg='foo') @@ -177,6 +175,26 @@ class CrawlSpiderTest(SpiderTest): </body></html>""" spider_class = CrawlSpider + def test_rule_without_link_extractor(self): + + response = HtmlResponse("http://example.org/somepage/index.html", body=self.test_body) + + class _CrawlSpider(self.spider_class): + name = "test" + allowed_domains = ['example.org'] + rules = ( + Rule(), + ) + + spider = _CrawlSpider() + output = list(spider._requests_to_follow(response)) + self.assertEqual(len(output), 3) + self.assertTrue(all(map(lambda r: isinstance(r, Request), output))) + self.assertEqual([r.url for r in output], + ['http://example.org/somepage/item/12.html', + 'http://example.org/about.html', + 'http://example.org/nofollow.html']) + def test_process_links(self): response = HtmlResponse("http://example.org/somepage/index.html", body=self.test_body) @@ -366,6 +384,14 @@ class CrawlSpiderTest(SpiderTest): self.assertTrue(hasattr(spider, '_follow_links')) self.assertFalse(spider._follow_links) + def test_start_url(self): + spider = self.spider_class("example.com") + spider.start_url = 'https://www.example.com' + + with self.assertRaisesRegex(AttributeError, + r'^Crawling could not start.*$'): + list(spider.start_requests()) + class SitemapSpiderTest(SpiderTest): diff --git a/tests/test_spiderloader/__init__.py b/tests/test_spiderloader/__init__.py index 106da798c..d8be6e277 100644 --- a/tests/test_spiderloader/__init__.py +++ b/tests/test_spiderloader/__init__.py @@ -109,6 +109,7 @@ class SpiderLoaderTest(unittest.TestCase): spiders = spider_loader.list() self.assertEqual(spiders, []) + class DuplicateSpiderNameLoaderTest(unittest.TestCase): def setUp(self): diff --git a/tests/test_spiderloader/test_spiders/nested/spider4.py b/tests/test_spiderloader/test_spiders/nested/spider4.py index 35b71870a..dbd1fb123 100644 --- a/tests/test_spiderloader/test_spiders/nested/spider4.py +++ b/tests/test_spiderloader/test_spiders/nested/spider4.py @@ -1,5 +1,6 @@ from scrapy.spiders import Spider + class Spider4(Spider): name = "spider4" allowed_domains = ['spider4.com'] diff --git a/tests/test_spiderloader/test_spiders/spider0.py b/tests/test_spiderloader/test_spiders/spider0.py index 75a90794e..af679dbd6 100644 --- a/tests/test_spiderloader/test_spiders/spider0.py +++ b/tests/test_spiderloader/test_spiders/spider0.py @@ -1,4 +1,5 @@ from scrapy.spiders import Spider + class Spider0(Spider): allowed_domains = ["scrapy1.org", "scrapy3.org"] diff --git a/tests/test_spiderloader/test_spiders/spider1.py b/tests/test_spiderloader/test_spiders/spider1.py index 76efddc7f..6b4317a90 100644 --- a/tests/test_spiderloader/test_spiders/spider1.py +++ b/tests/test_spiderloader/test_spiders/spider1.py @@ -1,5 +1,6 @@ from scrapy.spiders import Spider + class Spider1(Spider): name = "spider1" allowed_domains = ["scrapy1.org", "scrapy3.org"] diff --git a/tests/test_spiderloader/test_spiders/spider2.py b/tests/test_spiderloader/test_spiders/spider2.py index 0badd8437..352601863 100644 --- a/tests/test_spiderloader/test_spiders/spider2.py +++ b/tests/test_spiderloader/test_spiders/spider2.py @@ -1,5 +1,6 @@ from scrapy.spiders import Spider + class Spider2(Spider): name = "spider2" allowed_domains = ["scrapy2.org", "scrapy3.org"] diff --git a/tests/test_spiderloader/test_spiders/spider3.py b/tests/test_spiderloader/test_spiders/spider3.py index d406f2d4f..84998ba35 100644 --- a/tests/test_spiderloader/test_spiders/spider3.py +++ b/tests/test_spiderloader/test_spiders/spider3.py @@ -1,5 +1,6 @@ from scrapy.spiders import Spider + class Spider3(Spider): name = "spider3" allowed_domains = ['spider3.com'] diff --git a/tests/test_spidermiddleware.py b/tests/test_spidermiddleware.py index 832fd3330..55d665e79 100644 --- a/tests/test_spidermiddleware.py +++ b/tests/test_spidermiddleware.py @@ -1,3 +1,5 @@ +from unittest import mock + from twisted.trial.unittest import TestCase from twisted.python.failure import Failure @@ -6,7 +8,6 @@ from scrapy.http import Request, Response from scrapy.exceptions import _InvalidOutput from scrapy.utils.test import get_crawler from scrapy.core.spidermw import SpiderMiddlewareManager -from tests import mock class SpiderMiddlewareTestCase(TestCase): diff --git a/tests/test_spidermiddleware_depth.py b/tests/test_spidermiddleware_depth.py index 3685d5a6f..71cca2472 100644 --- a/tests/test_spidermiddleware_depth.py +++ b/tests/test_spidermiddleware_depth.py @@ -40,4 +40,3 @@ class TestDepthMiddleware(TestCase): def tearDown(self): self.stats.close_spider(self.spider, '') - diff --git a/tests/test_spidermiddleware_offsite.py b/tests/test_spidermiddleware_offsite.py index 7e4af0d4c..7511aa568 100644 --- a/tests/test_spidermiddleware_offsite.py +++ b/tests/test_spidermiddleware_offsite.py @@ -1,13 +1,12 @@ from unittest import TestCase - -from six.moves.urllib.parse import urlparse +from urllib.parse import urlparse +import warnings from scrapy.http import Response, Request from scrapy.spiders import Spider -from scrapy.spidermiddlewares.offsite import OffsiteMiddleware -from scrapy.spidermiddlewares.offsite import URLWarning +from scrapy.spidermiddlewares.offsite import OffsiteMiddleware, URLWarning from scrapy.utils.test import get_crawler -import warnings + class TestOffsiteMiddleware(TestCase): @@ -53,6 +52,7 @@ class TestOffsiteMiddleware2(TestOffsiteMiddleware): out = list(self.mw.process_spider_output(res, reqs, self.spider)) self.assertEqual(out, reqs) + class TestOffsiteMiddleware3(TestOffsiteMiddleware2): def _get_spider(self): @@ -73,7 +73,7 @@ class TestOffsiteMiddleware4(TestOffsiteMiddleware3): class TestOffsiteMiddleware5(TestOffsiteMiddleware4): - + def test_get_host_regex(self): self.spider.allowed_domains = ['http://scrapytest.org', 'scrapy.org', 'scrapy.test.org'] with warnings.catch_warnings(record=True) as w: diff --git a/tests/test_spidermiddleware_output_chain.py b/tests/test_spidermiddleware_output_chain.py index 6f8727a15..739cf1c2d 100644 --- a/tests/test_spidermiddleware_output_chain.py +++ b/tests/test_spidermiddleware_output_chain.py @@ -6,7 +6,6 @@ from twisted.internet import defer from scrapy import Spider, Request from scrapy.utils.test import get_crawler from tests.mockserver import MockServer -from tests.spiders import MockServerSpider class LogExceptionMiddleware: @@ -34,6 +33,7 @@ class RecoverySpider(Spider): if not response.meta.get('dont_fail'): raise TabError() + class RecoveryMiddleware: def process_spider_exception(self, response, exception, spider): spider.logger.info('Middleware: %s exception caught', exception.__class__.__name__) @@ -50,6 +50,7 @@ class FailProcessSpiderInputMiddleware: spider.logger.info('Middleware: will raise IndexError') raise IndexError() + class ProcessSpiderInputSpiderWithoutErrback(Spider): name = 'ProcessSpiderInputSpiderWithoutErrback' custom_settings = { @@ -155,7 +156,7 @@ class GeneratorFailMiddleware: r['processed'].append('{}.process_spider_output'.format(self.__class__.__name__)) yield r raise LookupError() - + def process_spider_exception(self, response, exception, spider): method = '{}.process_spider_exception'.format(self.__class__.__name__) spider.logger.info('%s: %s caught', method, exception.__class__.__name__) @@ -177,6 +178,7 @@ class GeneratorRecoverMiddleware: spider.logger.info('%s: %s caught', method, exception.__class__.__name__) yield {'processed': [method]} + class GeneratorDoNothingAfterRecoveryMiddleware(_GeneratorDoNothingMiddleware): pass @@ -247,6 +249,7 @@ class NotGeneratorRecoverMiddleware: spider.logger.info('%s: %s caught', method, exception.__class__.__name__) return [{'processed': [method]}] + class NotGeneratorDoNothingAfterRecoveryMiddleware(_NotGeneratorDoNothingMiddleware): pass @@ -261,7 +264,7 @@ class TestSpiderMiddleware(TestCase): @classmethod def tearDownClass(cls): cls.mockserver.__exit__(None, None, None) - + @defer.inlineCallbacks def crawl_log(self, spider): crawler = get_crawler(spider) @@ -305,7 +308,7 @@ class TestSpiderMiddleware(TestCase): self.assertIn("{'from': 'errback'}", str(log1)) self.assertNotIn("{'from': 'callback'}", str(log1)) self.assertIn("'item_scraped_count': 1", str(log1)) - + @defer.inlineCallbacks def test_generator_callback(self): """ @@ -316,7 +319,7 @@ class TestSpiderMiddleware(TestCase): log2 = yield self.crawl_log(GeneratorCallbackSpider) self.assertIn("Middleware: ImportError exception caught", str(log2)) self.assertIn("'item_scraped_count': 2", str(log2)) - + @defer.inlineCallbacks def test_not_a_generator_callback(self): """ diff --git a/tests/test_spidermiddleware_referer.py b/tests/test_spidermiddleware_referer.py index 21439c20e..7cc17600c 100644 --- a/tests/test_spidermiddleware_referer.py +++ b/tests/test_spidermiddleware_referer.py @@ -1,8 +1,7 @@ -from six.moves.urllib.parse import urlparse +from urllib.parse import urlparse from unittest import TestCase import warnings -from scrapy.exceptions import NotConfigured from scrapy.http import Response, Request from scrapy.settings import Settings from scrapy.spiders import Spider @@ -349,6 +348,7 @@ class TestSettingsCustomPolicy(TestRefererMiddleware): ] + # --- Tests using Request meta dict to set policy class TestRequestMetaDefault(MixinDefault, TestRefererMiddleware): req_meta = {'referrer_policy': POLICY_SCRAPY_DEFAULT} @@ -518,14 +518,17 @@ class TestPolicyHeaderPredecence001(MixinUnsafeUrl, TestRefererMiddleware): settings = {'REFERRER_POLICY': 'scrapy.spidermiddlewares.referer.SameOriginPolicy'} resp_headers = {'Referrer-Policy': POLICY_UNSAFE_URL.upper()} + class TestPolicyHeaderPredecence002(MixinNoReferrer, TestRefererMiddleware): settings = {'REFERRER_POLICY': 'scrapy.spidermiddlewares.referer.NoReferrerWhenDowngradePolicy'} resp_headers = {'Referrer-Policy': POLICY_NO_REFERRER.swapcase()} + class TestPolicyHeaderPredecence003(MixinNoReferrerWhenDowngrade, TestRefererMiddleware): settings = {'REFERRER_POLICY': 'scrapy.spidermiddlewares.referer.OriginWhenCrossOriginPolicy'} resp_headers = {'Referrer-Policy': POLICY_NO_REFERRER_WHEN_DOWNGRADE.title()} + class TestPolicyHeaderPredecence004(MixinNoReferrerWhenDowngrade, TestRefererMiddleware): """ The empty string means "no-referrer-when-downgrade" @@ -545,8 +548,8 @@ class TestReferrerOnRedirect(TestRefererMiddleware): (301, 'http://scrapytest.org/3'), (301, 'http://scrapytest.org/4'), ), - b'http://scrapytest.org/1', # expected initial referer - b'http://scrapytest.org/1', # expected referer for the redirection request + b'http://scrapytest.org/1', # expected initial referer + b'http://scrapytest.org/1', # expected referer for the redirection request ), ( 'https://scrapytest.org/1', 'https://scrapytest.org/2', @@ -606,8 +609,8 @@ class TestReferrerOnRedirectNoReferrer(TestReferrerOnRedirect): (301, 'http://scrapytest.org/3'), (301, 'http://scrapytest.org/4'), ), - None, # expected initial "Referer" - None, # expected "Referer" for the redirection request + None, # expected initial "Referer" + None, # expected "Referer" for the redirection request ), ( 'https://scrapytest.org/1', 'https://scrapytest.org/2', @@ -645,8 +648,8 @@ class TestReferrerOnRedirectSameOrigin(TestReferrerOnRedirect): (301, 'http://scrapytest.org/103'), (301, 'http://scrapytest.org/104'), ), - b'http://scrapytest.org/101', # expected initial "Referer" - b'http://scrapytest.org/101', # expected referer for the redirection request + b'http://scrapytest.org/101', # expected initial "Referer" + b'http://scrapytest.org/101', # expected referer for the redirection request ), ( 'https://scrapytest.org/201', 'https://scrapytest.org/202', @@ -754,8 +757,8 @@ class TestReferrerOnRedirectOriginWhenCrossOrigin(TestReferrerOnRedirect): (301, 'http://scrapytest.org/103'), (301, 'http://scrapytest.org/104'), ), - b'http://scrapytest.org/101', # expected initial referer - b'http://scrapytest.org/101', # expected referer for the redirection request + b'http://scrapytest.org/101', # expected initial referer + b'http://scrapytest.org/101', # expected referer for the redirection request ), ( 'https://scrapytest.org/201', 'https://scrapytest.org/202', @@ -824,8 +827,8 @@ class TestReferrerOnRedirectStrictOriginWhenCrossOrigin(TestReferrerOnRedirect): (301, 'http://scrapytest.org/103'), (301, 'http://scrapytest.org/104'), ), - b'http://scrapytest.org/101', # expected initial referer - b'http://scrapytest.org/101', # expected referer for the redirection request + b'http://scrapytest.org/101', # expected initial referer + b'http://scrapytest.org/101', # expected referer for the redirection request ), ( 'https://scrapytest.org/201', 'https://scrapytest.org/202', diff --git a/tests/test_spidermiddleware_urllength.py b/tests/test_spidermiddleware_urllength.py index a0aae0fdd..5ef2b23fd 100644 --- a/tests/test_spidermiddleware_urllength.py +++ b/tests/test_spidermiddleware_urllength.py @@ -18,4 +18,3 @@ class TestUrlLengthMiddleware(TestCase): spider = Spider('foo') out = list(mw.process_spider_output(res, reqs, spider)) self.assertEqual(out, [short_url_req]) - diff --git a/tests/test_squeues.py b/tests/test_squeues.py index 3ded5c027..d5fcf2f7f 100644 --- a/tests/test_squeues.py +++ b/tests/test_squeues.py @@ -7,16 +7,20 @@ from scrapy.http import Request from scrapy.loader import ItemLoader from scrapy.selector import Selector + class TestItem(Item): name = Field() + def _test_procesor(x): return x + x + class TestLoader(ItemLoader): default_item_class = TestItem name_out = staticmethod(_test_procesor) + def nonserializable_object_test(self): q = self.queue() try: @@ -35,6 +39,7 @@ def nonserializable_object_test(self): sel = Selector(text='<html><body><p>some text</p></body></html>') self.assertRaises(ValueError, q.push, sel) + class MarshalFifoDiskQueueTest(t.FifoDiskQueueTest): chunksize = 100000 @@ -53,15 +58,19 @@ class MarshalFifoDiskQueueTest(t.FifoDiskQueueTest): test_nonserializable_object = nonserializable_object_test + class ChunkSize1MarshalFifoDiskQueueTest(MarshalFifoDiskQueueTest): chunksize = 1 + class ChunkSize2MarshalFifoDiskQueueTest(MarshalFifoDiskQueueTest): chunksize = 2 + class ChunkSize3MarshalFifoDiskQueueTest(MarshalFifoDiskQueueTest): chunksize = 3 + class ChunkSize4MarshalFifoDiskQueueTest(MarshalFifoDiskQueueTest): chunksize = 4 @@ -100,15 +109,19 @@ class PickleFifoDiskQueueTest(MarshalFifoDiskQueueTest): self.assertEqual(r.url, r2.url) assert r2.meta['request'] is r2 + class ChunkSize1PickleFifoDiskQueueTest(PickleFifoDiskQueueTest): chunksize = 1 + class ChunkSize2PickleFifoDiskQueueTest(PickleFifoDiskQueueTest): chunksize = 2 + class ChunkSize3PickleFifoDiskQueueTest(PickleFifoDiskQueueTest): chunksize = 3 + class ChunkSize4PickleFifoDiskQueueTest(PickleFifoDiskQueueTest): chunksize = 4 diff --git a/tests/test_stats.py b/tests/test_stats.py index 9f950ebc9..2bbbb9e2c 100644 --- a/tests/test_stats.py +++ b/tests/test_stats.py @@ -1,10 +1,55 @@ +from datetime import datetime import unittest +from unittest import mock +from scrapy.extensions.corestats import CoreStats from scrapy.spiders import Spider from scrapy.statscollectors import StatsCollector, DummyStatsCollector from scrapy.utils.test import get_crawler +class CoreStatsExtensionTest(unittest.TestCase): + + def setUp(self): + self.crawler = get_crawler(Spider) + self.spider = self.crawler._create_spider('foo') + + @mock.patch('scrapy.extensions.corestats.datetime') + def test_core_stats_default_stats_collector(self, mock_datetime): + fixed_datetime = datetime(2019, 12, 1, 11, 38) + mock_datetime.utcnow = mock.Mock(return_value=fixed_datetime) + self.crawler.stats = StatsCollector(self.crawler) + ext = CoreStats.from_crawler(self.crawler) + ext.spider_opened(self.spider) + ext.item_scraped({}, self.spider) + ext.response_received(self.spider) + ext.item_dropped({}, self.spider, ZeroDivisionError()) + ext.spider_closed(self.spider, 'finished') + self.assertEqual( + ext.stats._stats, + { + 'start_time': fixed_datetime, + 'finish_time': fixed_datetime, + 'item_scraped_count': 1, + 'response_received_count': 1, + 'item_dropped_count': 1, + 'item_dropped_reasons_count/ZeroDivisionError': 1, + 'finish_reason': 'finished', + 'elapsed_time_seconds': 0.0, + } + ) + + def test_core_stats_dummy_stats_collector(self): + self.crawler.stats = DummyStatsCollector(self.crawler) + ext = CoreStats.from_crawler(self.crawler) + ext.spider_opened(self.spider) + ext.item_scraped({}, self.spider) + ext.response_received(self.spider) + ext.item_dropped({}, self.spider, ZeroDivisionError()) + ext.spider_closed(self.spider, 'finished') + self.assertEqual(ext.stats._stats, {}) + + class StatsCollectorTest(unittest.TestCase): def setUp(self): diff --git a/tests/test_toplevel.py b/tests/test_toplevel.py index 91bbe43bc..fdc5df166 100644 --- a/tests/test_toplevel.py +++ b/tests/test_toplevel.py @@ -1,12 +1,12 @@ from unittest import TestCase -import six + import scrapy class ToplevelTestCase(TestCase): def test_version(self): - self.assertIs(type(scrapy.__version__), six.text_type) + self.assertIs(type(scrapy.__version__), str) def test_version_info(self): self.assertIs(type(scrapy.version_info), tuple) diff --git a/tests/test_urlparse_monkeypatches.py b/tests/test_urlparse_monkeypatches.py index 22e39821c..bea0cf3e5 100644 --- a/tests/test_urlparse_monkeypatches.py +++ b/tests/test_urlparse_monkeypatches.py @@ -1,4 +1,4 @@ -from six.moves.urllib.parse import urlparse +from urllib.parse import urlparse import unittest diff --git a/tests/test_utils_asyncio.py b/tests/test_utils_asyncio.py new file mode 100644 index 000000000..44acc24af --- /dev/null +++ b/tests/test_utils_asyncio.py @@ -0,0 +1,17 @@ +from unittest import TestCase + +from pytest import mark + +from scrapy.utils.asyncio import is_asyncio_reactor_installed, install_asyncio_reactor + + +@mark.usefixtures('reactor_pytest') +class AsyncioTest(TestCase): + + def test_is_asyncio_reactor_installed(self): + # the result should depend only on the pytest --reactor argument + self.assertEqual(is_asyncio_reactor_installed(), self.reactor_pytest == 'asyncio') + + def test_install_asyncio_reactor(self): + # this should do nothing + install_asyncio_reactor() diff --git a/tests/test_utils_conf.py b/tests/test_utils_conf.py index 29937c189..02d8ba51e 100644 --- a/tests/test_utils_conf.py +++ b/tests/test_utils_conf.py @@ -79,7 +79,7 @@ class BuildComponentListTest(unittest.TestCase): self.assertRaises(ValueError, build_component_list, {}, d, convert=lambda x: x) d = {'one': {'a': 'a', 'b': 2}} self.assertRaises(ValueError, build_component_list, {}, d, convert=lambda x: x) - d = {'one': 'lorem ipsum',} + d = {'one': 'lorem ipsum'} self.assertRaises(ValueError, build_component_list, {}, d, convert=lambda x: x) diff --git a/tests/test_utils_console.py b/tests/test_utils_console.py index 65782747b..380c41367 100644 --- a/tests/test_utils_console.py +++ b/tests/test_utils_console.py @@ -14,6 +14,7 @@ try: except ImportError: ipy = False + class UtilsConsoleTestCase(unittest.TestCase): def test_get_shell_embed_func(self): @@ -21,7 +22,7 @@ class UtilsConsoleTestCase(unittest.TestCase): shell = get_shell_embed_func(['invalid']) self.assertEqual(shell, None) - shell = get_shell_embed_func(['invalid','python']) + shell = get_shell_embed_func(['invalid', 'python']) self.assertTrue(callable(shell)) self.assertEqual(shell.__name__, '_embed_standard_shell') diff --git a/tests/test_utils_curl.py b/tests/test_utils_curl.py new file mode 100644 index 000000000..50e1bfd5f --- /dev/null +++ b/tests/test_utils_curl.py @@ -0,0 +1,208 @@ +import unittest +import warnings + +from w3lib.http import basic_auth_header + +from scrapy import Request +from scrapy.utils.curl import curl_to_request_kwargs + + +class CurlToRequestKwargsTest(unittest.TestCase): + maxDiff = 5000 + + def _test_command(self, curl_command, expected_result): + result = curl_to_request_kwargs(curl_command) + self.assertEqual(result, expected_result) + try: + Request(**result) + except TypeError as e: + self.fail("Request kwargs are not correct {}".format(e)) + + def test_get(self): + curl_command = "curl http://example.org/" + expected_result = {"method": "GET", "url": "http://example.org/"} + self._test_command(curl_command, expected_result) + + def test_get_without_scheme(self): + curl_command = "curl www.example.org" + expected_result = {"method": "GET", "url": "http://www.example.org"} + self._test_command(curl_command, expected_result) + + def test_get_basic_auth(self): + curl_command = 'curl "https://api.test.com/" -u ' \ + '"some_username:some_password"' + expected_result = { + "method": "GET", + "url": "https://api.test.com/", + "headers": [ + ( + "Authorization", + basic_auth_header("some_username", "some_password") + ) + ], + } + self._test_command(curl_command, expected_result) + + def test_get_complex(self): + curl_command = ( + "curl 'http://httpbin.org/get' -H 'Accept-Encoding: gzip, deflate'" + " -H 'Accept-Language: en-US,en;q=0.9,ru;q=0.8,es;q=0.7' -H 'Upgra" + "de-Insecure-Requests: 1' -H 'User-Agent: Mozilla/5.0 (X11; Linux " + "x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Ubuntu Chromium/62" + ".0.3202.75 Chrome/62.0.3202.75 Safari/537.36' -H 'Accept: text/ht" + "ml,application/xhtml+xml,application/xml;q=0.9,image/webp,image/a" + "png,*/*;q=0.8' -H 'Referer: http://httpbin.org/' -H 'Cookie: _gau" + "ges_unique_year=1; _gauges_unique=1; _gauges_unique_month=1; _gau" + "ges_unique_hour=1; _gauges_unique_day=1' -H 'Connection: keep-ali" + "ve' --compressed" + ) + expected_result = { + "method": "GET", + "url": "http://httpbin.org/get", + "headers": [ + ("Accept-Encoding", "gzip, deflate"), + ("Accept-Language", "en-US,en;q=0.9,ru;q=0.8,es;q=0.7"), + ("Upgrade-Insecure-Requests", "1"), + ( + "User-Agent", + "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML" + ", like Gecko) Ubuntu Chromium/62.0.3202.75 Chrome/62.0.32" + "02.75 Safari/537.36", + ), + ( + "Accept", + "text/html,application/xhtml+xml,application/xml;q=0.9,ima" + "ge/webp,image/apng,*/*;q=0.8", + ), + ("Referer", "http://httpbin.org/"), + ("Connection", "keep-alive"), + ], + "cookies": { + '_gauges_unique_year': '1', + '_gauges_unique_hour': '1', + '_gauges_unique_day': '1', + '_gauges_unique': '1', + '_gauges_unique_month': '1' + }, + } + self._test_command(curl_command, expected_result) + + def test_post(self): + curl_command = ( + "curl 'http://httpbin.org/post' -X POST -H 'Cookie: _gauges_unique" + "_year=1; _gauges_unique=1; _gauges_unique_month=1; _gauges_unique" + "_hour=1; _gauges_unique_day=1' -H 'Origin: http://httpbin.org' -H" + " 'Accept-Encoding: gzip, deflate' -H 'Accept-Language: en-US,en;q" + "=0.9,ru;q=0.8,es;q=0.7' -H 'Upgrade-Insecure-Requests: 1' -H 'Use" + "r-Agent: Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTM" + "L, like Gecko) Ubuntu Chromium/62.0.3202.75 Chrome/62.0.3202.75 S" + "afari/537.36' -H 'Content-Type: application/x-www-form-urlencoded" + "' -H 'Accept: text/html,application/xhtml+xml,application/xml;q=0" + ".9,image/webp,image/apng,*/*;q=0.8' -H 'Cache-Control: max-age=0'" + " -H 'Referer: http://httpbin.org/forms/post' -H 'Connection: keep" + "-alive' --data 'custname=John+Smith&custtel=500&custemail=jsmith%" + "40example.org&size=small&topping=cheese&topping=onion&delivery=12" + "%3A15&comments=' --compressed" + ) + expected_result = { + "method": "POST", + "url": "http://httpbin.org/post", + "body": "custname=John+Smith&custtel=500&custemail=jsmith%40exampl" + "e.org&size=small&topping=cheese&topping=onion&delivery=12" + "%3A15&comments=", + "cookies": { + '_gauges_unique_year': '1', + '_gauges_unique_hour': '1', + '_gauges_unique_day': '1', + '_gauges_unique': '1', + '_gauges_unique_month': '1' + }, + "headers": [ + ("Origin", "http://httpbin.org"), + ("Accept-Encoding", "gzip, deflate"), + ("Accept-Language", "en-US,en;q=0.9,ru;q=0.8,es;q=0.7"), + ("Upgrade-Insecure-Requests", "1"), + ( + "User-Agent", + "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML" + ", like Gecko) Ubuntu Chromium/62.0.3202.75 Chrome/62.0.32" + "02.75 Safari/537.36", + ), + ("Content-Type", "application/x-www-form-urlencoded"), + ( + "Accept", + "text/html,application/xhtml+xml,application/xml;q=0.9,ima" + "ge/webp,image/apng,*/*;q=0.8", + ), + ("Cache-Control", "max-age=0"), + ("Referer", "http://httpbin.org/forms/post"), + ("Connection", "keep-alive"), + ], + } + self._test_command(curl_command, expected_result) + + def test_patch(self): + curl_command = ( + 'curl "https://example.com/api/fake" -u "username:password" -H "Ac' + 'cept: application/vnd.go.cd.v4+json" -H "Content-Type: applicatio' + 'n/json" -X PATCH -d \'{"hostname": "agent02.example.com", "agent' + '_config_state": "Enabled", "resources": ["Java","Linux"], "enviro' + 'nments": ["Dev"]}\'' + ) + expected_result = { + "method": "PATCH", + "url": "https://example.com/api/fake", + "headers": [ + ("Accept", "application/vnd.go.cd.v4+json"), + ("Content-Type", "application/json"), + ("Authorization", basic_auth_header("username", "password")), + ], + "body": '{"hostname": "agent02.example.com", "agent_config_state"' + ': "Enabled", "resources": ["Java","Linux"], "environments' + '": ["Dev"]}', + } + self._test_command(curl_command, expected_result) + + def test_delete(self): + curl_command = 'curl -X "DELETE" https://www.url.com/page' + expected_result = { + "method": "DELETE", "url": "https://www.url.com/page" + } + self._test_command(curl_command, expected_result) + + def test_get_silent(self): + curl_command = 'curl --silent "www.example.com"' + expected_result = {"method": "GET", "url": "http://www.example.com"} + self.assertEqual(curl_to_request_kwargs(curl_command), expected_result) + + def test_too_few_arguments_error(self): + self.assertRaisesRegex( + ValueError, + r"too few arguments|the following arguments are required:\s*url", + lambda: curl_to_request_kwargs("curl"), + ) + + def test_ignore_unknown_options(self): + # case 1: ignore_unknown_options=True: + with warnings.catch_warnings(): # avoid warning when executing tests + warnings.simplefilter('ignore') + curl_command = 'curl --bar --baz http://www.example.com' + expected_result = \ + {"method": "GET", "url": "http://www.example.com"} + self.assertEqual(curl_to_request_kwargs(curl_command), expected_result) + + # case 2: ignore_unknown_options=False (raise exception): + self.assertRaisesRegex( + ValueError, + "Unrecognized options:.*--bar.*--baz", + lambda: curl_to_request_kwargs( + "curl --bar --baz http://www.example.com", + ignore_unknown_options=False + ), + ) + + def test_must_start_with_curl_error(self): + self.assertRaises( + ValueError, + lambda: curl_to_request_kwargs("carl -X POST http://example.org") + ) diff --git a/tests/test_utils_datatypes.py b/tests/test_utils_datatypes.py index 535095b8d..e5aa56eb9 100644 --- a/tests/test_utils_datatypes.py +++ b/tests/test_utils_datatypes.py @@ -1,13 +1,10 @@ import copy import unittest +from collections.abc import Mapping, MutableMapping -import six -if six.PY2: - from collections import Mapping, MutableMapping -else: - from collections.abc import Mapping, MutableMapping - -from scrapy.utils.datatypes import CaselessDict, SequenceExclude +from scrapy.http import Request +from scrapy.utils.datatypes import CaselessDict, LocalCache, LocalWeakReferencedCache, SequenceExclude +from scrapy.utils.python import garbage_collect __doctests__ = ['scrapy.utils.datatypes'] @@ -197,14 +194,6 @@ class SequenceExcludeTest(unittest.TestCase): self.assertIn(20, d) self.assertNotIn(15, d) - def test_six_range(self): - import six.moves - seq = six.moves.range(10**3, 10**6) - d = SequenceExclude(seq) - self.assertIn(10**2, d) - self.assertIn(10**7, d) - self.assertNotIn(10**4, d) - def test_range_step(self): seq = range(10, 20, 3) d = SequenceExclude(seq) @@ -242,6 +231,93 @@ class SequenceExcludeTest(unittest.TestCase): for v in [-3, "test", 1.1]: self.assertNotIn(v, d) + +class LocalCacheTest(unittest.TestCase): + + def test_cache_with_limit(self): + cache = LocalCache(limit=2) + cache['a'] = 1 + cache['b'] = 2 + cache['c'] = 3 + self.assertEqual(len(cache), 2) + self.assertNotIn('a', cache) + self.assertIn('b', cache) + self.assertIn('c', cache) + self.assertEqual(cache['b'], 2) + self.assertEqual(cache['c'], 3) + + def test_cache_without_limit(self): + maximum = 10**4 + cache = LocalCache() + for x in range(maximum): + cache[str(x)] = x + self.assertEqual(len(cache), maximum) + for x in range(maximum): + self.assertIn(str(x), cache) + self.assertEqual(cache[str(x)], x) + + +class LocalWeakReferencedCacheTest(unittest.TestCase): + + def test_cache_with_limit(self): + cache = LocalWeakReferencedCache(limit=2) + r1 = Request('https://example.org') + r2 = Request('https://example.com') + r3 = Request('https://example.net') + cache[r1] = 1 + cache[r2] = 2 + cache[r3] = 3 + self.assertEqual(len(cache), 2) + self.assertNotIn(r1, cache) + self.assertIn(r2, cache) + self.assertIn(r3, cache) + self.assertEqual(cache[r2], 2) + self.assertEqual(cache[r3], 3) + del r2 + + # PyPy takes longer to collect dead references + garbage_collect() + + self.assertEqual(len(cache), 1) + + def test_cache_non_weak_referenceable_objects(self): + cache = LocalWeakReferencedCache() + k1 = None + k2 = 1 + k3 = [1, 2, 3] + cache[k1] = 1 + cache[k2] = 2 + cache[k3] = 3 + self.assertNotIn(k1, cache) + self.assertNotIn(k2, cache) + self.assertNotIn(k3, cache) + self.assertEqual(len(cache), 0) + + def test_cache_without_limit(self): + max = 10**4 + cache = LocalWeakReferencedCache() + refs = [] + for x in range(max): + refs.append(Request('https://example.org/{}'.format(x))) + cache[refs[-1]] = x + self.assertEqual(len(cache), max) + for i, r in enumerate(refs): + self.assertIn(r, cache) + self.assertEqual(cache[r], i) + del r # delete reference to the last object in the list + + # delete half of the objects, make sure that is reflected in the cache + for _ in range(max // 2): + refs.pop() + + # PyPy takes longer to collect dead references + garbage_collect() + + self.assertEqual(len(cache), max // 2) + for i, r in enumerate(refs): + self.assertIn(r, cache) + self.assertEqual(cache[r], i) + + if __name__ == "__main__": unittest.main() - diff --git a/tests/test_utils_defer.py b/tests/test_utils_defer.py index 003bb9b02..dfbe71ae2 100644 --- a/tests/test_utils_defer.py +++ b/tests/test_utils_defer.py @@ -5,8 +5,6 @@ from twisted.python.failure import Failure from scrapy.utils.defer import mustbe_deferred, process_chain, \ process_chain_both, process_parallel, iter_errback -from six.moves import xrange - class MustbeDeferredTest(unittest.TestCase): def test_success_function(self): @@ -16,8 +14,8 @@ class MustbeDeferredTest(unittest.TestCase): return steps dfd = mustbe_deferred(_append, 1) - dfd.addCallback(self.assertEqual, [1, 2]) # it is [1] with maybeDeferred - steps.append(2) # add another value, that should be catched by assertEqual + dfd.addCallback(self.assertEqual, [1, 2]) # it is [1] with maybeDeferred + steps.append(2) # add another value, that should be catched by assertEqual return dfd def test_unfired_deferred(self): @@ -29,18 +27,27 @@ class MustbeDeferredTest(unittest.TestCase): return dfd dfd = mustbe_deferred(_append, 1) - dfd.addCallback(self.assertEqual, [1, 2]) # it is [1] with maybeDeferred - steps.append(2) # add another value, that should be catched by assertEqual + dfd.addCallback(self.assertEqual, [1, 2]) # it is [1] with maybeDeferred + steps.append(2) # add another value, that should be catched by assertEqual return dfd + def cb1(value, arg1, arg2): return "(cb1 %s %s %s)" % (value, arg1, arg2) + + def cb2(value, arg1, arg2): return defer.succeed("(cb2 %s %s %s)" % (value, arg1, arg2)) + + def cb3(value, arg1, arg2): return "(cb3 %s %s %s)" % (value, arg1, arg2) + + def cb_fail(value, arg1, arg2): return Failure(TypeError()) + + def eb1(failure, arg1, arg2): return "(eb1 %s %s %s)" % (failure.value.__class__.__name__, arg1, arg2) @@ -83,7 +90,7 @@ class IterErrbackTest(unittest.TestCase): def test_iter_errback_good(self): def itergood(): - for x in xrange(10): + for x in range(10): yield x errors = [] @@ -93,7 +100,7 @@ class IterErrbackTest(unittest.TestCase): def test_iter_errback_bad(self): def iterbad(): - for x in xrange(10): + for x in range(10): if x == 5: a = 1/0 yield x diff --git a/tests/test_utils_deprecate.py b/tests/test_utils_deprecate.py index 3e7236fb1..159ef8f25 100644 --- a/tests/test_utils_deprecate.py +++ b/tests/test_utils_deprecate.py @@ -1,13 +1,11 @@ # -*- coding: utf-8 -*- -from __future__ import absolute_import import inspect import unittest +from unittest import mock import warnings from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.utils.deprecate import create_deprecated_class, update_classpath -from tests import mock - class MyWarning(UserWarning): pass diff --git a/tests/test_utils_http.py b/tests/test_utils_http.py index 583105673..2fac3da1f 100644 --- a/tests/test_utils_http.py +++ b/tests/test_utils_http.py @@ -2,6 +2,7 @@ import unittest from scrapy.utils.http import decode_chunked_transfer + class ChunkedTest(unittest.TestCase): def test_decode_chunked_transfer(self): @@ -12,9 +13,7 @@ class ChunkedTest(unittest.TestCase): chunked_body += "8\r\n" + "sequence\r\n" chunked_body += "0\r\n\r\n" body = decode_chunked_transfer(chunked_body) - self.assertEqual(body, \ - "This is the data in the first chunk\r\n" + - "and this is the second one\r\n" + - "consequence") - - + self.assertEqual(body, + "This is the data in the first chunk\r\n" + + "and this is the second one\r\n" + + "consequence") diff --git a/tests/test_utils_httpobj.py b/tests/test_utils_httpobj.py index 4f9f7a370..cf8ad1f23 100644 --- a/tests/test_utils_httpobj.py +++ b/tests/test_utils_httpobj.py @@ -1,9 +1,10 @@ import unittest -from six.moves.urllib.parse import urlparse +from urllib.parse import urlparse from scrapy.http import Request from scrapy.utils.httpobj import urlparse_cached + class HttpobjUtilsTest(unittest.TestCase): def test_urlparse_cached(self): diff --git a/tests/test_utils_iterators.py b/tests/test_utils_iterators.py index 2d845697e..9776dfb2a 100644 --- a/tests/test_utils_iterators.py +++ b/tests/test_utils_iterators.py @@ -1,12 +1,13 @@ # -*- coding: utf-8 -*- import os -import six + from twisted.trial import unittest from scrapy.utils.iterators import csviter, xmliter, _body_or_str, xmliter_lxml from scrapy.http import XmlResponse, TextResponse, Response from tests import get_testdata + FOOBAR_NL = u"foo\nbar" @@ -235,6 +236,7 @@ class LxmlXmliterTestCase(XmliterTestCase): i = self.xmliter(42, 'product') self.assertRaises(TypeError, next, i) + class UtilsCsvTestCase(unittest.TestCase): sample_feeds_dir = os.path.join(os.path.abspath(os.path.dirname(__file__)), 'sample_data', 'feeds') sample_feed_path = os.path.join(sample_feeds_dir, 'feed-sample3.csv') @@ -255,8 +257,8 @@ class UtilsCsvTestCase(unittest.TestCase): # explicit type check cuz' we no like stinkin' autocasting! yarrr for result_row in result: - self.assertTrue(all((isinstance(k, six.text_type) for k in result_row.keys()))) - self.assertTrue(all((isinstance(v, six.text_type) for v in result_row.values()))) + self.assertTrue(all((isinstance(k, str) for k in result_row.keys()))) + self.assertTrue(all((isinstance(v, str) for v in result_row.values()))) def test_csviter_delimiter(self): body = get_testdata('feeds', 'feed-sample3.csv').replace(b',', b'\t') diff --git a/tests/test_utils_log.py b/tests/test_utils_log.py index 742e04803..2c23f3616 100644 --- a/tests/test_utils_log.py +++ b/tests/test_utils_log.py @@ -1,5 +1,4 @@ # -*- coding: utf-8 -*- -from __future__ import print_function import sys import logging import unittest diff --git a/tests/test_utils_misc/__init__.py b/tests/test_utils_misc/__init__.py index e109d5343..6f945cd01 100644 --- a/tests/test_utils_misc/__init__.py +++ b/tests/test_utils_misc/__init__.py @@ -1,11 +1,11 @@ import sys import os import unittest +from unittest import mock from scrapy.item import Item, Field from scrapy.utils.misc import arg_to_iter, create_instance, load_object, set_environ, walk_modules -from tests import mock __doctests__ = ['scrapy.utils.misc'] @@ -74,7 +74,7 @@ class UtilsMiscTestCase(unittest.TestCase): self.assertEqual(list(arg_to_iter(100)), [100]) self.assertEqual(list(arg_to_iter(l for l in 'abc')), ['a', 'b', 'c']) self.assertEqual(list(arg_to_iter([1, 2, 3])), [1, 2, 3]) - self.assertEqual(list(arg_to_iter({'a':1})), [{'a': 1}]) + self.assertEqual(list(arg_to_iter({'a': 1})), [{'a': 1}]) self.assertEqual(list(arg_to_iter(TestItem(name="john"))), [TestItem(name="john")]) def test_create_instance(self): diff --git a/tests/test_utils_misc/test_return_with_argument_inside_generator.py b/tests/test_utils_misc/test_return_with_argument_inside_generator.py new file mode 100644 index 000000000..bdbec1beb --- /dev/null +++ b/tests/test_utils_misc/test_return_with_argument_inside_generator.py @@ -0,0 +1,37 @@ +import unittest + +from scrapy.utils.misc import is_generator_with_return_value + + +class UtilsMiscPy3TestCase(unittest.TestCase): + + def test_generators_with_return_statements(self): + def f(): + yield 1 + return 2 + + def g(): + yield 1 + return 'asdf' + + def h(): + yield 1 + return None + + def i(): + yield 1 + return + + def j(): + yield 1 + + def k(): + yield 1 + yield from g() + + assert is_generator_with_return_value(f) + assert is_generator_with_return_value(g) + assert not is_generator_with_return_value(h) + assert not is_generator_with_return_value(i) + assert not is_generator_with_return_value(j) + assert not is_generator_with_return_value(k) # not recursive diff --git a/tests/test_utils_python.py b/tests/test_utils_python.py index 3e1148354..b79e0ac1c 100644 --- a/tests/test_utils_python.py +++ b/tests/test_utils_python.py @@ -1,16 +1,17 @@ -import gc import functools +import gc import operator +import platform import unittest from itertools import count -import platform -import six +from warnings import catch_warnings from scrapy.utils.python import ( memoizemethod_noargs, binary_is_text, equal_attributes, - WeakKeyCache, stringify_dict, get_func_args, to_bytes, to_unicode, + WeakKeyCache, get_func_args, to_bytes, to_unicode, without_none_values, MutableChain) + __doctests__ = ['scrapy.utils.python'] @@ -21,8 +22,12 @@ class MutableChainTest(unittest.TestCase): m.extend([7, 8]) m.extend([9, 10], (11, 12)) self.assertEqual(next(m), 0) - self.assertEqual(m.next(), 1) - self.assertEqual(m.__next__(), 2) + self.assertEqual(m.__next__(), 1) + with catch_warnings(record=True) as warnings: + self.assertEqual(m.next(), 2) + self.assertEqual(len(warnings), 1) + self.assertIn('scrapy.utils.python.MutableChain.__next__', + str(warnings[0].message)) self.assertEqual(list(m), list(range(3, 13))) @@ -163,33 +168,6 @@ class UtilsPythonTestCase(unittest.TestCase): gc.collect() self.assertFalse(len(wk._weakdict)) - @unittest.skipUnless(six.PY2, "deprecated function") - def test_stringify_dict(self): - d = {'a': 123, u'b': b'c', u'd': u'e', object(): u'e'} - d2 = stringify_dict(d, keys_only=False) - self.assertEqual(d, d2) - self.assertIsNot(d, d2) # shouldn't modify in place - self.assertFalse(any(isinstance(x, six.text_type) for x in d2.keys())) - self.assertFalse(any(isinstance(x, six.text_type) for x in d2.values())) - - @unittest.skipUnless(six.PY2, "deprecated function") - def test_stringify_dict_tuples(self): - tuples = [('a', 123), (u'b', 'c'), (u'd', u'e'), (object(), u'e')] - d = dict(tuples) - d2 = stringify_dict(tuples, keys_only=False) - self.assertEqual(d, d2) - self.assertIsNot(d, d2) # shouldn't modify in place - self.assertFalse(any(isinstance(x, six.text_type) for x in d2.keys()), d2.keys()) - self.assertFalse(any(isinstance(x, six.text_type) for x in d2.values())) - - @unittest.skipUnless(six.PY2, "deprecated function") - def test_stringify_dict_keys_only(self): - d = {'a': 123, u'b': 'c', u'd': u'e', object(): u'e'} - d2 = stringify_dict(d) - self.assertEqual(d, d2) - self.assertIsNot(d, d2) # shouldn't modify in place - self.assertFalse(any(isinstance(x, six.text_type) for x in d2.keys())) - def test_get_func_args(self): def f1(a, b, c): pass @@ -227,16 +205,15 @@ class UtilsPythonTestCase(unittest.TestCase): if platform.python_implementation() == 'CPython': # TODO: how do we fix this to return the actual argument names? - self.assertEqual(get_func_args(six.text_type.split), []) + self.assertEqual(get_func_args(str.split), []) self.assertEqual(get_func_args(" ".join), []) self.assertEqual(get_func_args(operator.itemgetter(2)), []) else: - stripself = not six.PY2 # PyPy3 exposes them as methods self.assertEqual( - get_func_args(six.text_type.split, stripself), ['sep', 'maxsplit']) - self.assertEqual(get_func_args(" ".join, stripself), ['list']) + get_func_args(str.split, stripself=True), ['sep', 'maxsplit']) + self.assertEqual(get_func_args(" ".join, stripself=True), ['list']) self.assertEqual( - get_func_args(operator.itemgetter(2), stripself), ['obj']) + get_func_args(operator.itemgetter(2), stripself=True), ['obj']) def test_without_none_values(self): diff --git a/tests/test_utils_reqser.py b/tests/test_utils_reqser.py index 11ac56897..06d9c004c 100644 --- a/tests/test_utils_reqser.py +++ b/tests/test_utils_reqser.py @@ -1,8 +1,4 @@ -# -*- coding: utf-8 -*- import unittest -import sys - -import six from scrapy.http import Request, FormRequest from scrapy.spiders import Spider @@ -80,8 +76,6 @@ class RequestSerializationTest(unittest.TestCase): self._assert_serializes_ok(r, spider=self.spider) def test_mixin_private_callback_serialization(self): - if sys.version_info[0] < 3: - return r = Request("http://www.example.com", callback=self.spider._TestSpiderMixin__mixin_callback, errback=self.spider.handle_error) @@ -119,9 +113,8 @@ class RequestSerializationTest(unittest.TestCase): def test_private_name_mangling(self): self._assert_mangles_to( self.spider, '_TestSpider__parse_item_private') - if sys.version_info[0] >= 3: - self._assert_mangles_to( - self.spider, '_TestSpiderMixin__mixin_callback') + self._assert_mangles_to( + self.spider, '_TestSpiderMixin__mixin_callback') def test_unserializable_callback1(self): r = Request("http://www.example.com", callback=lambda x: x) diff --git a/tests/test_utils_request.py b/tests/test_utils_request.py index e8a4eb3ea..3e664fc74 100644 --- a/tests/test_utils_request.py +++ b/tests/test_utils_request.py @@ -1,9 +1,9 @@ -from __future__ import print_function import unittest from scrapy.http import Request from scrapy.utils.request import request_fingerprint, _fingerprint_cache, \ request_authenticate, request_httprepr + class UtilsRequestTest(unittest.TestCase): def test_request_fingerprint(self): @@ -17,7 +17,7 @@ class UtilsRequestTest(unittest.TestCase): self.assertNotEqual(request_fingerprint(r1), request_fingerprint(r2)) # make sure caching is working - self.assertEqual(request_fingerprint(r1), _fingerprint_cache[r1][None]) + self.assertEqual(request_fingerprint(r1), _fingerprint_cache[r1][(None, False)]) r1 = Request("http://www.example.com/members/offers.html") r2 = Request("http://www.example.com/members/offers.html") @@ -42,6 +42,13 @@ class UtilsRequestTest(unittest.TestCase): self.assertEqual(request_fingerprint(r3, include_headers=['accept-language', 'sessionid']), request_fingerprint(r3, include_headers=['SESSIONID', 'Accept-Language'])) + r1 = Request("http://www.example.com/test.html") + r2 = Request("http://www.example.com/test.html#fragment") + self.assertEqual(request_fingerprint(r1), request_fingerprint(r2)) + self.assertEqual(request_fingerprint(r1), request_fingerprint(r1, keep_fragments=True)) + self.assertNotEqual(request_fingerprint(r2), request_fingerprint(r2, keep_fragments=True)) + self.assertNotEqual(request_fingerprint(r1), request_fingerprint(r2, keep_fragments=True)) + r1 = Request("http://www.example.com") r2 = Request("http://www.example.com", method='POST') r3 = Request("http://www.example.com", method='POST', body=b'request body') diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index bea4dade3..6ebf290c0 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -1,12 +1,13 @@ import os import unittest -from six.moves.urllib.parse import urlparse +from urllib.parse import urlparse from scrapy.http import Response, TextResponse, HtmlResponse from scrapy.utils.python import to_bytes from scrapy.utils.response import (response_httprepr, open_in_browser, get_meta_refresh, get_base_url, response_status_message) + __doctests__ = ['scrapy.utils.response'] diff --git a/tests/test_utils_signal.py b/tests/test_utils_signal.py index 62edd420d..16b7c5c68 100644 --- a/tests/test_utils_signal.py +++ b/tests/test_utils_signal.py @@ -66,6 +66,7 @@ class SendCatchLogDeferredTest2(SendCatchLogTest): def _get_result(self, signal, *a, **kw): return send_catch_log_deferred(signal, *a, **kw) + class SendCatchLogTest2(unittest.TestCase): def test_error_logged_if_deferred_not_supported(self): diff --git a/tests/test_utils_sitemap.py b/tests/test_utils_sitemap.py index 716bb44eb..db323ab31 100644 --- a/tests/test_utils_sitemap.py +++ b/tests/test_utils_sitemap.py @@ -2,6 +2,7 @@ import unittest from scrapy.utils.sitemap import Sitemap, sitemap_urls_from_robots + class SitemapTest(unittest.TestCase): def test_sitemap(self): diff --git a/tests/test_utils_spider.py b/tests/test_utils_spider.py index 045e72117..ee7d17062 100644 --- a/tests/test_utils_spider.py +++ b/tests/test_utils_spider.py @@ -1,20 +1,19 @@ import unittest + +from scrapy import Spider from scrapy.http import Request from scrapy.item import BaseItem from scrapy.utils.spider import iterate_spider_output, iter_spider_classes -from scrapy.spiders import CrawlSpider - -class MyBaseSpider(CrawlSpider): - pass # abstract spider - -class MySpider1(MyBaseSpider): +class MySpider1(Spider): name = 'myspider1' -class MySpider2(MyBaseSpider): + +class MySpider2(Spider): name = 'myspider2' + class UtilsSpidersTestCase(unittest.TestCase): def test_iterate_spider_output(self): @@ -32,6 +31,6 @@ class UtilsSpidersTestCase(unittest.TestCase): it = iter_spider_classes(tests.test_utils_spider) self.assertEqual(set(it), {MySpider1, MySpider2}) + if __name__ == "__main__": unittest.main() - diff --git a/tests/test_utils_trackref.py b/tests/test_utils_trackref.py index c6072fc0d..16e02f919 100644 --- a/tests/test_utils_trackref.py +++ b/tests/test_utils_trackref.py @@ -1,7 +1,8 @@ -import six import unittest +from io import StringIO +from unittest import mock + from scrapy.utils import trackref -from tests import mock class Foo(trackref.object_ref): @@ -38,12 +39,12 @@ Live References Bar 1 oldest: 0s ago ''') - @mock.patch('sys.stdout', new_callable=six.StringIO) + @mock.patch('sys.stdout', new_callable=StringIO) def test_print_live_refs_empty(self, stdout): trackref.print_live_refs() self.assertEqual(stdout.getvalue(), 'Live References\n\n\n') - @mock.patch('sys.stdout', new_callable=six.StringIO) + @mock.patch('sys.stdout', new_callable=StringIO) def test_print_live_refs_with_objects(self, stdout): o1 = Foo() # NOQA trackref.print_live_refs() diff --git a/tests/test_utils_url.py b/tests/test_utils_url.py index e6588055c..21e9a056a 100644 --- a/tests/test_utils_url.py +++ b/tests/test_utils_url.py @@ -1,13 +1,10 @@ # -*- coding: utf-8 -*- import unittest -import six -from six.moves.urllib.parse import urlparse - from scrapy.spiders import Spider from scrapy.utils.url import (url_is_from_any_domain, url_is_from_spider, - add_http_if_no_scheme, guess_scheme, - parse_url, strip_url) + add_http_if_no_scheme, guess_scheme, strip_url) + __doctests__ = ['scrapy.utils.url'] @@ -187,6 +184,7 @@ class AddHttpIfNoScheme(unittest.TestCase): class GuessSchemeTest(unittest.TestCase): pass + def create_guess_scheme_t(args): def do_expected(self): url = guess_scheme(args[0]) @@ -195,6 +193,7 @@ def create_guess_scheme_t(args): args[0], url, args[1]) return do_expected + def create_skipped_scheme_t(args): def do_expected(self): raise unittest.SkipTest(args[2]) diff --git a/tests/test_webclient.py b/tests/test_webclient.py index 766329b57..746367b41 100644 --- a/tests/test_webclient.py +++ b/tests/test_webclient.py @@ -3,9 +3,9 @@ from twisted.internet import defer Tests borrowed from the twisted.web.client tests. """ import os -import six import shutil +import OpenSSL.SSL from twisted.trial import unittest from twisted.web import server, static, util, resource from twisted.internet import reactor, defer @@ -15,8 +15,12 @@ from twisted.protocols.policies import WrappingFactory from twisted.internet.defer import inlineCallbacks from scrapy.core.downloader import webclient as client +from scrapy.core.downloader.contextfactory import ScrapyClientContextFactory from scrapy.http import Request, Headers +from scrapy.settings import Settings +from scrapy.utils.misc import create_instance from scrapy.utils.python import to_bytes, to_unicode +from tests.mockserver import ssl_context_factory def getPage(url, contextFactory=None, response_transform=None, *args, **kwargs): @@ -73,26 +77,6 @@ class ParseUrlTestCase(unittest.TestCase): to_bytes(x) if not isinstance(x, int) else x for x in test) self.assertEqual(client._parse(url), test, url) - def test_externalUnicodeInterference(self): - """ - L{client._parse} should return C{str} for the scheme, host, and path - elements of its return tuple, even when passed an URL which has - previously been passed to L{urlparse} as a C{unicode} string. - """ - if not six.PY2: - raise unittest.SkipTest( - "Applies only to Py2, as urls can be ONLY unicode on Py3") - badInput = u'http://example.com/path' - goodInput = badInput.encode('ascii') - self._parse(badInput) # cache badInput in urlparse_cached - scheme, netloc, host, port, path = self._parse(goodInput) - self.assertTrue(isinstance(scheme, str)) - self.assertTrue(isinstance(netloc, str)) - self.assertTrue(isinstance(host, str)) - self.assertTrue(isinstance(path, str)) - self.assertTrue(isinstance(port, int)) - - class ScrapyHTTPPageGetterTests(unittest.TestCase): @@ -313,7 +297,7 @@ class WebClientTestCase(unittest.TestCase): def cleanup(passthrough): # Clean up the server which is hanging around not doing # anything. - connected = list(six.iterkeys(self.wrapper.protocols)) + connected = list(self.wrapper.protocols.keys()) # There might be nothing here if the server managed to already see # that the connection was lost. if connected: @@ -363,3 +347,55 @@ class WebClientTestCase(unittest.TestCase): self.assertEqual(content_encoding, EncodingResource.out_encoding) self.assertEqual( response.body.decode(content_encoding), to_unicode(original_body)) + + +class WebClientSSLTestCase(unittest.TestCase): + context_factory = None + + def _listen(self, site): + return reactor.listenSSL( + 0, site, + contextFactory=self.context_factory or ssl_context_factory(), + interface="127.0.0.1") + + def getURL(self, path): + return "https://127.0.0.1:%d/%s" % (self.portno, path) + + def setUp(self): + self.tmpname = self.mktemp() + os.mkdir(self.tmpname) + FilePath(self.tmpname).child("file").setContent(b"0123456789") + r = static.File(self.tmpname) + r.putChild(b"payload", PayloadResource()) + self.site = server.Site(r, timeout=None) + self.wrapper = WrappingFactory(self.site) + self.port = self._listen(self.wrapper) + self.portno = self.port.getHost().port + + @inlineCallbacks + def tearDown(self): + yield self.port.stopListening() + shutil.rmtree(self.tmpname) + + def testPayload(self): + s = "0123456789" * 10 + return getPage(self.getURL("payload"), body=s).addCallback( + self.assertEqual, to_bytes(s)) + + +class WebClientCustomCiphersSSLTestCase(WebClientSSLTestCase): + # we try to use a cipher that is not enabled by default in OpenSSL + custom_ciphers = 'CAMELLIA256-SHA' + context_factory = ssl_context_factory(cipher_string=custom_ciphers) + + def testPayload(self): + s = "0123456789" * 10 + settings = Settings({'DOWNLOADER_CLIENT_TLS_CIPHERS': self.custom_ciphers}) + client_context_factory = create_instance(ScrapyClientContextFactory, settings=settings, crawler=None) + return getPage(self.getURL("payload"), body=s, + contextFactory=client_context_factory).addCallback(self.assertEqual, to_bytes(s)) + + def testPayloadDefaultCiphers(self): + s = "0123456789" * 10 + d = getPage(self.getURL("payload"), body=s, contextFactory=ScrapyClientContextFactory()) + return self.assertFailure(d, OpenSSL.SSL.Error) diff --git a/tox.ini b/tox.ini index 157a8b3ed..b62100026 100644 --- a/tox.ini +++ b/tox.ini @@ -4,18 +4,16 @@ # and then run "tox" from this directory. [tox] -envlist = py27 +envlist = security,flake8,py3 +minversion = 1.7.0 [testenv] deps = -ctests/constraints.txt - -rrequirements-py2.txt + -rtests/requirements-py3.txt # Extras - botocore - google-cloud-storage - Pillow != 3.0.0 - leveldb - -rtests/requirements-py2.txt + botocore>=1.3.23 + Pillow>=3.4.2 passenv = S3_TEST_FILE_URI AWS_ACCESS_KEY_ID @@ -23,76 +21,55 @@ passenv = GCS_TEST_FILE_URI GCS_PROJECT_ID commands = - py.test --cov=scrapy --cov-report= {posargs:scrapy tests} + py.test --cov=scrapy --cov-report= {posargs:--durations=10 docs scrapy tests} -[testenv:trusty] -basepython = python2.7 +[testenv:security] +basepython = python3 deps = - pyOpenSSL==0.13 - lxml==3.3.3 - Twisted==13.2.0 - boto==2.20.1 - Pillow==2.3.0 - cssselect==0.9.1 - zope.interface==4.0.5 - -rtests/requirements-py2.txt - -[testenv:jessie] -# https://packages.debian.org/en/jessie/python/ -# https://packages.debian.org/en/jessie/zope/ -basepython = python2.7 -deps = - cryptography==0.6.1 - pyOpenSSL==0.14 - lxml==3.4.0 - Twisted==14.0.2 - boto==2.34.0 - Pillow==2.6.1 - cssselect==0.9.1 - zope.interface==4.1.1 - -rtests/requirements-py2.txt -# Not used directly but allows boto GCE plugins to load. -# https://github.com/GoogleCloudPlatform/compute-image-packages/issues/262 - google-compute-engine==2.8.12 - -[testenv:trunk] -basepython = python2.7 + bandit commands = - pip install -U https://github.com/scrapy/w3lib/archive/master.zip#egg=w3lib - pip install -U https://github.com/scrapy/queuelib/archive/master.zip#egg=queuelib - py.test --cov=scrapy --cov-report= {posargs:scrapy tests} + bandit -r -c .bandit.yml {posargs:scrapy} -[testenv:pypy] -basepython = pypy -commands = - py.test {posargs:scrapy tests} - -[testenv:py34] -basepython = python3.4 +[testenv:flake8] +basepython = python3 deps = - -ctests/constraints.txt - -rrequirements-py3.txt - # Extras - Pillow - -rtests/requirements-py3.txt - -[testenv:py35] -basepython = python3.5 -deps = {[testenv:py34]deps} - -[testenv:py36] -basepython = python3.6 -deps = {[testenv:py34]deps} - -[testenv:py37] -basepython = python3.7 -deps = {[testenv:py34]deps} + {[testenv]deps} + pytest-flake8 +commands = + py.test --flake8 {posargs:docs scrapy tests} [testenv:pypy3] basepython = pypy3 -deps = {[testenv:py34]deps} commands = - py.test {posargs:scrapy tests} + py.test {posargs:--durations=10 docs scrapy tests} + +[testenv:pinned] +basepython = python3 +deps = + -ctests/constraints.txt + cryptography==2.0 + cssselect==0.9.1 + lxml==3.5.0 + parsel==1.5.0 + Protego==0.1.15 + PyDispatcher==2.0.5 + pyOpenSSL==16.2.0 + queuelib==1.4.2 + service_identity==16.0.0 + six==1.10.0 + Twisted==17.9.0 + w3lib==1.17.0 + zope.interface==4.1.3 + -rtests/requirements-py3.txt + # Extras + botocore==1.3.23 + Pillow==3.4.2 + +[testenv:extra-deps] +deps = + {[testenv]deps} + reppy + robotexclusionrulesparser [docs] changedir = docs @@ -100,19 +77,36 @@ deps = -rdocs/requirements.txt [testenv:docs] +basepython = python3 changedir = {[docs]changedir} deps = {[docs]deps} commands = sphinx-build -W -b html . {envtmpdir}/html [testenv:docs-coverage] +basepython = python3 changedir = {[docs]changedir} deps = {[docs]deps} commands = sphinx-build -b coverage . {envtmpdir}/coverage [testenv:docs-links] +basepython = python3 changedir = {[docs]changedir} deps = {[docs]deps} commands = sphinx-build -W -b linkcheck . {envtmpdir}/linkcheck + +[asyncio] +commands = + {[testenv]commands} --reactor=asyncio + +[testenv:py35-asyncio] +basepython = python3.5 +deps = {[testenv]deps} +commands = {[asyncio]commands} + +[testenv:py38-asyncio] +basepython = python3.8 +deps = {[testenv]deps} +commands = {[asyncio]commands}