+ “The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”
+ by Albert Einstein
+ (about)
+
+
+ “There are only two ways to live your life. One is as though nothing is a miracle. The other is as though everything is a miracle.”
+ by Albert Einstein
+ (about)
+
+
+
+
+
\ No newline at end of file
diff --git a/docs/conf.py b/docs/conf.py
index 34dd5bcb7..6ec4582b1 100644
--- a/docs/conf.py
+++ b/docs/conf.py
@@ -27,10 +27,12 @@ sys.path.insert(0, path.dirname(path.dirname(__file__)))
# Add any Sphinx extension module names here, as strings. They can be extensions
# coming with Sphinx (named 'sphinx.ext.*') or your custom ones.
extensions = [
+ 'notfound.extension',
'scrapydocs',
'sphinx.ext.autodoc',
'sphinx.ext.coverage',
'sphinx.ext.intersphinx',
+ 'sphinx.ext.viewcode',
]
# Add any paths that contain templates here, relative to this directory.
@@ -237,7 +239,7 @@ coverage_ignore_pyobjects = [
r'\bContractsManager\b$',
# For default contracts we only want to document their general purpose in
- # their constructor, the methods they reimplement to achieve that purpose
+ # their __init__ method, the methods they reimplement to achieve that purpose
# should be irrelevant to developers using those contracts.
r'\w+Contract\.(adjust_request_args|(pre|post)_process)$',
@@ -273,4 +275,5 @@ coverage_ignore_pyobjects = [
intersphinx_mapping = {
'python': ('https://docs.python.org/3', None),
+ 'sphinx': ('https://www.sphinx-doc.org/en/stable', None),
}
diff --git a/docs/conftest.py b/docs/conftest.py
new file mode 100644
index 000000000..8c735e838
--- /dev/null
+++ b/docs/conftest.py
@@ -0,0 +1,29 @@
+import os
+from doctest import ELLIPSIS, NORMALIZE_WHITESPACE
+
+from scrapy.http.response.html import HtmlResponse
+from sybil import Sybil
+from sybil.parsers.codeblock import CodeBlockParser
+from sybil.parsers.doctest import DocTestParser
+from sybil.parsers.skip import skip
+
+
+def load_response(url, filename):
+ input_path = os.path.join(os.path.dirname(__file__), '_tests', filename)
+ with open(input_path, 'rb') as input_file:
+ return HtmlResponse(url, body=input_file.read())
+
+
+def setup(namespace):
+ namespace['load_response'] = load_response
+
+
+pytest_collect_file = Sybil(
+ parsers=[
+ DocTestParser(optionflags=ELLIPSIS | NORMALIZE_WHITESPACE),
+ CodeBlockParser(future_imports=['print_function']),
+ skip,
+ ],
+ pattern='*.rst',
+ setup=setup,
+).pytest()
diff --git a/docs/contributing.rst b/docs/contributing.rst
index 28dea74de..f084bd23d 100644
--- a/docs/contributing.rst
+++ b/docs/contributing.rst
@@ -177,20 +177,19 @@ Documentation policies
======================
For reference documentation of API members (classes, methods, etc.) use
-docstrings and make sure that the Sphinx documentation uses the autodoc_
-extension to pull the docstrings. API reference documentation should follow
-docstring conventions (`PEP 257`_) and be IDE-friendly: short, to the point,
-and it may provide short examples.
+docstrings and make sure that the Sphinx documentation uses the
+:mod:`~sphinx.ext.autodoc` extension to pull the docstrings. API reference
+documentation should follow docstring conventions (`PEP 257`_) and be
+IDE-friendly: short, to the point, and it may provide short examples.
Other types of documentation, such as tutorials or topics, should be covered in
files within the ``docs/`` directory. This includes documentation that is
specific to an API member, but goes beyond API reference documentation.
-In any case, if something is covered in a docstring, use the autodoc_
-extension to pull the docstring into the documentation instead of duplicating
-the docstring in files within the ``docs/`` directory.
-
-.. _autodoc: http://www.sphinx-doc.org/en/stable/ext/autodoc.html
+In any case, if something is covered in a docstring, use the
+:mod:`~sphinx.ext.autodoc` extension to pull the docstring into the
+documentation instead of duplicating the docstring in files within the
+``docs/`` directory.
Tests
=====
diff --git a/docs/faq.rst b/docs/faq.rst
index 9733471bf..080d81981 100644
--- a/docs/faq.rst
+++ b/docs/faq.rst
@@ -69,11 +69,11 @@ Here's an example spider using BeautifulSoup API, with ``lxml`` as the HTML pars
What Python versions does Scrapy support?
-----------------------------------------
-Scrapy is supported under Python 2.7 and Python 3.5+
+Scrapy is supported under Python 3.5+
under CPython (default Python implementation) and PyPy (starting with PyPy 5.9).
-Python 2.6 support was dropped starting at Scrapy 0.20.
Python 3 support was added in Scrapy 1.1.
PyPy support was added in Scrapy 1.4, PyPy3 support was added in Scrapy 1.5.
+Python 2 support was dropped in Scrapy 2.0.
.. note::
For Python 3 support on Windows, it is recommended to use
diff --git a/docs/intro/install.rst b/docs/intro/install.rst
index 2bf98dbdc..e924b5303 100644
--- a/docs/intro/install.rst
+++ b/docs/intro/install.rst
@@ -7,7 +7,7 @@ Installation guide
Installing Scrapy
=================
-Scrapy runs on Python 2.7 and Python 3.5 or above
+Scrapy runs on Python 3.5 or above
under CPython (default Python implementation) and PyPy (starting with PyPy 5.9).
If you're using `Anaconda`_ or `Miniconda`_, you can install the package from
@@ -102,10 +102,8 @@ just like any other Python package.
(See :ref:`platform-specific guides `
below for non-Python dependencies that you may need to install beforehand).
-Python virtualenvs can be created to use Python 2 by default, or Python 3 by default.
-
-* If you want to install scrapy with Python 3, install scrapy within a Python 3 virtualenv.
-* And if you want to install scrapy with Python 2, install scrapy within a Python 2 virtualenv.
+Python virtualenvs can be created to use Python 2 by default, or Python 3 by default. As Scrapy
+only supports Python 3, make sure you created a Python 3 virtualenv.
.. _virtualenv: https://virtualenv.pypa.io
.. _virtualenv installation instructions: https://virtualenv.pypa.io/en/stable/installation/
@@ -149,16 +147,12 @@ typically too old and slow to catch up with latest Scrapy.
To install scrapy on Ubuntu (or Ubuntu-based) systems, you need to install
these dependencies::
- sudo apt-get install python-dev python-pip libxml2-dev libxslt1-dev zlib1g-dev libffi-dev libssl-dev
+ sudo apt-get install python3 python3-dev python3-pip libxml2-dev libxslt1-dev zlib1g-dev libffi-dev libssl-dev
-- ``python-dev``, ``zlib1g-dev``, ``libxml2-dev`` and ``libxslt1-dev``
+- ``python3-dev``, ``zlib1g-dev``, ``libxml2-dev`` and ``libxslt1-dev``
are required for ``lxml``
- ``libssl-dev`` and ``libffi-dev`` are required for ``cryptography``
-If you want to install scrapy on Python 3, you’ll also need Python 3 development headers::
-
- sudo apt-get install python3 python3-dev
-
Inside a :ref:`virtualenv `,
you can install Scrapy with ``pip`` after that::
@@ -290,5 +284,5 @@ For details, see `Issue #2473 `_.
.. _zsh: https://www.zsh.org/
.. _Scrapinghub: https://scrapinghub.com
.. _Anaconda: https://docs.anaconda.com/anaconda/
-.. _Miniconda: https://conda.io/docs/user-guide/install/index.html
+.. _Miniconda: https://docs.conda.io/projects/conda/en/latest/user-guide/install/index.html
.. _conda-forge: https://conda-forge.org/
diff --git a/docs/intro/tutorial.rst b/docs/intro/tutorial.rst
index a190ce407..6b15a5fbd 100644
--- a/docs/intro/tutorial.rst
+++ b/docs/intro/tutorial.rst
@@ -78,9 +78,9 @@ Our first Spider
Spiders are classes that you define and that Scrapy uses to scrape information
from a website (or a group of websites). They must subclass
-:class:`scrapy.Spider` and define the initial requests to make, optionally how
-to follow links in the pages, and how to parse the downloaded page content to
-extract data.
+:class:`~scrapy.spiders.Spider` and define the initial requests to make,
+optionally how to follow links in the pages, and how to parse the downloaded
+page content to extract data.
This is the code for our first Spider. Save it in a file named
``quotes_spider.py`` under the ``tutorial/spiders`` directory in your project::
@@ -235,13 +235,16 @@ You will see something like::
[s] shelp() Shell help (print this help)
[s] fetch(req_or_url) Fetch request (or URL) and update local objects
[s] view(response) View response in a browser
- >>>
Using the shell, you can try selecting elements using `CSS`_ with the response
-object::
+object:
- >>> response.css('title')
- []
+.. invisible-code-block: python
+
+ response = load_response('http://quotes.toscrape.com/page/1/', 'quotes1.html')
+
+>>> response.css('title')
+[]
The result of running ``response.css('title')`` is a list-like object called
:class:`~scrapy.selector.SelectorList`, which represents a list of
@@ -372,6 +375,9 @@ we want::
We get a list of selectors for the quote HTML elements with::
>>> response.css("div.quote")
+ [,
+ ,
+ ...]
Each of the selectors returned by the query above allows us to run further
queries over their sub-elements. Let's assign the first selector to a
@@ -396,6 +402,12 @@ to get all of them::
>>> tags
['change', 'deep-thoughts', 'thinking', 'world']
+.. invisible-code-block: python
+
+ from sys import version_info
+
+.. skip: next if(version_info < (3, 6), reason="Only Python 3.6+ dictionaries match the output")
+
Having figured out how to extract each bit, we can now iterate over all the
quotes elements and put them together into a Python dictionary::
@@ -404,10 +416,9 @@ quotes elements and put them together into a Python dictionary::
... author = quote.css("small.author::text").get()
... tags = quote.css("div.tags a.tag::text").getall()
... print(dict(text=text, author=author, tags=tags))
- {'tags': ['change', 'deep-thoughts', 'thinking', 'world'], 'author': 'Albert Einstein', 'text': '“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”'}
- {'tags': ['abilities', 'choices'], 'author': 'J.K. Rowling', 'text': '“It is our choices, Harry, that show what we truly are, far more than our abilities.”'}
- ... a few more of these, omitted for brevity
- >>>
+ {'text': '“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”', 'author': 'Albert Einstein', 'tags': ['change', 'deep-thoughts', 'thinking', 'world']}
+ {'text': '“It is our choices, Harry, that show what we truly are, far more than our abilities.”', 'author': 'J.K. Rowling', 'tags': ['abilities', 'choices']}
+ ...
Extracting data in our spider
-----------------------------
@@ -521,7 +532,7 @@ There is also an ``attrib`` property available
(see :ref:`selecting-attributes` for more)::
>>> response.css('li.next a').attrib['href']
- '/page/2'
+ '/page/2/'
Let's see now our spider modified to recursively follow the link to the next
page, extracting data from it::
diff --git a/docs/news.rst b/docs/news.rst
index 2bcfe4d1c..9dfd28508 100644
--- a/docs/news.rst
+++ b/docs/news.rst
@@ -6,22 +6,246 @@ Release notes
.. note:: Scrapy 1.x will be the last series supporting Python 2. Scrapy 2.0,
planned for Q4 2019 or Q1 2020, will support **Python 3 only**.
+.. _release-1.8.0:
+
+Scrapy 1.8.0 (2019-10-28)
+-------------------------
+
+Highlights:
+
+* Dropped Python 3.4 support and updated minimum requirements; made Python 3.8
+ support official
+* New :meth:`Request.from_curl ` class method
+* New :setting:`ROBOTSTXT_PARSER` and :setting:`ROBOTSTXT_USER_AGENT` settings
+* New :setting:`DOWNLOADER_CLIENT_TLS_CIPHERS` and
+ :setting:`DOWNLOADER_CLIENT_TLS_VERBOSE_LOGGING` settings
+
+Backward-incompatible changes
+~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
+
+* Python 3.4 is no longer supported, and some of the minimum requirements of
+ Scrapy have also changed:
+
+ * cssselect_ 0.9.1
+ * cryptography_ 2.0
+ * lxml_ 3.5.0
+ * pyOpenSSL_ 16.2.0
+ * queuelib_ 1.4.2
+ * service_identity_ 16.0.0
+ * six_ 1.10.0
+ * Twisted_ 17.9.0 (16.0.0 with Python 2)
+ * zope.interface_ 4.1.3
+
+ (:issue:`3892`)
+
+* ``JSONRequest`` is now called :class:`~scrapy.http.JsonRequest` for
+ consistency with similar classes (:issue:`3929`, :issue:`3982`)
+
+* If you are using a custom context factory
+ (:setting:`DOWNLOADER_CLIENTCONTEXTFACTORY`), its ``__init__`` method must
+ accept two new parameters: ``tls_verbose_logging`` and ``tls_ciphers``
+ (:issue:`2111`, :issue:`3392`, :issue:`3442`, :issue:`3450`)
+
+* :class:`~scrapy.loader.ItemLoader` now turns the values of its input item
+ into lists::
+
+ >>> item = MyItem()
+ >>> item['field'] = 'value1'
+ >>> loader = ItemLoader(item=item)
+ >>> item['field']
+ ['value1']
+
+ This is needed to allow adding values to existing fields
+ (``loader.add_value('field', 'value2')``).
+
+ (:issue:`3804`, :issue:`3819`, :issue:`3897`, :issue:`3976`, :issue:`3998`,
+ :issue:`4036`)
+
+See also :ref:`1.8-deprecation-removals` below.
+
+
+New features
+~~~~~~~~~~~~
+
+* A new :meth:`Request.from_curl ` class
+ method allows :ref:`creating a request from a cURL command
+ ` (:issue:`2985`, :issue:`3862`)
+
+* A new :setting:`ROBOTSTXT_PARSER` setting allows choosing which robots.txt_
+ parser to use. It includes built-in support for
+ :ref:`RobotFileParser `,
+ :ref:`Protego ` (default), :ref:`Reppy `, and
+ :ref:`Robotexclusionrulesparser `, and allows you to
+ :ref:`implement support for additional parsers
+ ` (:issue:`754`, :issue:`2669`,
+ :issue:`3796`, :issue:`3935`, :issue:`3969`, :issue:`4006`)
+
+* A new :setting:`ROBOTSTXT_USER_AGENT` setting allows defining a separate
+ user agent string to use for robots.txt_ parsing (:issue:`3931`,
+ :issue:`3966`)
+
+* :class:`~scrapy.spiders.Rule` no longer requires a :class:`LinkExtractor
+ ` parameter
+ (:issue:`781`, :issue:`4016`)
+
+* Use the new :setting:`DOWNLOADER_CLIENT_TLS_CIPHERS` setting to customize
+ the TLS/SSL ciphers used by the default HTTP/1.1 downloader (:issue:`3392`,
+ :issue:`3442`)
+
+* Set the new :setting:`DOWNLOADER_CLIENT_TLS_VERBOSE_LOGGING` setting to
+ ``True`` to enable debug-level messages about TLS connection parameters
+ after establishing HTTPS connections (:issue:`2111`, :issue:`3450`)
+
+* Callbacks that receive keyword arguments
+ (see :attr:`Request.cb_kwargs `) can now be
+ tested using the new :class:`@cb_kwargs
+ `
+ :ref:`spider contract ` (:issue:`3985`, :issue:`3988`)
+
+* When a :class:`@scrapes ` spider
+ contract fails, all missing fields are now reported (:issue:`766`,
+ :issue:`3939`)
+
+* :ref:`Custom log formats ` can now drop messages by
+ having the corresponding methods of the configured :setting:`LOG_FORMATTER`
+ return ``None`` (:issue:`3984`, :issue:`3987`)
+
+* A much improved completion definition is now available for Zsh_
+ (:issue:`4069`)
+
+
+Bug fixes
+~~~~~~~~~
+
+* :meth:`ItemLoader.load_item() ` no
+ longer makes later calls to :meth:`ItemLoader.get_output_value()
+ ` or
+ :meth:`ItemLoader.load_item() ` return
+ empty data (:issue:`3804`, :issue:`3819`, :issue:`3897`, :issue:`3976`,
+ :issue:`3998`, :issue:`4036`)
+
+* Fixed :class:`~scrapy.statscollectors.DummyStatsCollector` raising a
+ :exc:`TypeError` exception (:issue:`4007`, :issue:`4052`)
+
+* :meth:`FilesPipeline.file_path
+ ` and
+ :meth:`ImagesPipeline.file_path
+ ` no longer choose
+ file extensions that are not `registered with IANA`_ (:issue:`1287`,
+ :issue:`3953`, :issue:`3954`)
+
+* When using botocore_ to persist files in S3, all botocore-supported headers
+ are properly mapped now (:issue:`3904`, :issue:`3905`)
+
+* FTP passwords in :setting:`FEED_URI` containing percent-escaped characters
+ are now properly decoded (:issue:`3941`)
+
+* A memory-handling and error-handling issue in
+ :func:`scrapy.utils.ssl.get_temp_key_info` has been fixed (:issue:`3920`)
+
+
+Documentation
+~~~~~~~~~~~~~
+
+* The documentation now covers how to define and configure a :ref:`custom log
+ format ` (:issue:`3616`, :issue:`3660`)
+
+* API documentation added for :class:`~scrapy.exporters.MarshalItemExporter`
+ and :class:`~scrapy.exporters.PythonItemExporter` (:issue:`3973`)
+
+* API documentation added for :class:`~scrapy.item.BaseItem` and
+ :class:`~scrapy.item.ItemMeta` (:issue:`3999`)
+
+* Minor documentation fixes (:issue:`2998`, :issue:`3398`, :issue:`3597`,
+ :issue:`3894`, :issue:`3934`, :issue:`3978`, :issue:`3993`, :issue:`4022`,
+ :issue:`4028`, :issue:`4033`, :issue:`4046`, :issue:`4050`, :issue:`4055`,
+ :issue:`4056`, :issue:`4061`, :issue:`4072`, :issue:`4071`, :issue:`4079`,
+ :issue:`4081`, :issue:`4089`, :issue:`4093`)
+
+
+.. _1.8-deprecation-removals:
+
+Deprecation removals
+~~~~~~~~~~~~~~~~~~~~
+
+* ``scrapy.xlib`` has been removed (:issue:`4015`)
+
+
+Deprecations
+~~~~~~~~~~~~
+
+* The LevelDB_ storage backend
+ (``scrapy.extensions.httpcache.LeveldbCacheStorage``) of
+ :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware` is
+ deprecated (:issue:`4085`, :issue:`4092`)
+
+* Use of the undocumented ``SCRAPY_PICKLED_SETTINGS_TO_OVERRIDE`` environment
+ variable is deprecated (:issue:`3910`)
+
+* ``scrapy.item.DictItem`` is deprecated, use :class:`~scrapy.item.Item`
+ instead (:issue:`3999`)
+
+
+Other changes
+~~~~~~~~~~~~~
+
+* Minimum versions of optional Scrapy requirements that are covered by
+ continuous integration tests have been updated:
+
+ * botocore_ 1.3.23
+ * Pillow_ 3.4.2
+
+ Lower versions of these optional requirements may work, but it is not
+ guaranteed (:issue:`3892`)
+
+* GitHub templates for bug reports and feature requests (:issue:`3126`,
+ :issue:`3471`, :issue:`3749`, :issue:`3754`)
+
+* Continuous integration fixes (:issue:`3923`)
+
+* Code cleanup (:issue:`3391`, :issue:`3907`, :issue:`3946`, :issue:`3950`,
+ :issue:`4023`, :issue:`4031`)
+
+
+.. _release-1.7.4:
+
+Scrapy 1.7.4 (2019-10-21)
+-------------------------
+
+Revert the fix for :issue:`3804` (:issue:`3819`), which has a few undesired
+side effects (:issue:`3897`, :issue:`3976`).
+
+As a result, when an item loader is initialized with an item,
+:meth:`ItemLoader.load_item() ` once again
+makes later calls to :meth:`ItemLoader.get_output_value()
+` or :meth:`ItemLoader.load_item()
+` return empty data.
+
+
+.. _release-1.7.3:
+
Scrapy 1.7.3 (2019-08-01)
-------------------------
Enforce lxml 4.3.5 or lower for Python 3.4 (:issue:`3912`, :issue:`3918`).
+
+.. _release-1.7.2:
+
Scrapy 1.7.2 (2019-07-23)
-------------------------
Fix Python 2 support (:issue:`3889`, :issue:`3893`, :issue:`3896`).
+.. _release-1.7.1:
+
Scrapy 1.7.1 (2019-07-18)
-------------------------
Re-packaging of Scrapy 1.7.0, which was missing some changes in PyPI.
+
.. _release-1.7.0:
Scrapy 1.7.0 (2019-07-18)
@@ -71,7 +295,7 @@ New features
~~~~~~~~~~~~
* A new scheduler priority queue,
- :class:`scrapy.pqueues.DownloaderAwarePriorityQueue`, may be
+ ``scrapy.pqueues.DownloaderAwarePriorityQueue``, may be
:ref:`enabled ` for a significant
scheduling improvement on crawls targetting multiple web domains, at the
cost of no :setting:`CONCURRENT_REQUESTS_PER_IP` support (:issue:`3520`)
@@ -84,12 +308,12 @@ New features
convenient way to build JSON requests (:issue:`3504`, :issue:`3505`)
* A ``process_request`` callback passed to the :class:`~scrapy.spiders.Rule`
- constructor now receives the :class:`~scrapy.http.Response` object that
+ ``__init__`` method now receives the :class:`~scrapy.http.Response` object that
originated the request as its second argument (:issue:`3682`)
* A new ``restrict_text`` parameter for the
:attr:`LinkExtractor `
- constructor allows filtering links by linking text (:issue:`3622`,
+ ``__init__`` method allows filtering links by linking text (:issue:`3622`,
:issue:`3635`)
* A new :setting:`FEED_STORAGE_S3_ACL` setting allows defining a custom ACL
@@ -150,9 +374,9 @@ Bug fixes
:setting:`AWS_REGION_NAME`, :setting:`AWS_USE_SSL`, :setting:`AWS_VERIFY`
(:issue:`3625`)
-* Fixed a memory leak in :class:`~scrapy.pipelines.media.MediaPipeline`
- affecting, for example, non-200 responses and exceptions from custom
- middlewares (:issue:`3813`)
+* Fixed a memory leak in ``scrapy.pipelines.media.MediaPipeline`` affecting,
+ for example, non-200 responses and exceptions from custom middlewares
+ (:issue:`3813`)
* Requests with private callbacks are now correctly unserialized from disk
(:issue:`3790`)
@@ -255,7 +479,7 @@ The following deprecated APIs have been removed (:issue:`3578`):
* From :class:`~scrapy.selector.Selector`:
- * ``_root`` (both the constructor argument and the object property, use
+ * ``_root`` (both the ``__init__`` method argument and the object property, use
``root``)
* ``extract_unquoted`` (use ``getall``)
@@ -301,7 +525,7 @@ Deprecations
* The ``queuelib.PriorityQueue`` value for the
:setting:`SCHEDULER_PRIORITY_QUEUE` setting is deprecated. Use
- :class:`scrapy.pqueues.ScrapyPriorityQueue` instead.
+ ``scrapy.pqueues.ScrapyPriorityQueue`` instead.
* ``process_request`` callbacks passed to :class:`~scrapy.spiders.Rule` that
do not accept two arguments are deprecated.
@@ -551,12 +775,12 @@ Scrapy 1.5.2 (2019-01-22)
*The fix is backward incompatible*, it enables telnet user-password
authentication by default with a random generated password. If you can't
- upgrade right away, please consider setting :setting:`TELNET_CONSOLE_PORT`
+ upgrade right away, please consider setting :setting:`TELNETCONSOLE_PORT`
out of its default value.
See :ref:`telnet console ` documentation for more info
-* Backport CI build failure under GCE environemnt due to boto import error.
+* Backport CI build failure under GCE environment due to boto import error.
.. _release-1.5.1:
@@ -758,7 +982,9 @@ Enjoy! (Or read on for the rest of changes in this release.)
Deprecations and Backward Incompatible Changes
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
-- Default to ``canonicalize=False`` in :class:`scrapy.linkextractors.LinkExtractor`
+- Default to ``canonicalize=False`` in
+ :class:`scrapy.linkextractors.LinkExtractor
+ `
(:issue:`2537`, fixes :issue:`1941` and :issue:`1982`):
**warning, this is technically backward-incompatible**
- Enable memusage extension by default (:issue:`2539`, fixes :issue:`2187`);
@@ -794,10 +1020,13 @@ New Features
- New ``data:`` URI download handler (:issue:`2334`, fixes :issue:`2156`)
- Log cache directory when HTTP Cache is used (:issue:`2611`, fixes :issue:`2604`)
- Warn users when project contains duplicate spider names (fixes :issue:`2181`)
-- :class:`CaselessDict` now accepts ``Mapping`` instances and not only dicts (:issue:`2646`)
-- :ref:`Media downloads `, with :class:`FilesPipelines`
- or :class:`ImagesPipelines`, can now optionally handle HTTP redirects
- using the new :setting:`MEDIA_ALLOW_REDIRECTS` setting (:issue:`2616`, fixes :issue:`2004`)
+- ``scrapy.utils.datatypes.CaselessDict`` now accepts ``Mapping`` instances and
+ not only dicts (:issue:`2646`)
+- :ref:`Media downloads `, with
+ :class:`~scrapy.pipelines.files.FilesPipeline` or
+ :class:`~scrapy.pipelines.images.ImagesPipeline`, can now optionally handle
+ HTTP redirects using the new :setting:`MEDIA_ALLOW_REDIRECTS` setting
+ (:issue:`2616`, fixes :issue:`2004`)
- Accept non-complete responses from websites using a new
:setting:`DOWNLOAD_FAIL_ON_DATALOSS` setting (:issue:`2590`, fixes :issue:`2586`)
- Optional pretty-printing of JSON and XML items via
@@ -817,8 +1046,8 @@ Bug fixes
- LinkExtractor now strips leading and trailing whitespaces from attributes
(:issue:`2547`, fixes :issue:`1614`)
-- Properly handle whitespaces in action attribute in :class:`FormRequest`
- (:issue:`2548`)
+- Properly handle whitespaces in action attribute in
+ :class:`~scrapy.http.FormRequest` (:issue:`2548`)
- Buffer CONNECT response bytes from proxy until all HTTP headers are received
(:issue:`2495`, fixes :issue:`2491`)
- FTP downloader now works on Python 3, provided you use Twisted>=17.1
@@ -851,7 +1080,8 @@ Cleanups & Refactoring
fixes :issue:`2560`)
- Add omitted ``self`` arguments in default project middleware template (:issue:`2595`)
- Remove redundant ``slot.add_request()`` call in ExecutionEngine (:issue:`2617`)
-- Catch more specific ``os.error`` exception in :class:`FSFilesStore` (:issue:`2644`)
+- Catch more specific ``os.error`` exception in
+ ``scrapy.pipelines.files.FSFilesStore`` (:issue:`2644`)
- Change "localhost" test server certificate (:issue:`2720`)
- Remove unused ``MEMUSAGE_REPORT`` setting (:issue:`2576`)
@@ -868,7 +1098,8 @@ Documentation
(:issue:`2477`, fixes :issue:`2475`)
- FAQ: rewrite note on Python 3 support on Windows (:issue:`2690`)
- Rearrange selector sections (:issue:`2705`)
-- Remove ``__nonzero__`` from :class:`SelectorList` docs (:issue:`2683`)
+- Remove ``__nonzero__`` from :class:`~scrapy.selector.SelectorList`
+ docs (:issue:`2683`)
- Mention how to disable request filtering in documentation of
:setting:`DUPEFILTER_CLASS` setting (:issue:`2714`)
- Add sphinx_rtd_theme to docs setup readme (:issue:`2668`)
@@ -2327,7 +2558,7 @@ Scrapy 0.18.0 (released 2013-08-09)
- Moved persistent (on disk) queues to a separate project (queuelib_) which scrapy now depends on
- Add scrapy commands using external libraries (:issue:`260`)
- Added ``--pdb`` option to ``scrapy`` command line tool
-- Added :meth:`XPathSelector.remove_namespaces` which allows to remove all namespaces from XML documents for convenience (to work with namespace-less XPaths). Documented in :ref:`topics-selectors`.
+- Added :meth:`XPathSelector.remove_namespaces ` which allows to remove all namespaces from XML documents for convenience (to work with namespace-less XPaths). Documented in :ref:`topics-selectors`.
- Several improvements to spider contracts
- New default middleware named MetaRefreshMiddldeware that handles meta-refresh html tag redirections,
- MetaRefreshMiddldeware and RedirectMiddleware have different priorities to address #62
@@ -2448,7 +2679,7 @@ Scrapy changes:
- added options ``-o`` and ``-t`` to the :command:`runspider` command
- documented :doc:`topics/autothrottle` and added to extensions installed by default. You still need to enable it with :setting:`AUTOTHROTTLE_ENABLED`
- major Stats Collection refactoring: removed separation of global/per-spider stats, removed stats-related signals (``stats_spider_opened``, etc). Stats are much simpler now, backward compatibility is kept on the Stats Collector API and signals.
-- added :meth:`~scrapy.contrib.spidermiddleware.SpiderMiddleware.process_start_requests` method to spider middlewares
+- added :meth:`~scrapy.spidermiddlewares.SpiderMiddleware.process_start_requests` method to spider middlewares
- dropped Signals singleton. Signals should now be accesed through the Crawler.signals attribute. See the signals documentation for more info.
- dropped Signals singleton. Signals should now be accesed through the Crawler.signals attribute. See the signals documentation for more info.
- dropped Stats Collector singleton. Stats can now be accessed through the Crawler.stats attribute. See the stats collection documentation for more info.
@@ -2472,7 +2703,7 @@ Scrapy changes:
- removed ``ENCODING_ALIASES`` setting, as encoding auto-detection has been moved to the `w3lib`_ library
- promoted :ref:`topics-djangoitem` to main contrib
- LogFormatter method now return dicts(instead of strings) to support lazy formatting (:issue:`164`, :commit:`dcef7b0`)
-- downloader handlers (:setting:`DOWNLOAD_HANDLERS` setting) now receive settings as the first argument of the constructor
+- downloader handlers (:setting:`DOWNLOAD_HANDLERS` setting) now receive settings as the first argument of the ``__init__`` method
- replaced memory usage acounting with (more portable) `resource`_ module, removed ``scrapy.utils.memory`` module
- removed signal: ``scrapy.mail.mail_sent``
- removed ``TRACK_REFS`` setting, now :ref:`trackrefs ` is always enabled
@@ -2609,7 +2840,8 @@ The numbers like #NNN reference tickets in the old issue tracker (Trac) which is
New features and improvements
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
-- Passed item is now sent in the ``item`` argument of the :signal:`item_passed` (#273)
+- Passed item is now sent in the ``item`` argument of the :signal:`item_passed
+ ` (#273)
- Added verbose option to ``scrapy version`` command, useful for bug reports (#298)
- HTTP cache now stored by default in the project data dir (#279)
- Added project data storage directory (#276, #277)
@@ -2685,7 +2917,7 @@ API changes
- ``Request.copy()`` and ``Request.replace()`` now also copies their ``callback`` and ``errback`` attributes (#231)
- Removed ``UrlFilterMiddleware`` from ``scrapy.contrib`` (already disabled by default)
- Offsite middelware doesn't filter out any request coming from a spider that doesn't have a allowed_domains attribute (#225)
-- Removed Spider Manager ``load()`` method. Now spiders are loaded in the constructor itself.
+- Removed Spider Manager ``load()`` method. Now spiders are loaded in the ``__init__`` method itself.
- Changes to Scrapy Manager (now called "Crawler"):
- ``scrapy.core.manager.ScrapyManager`` class renamed to ``scrapy.crawler.Crawler``
- ``scrapy.core.manager.scrapymanager`` singleton moved to ``scrapy.project.crawler``
@@ -2810,23 +3042,35 @@ First release of Scrapy.
.. _AJAX crawleable urls: https://developers.google.com/webmasters/ajax-crawling/docs/getting-started?csw=1
+.. _botocore: https://github.com/boto/botocore
.. _chunked transfer encoding: https://en.wikipedia.org/wiki/Chunked_transfer_encoding
.. _ClientForm: http://wwwsearch.sourceforge.net/old/ClientForm/
.. _Creating a pull request: https://help.github.com/en/articles/creating-a-pull-request
+.. _cryptography: https://cryptography.io/en/latest/
.. _cssselect: https://github.com/scrapy/cssselect/
.. _docstrings: https://docs.python.org/glossary.html#term-docstring
.. _KeyboardInterrupt: https://docs.python.org/library/exceptions.html#KeyboardInterrupt
+.. _LevelDB: https://github.com/google/leveldb
.. _lxml: http://lxml.de/
.. _marshal: https://docs.python.org/2/library/marshal.html
.. _parsel.csstranslator.GenericTranslator: https://parsel.readthedocs.io/en/latest/parsel.html#parsel.csstranslator.GenericTranslator
.. _parsel.csstranslator.HTMLTranslator: https://parsel.readthedocs.io/en/latest/parsel.html#parsel.csstranslator.HTMLTranslator
.. _parsel.csstranslator.XPathExpr: https://parsel.readthedocs.io/en/latest/parsel.html#parsel.csstranslator.XPathExpr
.. _PEP 257: https://www.python.org/dev/peps/pep-0257/
+.. _Pillow: https://python-pillow.org/
+.. _pyOpenSSL: https://www.pyopenssl.org/en/stable/
.. _queuelib: https://github.com/scrapy/queuelib
+.. _registered with IANA: https://www.iana.org/assignments/media-types/media-types.xhtml
.. _resource: https://docs.python.org/2/library/resource.html
+.. _robots.txt: http://www.robotstxt.org/
.. _scrapely: https://github.com/scrapy/scrapely
+.. _service_identity: https://service-identity.readthedocs.io/en/stable/
+.. _six: https://six.readthedocs.io/
.. _tox: https://pypi.python.org/pypi/tox
+.. _Twisted: https://twistedmatrix.com/trac/
.. _Twisted - hello, asynchronous programming: http://jessenoller.com/blog/2009/02/11/twisted-hello-asynchronous-programming/
.. _w3lib: https://github.com/scrapy/w3lib
.. _w3lib.encoding: https://github.com/scrapy/w3lib/blob/master/w3lib/encoding.py
.. _What is cacheable: https://www.w3.org/Protocols/rfc2616/rfc2616-sec14.html#sec14.9.1
+.. _zope.interface: https://zopeinterface.readthedocs.io/en/latest/
+.. _Zsh: https://www.zsh.org/
diff --git a/docs/requirements.txt b/docs/requirements.txt
index 379da9994..f9db85146 100644
--- a/docs/requirements.txt
+++ b/docs/requirements.txt
@@ -1,2 +1,3 @@
Sphinx>=2.1
-sphinx_rtd_theme
\ No newline at end of file
+sphinx-notfound-page
+sphinx_rtd_theme
diff --git a/docs/topics/contracts.rst b/docs/topics/contracts.rst
index 62f9a743b..371ae62d5 100644
--- a/docs/topics/contracts.rst
+++ b/docs/topics/contracts.rst
@@ -6,10 +6,6 @@ Spiders Contracts
.. versionadded:: 0.15
-.. note:: This is a new feature (introduced in Scrapy 0.15) and may be subject
- to minor functionality/API updates. Check the :ref:`release notes ` to
- be notified of updates.
-
Testing spiders can get particularly annoying and while nothing prevents you
from writing unit tests the task gets cumbersome quickly. Scrapy offers an
integrated way of testing your spiders by the means of contracts.
diff --git a/docs/topics/developer-tools.rst b/docs/topics/developer-tools.rst
index dcf8af365..bf14643be 100644
--- a/docs/topics/developer-tools.rst
+++ b/docs/topics/developer-tools.rst
@@ -203,7 +203,7 @@ where our quotes are coming from:
First click on the request with the name ``scroll``. On the right
you can now inspect the request. In ``Headers`` you'll find details
about the request headers, such as the URL, the method, the IP-address,
-and so on. We'll ignore the other tabs and click directly on ``Reponse``.
+and so on. We'll ignore the other tabs and click directly on ``Response``.
What you should see in the ``Preview`` pane is the rendered HTML-code,
that is exactly what we saw when we called ``view(response)`` in the
diff --git a/docs/topics/downloader-middleware.rst b/docs/topics/downloader-middleware.rst
index 539832618..ae6d41809 100644
--- a/docs/topics/downloader-middleware.rst
+++ b/docs/topics/downloader-middleware.rst
@@ -348,7 +348,6 @@ HttpCacheMiddleware
* :ref:`httpcache-storage-fs`
* :ref:`httpcache-storage-dbm`
- * :ref:`httpcache-storage-leveldb`
You can change the HTTP cache storage backend with the :setting:`HTTPCACHE_STORAGE`
setting. Or you can also :ref:`implement your own storage backend. `
@@ -475,30 +474,9 @@ DBM storage backend
A DBM_ storage backend is also available for the HTTP cache middleware.
- By default, it uses the anydbm_ module, but you can change it with the
+ By default, it uses the :mod:`dbm`, but you can change it with the
:setting:`HTTPCACHE_DBM_MODULE` setting.
-.. _httpcache-storage-leveldb:
-
-LevelDB storage backend
-~~~~~~~~~~~~~~~~~~~~~~~
-
-.. class:: LeveldbCacheStorage
-
- .. versionadded:: 0.23
-
- A LevelDB_ storage backend is also available for the HTTP cache middleware.
-
- This backend is not recommended for development because only one process
- can access LevelDB databases at the same time, so you can't run a crawl and
- open the scrapy shell in parallel for the same spider.
-
- In order to use this storage backend, install the `LevelDB python
- bindings`_ (e.g. ``pip install leveldb``).
-
- .. _LevelDB: https://github.com/google/leveldb
- .. _leveldb python bindings: https://pypi.python.org/pypi/leveldb
-
.. _httpcache-storage-custom:
Writing your own storage backend
@@ -534,7 +512,7 @@ defines the methods described below.
:param spider: the spider which generated the request
:type spider: :class:`~scrapy.spiders.Spider` object
- :param request: the request to find cached reponse for
+ :param request: the request to find cached response for
:type request: :class:`~scrapy.http.Request` object
.. method:: store_response(spider, request, response)
@@ -648,7 +626,7 @@ HTTPCACHE_DBM_MODULE
.. versionadded:: 0.13
-Default: ``'anydbm'``
+Default: ``'dbm'``
The database module to use in the :ref:`DBM storage backend
`. This setting is specific to the DBM backend.
@@ -1224,4 +1202,3 @@ The default encoding for proxy authentication on :class:`HttpProxyMiddleware`.
.. _DBM: https://en.wikipedia.org/wiki/Dbm
-.. _anydbm: https://docs.python.org/2/library/anydbm.html
diff --git a/docs/topics/email.rst b/docs/topics/email.rst
index 949cdc638..12eedf2cd 100644
--- a/docs/topics/email.rst
+++ b/docs/topics/email.rst
@@ -21,7 +21,7 @@ Quick example
=============
There are two ways to instantiate the mail sender. You can instantiate it using
-the standard constructor::
+the standard ``__init__`` method::
from scrapy.mail import MailSender
mailer = MailSender()
@@ -111,7 +111,7 @@ uses `Twisted non-blocking IO`_, like the rest of the framework.
Mail settings
=============
-These settings define the default constructor values of the :class:`MailSender`
+These settings define the default ``__init__`` method values of the :class:`MailSender`
class, and can be used to configure e-mail notifications in your project without
writing any code (for those extensions and code that uses :class:`MailSender`).
diff --git a/docs/topics/exporters.rst b/docs/topics/exporters.rst
index a698a6a4e..b8d898022 100644
--- a/docs/topics/exporters.rst
+++ b/docs/topics/exporters.rst
@@ -87,8 +87,8 @@ described next.
1. Declaring a serializer in the field
--------------------------------------
-If you use :class:`~.Item` you can declare a serializer in the
-:ref:`field metadata `. The serializer must be
+If you use :class:`~.Item` you can declare a serializer in the
+:ref:`field metadata `. The serializer must be
a callable which receives a value and returns its serialized form.
Example::
@@ -144,7 +144,7 @@ BaseItemExporter
defining what fields to export, whether to export empty fields, or which
encoding to use.
- These features can be configured through the constructor arguments which
+ These features can be configured through the ``__init__`` method arguments which
populate their respective instance attributes: :attr:`fields_to_export`,
:attr:`export_empty_fields`, :attr:`encoding`, :attr:`indent`.
@@ -246,8 +246,8 @@ XmlItemExporter
:param item_element: The name of each item element in the exported XML.
:type item_element: str
- The additional keyword arguments of this constructor are passed to the
- :class:`BaseItemExporter` constructor.
+ The additional keyword arguments of this ``__init__`` method are passed to the
+ :class:`BaseItemExporter` ``__init__`` method.
A typical output of this exporter would be::
@@ -306,9 +306,9 @@ CsvItemExporter
multi-valued fields, if found.
:type include_headers_line: str
- The additional keyword arguments of this constructor are passed to the
- :class:`BaseItemExporter` constructor, and the leftover arguments to the
- `csv.writer`_ constructor, so you can use any ``csv.writer`` constructor
+ The additional keyword arguments of this ``__init__`` method are passed to the
+ :class:`BaseItemExporter` ``__init__`` method, and the leftover arguments to the
+ `csv.writer`_ ``__init__`` method, so you can use any ``csv.writer`` ``__init__`` method
argument to customize this exporter.
A typical output of this exporter would be::
@@ -334,8 +334,8 @@ PickleItemExporter
For more information, refer to the `pickle module documentation`_.
- The additional keyword arguments of this constructor are passed to the
- :class:`BaseItemExporter` constructor.
+ The additional keyword arguments of this ``__init__`` method are passed to the
+ :class:`BaseItemExporter` ``__init__`` method.
Pickle isn't a human readable format, so no output examples are provided.
@@ -351,8 +351,8 @@ PprintItemExporter
:param file: the file-like object to use for exporting the data. Its ``write`` method should
accept ``bytes`` (a disk file opened in binary mode, a ``io.BytesIO`` object, etc)
- The additional keyword arguments of this constructor are passed to the
- :class:`BaseItemExporter` constructor.
+ The additional keyword arguments of this ``__init__`` method are passed to the
+ :class:`BaseItemExporter` ``__init__`` method.
A typical output of this exporter would be::
@@ -367,10 +367,10 @@ JsonItemExporter
.. class:: JsonItemExporter(file, \**kwargs)
Exports Items in JSON format to the specified file-like object, writing all
- objects as a list of objects. The additional constructor arguments are
- passed to the :class:`BaseItemExporter` constructor, and the leftover
- arguments to the `JSONEncoder`_ constructor, so you can use any
- `JSONEncoder`_ constructor argument to customize this exporter.
+ objects as a list of objects. The additional ``__init__`` method arguments are
+ passed to the :class:`BaseItemExporter` ``__init__`` method, and the leftover
+ arguments to the `JSONEncoder`_ ``__init__`` method, so you can use any
+ `JSONEncoder`_ ``__init__`` method argument to customize this exporter.
:param file: the file-like object to use for exporting the data. Its ``write`` method should
accept ``bytes`` (a disk file opened in binary mode, a ``io.BytesIO`` object, etc)
@@ -398,10 +398,10 @@ JsonLinesItemExporter
.. class:: JsonLinesItemExporter(file, \**kwargs)
Exports Items in JSON format to the specified file-like object, writing one
- JSON-encoded item per line. The additional constructor arguments are passed
- to the :class:`BaseItemExporter` constructor, and the leftover arguments to
- the `JSONEncoder`_ constructor, so you can use any `JSONEncoder`_
- constructor argument to customize this exporter.
+ JSON-encoded item per line. The additional ``__init__`` method arguments are passed
+ to the :class:`BaseItemExporter` ``__init__`` method, and the leftover arguments to
+ the `JSONEncoder`_ ``__init__`` method, so you can use any `JSONEncoder`_
+ ``__init__`` method argument to customize this exporter.
:param file: the file-like object to use for exporting the data. Its ``write`` method should
accept ``bytes`` (a disk file opened in binary mode, a ``io.BytesIO`` object, etc)
diff --git a/docs/topics/extensions.rst b/docs/topics/extensions.rst
index 72c2290b5..0a7455ec9 100644
--- a/docs/topics/extensions.rst
+++ b/docs/topics/extensions.rst
@@ -28,7 +28,7 @@ Loading & activating extensions
Extensions are loaded and activated at startup by instantiating a single
instance of the extension class. Therefore, all the extension initialization
-code must be performed in the class constructor (``__init__`` method).
+code must be performed in the class ``__init__`` method.
To make an extension available, add it to the :setting:`EXTENSIONS` setting in
your Scrapy settings. In :setting:`EXTENSIONS`, each extension is represented
diff --git a/docs/topics/feed-exports.rst b/docs/topics/feed-exports.rst
index af541db78..7481b1a99 100644
--- a/docs/topics/feed-exports.rst
+++ b/docs/topics/feed-exports.rst
@@ -99,12 +99,12 @@ The storages backends supported out of the box are:
* :ref:`topics-feed-storage-fs`
* :ref:`topics-feed-storage-ftp`
- * :ref:`topics-feed-storage-s3` (requires botocore_ or boto_)
+ * :ref:`topics-feed-storage-s3` (requires botocore_)
* :ref:`topics-feed-storage-stdout`
Some storage backends may be unavailable if the required external libraries are
not available. For example, the S3 backend is only available if the botocore_
-or boto_ library is installed (Scrapy supports boto_ only on Python 2).
+library is installed.
.. _topics-feed-uri-params:
@@ -182,7 +182,7 @@ The feeds are stored on `Amazon S3`_.
* ``s3://mybucket/path/to/export.csv``
* ``s3://aws_key:aws_secret@mybucket/path/to/export.csv``
- * Required external libraries: `botocore`_ (Python 2 and Python 3) or `boto`_ (Python 2 only)
+ * Required external libraries: `botocore`_
The AWS credentials can be passed as user/password in the URI, or they can be
passed through the following settings:
@@ -399,6 +399,5 @@ format in :setting:`FEED_EXPORTERS`. E.g., to disable the built-in CSV exporter
.. _URI: https://en.wikipedia.org/wiki/Uniform_Resource_Identifier
.. _Amazon S3: https://aws.amazon.com/s3/
-.. _boto: https://github.com/boto/boto
.. _botocore: https://github.com/boto/botocore
.. _Canned ACL: https://docs.aws.amazon.com/AmazonS3/latest/dev/acl-overview.html#canned-acl
diff --git a/docs/topics/items.rst b/docs/topics/items.rst
index 260f5882c..cdf60208e 100644
--- a/docs/topics/items.rst
+++ b/docs/topics/items.rst
@@ -16,12 +16,12 @@ especially in a larger project with many spiders.
To define common output data format Scrapy provides the :class:`Item` class.
:class:`Item` objects are simple containers used to collect the scraped data.
They provide a `dictionary-like`_ API with a convenient syntax for declaring
-their available fields.
+their available fields.
-Various Scrapy components use extra information provided by Items:
+Various Scrapy components use extra information provided by Items:
exporters look at declared fields to figure out columns to export,
serialization can be customized using Item fields metadata, :mod:`trackref`
-tracks Item instances to help find memory leaks
+tracks Item instances to help find memory leaks
(see :ref:`topics-leaks-trackrefs`), etc.
.. _dictionary-like: https://docs.python.org/2/library/stdtypes.html#dict
@@ -237,8 +237,12 @@ Item objects
Return a new Item optionally initialized from the given argument.
- Items replicate the standard `dict API`_, including its constructor. The
- only additional attribute provided by Items is:
+ Items replicate the standard `dict API`_, including its ``__init__`` method, and
+ also provide the following additional API members:
+
+ .. automethod:: copy
+
+ .. automethod:: deepcopy
.. attribute:: fields
diff --git a/docs/topics/jobs.rst b/docs/topics/jobs.rst
index 9fd311c69..f5542495b 100644
--- a/docs/topics/jobs.rst
+++ b/docs/topics/jobs.rst
@@ -71,34 +71,11 @@ on cookies.
Request serialization
---------------------
-Requests must be serializable by the ``pickle`` module, in order for persistence
-to work, so you should make sure that your requests are serializable.
-
-The most common issue here is to use ``lambda`` functions on request callbacks that
-can't be persisted.
-
-So, for example, this won't work::
-
- def some_callback(self, response):
- somearg = 'test'
- return scrapy.Request('http://www.example.com',
- callback=lambda r: self.other_callback(r, somearg))
-
- def other_callback(self, response, somearg):
- print("the argument passed is: %s" % somearg)
-
-But this will::
-
- def some_callback(self, response):
- somearg = 'test'
- return scrapy.Request('http://www.example.com',
- callback=self.other_callback, cb_kwargs={'somearg': somearg})
-
- def other_callback(self, response, somearg):
- print("the argument passed is: %s" % somearg)
+For persistence to work, :class:`~scrapy.http.Request` objects must be
+serializable with :mod:`pickle`, except for the ``callback`` and ``errback``
+values passed to their ``__init__`` method, which must be methods of the
+runnning :class:`~scrapy.spiders.Spider` class.
If you wish to log the requests that couldn't be serialized, you can set the
:setting:`SCHEDULER_DEBUG` setting to ``True`` in the project's settings page.
It is ``False`` by default.
-
-.. _pickle: https://docs.python.org/library/pickle.html
diff --git a/docs/topics/leaks.rst b/docs/topics/leaks.rst
index 8278e9849..793636f59 100644
--- a/docs/topics/leaks.rst
+++ b/docs/topics/leaks.rst
@@ -260,7 +260,7 @@ knowledge about Python internals. For more info about Guppy, refer to the
Debugging memory leaks with muppy
=================================
-If you're using Python 3, you can use muppy from `Pympler`_.
+You can use muppy from `Pympler`_.
.. _Pympler: https://pypi.org/project/Pympler/
diff --git a/docs/topics/loaders.rst b/docs/topics/loaders.rst
index 1c2f1da4d..12a5e5c60 100644
--- a/docs/topics/loaders.rst
+++ b/docs/topics/loaders.rst
@@ -26,7 +26,7 @@ Using Item Loaders to populate items
To use an Item Loader, you must first instantiate it. You can either
instantiate it with a dict-like object (e.g. Item or dict) or without one, in
-which case an Item is automatically instantiated in the Item Loader constructor
+which case an Item is automatically instantiated in the Item Loader ``__init__`` method
using the Item class specified in the :attr:`ItemLoader.default_item_class`
attribute.
@@ -35,6 +35,12 @@ Then, you start collecting values into the Item Loader, typically using
the same item field; the Item Loader will know how to "join" those values later
using a proper processing function.
+.. note:: Collected data is internally stored as lists,
+ allowing to add several values to the same field.
+ If an ``item`` argument is passed when creating a loader,
+ each of the item's values will be stored as-is if it's already
+ an iterable, or wrapped with a list if it's a single value.
+
Here is a typical Item Loader usage in a :ref:`Spider `, using
the :ref:`Product item ` declared in the :ref:`Items
chapter `::
@@ -128,9 +134,9 @@ So what happens is:
It's worth noticing that processors are just callable objects, which are called
with the data to be parsed, and return a parsed value. So you can use any
function as input or output processor. The only requirement is that they must
-accept one (and only one) positional argument, which will be an iterator.
+accept one (and only one) positional argument, which will be an iterable.
-.. note:: Both input and output processors must receive an iterator as their
+.. note:: Both input and output processors must receive an iterable as their
first argument. The output of those functions can be anything. The result of
input processors will be appended to an internal list (in the Loader)
containing the collected values (for that field). The result of the output
@@ -265,7 +271,7 @@ There are several ways to modify Item Loader context values:
loader.context['unit'] = 'cm'
2. On Item Loader instantiation (the keyword arguments of Item Loader
- constructor are stored in the Item Loader context)::
+ ``__init__`` method are stored in the Item Loader context)::
loader = ItemLoader(product, unit='cm')
@@ -494,7 +500,7 @@ ItemLoader objects
.. attribute:: default_item_class
An Item class (or factory), used to instantiate items when not given in
- the constructor.
+ the ``__init__`` method.
.. attribute:: default_input_processor
@@ -509,15 +515,15 @@ ItemLoader objects
.. attribute:: default_selector_class
The class used to construct the :attr:`selector` of this
- :class:`ItemLoader`, if only a response is given in the constructor.
- If a selector is given in the constructor this attribute is ignored.
+ :class:`ItemLoader`, if only a response is given in the ``__init__`` method.
+ If a selector is given in the ``__init__`` method this attribute is ignored.
This attribute is sometimes overridden in subclasses.
.. attribute:: selector
The :class:`~scrapy.selector.Selector` object to extract data from.
- It's either the selector given in the constructor or one created from
- the response given in the constructor using the
+ It's either the selector given in the ``__init__`` method or one created from
+ the response given in the ``__init__`` method using the
:attr:`default_selector_class`. This attribute is meant to be
read-only.
@@ -642,7 +648,7 @@ Here is a list of all built-in processors:
.. class:: Identity
The simplest processor, which doesn't do anything. It returns the original
- values unchanged. It doesn't receive any constructor arguments, nor does it
+ values unchanged. It doesn't receive any ``__init__`` method arguments, nor does it
accept Loader contexts.
Example::
@@ -656,7 +662,7 @@ Here is a list of all built-in processors:
Returns the first non-null/non-empty value from the values received,
so it's typically used as an output processor to single-valued fields.
- It doesn't receive any constructor arguments, nor does it accept Loader contexts.
+ It doesn't receive any ``__init__`` method arguments, nor does it accept Loader contexts.
Example::
@@ -667,7 +673,7 @@ Here is a list of all built-in processors:
.. class:: Join(separator=u' ')
- Returns the values joined with the separator given in the constructor, which
+ Returns the values joined with the separator given in the ``__init__`` method, which
defaults to ``u' '``. It doesn't accept Loader contexts.
When using the default separator, this processor is equivalent to the
@@ -705,7 +711,7 @@ Here is a list of all built-in processors:
those which do, this processor will pass the currently active :ref:`Loader
context ` through that parameter.
- The keyword arguments passed in the constructor are used as the default
+ The keyword arguments passed in the ``__init__`` method are used as the default
Loader context values passed to each function call. However, the final
Loader context values passed to functions are overridden with the currently
active Loader context accessible through the :meth:`ItemLoader.context`
@@ -749,12 +755,12 @@ Here is a list of all built-in processors:
['HELLO, 'THIS', 'IS', 'SCRAPY']
As with the Compose processor, functions can receive Loader contexts, and
- constructor keyword arguments are used as default context values. See
+ ``__init__`` method keyword arguments are used as default context values. See
:class:`Compose` processor for more info.
.. class:: SelectJmes(json_path)
- Queries the value using the json path provided to the constructor and returns the output.
+ Queries the value using the json path provided to the ``__init__`` method and returns the output.
Requires jmespath (https://github.com/jmespath/jmespath.py) to run.
This processor takes only one input at a time.
diff --git a/docs/topics/logging.rst b/docs/topics/logging.rst
index 87ea43c7d..dd09477b8 100644
--- a/docs/topics/logging.rst
+++ b/docs/topics/logging.rst
@@ -198,8 +198,9 @@ to override some of the Scrapy settings regarding logging.
Custom Log Formats
------------------
-A custom log format can be set for different actions by extending :class:`~scrapy.logformatter.LogFormatter` class
-and making :setting:`LOG_FORMATTER` point to your new class.
+A custom log format can be set for different actions by extending
+:class:`~scrapy.logformatter.LogFormatter` class and making
+:setting:`LOG_FORMATTER` point to your new class.
.. autoclass:: scrapy.logformatter.LogFormatter
:members:
@@ -254,18 +255,18 @@ scrapy.utils.log module
when running custom scripts using :class:`~scrapy.crawler.CrawlerRunner`.
In that case, its usage is not required but it's recommended.
- If you plan on configuring the handlers yourself is still recommended you
- call this function, passing ``install_root_handler=False``. Bear in mind
- there won't be any log output set by default in that case.
+ Another option when running custom scripts is to manually configure the logging.
+ To do this you can use `logging.basicConfig()`_ to set a basic root handler.
- To get you started on manually configuring logging's output, you can use
- `logging.basicConfig()`_ to set a basic root handler. This is an example
- on how to redirect ``INFO`` or higher messages to a file::
+ Note that :class:`~scrapy.crawler.CrawlerProcess` automatically calls ``configure_logging``,
+ so it is recommended to only use `logging.basicConfig()`_ together with
+ :class:`~scrapy.crawler.CrawlerRunner`.
+
+ This is an example on how to redirect ``INFO`` or higher messages to a file::
import logging
from scrapy.utils.log import configure_logging
- configure_logging(install_root_handler=False)
logging.basicConfig(
filename='log.txt',
format='%(levelname)s: %(message)s',
diff --git a/docs/topics/media-pipeline.rst b/docs/topics/media-pipeline.rst
index 0ce431ff5..431cc6027 100644
--- a/docs/topics/media-pipeline.rst
+++ b/docs/topics/media-pipeline.rst
@@ -171,7 +171,7 @@ policy::
For more information, see `canned ACLs`_ in the Amazon S3 Developer Guide.
-Because Scrapy uses ``boto`` / ``botocore`` internally you can also use other S3-like storages. Storages like
+Because Scrapy uses ``botocore`` internally you can also use other S3-like storages. Storages like
self-hosted `Minio`_ or `s3.scality`_. All you need to do is set endpoint option in you Scrapy settings::
AWS_ENDPOINT_URL = 'http://minio.example.com:9000'
diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst
index 9fe3c7518..8a6636474 100644
--- a/docs/topics/request-response.rst
+++ b/docs/topics/request-response.rst
@@ -137,7 +137,7 @@ Request objects
A string containing the URL of this request. Keep in mind that this
attribute contains the escaped URL, so it can differ from the URL passed in
- the constructor.
+ the ``__init__`` method.
This attribute is read-only. To change the URL of a Request use
:meth:`replace`.
@@ -400,7 +400,7 @@ fields with form data from :class:`Response` objects.
.. class:: FormRequest(url, [formdata, ...])
- The :class:`FormRequest` class adds a new keyword parameter to the constructor. The
+ The :class:`FormRequest` class adds a new keyword parameter to the ``__init__`` method. The
remaining arguments are the same as for the :class:`Request` class and are
not documented here.
@@ -473,7 +473,7 @@ fields with form data from :class:`Response` objects.
:type dont_click: boolean
The other parameters of this class method are passed directly to the
- :class:`FormRequest` constructor.
+ :class:`FormRequest` ``__init__`` method.
.. versionadded:: 0.10.3
The ``formname`` parameter.
@@ -547,7 +547,7 @@ dealing with JSON requests.
.. class:: JsonRequest(url, [... data, dumps_kwargs])
- The :class:`JsonRequest` class adds two new keyword parameters to the constructor. The
+ The :class:`JsonRequest` class adds two new keyword parameters to the ``__init__`` method. The
remaining arguments are the same as for the :class:`Request` class and are
not documented here.
@@ -556,7 +556,7 @@ dealing with JSON requests.
:param data: is any JSON serializable object that needs to be JSON encoded and assigned to body.
if :attr:`Request.body` argument is provided this parameter will be ignored.
- if :attr:`Request.body` argument is not provided and data argument is provided :attr:`Request.method` will be
+ if :attr:`Request.body` argument is not provided and data argument is provided :attr:`Request.method` will be
set to ``'POST'`` automatically.
:type data: JSON serializable object
@@ -596,8 +596,8 @@ Response objects
(for single valued headers) or lists (for multi-valued headers).
:type headers: dict
- :param body: the response body. To access the decoded text as str (unicode
- in Python 2) you can use ``response.text`` from an encoding-aware
+ :param body: the response body. To access the decoded text as str you can use
+ ``response.text`` from an encoding-aware
:ref:`Response subclass `,
such as :class:`TextResponse`.
:type body: bytes
@@ -723,7 +723,7 @@ TextResponse objects
:class:`Response` class, which is meant to be used only for binary data,
such as images, sounds or any media file.
- :class:`TextResponse` objects support a new constructor argument, in
+ :class:`TextResponse` objects support a new ``__init__`` method argument, in
addition to the base :class:`Response` objects. The remaining functionality
is the same as for the :class:`Response` class and is not documented here.
@@ -757,7 +757,7 @@ TextResponse objects
A string with the encoding of this response. The encoding is resolved by
trying the following mechanisms, in order:
- 1. the encoding passed in the constructor ``encoding`` argument
+ 1. the encoding passed in the ``__init__`` method ``encoding`` argument
2. the encoding declared in the Content-Type HTTP header. If this
encoding is not valid (ie. unknown), it is ignored and the next
diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst
index 75e0af63b..a1d15a760 100644
--- a/docs/topics/settings.rst
+++ b/docs/topics/settings.rst
@@ -188,7 +188,6 @@ AWS_ENDPOINT_URL
Default: ``None``
Endpoint URL used for S3-like storage, for example Minio or s3.scality.
-Only supported with ``botocore`` library.
.. setting:: AWS_USE_SSL
@@ -199,7 +198,6 @@ Default: ``None``
Use this option if you want to disable SSL connection for communication with
S3 or S3-like storage. By default SSL will be used.
-Only supported with ``botocore`` library.
.. setting:: AWS_VERIFY
@@ -209,7 +207,7 @@ AWS_VERIFY
Default: ``None``
Verify SSL connection between Scrapy and S3 or S3-like storage. By default
-SSL verification will occur. Only supported with ``botocore`` library.
+SSL verification will occur.
.. setting:: AWS_REGION_NAME
@@ -219,7 +217,6 @@ AWS_REGION_NAME
Default: ``None``
The name of the region associated with the AWS client.
-Only supported with ``botocore`` library.
.. setting:: BOT_NAME
@@ -229,8 +226,7 @@ BOT_NAME
Default: ``'scrapybot'``
The name of the bot implemented by this Scrapy project (also known as the
-project name). This will be used to construct the User-Agent by default, and
-also for logging.
+project name). This name will be used for the logging too.
It's automatically populated with your project name when you create your
project with the :command:`startproject` command.
@@ -796,6 +792,7 @@ Default: ``True``
Whether or not to use passive mode when initiating FTP transfers.
+.. reqmeta:: ftp_password
.. setting:: FTP_PASSWORD
FTP_PASSWORD
@@ -814,6 +811,7 @@ in ``Request`` meta.
.. _RFC 1635: https://tools.ietf.org/html/rfc1635
+.. reqmeta:: ftp_user
.. setting:: FTP_USER
FTP_USER
diff --git a/docs/topics/spiders.rst b/docs/topics/spiders.rst
index d60c93be6..d65a43afd 100644
--- a/docs/topics/spiders.rst
+++ b/docs/topics/spiders.rst
@@ -72,8 +72,6 @@ scrapy.Spider
spider that crawls ``mywebsite.com`` would often be called
``mywebsite``.
- .. note:: In Python 2 this must be ASCII only.
-
.. attribute:: allowed_domains
An optional list of strings containing domains that this spider is
diff --git a/extras/scrapy_zsh_completion b/extras/scrapy_zsh_completion
index 564991aa8..e995947cb 100644
--- a/extras/scrapy_zsh_completion
+++ b/extras/scrapy_zsh_completion
@@ -1,25 +1,210 @@
#compdef scrapy
-
-# zsh completion for the Scrapy command-line tool
-
_scrapy() {
- local curcontext="$curcontext" cmd spiders
+ local context state state_descr line
typeset -A opt_args
- cmd=$words[2]
-
- case "$cmd" in
- crawl|edit|check)
- spiders=$(scrapy list 2>/dev/null) || spiders=""
- if [[ -n "$spiders" ]]; then
- compadd `echo $spiders`
- fi
- ;;
- *)
- if [[ CURRENT -eq 2 ]]; then
- _arguments '*: :(check crawl edit fetch genspider list parse runspider settings shell startproject version view)'
- fi
- ;;
+ _arguments \
+ "(- 1 *)--help[Help]" \
+ "1: :->command" \
+ "*:: :->args"
+
+ case $state in
+ command)
+ _scrapy_cmds
+ ;;
+ args)
+ case $words[1] in
+ bench)
+ _scrapy_glb_opts
+ ;;
+ fetch)
+ local options=(
+ '--headers[print response HTTP headers instead of body]'
+ '--no-redirect[do not handle HTTP 3xx status codes and print response as-is]'
+ '--spider[use this spider]:spider:_scrapy_spiders'
+ '1::URL:_httpie_urls'
+ )
+ _scrapy_glb_opts $options
+ ;;
+ genspider)
+ local options=(
+ {-l,--list}'[List available templates]'
+ {-e,--edit}'[Edit spider after creating it]'
+ '--force[If the spider already exists, overwrite it with the template]'
+ {-d,--dump=}'[Dump template to standard output]:template:(basic crawl csvfeed xmlfeed)'
+ {-t,--template=}'[Uses a custom template]:template:(basic crawl csvfeed xmlfeed)'
+ '1:name:(NAME)'
+ '2:domain:_httpie_urls'
+ )
+ _scrapy_glb_opts $options
+ ;;
+ runspider)
+ local options=(
+ {-o,--output}'[dump scraped items into FILE (use - for stdout)]:file:_files'
+ {-t,--output-format}'[format to use for dumping items with -o]:format:(FORMAT)'
+ '*-a[set spider argument (may be repeated)]:value pair:(NAME=VALUE)'
+ '1:spider file:_files -g \*.py'
+ )
+ _scrapy_glb_opts $options
+ ;;
+ settings)
+ local options=(
+ '--get=[print raw setting value]:option:(SETTING)'
+ '--getbool=[print setting value, interpreted as a boolean]:option:(SETTING)'
+ '--getint=[print setting value, interpreted as an integer]:option:(SETTING)'
+ '--getfloat=[print setting value, interpreted as a float]:option:(SETTING)'
+ '--getlist=[print setting value, interpreted as a list]:option:(SETTING)'
+ )
+ _scrapy_glb_opts $options
+ ;;
+ shell)
+ local options=(
+ '-c[evaluate the code in the shell, print the result and exit]:code:(CODE)'
+ '--no-redirect[do not handle HTTP 3xx status codes and print response as-is]'
+ '--spider[use this spider]:spider:_scrapy_spiders'
+ '::file:_files -g \*.html'
+ '::URL:_httpie_urls'
+ )
+ _scrapy_glb_opts $options
+ ;;
+ startproject)
+ local options=(
+ '1:name:(NAME)'
+ '2:dir:_dir_list'
+ )
+ _scrapy_glb_opts $options
+ ;;
+ version)
+ local options=(
+ {-v,--verbose}'[also display twisted/python/platform info (useful for bug reports)]'
+ )
+ _scrapy_glb_opts $options
+ ;;
+ view)
+ local options=(
+ '--no-redirect[do not handle HTTP 3xx status codes and print response as-is]'
+ '--spider[use this spider]:spider:_scrapy_spiders'
+ '1:URL:_httpie_urls'
+ )
+ _scrapy_glb_opts $options
+ ;;
+ check)
+ local options=(
+ '(- 1 *)'{-l,--list}'[only list contracts, without checking them]'
+ {-v,--verbose}'[print contract tests for all spiders]'
+ '1:spider:_scrapy_spiders'
+ )
+ _scrapy_glb_opts $options
+ ;;
+ crawl)
+ local options=(
+ {-o,--output}'[dump scraped items into FILE (use - for stdout)]:file:_files'
+ {-t,--output-format}'[format to use for dumping items with -o]:format:(FORMAT)'
+ '*-a[set spider argument (may be repeated)]:value pair:(NAME=VALUE)'
+ '1:spider:_scrapy_spiders'
+ )
+ _scrapy_glb_opts $options
+ ;;
+ edit)
+ local options=(
+ '1:spider:_scrapy_spiders'
+ )
+ _scrapy_glb_opts $options
+ ;;
+ list)
+ _scrapy_glb_opts
+ ;;
+ parse)
+ local options=(
+ '*-a[set spider argument (may be repeated)]:value pair:(NAME=VALUE)'
+ '--spider[use this spider without looking for one]:spider:_scrapy_spiders'
+ '--pipelines[process items through pipelines]'
+ "--nolinks[don't show links to follow (extracted requests)]"
+ "--noitems[don't show scraped items]"
+ '--nocolour[avoid using pygments to colorize the output]'
+ {-r,--rules}'[use CrawlSpider rules to discover the callback]'
+ {-c,--callback=}'[use this callback for parsing, instead looking for a callback]:callback:(CALLBACK)'
+ {-m,--meta=}'[inject extra meta into the Request, it must be a valid raw json string]:meta:(META)'
+ '--cbkwargs=[inject extra callback kwargs into the Request, it must be a valid raw json string]:arguments:(CBKWARGS)'
+ {-d,--depth=}'[maximum depth for parsing requests (default: 1)]:depth:(DEPTH)'
+ {-v,--verbose}'[print each depth level one by one]'
+ '1:URL:_httpie_urls'
+ )
+ _scrapy_glb_opts $options
+ ;;
+ esac
+ ;;
esac
}
-_scrapy
\ No newline at end of file
+_scrapy_cmds() {
+ local -a commands project_commands
+ commands=(
+ 'bench:Run quick benchmark test'
+ 'fetch:Fetch a URL using the Scrapy downloader'
+ 'genspider:Generate new spider using pre-defined templates'
+ 'runspider:Run a self-contained spider (without creating a project)'
+ 'settings:Get settings values'
+ 'shell:Interactive scraping console'
+ 'startproject:Create new project'
+ 'version:Print Scrapy version'
+ 'view:Open URL in browser, as seen by Scrapy'
+ )
+ project_commands=(
+ 'check:Check spider contracts'
+ 'crawl:Run a spider'
+ 'edit:Edit spider'
+ 'list:List available spiders'
+ 'parse:Parse URL (using its spider) and print the results'
+ )
+ if [[ $(scrapy -h | grep -s "no active project") == "" ]]; then
+ commands=(${commands[@]} ${project_commands[@]})
+ fi
+ _describe -t common-commands 'common commands' commands
+}
+
+_scrapy_glb_opts() {
+ local -a options
+ options=(
+ '(- *)'{-h,--help}'[show this help message and exit]'
+ '(--nolog)--logfile=[log file. if omitted stderr will be used]:file:_files'
+ '--pidfile=[write process ID to FILE]:file:_files'
+ '--profile=[write python cProfile stats to FILE]:file:_files'
+ '(--nolog)'{-L,--loglevel=}'[log level (default: INFO)]:log level:(DEBUG INFO WARN ERROR)'
+ '(-L --loglevel --logfile)--nolog[disable logging completely]'
+ '--pdb[enable pdb on failure]'
+ '*'{-s,--set=}'[set/override setting (may be repeated)]:value pair:(NAME=VALUE)'
+ )
+ options=(${options[@]} "$@")
+ _arguments $options
+}
+
+_httpie_urls() {
+
+ local ret=1
+
+ if ! [[ -prefix [-+.a-z0-9]#:// ]]; then
+ local expl
+ compset -S '[^:/]*' && compstate[to_end]=''
+ _wanted url-schemas expl 'URL schema' compadd -S '' http:// https:// && ret=0
+ else
+ _urls && ret=0
+ fi
+
+ return $ret
+
+}
+
+_scrapy_spiders() {
+
+ local ret=1
+
+ if [[ $(scrapy -h | grep -s "no active project") == "" ]]; then
+ compadd -S '' $(scrapy list) && ret=0
+ else
+ compadd -S '' SPIDER && ret=0
+ fi
+
+ return $ret
+}
+
+_scrapy $@
diff --git a/pytest.ini b/pytest.ini
index 73d169601..529ad5d27 100644
--- a/pytest.ini
+++ b/pytest.ini
@@ -2,5 +2,270 @@
usefixtures = chdir
python_files=test_*.py __init__.py
python_classes=
-addopts = --doctest-modules --assert=plain
+addopts =
+ --assert=plain
+ --doctest-modules
+ --ignore=docs/_ext
+ --ignore=docs/conf.py
+ --ignore=docs/news.rst
+ --ignore=docs/topics/commands.rst
+ --ignore=docs/topics/debug.rst
+ --ignore=docs/topics/developer-tools.rst
+ --ignore=docs/topics/dynamic-content.rst
+ --ignore=docs/topics/items.rst
+ --ignore=docs/topics/leaks.rst
+ --ignore=docs/topics/loaders.rst
+ --ignore=docs/topics/selectors.rst
+ --ignore=docs/topics/shell.rst
+ --ignore=docs/topics/stats.rst
+ --ignore=docs/topics/telnetconsole.rst
+ --ignore=docs/utils
twisted = 1
+flake8-ignore =
+ # extras
+ extras/qps-bench-server.py E261 E501
+ extras/qpsclient.py E501 E261 E501
+ # scrapy/commands
+ scrapy/commands/__init__.py E128 E501
+ scrapy/commands/check.py F401 E501
+ scrapy/commands/crawl.py E501
+ scrapy/commands/edit.py E501
+ scrapy/commands/fetch.py E401 E302 E501 E128 E502 E731
+ scrapy/commands/genspider.py E128 E501 E502
+ scrapy/commands/list.py E302
+ scrapy/commands/parse.py E128 E501 E731 E226
+ scrapy/commands/runspider.py E501
+ scrapy/commands/settings.py E302 E128
+ scrapy/commands/shell.py E128 E501 E502
+ scrapy/commands/startproject.py E502 E127 E501 E128
+ scrapy/commands/version.py E501 E128
+ scrapy/commands/view.py F401 E302
+ # scrapy/contracts
+ scrapy/contracts/__init__.py E501 W504
+ scrapy/contracts/default.py E502 E128
+ # scrapy/core
+ scrapy/core/engine.py E261 E501 E128 E127 E306 E502
+ scrapy/core/scheduler.py E501
+ scrapy/core/scraper.py E501 E306 E261 E128 W504
+ scrapy/core/spidermw.py E501 E731 E502 E126 E226
+ scrapy/core/downloader/__init__.py F401 E501
+ scrapy/core/downloader/contextfactory.py E501 E128 E126
+ scrapy/core/downloader/middleware.py E501 E502
+ scrapy/core/downloader/tls.py E501 E305 E241
+ scrapy/core/downloader/webclient.py E731 E501 E261 E502 E128 E126 E226
+ scrapy/core/downloader/handlers/__init__.py E501
+ scrapy/core/downloader/handlers/ftp.py E501 E305 E128 E127
+ scrapy/core/downloader/handlers/http.py F401
+ scrapy/core/downloader/handlers/http10.py E501
+ scrapy/core/downloader/handlers/http11.py E501
+ scrapy/core/downloader/handlers/s3.py E501 F401 E502 E128 E126
+ # scrapy/downloadermiddlewares
+ scrapy/downloadermiddlewares/ajaxcrawl.py E302 E501 E226
+ scrapy/downloadermiddlewares/decompression.py E501
+ scrapy/downloadermiddlewares/defaultheaders.py E501
+ scrapy/downloadermiddlewares/httpcache.py E501 E126
+ scrapy/downloadermiddlewares/httpcompression.py E502 E128
+ scrapy/downloadermiddlewares/httpproxy.py E501
+ scrapy/downloadermiddlewares/redirect.py E501 W504
+ scrapy/downloadermiddlewares/retry.py E501 E126
+ scrapy/downloadermiddlewares/robotstxt.py F401 E501
+ scrapy/downloadermiddlewares/stats.py E501
+ # scrapy/extensions
+ scrapy/extensions/closespider.py E501 E502 E128 E123
+ scrapy/extensions/corestats.py E302 E501
+ scrapy/extensions/feedexport.py E128 E501
+ scrapy/extensions/httpcache.py E128 E501 E303 F401
+ scrapy/extensions/memdebug.py E501
+ scrapy/extensions/spiderstate.py E302 E501
+ scrapy/extensions/telnet.py E501 W504
+ scrapy/extensions/throttle.py E501
+ # scrapy/http
+ scrapy/http/__init__.py F401
+ scrapy/http/common.py E501
+ scrapy/http/cookies.py E501
+ scrapy/http/request/__init__.py E501
+ scrapy/http/request/form.py E501 E123
+ scrapy/http/request/json_request.py E501
+ scrapy/http/response/__init__.py E501 E128 W293 W291
+ scrapy/http/response/html.py E302
+ scrapy/http/response/text.py E501 W293 E128 E124
+ scrapy/http/response/xml.py E302
+ # scrapy/linkextractors
+ scrapy/linkextractors/__init__.py E731 E502 E501 E402 F401
+ scrapy/linkextractors/lxmlhtml.py E501 E731 E226
+ # scrapy/loader
+ scrapy/loader/__init__.py E501 E502 E128
+ scrapy/loader/common.py E302
+ scrapy/loader/processors.py E501
+ # scrapy/pipelines
+ scrapy/pipelines/__init__.py E302
+ scrapy/pipelines/files.py E116 E501 E266
+ scrapy/pipelines/images.py E265 E501
+ scrapy/pipelines/media.py E125 E501 E266
+ # scrapy/selector
+ scrapy/selector/__init__.py F403 F401
+ scrapy/selector/unified.py F401 E501 E111
+ # scrapy/settings
+ scrapy/settings/__init__.py E501
+ scrapy/settings/default_settings.py E501 E261 E114 E116 E226
+ scrapy/settings/deprecated.py E501
+ # scrapy/spidermiddlewares
+ scrapy/spidermiddlewares/httperror.py E501
+ scrapy/spidermiddlewares/offsite.py E501
+ scrapy/spidermiddlewares/referer.py F401 E501 E129 W503 W504
+ scrapy/spidermiddlewares/urllength.py E501
+ # scrapy/spiders
+ scrapy/spiders/__init__.py F401 E501 E402
+ scrapy/spiders/crawl.py E501
+ scrapy/spiders/feed.py E501 E261
+ scrapy/spiders/sitemap.py E501
+ # scrapy/utils
+ scrapy/utils/benchserver.py E501
+ scrapy/utils/boto.py F401
+ scrapy/utils/conf.py E402 E502 E501
+ scrapy/utils/console.py E302 E261 F401 E306 E305
+ scrapy/utils/curl.py F401
+ scrapy/utils/datatypes.py E501 E226
+ scrapy/utils/decorators.py E501 E302
+ scrapy/utils/defer.py E501 E302 E128
+ scrapy/utils/deprecate.py E128 E501 E127 E502
+ scrapy/utils/display.py E302
+ scrapy/utils/engine.py F401 E261 E302
+ scrapy/utils/ftp.py E302
+ scrapy/utils/gz.py E305 E501 E302 W504
+ scrapy/utils/http.py F403 F401 E226
+ scrapy/utils/httpobj.py E302 E501
+ scrapy/utils/iterators.py E501 E701
+ scrapy/utils/job.py E302
+ scrapy/utils/log.py E128 W503
+ scrapy/utils/markup.py F403 F401 W292
+ scrapy/utils/misc.py E501 E226
+ scrapy/utils/multipart.py F403 F401 W292
+ scrapy/utils/project.py E501
+ scrapy/utils/python.py E501 E302
+ scrapy/utils/reactor.py E302 E226
+ scrapy/utils/reqser.py E501
+ scrapy/utils/request.py E302 E127 E501
+ scrapy/utils/response.py E501 E302 E128
+ scrapy/utils/signal.py E501 E128
+ scrapy/utils/sitemap.py E501
+ scrapy/utils/spider.py E271 E302 E501
+ scrapy/utils/ssl.py E501
+ scrapy/utils/template.py E302
+ scrapy/utils/test.py E302 E501
+ scrapy/utils/url.py E501 F403 F401 E128 F405
+ # scrapy
+ scrapy/__init__.py E402 E501
+ scrapy/_monkeypatches.py W293
+ scrapy/cmdline.py E502 E501
+ scrapy/crawler.py E501
+ scrapy/dupefilters.py E302 E501 E202
+ scrapy/exceptions.py E302 E501
+ scrapy/exporters.py E501 E261 E226
+ scrapy/extension.py E302
+ scrapy/interfaces.py E302 E501
+ scrapy/item.py E501 E128
+ scrapy/link.py E501
+ scrapy/logformatter.py E501 W293
+ scrapy/mail.py E402 E128 E501 E502
+ scrapy/middleware.py E502 E128 E501
+ scrapy/pqueues.py E501
+ scrapy/resolver.py E302
+ scrapy/responsetypes.py E128 E501 E305
+ scrapy/robotstxt.py E302 E501
+ scrapy/shell.py E501
+ scrapy/signalmanager.py E501
+ scrapy/spiderloader.py E225 F841 E501 E126
+ scrapy/squeues.py E128
+ scrapy/statscollectors.py E501
+ # tests
+ tests/__init__.py F401 E402 E501
+ tests/mockserver.py E401 E501 E126 E123 F401
+ tests/pipelines.py E302 F841 E226
+ tests/spiders.py E302 E501 E127
+ tests/test_closespider.py E501 E127
+ tests/test_command_fetch.py E501 E261
+ tests/test_command_parse.py F401 E302 E501 E128 E303 E226
+ tests/test_command_shell.py E501 E128
+ tests/test_commands.py F401 E128 E501
+ tests/test_contracts.py E501 E128 W293
+ tests/test_crawl.py E501 E741 E265
+ tests/test_crawler.py F841 E306 E501
+ tests/test_dependencies.py E302 F841 E501 E305
+ tests/test_downloader_handlers.py E124 E127 E128 E225 E261 E265 F401 E501 E502 E701 E126 E226 E123
+ tests/test_downloadermiddleware.py E501
+ tests/test_downloadermiddleware_ajaxcrawlable.py E302 E501
+ tests/test_downloadermiddleware_cookies.py E731 E741 E501 E128 E303 E265 E126
+ tests/test_downloadermiddleware_decompression.py E127
+ tests/test_downloadermiddleware_defaultheaders.py E501
+ tests/test_downloadermiddleware_downloadtimeout.py E501
+ tests/test_downloadermiddleware_httpcache.py E501 E302 E305 F401
+ tests/test_downloadermiddleware_httpcompression.py E501 F401 E251 E126 E123
+ tests/test_downloadermiddleware_httpproxy.py F401 E501 E128
+ tests/test_downloadermiddleware_redirect.py E501 E303 E128 E306 E127 E305
+ tests/test_downloadermiddleware_retry.py E501 E128 W293 E251 E502 E303 E126
+ tests/test_downloadermiddleware_robotstxt.py E501
+ tests/test_downloadermiddleware_stats.py E501
+ tests/test_dupefilters.py E302 E221 E501 E741 W293 W291 E128 E124
+ tests/test_engine.py E401 E501 E502 E128 E261
+ tests/test_exporters.py E501 E731 E306 E128 E124
+ tests/test_extension_telnet.py F401 F841
+ tests/test_feedexport.py E501 F401 F841 E241
+ tests/test_http_cookies.py E501
+ tests/test_http_headers.py E302 E501
+ tests/test_http_request.py F401 E402 E501 E261 E127 E128 W293 E502 E128 E502 E126 E123
+ tests/test_http_response.py E501 E301 E502 E128 E265
+ tests/test_item.py E701 E128 F841 E306
+ tests/test_link.py E501
+ tests/test_linkextractors.py E501 E128 E124
+ tests/test_loader.py E302 E501 E731 E303 E741 E128 E117 E241
+ tests/test_logformatter.py E128 E501 E122 E302
+ tests/test_mail.py E302 E128 E501 E305
+ tests/test_middleware.py E302 E501 E128
+ tests/test_pipeline_crawl.py E131 E501 E128 E126
+ tests/test_pipeline_files.py F401 E501 W293 E303 E272 E226
+ tests/test_pipeline_images.py F401 F841 E501 E303
+ tests/test_pipeline_media.py E501 E741 E731 E128 E261 E306 E502
+ tests/test_request_cb_kwargs.py E501
+ tests/test_responsetypes.py E501 E302 E305
+ tests/test_robotstxt_interface.py F401 E302 E501 W291 E501
+ tests/test_scheduler.py E501 E126 E123
+ tests/test_selector.py F401 E501 E127
+ tests/test_spider.py E501 F401
+ tests/test_spidermiddleware.py E501 E226
+ tests/test_spidermiddleware_httperror.py E128 E501 E127 E121
+ tests/test_spidermiddleware_offsite.py E302 E501 E128 E111 W293
+ tests/test_spidermiddleware_output_chain.py F401 E501 E302 W293 E226
+ tests/test_spidermiddleware_referer.py F401 E501 E302 F841 E125 E201 E261 E124 E501 E241 E121
+ tests/test_squeues.py E501 E302 E701 E741
+ tests/test_utils_conf.py E501 E303 E128
+ tests/test_utils_console.py E302
+ tests/test_utils_curl.py E501
+ tests/test_utils_datatypes.py E402 E501 E305
+ tests/test_utils_defer.py E306 E261 E501 E302 F841 E226
+ tests/test_utils_deprecate.py F841 E306 E501
+ tests/test_utils_http.py E302 E501 E502 E128 W504
+ tests/test_utils_httpobj.py E302
+ tests/test_utils_iterators.py E501 E128 E129 E302 E303 E241
+ tests/test_utils_log.py E741 E226
+ tests/test_utils_python.py E501 E303 E731 E701 E305
+ tests/test_utils_reqser.py F401 E501 E128
+ tests/test_utils_request.py E302 E501 E128 E305
+ tests/test_utils_response.py E501
+ tests/test_utils_signal.py E741 F841 E302 E731 E226
+ tests/test_utils_sitemap.py E302 E128 E501 E124
+ tests/test_utils_spider.py E261 E302 E305
+ tests/test_utils_template.py E305
+ tests/test_utils_url.py F401 E501 E127 E302 E305 E211 E125 E501 E226 E241 E126 E123
+ tests/test_webclient.py E501 E128 E122 E303 E402 E306 E226 E241 E123 E126
+ tests/mocks/dummydbm.py E302
+ tests/test_cmdline/__init__.py E502 E501
+ tests/test_cmdline/extensions.py E302
+ tests/test_settings/__init__.py F401 E501 E128
+ tests/test_spiderloader/__init__.py E128 E501 E302
+ tests/test_spiderloader/test_spiders/spider0.py E302
+ tests/test_spiderloader/test_spiders/spider1.py E302
+ tests/test_spiderloader/test_spiders/spider2.py E302
+ tests/test_spiderloader/test_spiders/spider3.py E302
+ tests/test_spiderloader/test_spiders/nested/spider4.py E302
+ tests/test_utils_misc/__init__.py E501
diff --git a/requirements-py2.txt b/requirements-py2.txt
deleted file mode 100644
index dde8d1c9c..000000000
--- a/requirements-py2.txt
+++ /dev/null
@@ -1,18 +0,0 @@
-parsel>=1.5.0
-PyDispatcher>=2.0.5
-w3lib>=1.17.0
-protego>=0.1.15
-
-pyOpenSSL>=16.2.0 # Earlier versions fail with "AttributeError: module 'lib' has no attribute 'SSL_ST_INIT'"
-queuelib>=1.4.2 # Earlier versions fail with "AttributeError: '...QueueTest' object has no attribute 'qpath'"
-cryptography>=2.0 # Earlier versions would fail to install
-
-# Reference versions taken from
-# https://packages.ubuntu.com/xenial/python/
-# https://packages.ubuntu.com/xenial/zope/
-cssselect>=0.9.1
-lxml>=3.5.0
-service_identity>=16.0.0
-six>=1.10.0
-Twisted>=16.0.0
-zope.interface>=4.1.3
diff --git a/scrapy/VERSION b/scrapy/VERSION
index bd8bf882d..27f9cd322 100644
--- a/scrapy/VERSION
+++ b/scrapy/VERSION
@@ -1 +1 @@
-1.7.0
+1.8.0
diff --git a/scrapy/__init__.py b/scrapy/__init__.py
index 03ec6c667..230e5cee3 100644
--- a/scrapy/__init__.py
+++ b/scrapy/__init__.py
@@ -14,8 +14,8 @@ del pkgutil
# Check minimum required Python version
import sys
-if sys.version_info < (2, 7):
- print("Scrapy %s requires Python 2.7" % __version__)
+if sys.version_info < (3, 5):
+ print("Scrapy %s requires Python 3.5" % __version__)
sys.exit(1)
# Ignore noisy twisted deprecation warnings
diff --git a/scrapy/_monkeypatches.py b/scrapy/_monkeypatches.py
index b68099cad..1f8067b35 100644
--- a/scrapy/_monkeypatches.py
+++ b/scrapy/_monkeypatches.py
@@ -1,16 +1,6 @@
-import six
from six.moves import copyreg
-if six.PY2:
- from urlparse import urlparse
-
- # workaround for https://bugs.python.org/issue9374 - Python < 2.7.4
- if urlparse('s3://bucket/key?key=value').query != 'key=value':
- from urlparse import uses_query
- uses_query.append('s3')
-
-
# Undo what Twisted's perspective broker adds to pickle register
# to prevent bugs like Twisted#7989 while serializing requests
import twisted.persisted.styles # NOQA
diff --git a/scrapy/commands/check.py b/scrapy/commands/check.py
index ab73e85e7..3e6c11b7d 100644
--- a/scrapy/commands/check.py
+++ b/scrapy/commands/check.py
@@ -96,4 +96,3 @@ class Command(ScrapyCommand):
result.printErrors()
result.printSummary(start, stop)
self.exitcode = int(not result.wasSuccessful())
-
diff --git a/scrapy/commands/fetch.py b/scrapy/commands/fetch.py
index 7d4840529..d45133e0e 100644
--- a/scrapy/commands/fetch.py
+++ b/scrapy/commands/fetch.py
@@ -1,5 +1,5 @@
from __future__ import print_function
-import sys, six
+import sys
from w3lib.url import is_url
from scrapy.commands import ScrapyCommand
@@ -45,8 +45,7 @@ class Command(ScrapyCommand):
self._print_bytes(response.body)
def _print_bytes(self, bytes_):
- bytes_writer = sys.stdout if six.PY2 else sys.stdout.buffer
- bytes_writer.write(bytes_ + b'\n')
+ sys.stdout.buffer.write(bytes_ + b'\n')
def run(self, args, opts):
if len(args) != 1 or not is_url(args[0]):
diff --git a/scrapy/commands/startproject.py b/scrapy/commands/startproject.py
index 67337c26e..3b9f6eabb 100644
--- a/scrapy/commands/startproject.py
+++ b/scrapy/commands/startproject.py
@@ -119,4 +119,3 @@ class Command(ScrapyCommand):
_templates_base_dir = self.settings['TEMPLATES_DIR'] or \
join(scrapy.__path__[0], 'templates')
return join(_templates_base_dir, 'project')
-
diff --git a/scrapy/commands/version.py b/scrapy/commands/version.py
index 577365c3b..8651948f7 100644
--- a/scrapy/commands/version.py
+++ b/scrapy/commands/version.py
@@ -30,4 +30,3 @@ class Command(ScrapyCommand):
print(patt % (name, version))
else:
print("Scrapy %s" % scrapy.__version__)
-
diff --git a/scrapy/core/downloader/handlers/ftp.py b/scrapy/core/downloader/handlers/ftp.py
index 806a537d4..39ed67a1a 100644
--- a/scrapy/core/downloader/handlers/ftp.py
+++ b/scrapy/core/downloader/handlers/ftp.py
@@ -112,4 +112,3 @@ class FTPDownloadHandler(object):
httpcode = self.CODE_MAPPING.get(ftpcode, self.CODE_MAPPING["default"])
return Response(url=request.url, status=httpcode, body=to_bytes(message))
raise result.type(result.value)
-
diff --git a/scrapy/core/downloader/handlers/http11.py b/scrapy/core/downloader/handlers/http11.py
index 91b45a8fc..7d917cb74 100644
--- a/scrapy/core/downloader/handlers/http11.py
+++ b/scrapy/core/downloader/handlers/http11.py
@@ -174,7 +174,7 @@ def tunnel_request_data(host, port, proxy_auth_header=None):
r"""
Return binary content of a CONNECT request.
- >>> from scrapy.utils.python import to_native_str as s
+ >>> from scrapy.utils.python import to_unicode as s
>>> s(tunnel_request_data("example.com", 8080))
'CONNECT example.com:8080 HTTP/1.1\r\nHost: example.com:8080\r\n\r\n'
>>> s(tunnel_request_data("example.com", 8080, b"123"))
diff --git a/scrapy/core/downloader/webclient.py b/scrapy/core/downloader/webclient.py
index 3a5890ed0..3fe13414a 100644
--- a/scrapy/core/downloader/webclient.py
+++ b/scrapy/core/downloader/webclient.py
@@ -157,4 +157,3 @@ class ScrapyHTTPClientFactory(HTTPClientFactory):
def gotHeaders(self, headers):
self.headers_time = time()
self.response_headers = headers
-
diff --git a/scrapy/core/scraper.py b/scrapy/core/scraper.py
index 1f389cf2e..db463f989 100644
--- a/scrapy/core/scraper.py
+++ b/scrapy/core/scraper.py
@@ -231,9 +231,9 @@ class Scraper(object):
signal=signals.item_dropped, item=item, response=response,
spider=spider, exception=output.value)
else:
- logger.error('Error processing %(item)s', {'item': item},
- exc_info=failure_to_exc_info(output),
- extra={'spider': spider})
+ logkws = self.logformatter.error(item, ex, response, spider)
+ logger.log(*logformatter_adapter(logkws), extra={'spider': spider},
+ exc_info=failure_to_exc_info(output))
return self.signals.send_catch_log_deferred(
signal=signals.item_error, item=item, response=response,
spider=spider, failure=output)
@@ -244,4 +244,3 @@ class Scraper(object):
return self.signals.send_catch_log_deferred(
signal=signals.item_scraped, item=output, response=response,
spider=spider)
-
diff --git a/scrapy/core/spidermw.py b/scrapy/core/spidermw.py
index b5f9837ff..00cee3ada 100644
--- a/scrapy/core/spidermw.py
+++ b/scrapy/core/spidermw.py
@@ -36,7 +36,7 @@ class SpiderMiddlewareManager(MiddlewareManager):
self.methods['process_spider_exception'].appendleft(getattr(mw, 'process_spider_exception', None))
def scrape_response(self, scrape_func, response, request, spider):
- fname = lambda f:'%s.%s' % (
+ fname = lambda f: '%s.%s' % (
six.get_method_self(f).__class__.__name__,
six.get_method_function(f).__name__)
diff --git a/scrapy/crawler.py b/scrapy/crawler.py
index ded3c082b..8868a985b 100644
--- a/scrapy/crawler.py
+++ b/scrapy/crawler.py
@@ -3,7 +3,6 @@ import signal
import logging
import warnings
-import sys
from twisted.internet import reactor, defer
from zope.interface.verify import verifyClass, DoesNotImplement
@@ -88,20 +87,9 @@ class Crawler(object):
yield self.engine.open_spider(self.spider, start_requests)
yield defer.maybeDeferred(self.engine.start)
except Exception:
- # In Python 2 reraising an exception after yield discards
- # the original traceback (see https://bugs.python.org/issue7563),
- # so sys.exc_info() workaround is used.
- # This workaround also works in Python 3, but it is not needed,
- # and it is slower, so in Python 3 we use native `raise`.
- if six.PY2:
- exc_info = sys.exc_info()
-
self.crawling = False
if self.engine is not None:
yield self.engine.close()
-
- if six.PY2:
- six.reraise(*exc_info)
raise
def _create_spider(self, *args, **kwargs):
diff --git a/scrapy/downloadermiddlewares/cookies.py b/scrapy/downloadermiddlewares/cookies.py
index 321c0171b..0d2b9900c 100644
--- a/scrapy/downloadermiddlewares/cookies.py
+++ b/scrapy/downloadermiddlewares/cookies.py
@@ -6,7 +6,7 @@ from collections import defaultdict
from scrapy.exceptions import NotConfigured
from scrapy.http import Response
from scrapy.http.cookies import CookieJar
-from scrapy.utils.python import to_native_str
+from scrapy.utils.python import to_unicode
logger = logging.getLogger(__name__)
@@ -53,7 +53,7 @@ class CookiesMiddleware(object):
def _debug_cookie(self, request, spider):
if self.debug:
- cl = [to_native_str(c, errors='replace')
+ cl = [to_unicode(c, errors='replace')
for c in request.headers.getlist('Cookie')]
if cl:
cookies = "\n".join("Cookie: {}\n".format(c) for c in cl)
@@ -62,7 +62,7 @@ class CookiesMiddleware(object):
def _debug_set_cookie(self, response, spider):
if self.debug:
- cl = [to_native_str(c, errors='replace')
+ cl = [to_unicode(c, errors='replace')
for c in response.headers.getlist('Set-Cookie')]
if cl:
cookies = "\n".join("Set-Cookie: {}\n".format(c) for c in cl)
diff --git a/scrapy/downloadermiddlewares/decompression.py b/scrapy/downloadermiddlewares/decompression.py
index 49313cc04..e2d73f347 100644
--- a/scrapy/downloadermiddlewares/decompression.py
+++ b/scrapy/downloadermiddlewares/decompression.py
@@ -4,6 +4,7 @@ and extract the potentially compressed responses that may arrive.
import bz2
import gzip
+from io import BytesIO
import zipfile
import tarfile
import logging
@@ -11,11 +12,6 @@ from tempfile import mktemp
import six
-try:
- from cStringIO import StringIO as BytesIO
-except ImportError:
- from io import BytesIO
-
from scrapy.responsetypes import responsetypes
logger = logging.getLogger(__name__)
diff --git a/scrapy/downloadermiddlewares/httpproxy.py b/scrapy/downloadermiddlewares/httpproxy.py
index 2c35d1b90..2212d9688 100644
--- a/scrapy/downloadermiddlewares/httpproxy.py
+++ b/scrapy/downloadermiddlewares/httpproxy.py
@@ -1,10 +1,7 @@
import base64
from six.moves.urllib.parse import unquote, urlunparse
from six.moves.urllib.request import getproxies, proxy_bypass
-try:
- from urllib2 import _parse_proxy
-except ImportError:
- from urllib.request import _parse_proxy
+from urllib.request import _parse_proxy
from scrapy.exceptions import NotConfigured
from scrapy.utils.httpobj import urlparse_cached
diff --git a/scrapy/downloadermiddlewares/redirect.py b/scrapy/downloadermiddlewares/redirect.py
index 49468a2e4..b73f864dd 100644
--- a/scrapy/downloadermiddlewares/redirect.py
+++ b/scrapy/downloadermiddlewares/redirect.py
@@ -1,5 +1,5 @@
import logging
-from six.moves.urllib.parse import urljoin
+from six.moves.urllib.parse import urljoin, urlparse
from w3lib.url import safe_url_string
@@ -70,7 +70,10 @@ class RedirectMiddleware(BaseRedirectMiddleware):
if 'Location' not in response.headers or response.status not in allowed_status:
return response
- location = safe_url_string(response.headers['location'])
+ location = safe_url_string(response.headers['Location'])
+ if response.headers['Location'].startswith(b'//'):
+ request_scheme = urlparse(request.url).scheme
+ location = request_scheme + '://' + location.lstrip('/')
redirected_url = urljoin(request.url, location)
diff --git a/scrapy/downloadermiddlewares/robotstxt.py b/scrapy/downloadermiddlewares/robotstxt.py
index 6a5dfb79c..251706c50 100644
--- a/scrapy/downloadermiddlewares/robotstxt.py
+++ b/scrapy/downloadermiddlewares/robotstxt.py
@@ -5,15 +5,12 @@ enable this middleware and enable the ROBOTSTXT_OBEY setting.
"""
import logging
-import sys
-import re
from twisted.internet.defer import Deferred, maybeDeferred
from scrapy.exceptions import NotConfigured, IgnoreRequest
from scrapy.http import Request
from scrapy.utils.httpobj import urlparse_cached
from scrapy.utils.log import failure_to_exc_info
-from scrapy.utils.python import to_native_str
from scrapy.utils.misc import load_object
logger = logging.getLogger(__name__)
diff --git a/scrapy/exporters.py b/scrapy/exporters.py
index 6fc87ed18..3defafd60 100644
--- a/scrapy/exporters.py
+++ b/scrapy/exporters.py
@@ -4,7 +4,6 @@ Item Exporters are used to export/serialize items into different formats.
import csv
import io
-import sys
import pprint
import marshal
import six
@@ -12,7 +11,7 @@ from six.moves import cPickle as pickle
from xml.sax.saxutils import XMLGenerator
from scrapy.utils.serialize import ScrapyJSONEncoder
-from scrapy.utils.python import to_bytes, to_unicode, to_native_str, is_listlike
+from scrapy.utils.python import to_bytes, to_unicode, is_listlike
from scrapy.item import BaseItem
from scrapy.exceptions import ScrapyDeprecationWarning
import warnings
@@ -31,7 +30,7 @@ class BaseItemExporter(object):
def _configure(self, options, dont_fail=False):
"""Configure the exporter by poping options from the ``options`` dict.
If dont_fail is set, it won't raise an exception on unexpected options
- (useful for using with keyword arguments in subclasses constructors)
+ (useful for using with keyword arguments in subclasses ``__init__`` methods)
"""
self.encoding = options.pop('encoding', None)
self.fields_to_export = options.pop('fields_to_export', None)
@@ -143,11 +142,11 @@ class XmlItemExporter(BaseItemExporter):
def _beautify_newline(self, new_item=False):
if self.indent is not None and (self.indent > 0 or new_item):
- self._xg_characters('\n')
+ self.xg.characters('\n')
def _beautify_indent(self, depth=1):
if self.indent:
- self._xg_characters(' ' * self.indent * depth)
+ self.xg.characters(' ' * self.indent * depth)
def start_exporting(self):
self.xg.startDocument()
@@ -182,26 +181,12 @@ class XmlItemExporter(BaseItemExporter):
self._export_xml_field('value', value, depth=depth+1)
self._beautify_indent(depth=depth)
elif isinstance(serialized_value, six.text_type):
- self._xg_characters(serialized_value)
+ self.xg.characters(serialized_value)
else:
- self._xg_characters(str(serialized_value))
+ self.xg.characters(str(serialized_value))
self.xg.endElement(name)
self._beautify_newline()
- # Workaround for https://bugs.python.org/issue17606
- # Before Python 2.7.4 xml.sax.saxutils required bytes;
- # since 2.7.4 it requires unicode. The bug is likely to be
- # fixed in 2.7.6, but 2.7.6 will still support unicode,
- # and Python 3.x will require unicode, so ">= 2.7.4" should be fine.
- if sys.version_info[:3] >= (2, 7, 4):
- def _xg_characters(self, serialized_value):
- if not isinstance(serialized_value, six.text_type):
- serialized_value = serialized_value.decode(self.encoding)
- return self.xg.characters(serialized_value)
- else: # pragma: no cover
- def _xg_characters(self, serialized_value):
- return self.xg.characters(serialized_value)
-
class CsvItemExporter(BaseItemExporter):
@@ -216,7 +201,7 @@ class CsvItemExporter(BaseItemExporter):
write_through=True,
encoding=self.encoding,
newline='' # Windows needs this https://github.com/scrapy/scrapy/issues/3034
- ) if six.PY3 else file
+ )
self.csv_writer = csv.writer(self.stream, **kwargs)
self._headers_not_written = True
self._join_multivalued = join_multivalued
@@ -246,7 +231,7 @@ class CsvItemExporter(BaseItemExporter):
def _build_row(self, values):
for s in values:
try:
- yield to_native_str(s, self.encoding)
+ yield to_unicode(s, self.encoding)
except TypeError:
yield s
diff --git a/scrapy/extensions/feedexport.py b/scrapy/extensions/feedexport.py
index ce2846eba..e2492d506 100644
--- a/scrapy/extensions/feedexport.py
+++ b/scrapy/extensions/feedexport.py
@@ -10,7 +10,6 @@ import logging
import posixpath
from tempfile import NamedTemporaryFile
from datetime import datetime
-import six
from six.moves.urllib.parse import urlparse, unquote
from ftplib import FTP
@@ -65,7 +64,7 @@ class StdoutFeedStorage(object):
def __init__(self, uri, _stdout=None):
if not _stdout:
- _stdout = sys.stdout if six.PY2 else sys.stdout.buffer
+ _stdout = sys.stdout.buffer
self._stdout = _stdout
def open(self, spider):
@@ -199,9 +198,9 @@ class FeedExporter(object):
def __init__(self, settings):
self.settings = settings
- self.urifmt = settings['FEED_URI']
- if not self.urifmt:
+ if not settings['FEED_URI']:
raise NotConfigured
+ self.urifmt = str(settings['FEED_URI'])
self.format = settings['FEED_FORMAT'].lower()
self.export_encoding = settings['FEED_EXPORT_ENCODING']
self.storages = self._load_components('FEED_STORAGES')
@@ -242,7 +241,9 @@ class FeedExporter(object):
def close_spider(self, spider):
slot = self.slot
if not slot.itemcount and not self.store_empty:
- return
+ # We need to call slot.storage.store nonetheless to get the file
+ # properly closed.
+ return defer.maybeDeferred(slot.storage.store, slot.file)
if self._exporting:
slot.exporter.finish_exporting()
self._exporting = False
diff --git a/scrapy/extensions/httpcache.py b/scrapy/extensions/httpcache.py
index c6094643d..f3fabf710 100644
--- a/scrapy/extensions/httpcache.py
+++ b/scrapy/extensions/httpcache.py
@@ -1,19 +1,24 @@
from __future__ import print_function
-import os
+
import gzip
import logging
-from six.moves import cPickle as pickle
+import os
+from email.utils import mktime_tz, parsedate_tz
from importlib import import_module
from time import time
+from warnings import warn
from weakref import WeakKeyDictionary
-from email.utils import mktime_tz, parsedate_tz
+
+from six.moves import cPickle as pickle
from w3lib.http import headers_raw_to_dict, headers_dict_to_raw
+
+from scrapy.exceptions import ScrapyDeprecationWarning
from scrapy.http import Headers, Response
from scrapy.responsetypes import responsetypes
-from scrapy.utils.request import request_fingerprint
-from scrapy.utils.project import data_path
from scrapy.utils.httpobj import urlparse_cached
+from scrapy.utils.project import data_path
from scrapy.utils.python import to_bytes, to_unicode, garbage_collect
+from scrapy.utils.request import request_fingerprint
logger = logging.getLogger(__name__)
@@ -342,75 +347,6 @@ class FilesystemCacheStorage(object):
return pickle.load(f)
-class LeveldbCacheStorage(object):
-
- def __init__(self, settings):
- import leveldb
- self._leveldb = leveldb
- self.cachedir = data_path(settings['HTTPCACHE_DIR'], createdir=True)
- self.expiration_secs = settings.getint('HTTPCACHE_EXPIRATION_SECS')
- self.db = None
-
- def open_spider(self, spider):
- dbpath = os.path.join(self.cachedir, '%s.leveldb' % spider.name)
- self.db = self._leveldb.LevelDB(dbpath)
-
- logger.debug("Using LevelDB cache storage in %(cachepath)s" % {'cachepath': dbpath}, extra={'spider': spider})
-
- def close_spider(self, spider):
- # Do compactation each time to save space and also recreate files to
- # avoid them being removed in storages with timestamp-based autoremoval.
- self.db.CompactRange()
- del self.db
- garbage_collect()
-
- def retrieve_response(self, spider, request):
- data = self._read_data(spider, request)
- if data is None:
- return # not cached
- url = data['url']
- status = data['status']
- headers = Headers(data['headers'])
- body = data['body']
- respcls = responsetypes.from_args(headers=headers, url=url)
- response = respcls(url=url, headers=headers, status=status, body=body)
- return response
-
- def store_response(self, spider, request, response):
- key = self._request_key(request)
- data = {
- 'status': response.status,
- 'url': response.url,
- 'headers': dict(response.headers),
- 'body': response.body,
- }
- batch = self._leveldb.WriteBatch()
- batch.Put(key + b'_data', pickle.dumps(data, protocol=2))
- batch.Put(key + b'_time', to_bytes(str(time())))
- self.db.Write(batch)
-
- def _read_data(self, spider, request):
- key = self._request_key(request)
- try:
- ts = self.db.Get(key + b'_time')
- except KeyError:
- return # not found or invalid entry
-
- if 0 < self.expiration_secs < time() - float(ts):
- return # expired
-
- try:
- data = self.db.Get(key + b'_data')
- except KeyError:
- return # invalid entry
- else:
- return pickle.loads(data)
-
- def _request_key(self, request):
- return to_bytes(request_fingerprint(request))
-
-
-
def parse_cachecontrol(header):
"""Parse Cache-Control header
diff --git a/scrapy/http/cookies.py b/scrapy/http/cookies.py
index 4e8056750..60a14c6f8 100644
--- a/scrapy/http/cookies.py
+++ b/scrapy/http/cookies.py
@@ -3,7 +3,7 @@ from six.moves.http_cookiejar import (
CookieJar as _CookieJar, DefaultCookiePolicy, IPV4_RE
)
from scrapy.utils.httpobj import urlparse_cached
-from scrapy.utils.python import to_native_str
+from scrapy.utils.python import to_unicode
class CookieJar(object):
@@ -165,13 +165,13 @@ class WrappedRequest(object):
return name in self.request.headers
def get_header(self, name, default=None):
- return to_native_str(self.request.headers.get(name, default),
- errors='replace')
+ return to_unicode(self.request.headers.get(name, default),
+ errors='replace')
def header_items(self):
return [
- (to_native_str(k, errors='replace'),
- [to_native_str(x, errors='replace') for x in v])
+ (to_unicode(k, errors='replace'),
+ [to_unicode(x, errors='replace') for x in v])
for k, v in self.request.headers.items()
]
@@ -189,7 +189,7 @@ class WrappedResponse(object):
# python3 cookiejars calls get_all
def get_all(self, name, default=None):
- return [to_native_str(v, errors='replace')
+ return [to_unicode(v, errors='replace')
for v in self.response.headers.getlist(name)]
# python2 cookiejars calls getheaders
getheaders = get_all
diff --git a/scrapy/http/headers.py b/scrapy/http/headers.py
index 62507eb19..f3b46b994 100644
--- a/scrapy/http/headers.py
+++ b/scrapy/http/headers.py
@@ -91,5 +91,3 @@ class Headers(CaselessDict):
def __copy__(self):
return self.__class__(self)
copy = __copy__
-
-
diff --git a/scrapy/http/request/__init__.py b/scrapy/http/request/__init__.py
index d09eaf849..76a428199 100644
--- a/scrapy/http/request/__init__.py
+++ b/scrapy/http/request/__init__.py
@@ -66,7 +66,7 @@ class Request(object_ref):
s = safe_url_string(url, self.encoding)
self._url = escape_ajax(s)
- if ':' not in self._url:
+ if ('://' not in self._url) and (not self._url.startswith('data:')):
raise ValueError('Missing scheme in request url: %s' % self._url)
url = property(_get_url, obsolete_setter(_set_url, 'url'))
diff --git a/scrapy/http/request/form.py b/scrapy/http/request/form.py
index 3ce8fc48e..b6feede07 100644
--- a/scrapy/http/request/form.py
+++ b/scrapy/http/request/form.py
@@ -104,8 +104,7 @@ def _get_form(response, formname, formid, formnumber, formxpath):
el = el.getparent()
if el is None:
break
- encoded = formxpath if six.PY3 else formxpath.encode('unicode_escape')
- raise ValueError('No """)
- r1 = self.request_class.from_response(response, formdata={'two':'3'})
+ r1 = self.request_class.from_response(response, formdata={'two': '3'})
self.assertEqual(r1.method, 'POST')
self.assertEqual(r1.headers['Content-type'], b'application/x-www-form-urlencoded')
fs = _qs(r1)
@@ -1064,8 +1063,7 @@ class FormRequestTest(RequestTest):
self.assertEqual(fs, {})
xpath = u"//form[@name='\u03b1']"
- encoded = xpath if six.PY3 else xpath.encode('unicode_escape')
- self.assertRaisesRegex(ValueError, re.escape(encoded),
+ self.assertRaisesRegex(ValueError, re.escape(xpath),
self.request_class.from_response,
response, formxpath=xpath)
@@ -1208,10 +1206,7 @@ def _qs(req, encoding='utf-8', to_unicode=False):
qs = req.body
else:
qs = req.url.partition('?')[2]
- if six.PY2:
- uqs = unquote(to_native_str(qs, encoding))
- elif six.PY3:
- uqs = unquote_to_bytes(qs)
+ uqs = unquote_to_bytes(qs)
if to_unicode:
uqs = uqs.decode(encoding)
return parse_qs(uqs, True)
diff --git a/tests/test_http_response.py b/tests/test_http_response.py
index 0ae1612b5..36ccdfa1f 100644
--- a/tests/test_http_response.py
+++ b/tests/test_http_response.py
@@ -7,7 +7,7 @@ from w3lib.encoding import resolve_encoding
from scrapy.http import (Request, Response, TextResponse, HtmlResponse,
XmlResponse, Headers)
from scrapy.selector import Selector
-from scrapy.utils.python import to_native_str
+from scrapy.utils.python import to_unicode
from scrapy.exceptions import NotSupported
from scrapy.link import Link
from tests import get_testdata
@@ -21,8 +21,7 @@ class BaseResponseTest(unittest.TestCase):
# Response requires url in the consturctor
self.assertRaises(Exception, self.response_class)
self.assertTrue(isinstance(self.response_class('http://example.com/'), self.response_class))
- if not six.PY2:
- self.assertRaises(TypeError, self.response_class, b"http://example.com")
+ self.assertRaises(TypeError, self.response_class, b"http://example.com")
# body can be str or None
self.assertTrue(isinstance(self.response_class('http://example.com/', body=b''), self.response_class))
self.assertTrue(isinstance(self.response_class('http://example.com/', body=b'body'), self.response_class))
@@ -286,11 +285,11 @@ class TextResponseTest(BaseResponseTest):
assert isinstance(resp.url, str)
resp = self.response_class(url=u"http://www.example.com/price/\xa3", encoding='utf-8')
- self.assertEqual(resp.url, to_native_str(b'http://www.example.com/price/\xc2\xa3'))
+ self.assertEqual(resp.url, to_unicode(b'http://www.example.com/price/\xc2\xa3'))
resp = self.response_class(url=u"http://www.example.com/price/\xa3", encoding='latin-1')
self.assertEqual(resp.url, 'http://www.example.com/price/\xa3')
resp = self.response_class(u"http://www.example.com/price/\xa3", headers={"Content-type": ["text/html; charset=utf-8"]})
- self.assertEqual(resp.url, to_native_str(b'http://www.example.com/price/\xc2\xa3'))
+ self.assertEqual(resp.url, to_unicode(b'http://www.example.com/price/\xc2\xa3'))
resp = self.response_class(u"http://www.example.com/price/\xa3", headers={"Content-type": ["text/html; charset=iso-8859-1"]})
self.assertEqual(resp.url, 'http://www.example.com/price/\xa3')
@@ -658,7 +657,7 @@ class XmlResponseTest(TextResponseTest):
r2 = self.response_class("http://www.example.com", body=body)
self._assert_response_values(r2, 'iso-8859-1', body)
- # make sure replace() preserves the explicit encoding passed in the constructor
+ # make sure replace() preserves the explicit encoding passed in the __init__ method
body = b""""""
r3 = self.response_class("http://www.example.com", body=body, encoding='utf-8')
body2 = b"New body"
diff --git a/tests/test_item.py b/tests/test_item.py
index 947566686..49117ef04 100644
--- a/tests/test_item.py
+++ b/tests/test_item.py
@@ -1,12 +1,12 @@
import sys
import unittest
+from unittest import mock
from warnings import catch_warnings
import six
from scrapy.exceptions import ScrapyDeprecationWarning
from scrapy.item import ABCMeta, DictItem, Field, Item, ItemMeta
-from tests import mock
PY36_PLUS = (sys.version_info.major >= 3) and (sys.version_info.minor >= 6)
@@ -62,12 +62,8 @@ class ItemTest(unittest.TestCase):
i['number'] = 123
itemrepr = repr(i)
- if six.PY2:
- self.assertEqual(itemrepr,
- "{'name': u'John Doe', 'number': 123}")
- else:
- self.assertEqual(itemrepr,
- "{'name': 'John Doe', 'number': 123}")
+ self.assertEqual(itemrepr,
+ "{'name': 'John Doe', 'number': 123}")
i2 = eval(itemrepr)
self.assertEqual(i2['name'], 'John Doe')
@@ -245,7 +241,7 @@ class ItemTest(unittest.TestCase):
def test_copy(self):
class TestItem(Item):
name = Field()
- item = TestItem({'name':'lower'})
+ item = TestItem({'name': 'lower'})
copied_item = item.copy()
self.assertNotEqual(id(item), id(copied_item))
copied_item['name'] = copied_item['name'].upper()
diff --git a/tests/test_link.py b/tests/test_link.py
index 955430b37..e0f1efffa 100644
--- a/tests/test_link.py
+++ b/tests/test_link.py
@@ -1,6 +1,4 @@
import unittest
-import warnings
-import six
from scrapy.link import Link
@@ -45,13 +43,6 @@ class LinkTest(unittest.TestCase):
l2 = eval(repr(l1))
self._assert_same_links(l1, l2)
- def test_non_str_url_py2(self):
- if six.PY2:
- with warnings.catch_warnings(record=True) as w:
- link = Link(u"http://www.example.com/\xa3")
- self.assertIsInstance(link.url, str)
- self.assertEqual(link.url, b'http://www.example.com/\xc2\xa3')
- assert len(w) == 1, "warning not issued"
- else:
- with self.assertRaises(TypeError):
- Link(b"http://www.example.com/\xc2\xa3")
+ def test_bytes_url(self):
+ with self.assertRaises(TypeError):
+ Link(b"http://www.example.com/\xc2\xa3")
diff --git a/tests/test_linkextractors.py b/tests/test_linkextractors.py
index d96e259f6..57ef1694a 100644
--- a/tests/test_linkextractors.py
+++ b/tests/test_linkextractors.py
@@ -322,7 +322,7 @@ class Base:
Link(url=page4_url, text=u'href with whitespaces'),
])
- lx = self.extractor_cls(attrs=("href","src"), tags=("a","area","img"), deny_extensions=())
+ lx = self.extractor_cls(attrs=("href", "src"), tags=("a", "area", "img"), deny_extensions=())
self.assertEqual(lx.extract_links(self.response), [
Link(url='http://example.com/sample1.html', text=u''),
Link(url='http://example.com/sample2.html', text=u'sample 2'),
@@ -360,7 +360,7 @@ class Base:
Link(url='http://example.com/sample2.html', text=u'sample 2'),
])
- lx = self.extractor_cls(tags=("a","img"), attrs=("href", "src"), deny_extensions=())
+ lx = self.extractor_cls(tags=("a", "img"), attrs=("href", "src"), deny_extensions=())
self.assertEqual(lx.extract_links(response), [
Link(url='http://example.com/sample2.html', text=u'sample 2'),
Link(url='http://example.com/sample2.jpg', text=u''),
diff --git a/tests/test_loader.py b/tests/test_loader.py
index 2725b001a..b87602809 100644
--- a/tests/test_loader.py
+++ b/tests/test_loader.py
@@ -1,13 +1,15 @@
-import unittest
-import six
from functools import partial
+import unittest
+
+import six
-from scrapy.loader import ItemLoader
-from scrapy.loader.processors import Join, Identity, TakeFirst, \
- Compose, MapCompose, SelectJmes
-from scrapy.item import Item, Field
-from scrapy.selector import Selector
from scrapy.http import HtmlResponse
+from scrapy.item import Item, Field
+from scrapy.loader import ItemLoader
+from scrapy.loader.processors import (Compose, Identity, Join,
+ MapCompose, SelectJmes, TakeFirst)
+from scrapy.selector import Selector
+
# test items
class NameItem(Item):
@@ -61,7 +63,7 @@ class BasicItemLoaderTest(unittest.TestCase):
il.add_value('name', u'marta')
item = il.load_item()
assert item is i
- self.assertEqual(item['summary'], u'lala')
+ self.assertEqual(item['summary'], [u'lala'])
self.assertEqual(item['name'], [u'marta'])
def test_load_item_using_custom_loader(self):
@@ -419,43 +421,6 @@ class BasicItemLoaderTest(unittest.TestCase):
self.assertEqual(item['url'], u'rabbit.hole')
self.assertEqual(item['summary'], u'rabbithole')
- def test_create_item_from_dict(self):
- class TestItem(Item):
- title = Field()
-
- class TestItemLoader(ItemLoader):
- default_item_class = TestItem
-
- input_item = {'title': 'Test item title 1'}
- il = TestItemLoader(item=input_item)
- # Getting output value mustn't remove value from item
- self.assertEqual(il.load_item(), {
- 'title': 'Test item title 1',
- })
- self.assertEqual(il.get_output_value('title'), 'Test item title 1')
- self.assertEqual(il.load_item(), {
- 'title': 'Test item title 1',
- })
-
- input_item = {'title': 'Test item title 2'}
- il = TestItemLoader(item=input_item)
- # Values from dict must be added to item _values
- self.assertEqual(il._values.get('title'), 'Test item title 2')
-
- input_item = {'title': [u'Test item title 3', u'Test item 4']}
- il = TestItemLoader(item=input_item)
- # Same rules must work for lists
- self.assertEqual(il._values.get('title'),
- [u'Test item title 3', u'Test item 4'])
- self.assertEqual(il.load_item(), {
- 'title': [u'Test item title 3', u'Test item 4'],
- })
- self.assertEqual(il.get_output_value('title'),
- [u'Test item title 3', u'Test item 4'])
- self.assertEqual(il.load_item(), {
- 'title': [u'Test item title 3', u'Test item 4'],
- })
-
def test_error_input_processor(self):
class TestItem(Item):
name = Field()
@@ -493,6 +458,220 @@ class BasicItemLoaderTest(unittest.TestCase):
[u'marta', u'other'], Compose(float))
+class InitializationTestMixin(object):
+
+ item_class = None
+
+ def test_keep_single_value(self):
+ """Loaded item should contain values from the initial item"""
+ input_item = self.item_class(name='foo')
+ il = ItemLoader(item=input_item)
+ loaded_item = il.load_item()
+ self.assertIsInstance(loaded_item, self.item_class)
+ self.assertEqual(dict(loaded_item), {'name': ['foo']})
+
+ def test_keep_list(self):
+ """Loaded item should contain values from the initial item"""
+ input_item = self.item_class(name=['foo', 'bar'])
+ il = ItemLoader(item=input_item)
+ loaded_item = il.load_item()
+ self.assertIsInstance(loaded_item, self.item_class)
+ self.assertEqual(dict(loaded_item), {'name': ['foo', 'bar']})
+
+ def test_add_value_singlevalue_singlevalue(self):
+ """Values added after initialization should be appended"""
+ input_item = self.item_class(name='foo')
+ il = ItemLoader(item=input_item)
+ il.add_value('name', 'bar')
+ loaded_item = il.load_item()
+ self.assertIsInstance(loaded_item, self.item_class)
+ self.assertEqual(dict(loaded_item), {'name': ['foo', 'bar']})
+
+ def test_add_value_singlevalue_list(self):
+ """Values added after initialization should be appended"""
+ input_item = self.item_class(name='foo')
+ il = ItemLoader(item=input_item)
+ il.add_value('name', ['item', 'loader'])
+ loaded_item = il.load_item()
+ self.assertIsInstance(loaded_item, self.item_class)
+ self.assertEqual(dict(loaded_item), {'name': ['foo', 'item', 'loader']})
+
+ def test_add_value_list_singlevalue(self):
+ """Values added after initialization should be appended"""
+ input_item = self.item_class(name=['foo', 'bar'])
+ il = ItemLoader(item=input_item)
+ il.add_value('name', 'qwerty')
+ loaded_item = il.load_item()
+ self.assertIsInstance(loaded_item, self.item_class)
+ self.assertEqual(dict(loaded_item), {'name': ['foo', 'bar', 'qwerty']})
+
+ def test_add_value_list_list(self):
+ """Values added after initialization should be appended"""
+ input_item = self.item_class(name=['foo', 'bar'])
+ il = ItemLoader(item=input_item)
+ il.add_value('name', ['item', 'loader'])
+ loaded_item = il.load_item()
+ self.assertIsInstance(loaded_item, self.item_class)
+ self.assertEqual(dict(loaded_item), {'name': ['foo', 'bar', 'item', 'loader']})
+
+ def test_get_output_value_singlevalue(self):
+ """Getting output value must not remove value from item"""
+ input_item = self.item_class(name='foo')
+ il = ItemLoader(item=input_item)
+ self.assertEqual(il.get_output_value('name'), ['foo'])
+ loaded_item = il.load_item()
+ self.assertIsInstance(loaded_item, self.item_class)
+ self.assertEqual(loaded_item, dict({'name': ['foo']}))
+
+ def test_get_output_value_list(self):
+ """Getting output value must not remove value from item"""
+ input_item = self.item_class(name=['foo', 'bar'])
+ il = ItemLoader(item=input_item)
+ self.assertEqual(il.get_output_value('name'), ['foo', 'bar'])
+ loaded_item = il.load_item()
+ self.assertIsInstance(loaded_item, self.item_class)
+ self.assertEqual(loaded_item, dict({'name': ['foo', 'bar']}))
+
+ def test_values_single(self):
+ """Values from initial item must be added to loader._values"""
+ input_item = self.item_class(name='foo')
+ il = ItemLoader(item=input_item)
+ self.assertEqual(il._values.get('name'), ['foo'])
+
+ def test_values_list(self):
+ """Values from initial item must be added to loader._values"""
+ input_item = self.item_class(name=['foo', 'bar'])
+ il = ItemLoader(item=input_item)
+ self.assertEqual(il._values.get('name'), ['foo', 'bar'])
+
+
+class InitializationFromDictTest(InitializationTestMixin, unittest.TestCase):
+ item_class = dict
+
+
+class InitializationFromItemTest(InitializationTestMixin, unittest.TestCase):
+ item_class = NameItem
+
+
+class BaseNoInputReprocessingLoader(ItemLoader):
+ title_in = MapCompose(str.upper)
+ title_out = TakeFirst()
+
+
+class NoInputReprocessingDictLoader(BaseNoInputReprocessingLoader):
+ default_item_class = dict
+
+
+class NoInputReprocessingFromDictTest(unittest.TestCase):
+ """
+ Loaders initialized from loaded items must not reprocess fields (dict instances)
+ """
+ def test_avoid_reprocessing_with_initial_values_single(self):
+ il = NoInputReprocessingDictLoader(item=dict(title='foo'))
+ il_loaded = il.load_item()
+ self.assertEqual(il_loaded, dict(title='foo'))
+ self.assertEqual(NoInputReprocessingDictLoader(item=il_loaded).load_item(), dict(title='foo'))
+
+ def test_avoid_reprocessing_with_initial_values_list(self):
+ il = NoInputReprocessingDictLoader(item=dict(title=['foo', 'bar']))
+ il_loaded = il.load_item()
+ self.assertEqual(il_loaded, dict(title='foo'))
+ self.assertEqual(NoInputReprocessingDictLoader(item=il_loaded).load_item(), dict(title='foo'))
+
+ def test_avoid_reprocessing_without_initial_values_single(self):
+ il = NoInputReprocessingDictLoader()
+ il.add_value('title', 'foo')
+ il_loaded = il.load_item()
+ self.assertEqual(il_loaded, dict(title='FOO'))
+ self.assertEqual(NoInputReprocessingDictLoader(item=il_loaded).load_item(), dict(title='FOO'))
+
+ def test_avoid_reprocessing_without_initial_values_list(self):
+ il = NoInputReprocessingDictLoader()
+ il.add_value('title', ['foo', 'bar'])
+ il_loaded = il.load_item()
+ self.assertEqual(il_loaded, dict(title='FOO'))
+ self.assertEqual(NoInputReprocessingDictLoader(item=il_loaded).load_item(), dict(title='FOO'))
+
+
+class NoInputReprocessingItem(Item):
+ title = Field()
+
+
+class NoInputReprocessingItemLoader(BaseNoInputReprocessingLoader):
+ default_item_class = NoInputReprocessingItem
+
+
+class NoInputReprocessingFromItemTest(unittest.TestCase):
+ """
+ Loaders initialized from loaded items must not reprocess fields (BaseItem instances)
+ """
+ def test_avoid_reprocessing_with_initial_values_single(self):
+ il = NoInputReprocessingItemLoader(item=NoInputReprocessingItem(title='foo'))
+ il_loaded = il.load_item()
+ self.assertEqual(il_loaded, {'title': 'foo'})
+ self.assertEqual(NoInputReprocessingItemLoader(item=il_loaded).load_item(), {'title': 'foo'})
+
+ def test_avoid_reprocessing_with_initial_values_list(self):
+ il = NoInputReprocessingItemLoader(item=NoInputReprocessingItem(title=['foo', 'bar']))
+ il_loaded = il.load_item()
+ self.assertEqual(il_loaded, {'title': 'foo'})
+ self.assertEqual(NoInputReprocessingItemLoader(item=il_loaded).load_item(), {'title': 'foo'})
+
+ def test_avoid_reprocessing_without_initial_values_single(self):
+ il = NoInputReprocessingItemLoader()
+ il.add_value('title', 'FOO')
+ il_loaded = il.load_item()
+ self.assertEqual(il_loaded, {'title': 'FOO'})
+ self.assertEqual(NoInputReprocessingItemLoader(item=il_loaded).load_item(), {'title': 'FOO'})
+
+ def test_avoid_reprocessing_without_initial_values_list(self):
+ il = NoInputReprocessingItemLoader()
+ il.add_value('title', ['foo', 'bar'])
+ il_loaded = il.load_item()
+ self.assertEqual(il_loaded, {'title': 'FOO'})
+ self.assertEqual(NoInputReprocessingItemLoader(item=il_loaded).load_item(), {'title': 'FOO'})
+
+
+class TestOutputProcessorDict(unittest.TestCase):
+ def test_output_processor(self):
+
+ class TempDict(dict):
+ def __init__(self, *args, **kwargs):
+ super(TempDict, self).__init__(self, *args, **kwargs)
+ self.setdefault('temp', 0.3)
+
+ class TempLoader(ItemLoader):
+ default_item_class = TempDict
+ default_input_processor = Identity()
+ default_output_processor = Compose(TakeFirst())
+
+ loader = TempLoader()
+ item = loader.load_item()
+ self.assertIsInstance(item, TempDict)
+ self.assertEqual(dict(item), {'temp': 0.3})
+
+
+class TestOutputProcessorItem(unittest.TestCase):
+ def test_output_processor(self):
+
+ class TempItem(Item):
+ temp = Field()
+
+ def __init__(self, *args, **kwargs):
+ super(TempItem, self).__init__(self, *args, **kwargs)
+ self.setdefault('temp', 0.3)
+
+ class TempLoader(ItemLoader):
+ default_item_class = TempItem
+ default_input_processor = Identity()
+ default_output_processor = Compose(TakeFirst())
+
+ loader = TempLoader()
+ item = loader.load_item()
+ self.assertIsInstance(item, TempItem)
+ self.assertEqual(dict(item), {'temp': 0.3})
+
+
class ProcessorsTest(unittest.TestCase):
def test_take_first(self):
@@ -523,7 +702,8 @@ class ProcessorsTest(unittest.TestCase):
self.assertRaises(ValueError, proc, 'hello')
def test_mapcompose(self):
- filter_world = lambda x: None if x == 'world' else x
+ def filter_world(x):
+ return None if x == 'world' else x
proc = MapCompose(filter_world, six.text_type.upper)
self.assertEqual(proc([u'hello', u'world', u'this', u'is', u'scrapy']),
[u'HELLO', u'THIS', u'IS', u'SCRAPY'])
@@ -535,7 +715,6 @@ class ProcessorsTest(unittest.TestCase):
self.assertRaises(ValueError, proc, 'hello')
-
class SelectortemLoaderTest(unittest.TestCase):
response = HtmlResponse(url="", encoding='utf-8', body=b"""
@@ -548,11 +727,11 @@ class SelectortemLoaderTest(unittest.TestCase):
""")
- def test_constructor(self):
+ def test_init_method(self):
l = TestItemLoader()
self.assertEqual(l.selector, None)
- def test_constructor_errors(self):
+ def test_init_method_errors(self):
l = TestItemLoader()
self.assertRaises(RuntimeError, l.add_xpath, 'url', '//a/@href')
self.assertRaises(RuntimeError, l.replace_xpath, 'url', '//a/@href')
@@ -561,7 +740,7 @@ class SelectortemLoaderTest(unittest.TestCase):
self.assertRaises(RuntimeError, l.replace_css, 'name', '#name::text')
self.assertRaises(RuntimeError, l.get_css, '#name::text')
- def test_constructor_with_selector(self):
+ def test_init_method_with_selector(self):
sel = Selector(text=u"