mirror of https://github.com/scrapy/scrapy.git
Merge remote-tracking branch 'whalebot-helmsman/eagerness' into asyncio-eagerness
This commit is contained in:
commit
42489646fe
|
|
@ -1,5 +1,5 @@
|
|||
[bumpversion]
|
||||
current_version = 2.1.0
|
||||
current_version = 2.2.0
|
||||
commit = True
|
||||
tag = True
|
||||
tag_name = {new_version}
|
||||
|
|
|
|||
|
|
@ -15,6 +15,7 @@ htmlcov/
|
|||
.pytest_cache/
|
||||
.coverage.*
|
||||
.cache/
|
||||
.mypy_cache/
|
||||
|
||||
# Windows
|
||||
Thumbs.db
|
||||
|
|
|
|||
|
|
@ -15,14 +15,14 @@ matrix:
|
|||
python: 3.8
|
||||
- env: TOXENV=docs
|
||||
python: 3.7 # Keep in sync with .readthedocs.yml
|
||||
- env: TOXENV=typing
|
||||
python: 3.8
|
||||
|
||||
- env: TOXENV=pypy3
|
||||
- env: TOXENV=pinned
|
||||
python: 3.5.1
|
||||
dist: trusty
|
||||
python: 3.5.2
|
||||
- env: TOXENV=asyncio
|
||||
python: 3.5.1 # We use additional code to support 3.5.3 and earlier
|
||||
dist: trusty
|
||||
python: 3.5.2 # We use additional code to support 3.5.3 and earlier
|
||||
- env: TOXENV=py
|
||||
python: 3.5
|
||||
- env: TOXENV=asyncio
|
||||
|
|
|
|||
|
|
@ -40,7 +40,7 @@ including a list of features.
|
|||
Requirements
|
||||
============
|
||||
|
||||
* Python 3.5.1+
|
||||
* Python 3.5.2+
|
||||
* Works on Linux, Windows, macOS, BSD
|
||||
|
||||
Install
|
||||
|
|
|
|||
|
|
@ -281,6 +281,7 @@ coverage_ignore_pyobjects = [
|
|||
# -------------------------------------
|
||||
|
||||
intersphinx_mapping = {
|
||||
'attrs': ('https://www.attrs.org/en/stable/', None),
|
||||
'coverage': ('https://coverage.readthedocs.io/en/stable', None),
|
||||
'cssselect': ('https://cssselect.readthedocs.io/en/latest', None),
|
||||
'pytest': ('https://docs.pytest.org/en/latest', None),
|
||||
|
|
|
|||
|
|
@ -155,6 +155,9 @@ Finally, try to keep aesthetic changes (:pep:`8` compliance, unused imports
|
|||
removal, etc) in separate commits from functional changes. This will make pull
|
||||
requests easier to review and more likely to get merged.
|
||||
|
||||
|
||||
.. _coding-style:
|
||||
|
||||
Coding style
|
||||
============
|
||||
|
||||
|
|
@ -163,7 +166,7 @@ Scrapy:
|
|||
|
||||
* Unless otherwise specified, follow :pep:`8`.
|
||||
|
||||
* It's OK to use lines longer than 80 chars if it improves the code
|
||||
* It's OK to use lines longer than 79 chars if it improves the code
|
||||
readability.
|
||||
|
||||
* Don't put your name in the code you contribute; git provides enough
|
||||
|
|
|
|||
10
docs/faq.rst
10
docs/faq.rst
|
|
@ -69,7 +69,7 @@ Here's an example spider using BeautifulSoup API, with ``lxml`` as the HTML pars
|
|||
What Python versions does Scrapy support?
|
||||
-----------------------------------------
|
||||
|
||||
Scrapy is supported under Python 3.5.1+
|
||||
Scrapy is supported under Python 3.5.2+
|
||||
under CPython (default Python implementation) and PyPy (starting with PyPy 5.9).
|
||||
Python 3 support was added in Scrapy 1.1.
|
||||
PyPy support was added in Scrapy 1.4, PyPy3 support was added in Scrapy 1.5.
|
||||
|
|
@ -342,15 +342,15 @@ method for this purpose. For example::
|
|||
|
||||
from copy import deepcopy
|
||||
|
||||
from scrapy.item import Item
|
||||
|
||||
from itemadapter import is_item, ItemAdapter
|
||||
|
||||
class MultiplyItemsMiddleware:
|
||||
|
||||
def process_spider_output(self, response, result, spider):
|
||||
for item in result:
|
||||
if isinstance(item, (Item, dict)):
|
||||
for _ in range(item['multiply_by']):
|
||||
if is_item(item):
|
||||
adapter = ItemAdapter(item)
|
||||
for _ in range(adapter['multiply_by']):
|
||||
yield deepcopy(item)
|
||||
|
||||
Does Scrapy support IPv6 addresses?
|
||||
|
|
|
|||
|
|
@ -7,7 +7,7 @@ Installation guide
|
|||
Installing Scrapy
|
||||
=================
|
||||
|
||||
Scrapy runs on Python 3.5.1 or above under CPython (default Python
|
||||
Scrapy runs on Python 3.5.2 or above under CPython (default Python
|
||||
implementation) and PyPy (starting with PyPy 5.9).
|
||||
|
||||
If you're using `Anaconda`_ or `Miniconda`_, you can install the package from
|
||||
|
|
|
|||
195
docs/news.rst
195
docs/news.rst
|
|
@ -3,6 +3,201 @@
|
|||
Release notes
|
||||
=============
|
||||
|
||||
.. _release-2.2.0:
|
||||
|
||||
Scrapy 2.2.0 (2020-06-24)
|
||||
-------------------------
|
||||
|
||||
Highlights:
|
||||
|
||||
* Python 3.5.2+ is required now
|
||||
* :ref:`dataclass objects <dataclass-items>` and
|
||||
:ref:`attrs objects <attrs-items>` are now valid :ref:`item types
|
||||
<item-types>`
|
||||
* New :meth:`TextResponse.json <scrapy.http.TextResponse.json>` method
|
||||
* New :signal:`bytes_received` signal that allows canceling response download
|
||||
* :class:`~scrapy.downloadermiddlewares.cookies.CookiesMiddleware` fixes
|
||||
|
||||
Backward-incompatible changes
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
* Support for Python 3.5.0 and 3.5.1 has been dropped; Scrapy now refuses to
|
||||
run with a Python version lower than 3.5.2, which introduced
|
||||
:class:`typing.Type` (:issue:`4615`)
|
||||
|
||||
|
||||
Deprecations
|
||||
~~~~~~~~~~~~
|
||||
|
||||
* :meth:`TextResponse.body_as_unicode
|
||||
<scrapy.http.TextResponse.body_as_unicode>` is now deprecated, use
|
||||
:attr:`TextResponse.text <scrapy.http.TextResponse.text>` instead
|
||||
(:issue:`4546`, :issue:`4555`, :issue:`4579`)
|
||||
|
||||
* :class:`scrapy.item.BaseItem` is now deprecated, use
|
||||
:class:`scrapy.item.Item` instead (:issue:`4534`)
|
||||
|
||||
|
||||
New features
|
||||
~~~~~~~~~~~~
|
||||
|
||||
* :ref:`dataclass objects <dataclass-items>` and
|
||||
:ref:`attrs objects <attrs-items>` are now valid :ref:`item types
|
||||
<item-types>`, and a new itemadapter_ library makes it easy to
|
||||
write code that :ref:`supports any item type <supporting-item-types>`
|
||||
(:issue:`2749`, :issue:`2807`, :issue:`3761`, :issue:`3881`, :issue:`4642`)
|
||||
|
||||
* A new :meth:`TextResponse.json <scrapy.http.TextResponse.json>` method
|
||||
allows to deserialize JSON responses (:issue:`2444`, :issue:`4460`,
|
||||
:issue:`4574`)
|
||||
|
||||
* A new :signal:`bytes_received` signal allows monitoring response download
|
||||
progress and :ref:`stopping downloads <topics-stop-response-download>`
|
||||
(:issue:`4205`, :issue:`4559`)
|
||||
|
||||
* The dictionaries in the result list of a :ref:`media pipeline
|
||||
<topics-media-pipeline>` now include a new key, ``status``, which indicates
|
||||
if the file was downloaded or, if the file was not downloaded, why it was
|
||||
not downloaded; see :meth:`FilesPipeline.get_media_requests
|
||||
<scrapy.pipelines.files.FilesPipeline.get_media_requests>` for more
|
||||
information (:issue:`2893`, :issue:`4486`)
|
||||
|
||||
* When using :ref:`Google Cloud Storage <media-pipeline-gcs>` for
|
||||
a :ref:`media pipeline <topics-media-pipeline>`, a warning is now logged if
|
||||
the configured credentials do not grant the required permissions
|
||||
(:issue:`4346`, :issue:`4508`)
|
||||
|
||||
* :ref:`Link extractors <topics-link-extractors>` are now serializable,
|
||||
as long as you do not use :ref:`lambdas <lambda>` for parameters; for
|
||||
example, you can now pass link extractors in :attr:`Request.cb_kwargs
|
||||
<scrapy.http.Request.cb_kwargs>` or
|
||||
:attr:`Request.meta <scrapy.http.Request.meta>` when :ref:`persisting
|
||||
scheduled requests <topics-jobs>` (:issue:`4554`)
|
||||
|
||||
* Upgraded the :ref:`pickle protocol <pickle-protocols>` that Scrapy uses
|
||||
from protocol 2 to protocol 4, improving serialization capabilities and
|
||||
performance (:issue:`4135`, :issue:`4541`)
|
||||
|
||||
* :func:`scrapy.utils.misc.create_instance` now raises a :exc:`TypeError`
|
||||
exception if the resulting instance is ``None`` (:issue:`4528`,
|
||||
:issue:`4532`)
|
||||
|
||||
.. _itemadapter: https://github.com/scrapy/itemadapter
|
||||
|
||||
|
||||
Bug fixes
|
||||
~~~~~~~~~
|
||||
|
||||
* :class:`~scrapy.downloadermiddlewares.cookies.CookiesMiddleware` no longer
|
||||
discards cookies defined in :attr:`Request.headers
|
||||
<scrapy.http.Request.headers>` (:issue:`1992`, :issue:`2400`)
|
||||
|
||||
* :class:`~scrapy.downloadermiddlewares.cookies.CookiesMiddleware` no longer
|
||||
re-encodes cookies defined as :class:`bytes` in the ``cookies`` parameter
|
||||
of the ``__init__`` method of :class:`~scrapy.http.Request`
|
||||
(:issue:`2400`, :issue:`3575`)
|
||||
|
||||
* When :setting:`FEEDS` defines multiple URIs, :setting:`FEED_STORE_EMPTY` is
|
||||
``False`` and the crawl yields no items, Scrapy no longer stops feed
|
||||
exports after the first URI (:issue:`4621`, :issue:`4626`)
|
||||
|
||||
* :class:`~scrapy.spiders.Spider` callbacks defined using :doc:`coroutine
|
||||
syntax <topics/coroutines>` no longer need to return an iterable, and may
|
||||
instead return a :class:`~scrapy.http.Request` object, an
|
||||
:ref:`item <topics-items>`, or ``None`` (:issue:`4609`)
|
||||
|
||||
* The :command:`startproject` command now ensures that the generated project
|
||||
folders and files have the right permissions (:issue:`4604`)
|
||||
|
||||
* Fix a :exc:`KeyError` exception being sometimes raised from
|
||||
:class:`scrapy.utils.datatypes.LocalWeakReferencedCache` (:issue:`4597`,
|
||||
:issue:`4599`)
|
||||
|
||||
* When :setting:`FEEDS` defines multiple URIs, log messages about items being
|
||||
stored now contain information from the corresponding feed, instead of
|
||||
always containing information about only one of the feeds (:issue:`4619`,
|
||||
:issue:`4629`)
|
||||
|
||||
|
||||
Documentation
|
||||
~~~~~~~~~~~~~
|
||||
|
||||
* Added a new section about :ref:`accessing cb_kwargs from errbacks
|
||||
<errback-cb_kwargs>` (:issue:`4598`, :issue:`4634`)
|
||||
|
||||
* Covered chompjs_ in :ref:`topics-parsing-javascript` (:issue:`4556`,
|
||||
:issue:`4562`)
|
||||
|
||||
* Removed from :doc:`topics/coroutines` the warning about the API being
|
||||
experimental (:issue:`4511`, :issue:`4513`)
|
||||
|
||||
* Removed references to unsupported versions of :doc:`Twisted
|
||||
<twisted:index>` (:issue:`4533`)
|
||||
|
||||
* Updated the description of the :ref:`screenshot pipeline example
|
||||
<ScreenshotPipeline>`, which now uses :doc:`coroutine syntax
|
||||
<topics/coroutines>` instead of returning a
|
||||
:class:`~twisted.internet.defer.Deferred` (:issue:`4514`, :issue:`4593`)
|
||||
|
||||
* Removed a misleading import line from the
|
||||
:func:`scrapy.utils.log.configure_logging` code example (:issue:`4510`,
|
||||
:issue:`4587`)
|
||||
|
||||
* The display-on-hover behavior of internal documentation references now also
|
||||
covers links to :ref:`commands <topics-commands>`, :attr:`Request.meta
|
||||
<scrapy.http.Request.meta>` keys, :ref:`settings <topics-settings>` and
|
||||
:ref:`signals <topics-signals>` (:issue:`4495`, :issue:`4563`)
|
||||
|
||||
* It is again possible to download the documentation for offline reading
|
||||
(:issue:`4578`, :issue:`4585`)
|
||||
|
||||
* Removed backslashes preceding ``*args`` and ``**kwargs`` in some function
|
||||
and method signatures (:issue:`4592`, :issue:`4596`)
|
||||
|
||||
.. _chompjs: https://github.com/Nykakin/chompjs
|
||||
|
||||
|
||||
Quality assurance
|
||||
~~~~~~~~~~~~~~~~~
|
||||
|
||||
* Adjusted the code base further to our :ref:`style guidelines
|
||||
<coding-style>` (:issue:`4237`, :issue:`4525`, :issue:`4538`,
|
||||
:issue:`4539`, :issue:`4540`, :issue:`4542`, :issue:`4543`, :issue:`4544`,
|
||||
:issue:`4545`, :issue:`4557`, :issue:`4558`, :issue:`4566`, :issue:`4568`,
|
||||
:issue:`4572`)
|
||||
|
||||
* Removed remnants of Python 2 support (:issue:`4550`, :issue:`4553`,
|
||||
:issue:`4568`)
|
||||
|
||||
* Improved code sharing between the :command:`crawl` and :command:`runspider`
|
||||
commands (:issue:`4548`, :issue:`4552`)
|
||||
|
||||
* Replaced ``chain(*iterable)`` with ``chain.from_iterable(iterable)``
|
||||
(:issue:`4635`)
|
||||
|
||||
* You may now run the :mod:`asyncio` tests with Tox on any Python version
|
||||
(:issue:`4521`)
|
||||
|
||||
* Updated test requirements to reflect an incompatibility with pytest 5.4 and
|
||||
5.4.1 (:issue:`4588`)
|
||||
|
||||
* Improved :class:`~scrapy.spiderloader.SpiderLoader` test coverage for
|
||||
scenarios involving duplicate spider names (:issue:`4549`, :issue:`4560`)
|
||||
|
||||
* Configured Travis CI to also run the tests with Python 3.5.2
|
||||
(:issue:`4518`, :issue:`4615`)
|
||||
|
||||
* Added a `Pylint <https://www.pylint.org/>`_ job to Travis CI
|
||||
(:issue:`3727`)
|
||||
|
||||
* Added a `Mypy <http://mypy-lang.org/>`_ job to Travis CI (:issue:`4637`)
|
||||
|
||||
* Made use of set literals in tests (:issue:`4573`)
|
||||
|
||||
* Cleaned up the Travis CI configuration (:issue:`4517`, :issue:`4519`,
|
||||
:issue:`4522`, :issue:`4537`)
|
||||
|
||||
|
||||
.. _release-2.1.0:
|
||||
|
||||
Scrapy 2.1.0 (2020-04-24)
|
||||
|
|
|
|||
|
|
@ -104,7 +104,7 @@ Spiders
|
|||
-------
|
||||
|
||||
Spiders are custom classes written by Scrapy users to parse responses and
|
||||
extract items (aka scraped items) from them or additional requests to
|
||||
extract :ref:`items <topics-items>` from them or additional requests to
|
||||
follow. For more information see :ref:`topics-spiders`.
|
||||
|
||||
.. _component-pipelines:
|
||||
|
|
|
|||
|
|
@ -123,21 +123,28 @@ There are several use cases for coroutines in Scrapy. Code that would
|
|||
return Deferreds when written for previous Scrapy versions, such as downloader
|
||||
middlewares and signal handlers, can be rewritten to be shorter and cleaner::
|
||||
|
||||
from itemadapter import ItemAdapter
|
||||
|
||||
class DbPipeline:
|
||||
def _update_item(self, data, item):
|
||||
item['field'] = data
|
||||
adapter = ItemAdapter(item)
|
||||
adapter['field'] = data
|
||||
return item
|
||||
|
||||
def process_item(self, item, spider):
|
||||
dfd = db.get_some_data(item['id'])
|
||||
adapter = ItemAdapter(item)
|
||||
dfd = db.get_some_data(adapter['id'])
|
||||
dfd.addCallback(self._update_item, item)
|
||||
return dfd
|
||||
|
||||
becomes::
|
||||
|
||||
from itemadapter import ItemAdapter
|
||||
|
||||
class DbPipeline:
|
||||
async def process_item(self, item, spider):
|
||||
item['field'] = await db.get_some_data(item['id'])
|
||||
adapter = ItemAdapter(item)
|
||||
adapter['field'] = await db.get_some_data(adapter['id'])
|
||||
return item
|
||||
|
||||
Coroutines may be used to call asynchronous code. This includes other
|
||||
|
|
|
|||
|
|
@ -40,6 +40,7 @@ Here you can see an :doc:`Item Pipeline <item-pipeline>` which uses multiple
|
|||
Item Exporters to group scraped items to different files according to the
|
||||
value of one of their fields::
|
||||
|
||||
from itemadapter import ItemAdapter
|
||||
from scrapy.exporters import XmlItemExporter
|
||||
|
||||
class PerYearXmlExportPipeline:
|
||||
|
|
@ -53,7 +54,8 @@ value of one of their fields::
|
|||
exporter.finish_exporting()
|
||||
|
||||
def _exporter_for_item(self, item):
|
||||
year = item['year']
|
||||
adapter = ItemAdapter(item)
|
||||
year = adapter['year']
|
||||
if year not in self.year_to_exporter:
|
||||
f = open('{}.xml'.format(year), 'wb')
|
||||
exporter = XmlItemExporter(f)
|
||||
|
|
@ -167,9 +169,10 @@ BaseItemExporter
|
|||
value unchanged except for ``unicode`` values which are encoded to
|
||||
``str`` using the encoding declared in the :attr:`encoding` attribute.
|
||||
|
||||
:param field: the field being serialized. If a raw dict is being
|
||||
exported (not :class:`~.Item`) *field* value is an empty dict.
|
||||
:type field: :class:`~scrapy.item.Field` object or an empty dict
|
||||
:param field: the field being serialized. If the source :ref:`item object
|
||||
<item-types>` does not define field metadata, *field* is an empty
|
||||
:class:`dict`.
|
||||
:type field: :class:`~scrapy.item.Field` object or a :class:`dict` instance
|
||||
|
||||
:param name: the name of the field being serialized
|
||||
:type name: str
|
||||
|
|
@ -192,14 +195,17 @@ BaseItemExporter
|
|||
|
||||
.. attribute:: fields_to_export
|
||||
|
||||
A list with the name of the fields that will be exported, or None if you
|
||||
want to export all fields. Defaults to None.
|
||||
A list with the name of the fields that will be exported, or ``None`` if
|
||||
you want to export all fields. Defaults to ``None``.
|
||||
|
||||
Some exporters (like :class:`CsvItemExporter`) respect the order of the
|
||||
fields defined in this attribute.
|
||||
|
||||
Some exporters may require fields_to_export list in order to export the
|
||||
data properly when spiders return dicts (not :class:`~Item` instances).
|
||||
When using :ref:`item objects <item-types>` that do not expose all their
|
||||
possible fields, exporters that do not support exporting a different
|
||||
subset of fields per item will only export the fields found in the first
|
||||
item exported. Use ``fields_to_export`` to define all the fields to be
|
||||
exported.
|
||||
|
||||
.. attribute:: export_empty_fields
|
||||
|
||||
|
|
@ -238,7 +244,7 @@ XmlItemExporter
|
|||
|
||||
.. class:: XmlItemExporter(file, item_element='item', root_element='items', **kwargs)
|
||||
|
||||
Exports Items in XML format to the specified file object.
|
||||
Exports items in XML format to the specified file object.
|
||||
|
||||
:param file: the file-like object to use for exporting the data. Its ``write`` method should
|
||||
accept ``bytes`` (a disk file opened in binary mode, a ``io.BytesIO`` object, etc)
|
||||
|
|
@ -292,7 +298,7 @@ CsvItemExporter
|
|||
|
||||
.. class:: CsvItemExporter(file, include_headers_line=True, join_multivalued=',', **kwargs)
|
||||
|
||||
Exports Items in CSV format to the given file-like object. If the
|
||||
Exports items in CSV format to the given file-like object. If the
|
||||
:attr:`fields_to_export` attribute is set, it will be used to define the
|
||||
CSV columns and their order. The :attr:`export_empty_fields` attribute has
|
||||
no effect on this exporter.
|
||||
|
|
@ -325,7 +331,7 @@ PickleItemExporter
|
|||
|
||||
.. class:: PickleItemExporter(file, protocol=0, **kwargs)
|
||||
|
||||
Exports Items in pickle format to the given file-like object.
|
||||
Exports items in pickle format to the given file-like object.
|
||||
|
||||
:param file: the file-like object to use for exporting the data. Its ``write`` method should
|
||||
accept ``bytes`` (a disk file opened in binary mode, a ``io.BytesIO`` object, etc)
|
||||
|
|
@ -345,7 +351,7 @@ PprintItemExporter
|
|||
|
||||
.. class:: PprintItemExporter(file, **kwargs)
|
||||
|
||||
Exports Items in pretty print format to the specified file object.
|
||||
Exports items in pretty print format to the specified file object.
|
||||
|
||||
:param file: the file-like object to use for exporting the data. Its ``write`` method should
|
||||
accept ``bytes`` (a disk file opened in binary mode, a ``io.BytesIO`` object, etc)
|
||||
|
|
@ -365,7 +371,7 @@ JsonItemExporter
|
|||
|
||||
.. class:: JsonItemExporter(file, **kwargs)
|
||||
|
||||
Exports Items in JSON format to the specified file-like object, writing all
|
||||
Exports items in JSON format to the specified file-like object, writing all
|
||||
objects as a list of objects. The additional ``__init__`` method arguments are
|
||||
passed to the :class:`BaseItemExporter` ``__init__`` method, and the leftover
|
||||
arguments to the :class:`~json.JSONEncoder` ``__init__`` method, so you can use any
|
||||
|
|
@ -394,7 +400,7 @@ JsonLinesItemExporter
|
|||
|
||||
.. class:: JsonLinesItemExporter(file, **kwargs)
|
||||
|
||||
Exports Items in JSON format to the specified file-like object, writing one
|
||||
Exports items in JSON format to the specified file-like object, writing one
|
||||
JSON-encoded item per line. The additional ``__init__`` method arguments are passed
|
||||
to the :class:`BaseItemExporter` ``__init__`` method, and the leftover arguments to
|
||||
the :class:`~json.JSONEncoder` ``__init__`` method, so you can use any
|
||||
|
|
|
|||
|
|
@ -298,8 +298,8 @@ Example: ``FEED_EXPORT_FIELDS = ["foo", "bar", "baz"]``.
|
|||
|
||||
Use FEED_EXPORT_FIELDS option to define fields to export and their order.
|
||||
|
||||
When FEED_EXPORT_FIELDS is empty or None (default), Scrapy uses fields
|
||||
defined in dicts or :class:`~.Item` subclasses a spider is yielding.
|
||||
When FEED_EXPORT_FIELDS is empty or None (default), Scrapy uses the fields
|
||||
defined in :ref:`item objects <topics-items>` yielded by your spider.
|
||||
|
||||
If an exporter requires a fixed set of fields (this is the case for
|
||||
:ref:`CSV <topics-feed-format-csv>` export format) and FEED_EXPORT_FIELDS
|
||||
|
|
|
|||
|
|
@ -27,15 +27,19 @@ Each item pipeline component is a Python class that must implement the following
|
|||
|
||||
.. method:: process_item(self, item, spider)
|
||||
|
||||
This method is called for every item pipeline component. :meth:`process_item`
|
||||
must either: return a dict with data, return an :class:`~scrapy.item.Item`
|
||||
(or any descendant class) object, return a
|
||||
:class:`~twisted.internet.defer.Deferred` or raise
|
||||
:exc:`~scrapy.exceptions.DropItem` exception. Dropped items are no longer
|
||||
processed by further pipeline components.
|
||||
This method is called for every item pipeline component.
|
||||
|
||||
:param item: the item scraped
|
||||
:type item: :class:`~scrapy.item.Item` object or a dict
|
||||
`item` is an :ref:`item object <item-types>`, see
|
||||
:ref:`supporting-item-types`.
|
||||
|
||||
:meth:`process_item` must either: return an :ref:`item object <item-types>`,
|
||||
return a :class:`~twisted.internet.defer.Deferred` or raise a
|
||||
:exc:`~scrapy.exceptions.DropItem` exception.
|
||||
|
||||
Dropped items are no longer processed by further pipeline components.
|
||||
|
||||
:param item: the scraped item
|
||||
:type item: :ref:`item object <item-types>`
|
||||
|
||||
:param spider: the spider which scraped the item
|
||||
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||
|
|
@ -79,16 +83,17 @@ Let's take a look at the following hypothetical pipeline that adjusts the
|
|||
(``price_excludes_vat`` attribute), and drops those items which don't
|
||||
contain a price::
|
||||
|
||||
from itemadapter import ItemAdapter
|
||||
from scrapy.exceptions import DropItem
|
||||
|
||||
class PricePipeline:
|
||||
|
||||
vat_factor = 1.15
|
||||
|
||||
def process_item(self, item, spider):
|
||||
if item.get('price'):
|
||||
if item.get('price_excludes_vat'):
|
||||
item['price'] = item['price'] * self.vat_factor
|
||||
adapter = ItemAdapter(item)
|
||||
if adapter.get('price'):
|
||||
if adapter.get('price_excludes_vat'):
|
||||
adapter['price'] = adapter['price'] * self.vat_factor
|
||||
return item
|
||||
else:
|
||||
raise DropItem("Missing price in %s" % item)
|
||||
|
|
@ -103,6 +108,8 @@ format::
|
|||
|
||||
import json
|
||||
|
||||
from itemadapter import ItemAdapter
|
||||
|
||||
class JsonWriterPipeline:
|
||||
|
||||
def open_spider(self, spider):
|
||||
|
|
@ -112,7 +119,7 @@ format::
|
|||
self.file.close()
|
||||
|
||||
def process_item(self, item, spider):
|
||||
line = json.dumps(dict(item)) + "\n"
|
||||
line = json.dumps(ItemAdapter(item).asdict()) + "\n"
|
||||
self.file.write(line)
|
||||
return item
|
||||
|
||||
|
|
@ -131,6 +138,7 @@ The main point of this example is to show how to use :meth:`from_crawler`
|
|||
method and how to clean up the resources properly.::
|
||||
|
||||
import pymongo
|
||||
from itemadapter import ItemAdapter
|
||||
|
||||
class MongoPipeline:
|
||||
|
||||
|
|
@ -155,7 +163,7 @@ method and how to clean up the resources properly.::
|
|||
self.client.close()
|
||||
|
||||
def process_item(self, item, spider):
|
||||
self.db[self.collection_name].insert_one(dict(item))
|
||||
self.db[self.collection_name].insert_one(ItemAdapter(item).asdict())
|
||||
return item
|
||||
|
||||
.. _MongoDB: https://www.mongodb.com/
|
||||
|
|
@ -177,10 +185,11 @@ item.
|
|||
|
||||
::
|
||||
|
||||
import scrapy
|
||||
import hashlib
|
||||
from urllib.parse import quote
|
||||
|
||||
import scrapy
|
||||
from itemadapter import ItemAdapter
|
||||
|
||||
class ScreenshotPipeline:
|
||||
"""Pipeline that uses Splash to render screenshot of
|
||||
|
|
@ -189,7 +198,8 @@ item.
|
|||
SPLASH_URL = "http://localhost:8050/render.png?url={}"
|
||||
|
||||
async def process_item(self, item, spider):
|
||||
encoded_item_url = quote(item["url"])
|
||||
adapter = ItemAdapter(item)
|
||||
encoded_item_url = quote(adapter["url"])
|
||||
screenshot_url = self.SPLASH_URL.format(encoded_item_url)
|
||||
request = scrapy.Request(screenshot_url)
|
||||
response = await spider.crawler.engine.download(request, spider)
|
||||
|
|
@ -199,14 +209,14 @@ item.
|
|||
return item
|
||||
|
||||
# Save screenshot to file, filename will be hash of url.
|
||||
url = item["url"]
|
||||
url = adapter["url"]
|
||||
url_hash = hashlib.md5(url.encode("utf8")).hexdigest()
|
||||
filename = "{}.png".format(url_hash)
|
||||
with open(filename, "wb") as f:
|
||||
f.write(response.body)
|
||||
|
||||
# Store filename in item.
|
||||
item["screenshot_filename"] = filename
|
||||
adapter["screenshot_filename"] = filename
|
||||
return item
|
||||
|
||||
.. _Splash: https://splash.readthedocs.io/en/stable/
|
||||
|
|
@ -219,6 +229,7 @@ already processed. Let's say that our items have a unique id, but our spider
|
|||
returns multiples items with the same id::
|
||||
|
||||
|
||||
from itemadapter import ItemAdapter
|
||||
from scrapy.exceptions import DropItem
|
||||
|
||||
class DuplicatesPipeline:
|
||||
|
|
@ -227,10 +238,11 @@ returns multiples items with the same id::
|
|||
self.ids_seen = set()
|
||||
|
||||
def process_item(self, item, spider):
|
||||
if item['id'] in self.ids_seen:
|
||||
raise DropItem("Duplicate item found: %s" % item)
|
||||
adapter = ItemAdapter(item)
|
||||
if adapter['id'] in self.ids_seen:
|
||||
raise DropItem("Duplicate item found: %r" % item)
|
||||
else:
|
||||
self.ids_seen.add(item['id'])
|
||||
self.ids_seen.add(adapter['id'])
|
||||
return item
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -8,29 +8,155 @@ Items
|
|||
:synopsis: Item and Field classes
|
||||
|
||||
The main goal in scraping is to extract structured data from unstructured
|
||||
sources, typically, web pages. Scrapy spiders can return the extracted data
|
||||
as Python dicts. While convenient and familiar, Python dicts lack structure:
|
||||
it is easy to make a typo in a field name or return inconsistent data,
|
||||
especially in a larger project with many spiders.
|
||||
sources, typically, web pages. :ref:`Spiders <topics-spiders>` may return the
|
||||
extracted data as `items`, Python objects that define key-value pairs.
|
||||
|
||||
To define common output data format Scrapy provides the :class:`Item` class.
|
||||
:class:`Item` objects are simple containers used to collect the scraped data.
|
||||
They provide an API similar to :class:`dict` API with a convenient syntax
|
||||
for declaring their available fields.
|
||||
Scrapy supports :ref:`multiple types of items <item-types>`. When you create an
|
||||
item, you may use whichever type of item you want. When you write code that
|
||||
receives an item, your code should :ref:`work for any item type
|
||||
<supporting-item-types>`.
|
||||
|
||||
Various Scrapy components use extra information provided by Items:
|
||||
exporters look at declared fields to figure out columns to export,
|
||||
serialization can be customized using Item fields metadata, :mod:`trackref`
|
||||
tracks Item instances to help find memory leaks
|
||||
(see :ref:`topics-leaks-trackrefs`), etc.
|
||||
.. _item-types:
|
||||
|
||||
Item Types
|
||||
==========
|
||||
|
||||
Scrapy supports the following types of items, via the `itemadapter`_ library:
|
||||
:ref:`dictionaries <dict-items>`, :ref:`Item objects <item-objects>`,
|
||||
:ref:`dataclass objects <dataclass-items>`, and :ref:`attrs objects <attrs-items>`.
|
||||
|
||||
.. _itemadapter: https://github.com/scrapy/itemadapter
|
||||
|
||||
.. _dict-items:
|
||||
|
||||
Dictionaries
|
||||
------------
|
||||
|
||||
As an item type, :class:`dict` is convenient and familiar.
|
||||
|
||||
.. _item-objects:
|
||||
|
||||
Item objects
|
||||
------------
|
||||
|
||||
:class:`Item` provides a :class:`dict`-like API plus additional features that
|
||||
make it the most feature-complete item type:
|
||||
|
||||
.. class:: Item([arg])
|
||||
|
||||
:class:`Item` objects replicate the standard :class:`dict` API, including
|
||||
its ``__init__`` method.
|
||||
|
||||
:class:`Item` allows defining field names, so that:
|
||||
|
||||
- :class:`KeyError` is raised when using undefined field names (i.e.
|
||||
prevents typos going unnoticed)
|
||||
|
||||
- :ref:`Item exporters <topics-exporters>` can export all fields by
|
||||
default even if the first scraped object does not have values for all
|
||||
of them
|
||||
|
||||
:class:`Item` also allows defining field metadata, which can be used to
|
||||
:ref:`customize serialization <topics-exporters-field-serialization>`.
|
||||
|
||||
:mod:`trackref` tracks :class:`Item` objects to help find memory leaks
|
||||
(see :ref:`topics-leaks-trackrefs`).
|
||||
|
||||
:class:`Item` objects also provide the following additional API members:
|
||||
|
||||
.. automethod:: copy
|
||||
|
||||
.. automethod:: deepcopy
|
||||
|
||||
.. attribute:: fields
|
||||
|
||||
A dictionary containing *all declared fields* for this Item, not only
|
||||
those populated. The keys are the field names and the values are the
|
||||
:class:`Field` objects used in the :ref:`Item declaration
|
||||
<topics-items-declaring>`.
|
||||
|
||||
Example::
|
||||
|
||||
from scrapy.item import Item, Field
|
||||
|
||||
class CustomItem(Item):
|
||||
one_field = Field()
|
||||
another_field = Field()
|
||||
|
||||
.. _dataclass-items:
|
||||
|
||||
Dataclass objects
|
||||
-----------------
|
||||
|
||||
.. versionadded:: 2.2
|
||||
|
||||
:func:`~dataclasses.dataclass` allows defining item classes with field names,
|
||||
so that :ref:`item exporters <topics-exporters>` can export all fields by
|
||||
default even if the first scraped object does not have values for all of them.
|
||||
|
||||
Additionally, ``dataclass`` items also allow to:
|
||||
|
||||
* define the type and default value of each defined field.
|
||||
|
||||
* define custom field metadata through :func:`dataclasses.field`, which can be used to
|
||||
:ref:`customize serialization <topics-exporters-field-serialization>`.
|
||||
|
||||
They work natively in Python 3.7 or later, or using the `dataclasses
|
||||
backport`_ in Python 3.6.
|
||||
|
||||
.. _dataclasses backport: https://pypi.org/project/dataclasses/
|
||||
|
||||
Example::
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
@dataclass
|
||||
class CustomItem:
|
||||
one_field: str
|
||||
another_field: int
|
||||
|
||||
.. note:: Field types are not enforced at run time.
|
||||
|
||||
.. _attrs-items:
|
||||
|
||||
attr.s objects
|
||||
--------------
|
||||
|
||||
.. versionadded:: 2.2
|
||||
|
||||
:func:`attr.s` allows defining item classes with field names,
|
||||
so that :ref:`item exporters <topics-exporters>` can export all fields by
|
||||
default even if the first scraped object does not have values for all of them.
|
||||
|
||||
Additionally, ``attr.s`` items also allow to:
|
||||
|
||||
* define the type and default value of each defined field.
|
||||
|
||||
* define custom field :ref:`metadata <attrs:metadata>`, which can be used to
|
||||
:ref:`customize serialization <topics-exporters-field-serialization>`.
|
||||
|
||||
In order to use this type, the :doc:`attrs package <attrs:index>` needs to be installed.
|
||||
|
||||
Example::
|
||||
|
||||
import attr
|
||||
|
||||
@attr.s
|
||||
class CustomItem:
|
||||
one_field = attr.ib()
|
||||
another_field = attr.ib()
|
||||
|
||||
|
||||
Working with Item objects
|
||||
=========================
|
||||
|
||||
.. _topics-items-declaring:
|
||||
|
||||
Declaring Items
|
||||
===============
|
||||
Declaring Item subclasses
|
||||
-------------------------
|
||||
|
||||
Items are declared using a simple class definition syntax and :class:`Field`
|
||||
objects. Here is an example::
|
||||
Item subclasses are declared using a simple class definition syntax and
|
||||
:class:`Field` objects. Here is an example::
|
||||
|
||||
import scrapy
|
||||
|
||||
|
|
@ -48,10 +174,11 @@ objects. Here is an example::
|
|||
.. _Django: https://www.djangoproject.com/
|
||||
.. _Django Models: https://docs.djangoproject.com/en/dev/topics/db/models/
|
||||
|
||||
|
||||
.. _topics-items-fields:
|
||||
|
||||
Item Fields
|
||||
===========
|
||||
Declaring fields
|
||||
----------------
|
||||
|
||||
:class:`Field` objects are used to specify metadata for each field. For
|
||||
example, the serializer function for the ``last_updated`` field illustrated in
|
||||
|
|
@ -72,15 +199,31 @@ It's important to note that the :class:`Field` objects used to declare the item
|
|||
do not stay assigned as class attributes. Instead, they can be accessed through
|
||||
the :attr:`Item.fields` attribute.
|
||||
|
||||
Working with Items
|
||||
==================
|
||||
.. class:: Field([arg])
|
||||
|
||||
The :class:`Field` class is just an alias to the built-in :class:`dict` class and
|
||||
doesn't provide any extra functionality or attributes. In other words,
|
||||
:class:`Field` objects are plain-old Python dicts. A separate class is used
|
||||
to support the :ref:`item declaration syntax <topics-items-declaring>`
|
||||
based on class attributes.
|
||||
|
||||
.. note:: Field metadata can also be declared for ``dataclass`` and ``attrs``
|
||||
items. Please refer to the documentation for `dataclasses.field`_ and
|
||||
`attr.ib`_ for additional information.
|
||||
|
||||
.. _dataclasses.field: https://docs.python.org/3/library/dataclasses.html#dataclasses.field
|
||||
.. _attr.ib: https://www.attrs.org/en/stable/api.html#attr.ib
|
||||
|
||||
|
||||
Working with Item objects
|
||||
-------------------------
|
||||
|
||||
Here are some examples of common tasks performed with items, using the
|
||||
``Product`` item :ref:`declared above <topics-items-declaring>`. You will
|
||||
notice the API is very similar to the :class:`dict` API.
|
||||
|
||||
Creating items
|
||||
--------------
|
||||
''''''''''''''
|
||||
|
||||
>>> product = Product(name='Desktop PC', price=1000)
|
||||
>>> print(product)
|
||||
|
|
@ -88,7 +231,7 @@ Product(name='Desktop PC', price=1000)
|
|||
|
||||
|
||||
Getting field values
|
||||
--------------------
|
||||
''''''''''''''''''''
|
||||
|
||||
>>> product['name']
|
||||
Desktop PC
|
||||
|
|
@ -128,7 +271,7 @@ False
|
|||
|
||||
|
||||
Setting field values
|
||||
--------------------
|
||||
''''''''''''''''''''
|
||||
|
||||
>>> product['last_updated'] = 'today'
|
||||
>>> product['last_updated']
|
||||
|
|
@ -141,7 +284,7 @@ KeyError: 'Product does not support field: lala'
|
|||
|
||||
|
||||
Accessing all populated values
|
||||
------------------------------
|
||||
''''''''''''''''''''''''''''''
|
||||
|
||||
To access all populated values, just use the typical :class:`dict` API:
|
||||
|
||||
|
|
@ -155,7 +298,7 @@ To access all populated values, just use the typical :class:`dict` API:
|
|||
.. _copying-items:
|
||||
|
||||
Copying items
|
||||
-------------
|
||||
'''''''''''''
|
||||
|
||||
To copy an item, you must first decide whether you want a shallow copy or a
|
||||
deep copy.
|
||||
|
|
@ -183,7 +326,7 @@ To create a deep copy, call :meth:`~scrapy.item.Item.deepcopy` instead
|
|||
|
||||
|
||||
Other common tasks
|
||||
------------------
|
||||
''''''''''''''''''
|
||||
|
||||
Creating dicts from items:
|
||||
|
||||
|
|
@ -201,8 +344,8 @@ Traceback (most recent call last):
|
|||
KeyError: 'Product does not support field: lala'
|
||||
|
||||
|
||||
Extending Items
|
||||
===============
|
||||
Extending Item subclasses
|
||||
-------------------------
|
||||
|
||||
You can extend Items (to add more fields or to change some metadata for some
|
||||
fields) by declaring a subclass of your original Item.
|
||||
|
|
@ -222,39 +365,25 @@ appending more values, or changing existing values, like this::
|
|||
That adds (or replaces) the ``serializer`` metadata key for the ``name`` field,
|
||||
keeping all the previously existing metadata values.
|
||||
|
||||
Item objects
|
||||
============
|
||||
|
||||
.. class:: Item([arg])
|
||||
.. _supporting-item-types:
|
||||
|
||||
Return a new Item optionally initialized from the given argument.
|
||||
Supporting All Item Types
|
||||
=========================
|
||||
|
||||
Items replicate the standard :class:`dict` API, including its ``__init__``
|
||||
method, and also provide the following additional API members:
|
||||
In code that receives an item, such as methods of :ref:`item pipelines
|
||||
<topics-item-pipeline>` or :ref:`spider middlewares
|
||||
<topics-spider-middleware>`, it is a good practice to use the
|
||||
:class:`~itemadapter.ItemAdapter` class and the
|
||||
:func:`~itemadapter.is_item` function to write code that works for
|
||||
any :ref:`supported item type <item-types>`:
|
||||
|
||||
.. automethod:: copy
|
||||
.. autoclass:: itemadapter.ItemAdapter
|
||||
|
||||
.. automethod:: deepcopy
|
||||
.. autofunction:: itemadapter.is_item
|
||||
|
||||
.. attribute:: fields
|
||||
|
||||
A dictionary containing *all declared fields* for this Item, not only
|
||||
those populated. The keys are the field names and the values are the
|
||||
:class:`Field` objects used in the :ref:`Item declaration
|
||||
<topics-items-declaring>`.
|
||||
|
||||
Field objects
|
||||
=============
|
||||
|
||||
.. class:: Field([arg])
|
||||
|
||||
The :class:`Field` class is just an alias to the built-in :class:`dict` class and
|
||||
doesn't provide any extra functionality or attributes. In other words,
|
||||
:class:`Field` objects are plain-old Python dicts. A separate class is used
|
||||
to support the :ref:`item declaration syntax <topics-items-declaring>`
|
||||
based on class attributes.
|
||||
|
||||
Other classes related to Item
|
||||
=============================
|
||||
Other classes related to items
|
||||
==============================
|
||||
|
||||
.. autoclass:: ItemMeta
|
||||
|
|
|
|||
|
|
@ -4,7 +4,7 @@
|
|||
Debugging memory leaks
|
||||
======================
|
||||
|
||||
In Scrapy, objects such as Requests, Responses and Items have a finite
|
||||
In Scrapy, objects such as requests, responses and items have a finite
|
||||
lifetime: they are created, used for a while, and finally destroyed.
|
||||
|
||||
From all those objects, the Request is probably the one with the longest
|
||||
|
|
@ -61,8 +61,8 @@ Debugging memory leaks with ``trackref``
|
|||
========================================
|
||||
|
||||
:mod:`trackref` is a module provided by Scrapy to debug the most common cases of
|
||||
memory leaks. It basically tracks the references to all live Requests,
|
||||
Responses, Item and Selector objects.
|
||||
memory leaks. It basically tracks the references to all live Request,
|
||||
Response, Item, Spider and Selector objects.
|
||||
|
||||
You can enter the telnet console and inspect how many objects (of the classes
|
||||
mentioned above) are currently alive using the ``prefs()`` function which is an
|
||||
|
|
@ -200,11 +200,10 @@ Debugging memory leaks with muppy
|
|||
|
||||
``trackref`` provides a very convenient mechanism for tracking down memory
|
||||
leaks, but it only keeps track of the objects that are more likely to cause
|
||||
memory leaks (Requests, Responses, Items, and Selectors). However, there are
|
||||
other cases where the memory leaks could come from other (more or less obscure)
|
||||
objects. If this is your case, and you can't find your leaks using ``trackref``,
|
||||
you still have another resource: the muppy library.
|
||||
|
||||
memory leaks. However, there are other cases where the memory leaks could come
|
||||
from other (more or less obscure) objects. If this is your case, and you can't
|
||||
find your leaks using ``trackref``, you still have another resource: the muppy
|
||||
library.
|
||||
|
||||
You can use muppy from `Pympler`_.
|
||||
|
||||
|
|
|
|||
|
|
@ -7,13 +7,12 @@ Item Loaders
|
|||
.. module:: scrapy.loader
|
||||
:synopsis: Item Loader class
|
||||
|
||||
Item Loaders provide a convenient mechanism for populating scraped :ref:`Items
|
||||
<topics-items>`. Even though Items can be populated using their own
|
||||
dictionary-like API, Item Loaders provide a much more convenient API for
|
||||
populating them from a scraping process, by automating some common tasks like
|
||||
parsing the raw extracted data before assigning it.
|
||||
Item Loaders provide a convenient mechanism for populating scraped :ref:`items
|
||||
<topics-items>`. Even though items can be populated directly, Item Loaders provide a
|
||||
much more convenient API for populating them from a scraping process, by automating
|
||||
some common tasks like parsing the raw extracted data before assigning it.
|
||||
|
||||
In other words, :ref:`Items <topics-items>` provide the *container* of
|
||||
In other words, :ref:`items <topics-items>` provide the *container* of
|
||||
scraped data, while Item Loaders provide the mechanism for *populating* that
|
||||
container.
|
||||
|
||||
|
|
@ -25,10 +24,10 @@ Using Item Loaders to populate items
|
|||
====================================
|
||||
|
||||
To use an Item Loader, you must first instantiate it. You can either
|
||||
instantiate it with a dict-like object (e.g. Item or dict) or without one, in
|
||||
which case an Item is automatically instantiated in the Item Loader ``__init__`` method
|
||||
using the Item class specified in the :attr:`ItemLoader.default_item_class`
|
||||
attribute.
|
||||
instantiate it with an :ref:`item object <topics-items>` or without one, in which
|
||||
case an :ref:`item object <topics-items>` is automatically created in the
|
||||
Item Loader ``__init__`` method using the :ref:`item <topics-items>` class
|
||||
specified in the :attr:`ItemLoader.default_item_class` attribute.
|
||||
|
||||
Then, you start collecting values into the Item Loader, typically using
|
||||
:ref:`Selectors <topics-selectors>`. You can add more than one value to
|
||||
|
|
@ -77,6 +76,43 @@ called which actually returns the item populated with the data
|
|||
previously extracted and collected with the :meth:`~ItemLoader.add_xpath`,
|
||||
:meth:`~ItemLoader.add_css`, and :meth:`~ItemLoader.add_value` calls.
|
||||
|
||||
|
||||
.. _topics-loaders-dataclass:
|
||||
|
||||
Working with dataclass items
|
||||
============================
|
||||
|
||||
By default, :ref:`dataclass items <dataclass-items>` require all fields to be
|
||||
passed when created. This could be an issue when using dataclass items with
|
||||
item loaders: unless a pre-populated item is passed to the loader, fields
|
||||
will be populated incrementally using the loader's :meth:`~ItemLoader.add_xpath`,
|
||||
:meth:`~ItemLoader.add_css` and :meth:`~ItemLoader.add_value` methods.
|
||||
|
||||
Given the way that item loaders store data internally, one approach
|
||||
to overcome this is to define items using the :func:`~dataclasses.field`
|
||||
function, with ``list`` as the ``default_factory`` argument::
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
@dataclass
|
||||
class InventoryItem:
|
||||
name: str = field(default_factory=list)
|
||||
price: float = field(default_factory=list)
|
||||
stock: int = field(default_factory=list)
|
||||
|
||||
Note that in order to keep the example simple, the types do not match
|
||||
completely. A more accurate but verbose definition would be::
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from typing import List, Union
|
||||
|
||||
@dataclass
|
||||
class InventoryItem:
|
||||
name: Union[str, List[str]] = field(default_factory=list)
|
||||
price: Union[float, List[float]] = field(default_factory=list)
|
||||
stock: Union[int, List[int]] = field(default_factory=list)
|
||||
|
||||
|
||||
.. _topics-loaders-processors:
|
||||
|
||||
Input and Output processors
|
||||
|
|
@ -88,7 +124,7 @@ received (through the :meth:`~ItemLoader.add_xpath`, :meth:`~ItemLoader.add_css`
|
|||
:meth:`~ItemLoader.add_value` methods) and the result of the input processor is
|
||||
collected and kept inside the ItemLoader. After collecting all data, the
|
||||
:meth:`ItemLoader.load_item` method is called to populate and get the populated
|
||||
:class:`~scrapy.item.Item` object. That's when the output processor is
|
||||
:ref:`item object <topics-items>`. That's when the output processor is
|
||||
called with the data previously collected (and processed using the input
|
||||
processor). The result of the output processor is the final value that gets
|
||||
assigned to the item.
|
||||
|
|
@ -153,12 +189,10 @@ Last, but not least, Scrapy comes with some :ref:`commonly used processors
|
|||
<topics-loaders-available-processors>` built-in for convenience.
|
||||
|
||||
|
||||
|
||||
Declaring Item Loaders
|
||||
======================
|
||||
|
||||
Item Loaders are declared like Items, by using a class definition syntax. Here
|
||||
is an example::
|
||||
Item Loaders are declared using a class definition syntax. Here is an example::
|
||||
|
||||
from scrapy.loader import ItemLoader
|
||||
from scrapy.loader.processors import TakeFirst, MapCompose, Join
|
||||
|
|
@ -275,9 +309,9 @@ ItemLoader objects
|
|||
|
||||
.. class:: ItemLoader([item, selector, response], **kwargs)
|
||||
|
||||
Return a new Item Loader for populating the given Item. If no item is
|
||||
given, one is instantiated automatically using the class in
|
||||
:attr:`default_item_class`.
|
||||
Return a new Item Loader for populating the given :ref:`item object
|
||||
<topics-items>`. If no item object is given, one is instantiated
|
||||
automatically using the class in :attr:`default_item_class`.
|
||||
|
||||
When instantiated with a ``selector`` or a ``response`` parameters
|
||||
the :class:`ItemLoader` class provides convenient mechanisms for extracting
|
||||
|
|
@ -286,7 +320,7 @@ ItemLoader objects
|
|||
:param item: The item instance to populate using subsequent calls to
|
||||
:meth:`~ItemLoader.add_xpath`, :meth:`~ItemLoader.add_css`,
|
||||
or :meth:`~ItemLoader.add_value`.
|
||||
:type item: :class:`~scrapy.item.Item` object
|
||||
:type item: :ref:`item object <topics-items>`
|
||||
|
||||
:param selector: The selector to extract data from, when using the
|
||||
:meth:`add_xpath` (resp. :meth:`add_css`) or :meth:`replace_xpath`
|
||||
|
|
@ -444,17 +478,19 @@ ItemLoader objects
|
|||
|
||||
Create a nested loader with an xpath selector.
|
||||
The supplied selector is applied relative to selector associated
|
||||
with this :class:`ItemLoader`. The nested loader shares the :class:`Item`
|
||||
with the parent :class:`ItemLoader` so calls to :meth:`add_xpath`,
|
||||
:meth:`add_value`, :meth:`replace_value`, etc. will behave as expected.
|
||||
with this :class:`ItemLoader`. The nested loader shares the :ref:`item
|
||||
object <topics-items>` with the parent :class:`ItemLoader` so calls to
|
||||
:meth:`add_xpath`, :meth:`add_value`, :meth:`replace_value`, etc. will
|
||||
behave as expected.
|
||||
|
||||
.. method:: nested_css(css)
|
||||
|
||||
Create a nested loader with a css selector.
|
||||
The supplied selector is applied relative to selector associated
|
||||
with this :class:`ItemLoader`. The nested loader shares the :class:`Item`
|
||||
with the parent :class:`ItemLoader` so calls to :meth:`add_xpath`,
|
||||
:meth:`add_value`, :meth:`replace_value`, etc. will behave as expected.
|
||||
with this :class:`ItemLoader`. The nested loader shares the :ref:`item
|
||||
object <topics-items>` with the parent :class:`ItemLoader` so calls to
|
||||
:meth:`add_xpath`, :meth:`add_value`, :meth:`replace_value`, etc. will
|
||||
behave as expected.
|
||||
|
||||
.. method:: get_collected_values(field_name)
|
||||
|
||||
|
|
@ -477,7 +513,7 @@ ItemLoader objects
|
|||
|
||||
.. attribute:: item
|
||||
|
||||
The :class:`~scrapy.item.Item` object being parsed by this Item Loader.
|
||||
The :ref:`item object <topics-items>` being parsed by this Item Loader.
|
||||
This is mostly used as a property so when attempting to override this
|
||||
value, you may want to check out :attr:`default_item_class` first.
|
||||
|
||||
|
|
@ -488,8 +524,8 @@ ItemLoader objects
|
|||
|
||||
.. attribute:: default_item_class
|
||||
|
||||
An Item class (or factory), used to instantiate items when not given in
|
||||
the ``__init__`` method.
|
||||
An :ref:`item object <topics-items>` class or factory, used to
|
||||
instantiate items when not given in the ``__init__`` method.
|
||||
|
||||
.. attribute:: default_input_processor
|
||||
|
||||
|
|
|
|||
|
|
@ -156,7 +156,7 @@ following forms::
|
|||
|
||||
ftp://username:password@address:port/path
|
||||
ftp://address:port/path
|
||||
|
||||
|
||||
If ``username`` and ``password`` are not provided, they are taken from the :setting:`FTP_USER` and
|
||||
:setting:`FTP_PASSWORD` settings respectively.
|
||||
|
||||
|
|
@ -201,6 +201,9 @@ For self-hosting you also might feel the need not to use SSL and not to verify S
|
|||
.. _s3.scality: https://s3.scality.com/
|
||||
.. _canned ACLs: https://docs.aws.amazon.com/AmazonS3/latest/dev/acl-overview.html#canned-acl
|
||||
|
||||
|
||||
.. _media-pipeline-gcs:
|
||||
|
||||
Google Cloud Storage
|
||||
---------------------
|
||||
|
||||
|
|
@ -243,20 +246,22 @@ Usage example
|
|||
.. setting:: IMAGES_URLS_FIELD
|
||||
.. setting:: IMAGES_RESULT_FIELD
|
||||
|
||||
In order to use a media pipeline first, :ref:`enable it
|
||||
In order to use a media pipeline, first :ref:`enable it
|
||||
<topics-media-pipeline-enabling>`.
|
||||
|
||||
Then, if a spider returns a dict with the URLs key (``file_urls`` or
|
||||
``image_urls``, for the Files or Images Pipeline respectively), the pipeline will
|
||||
put the results under respective key (``files`` or ``images``).
|
||||
Then, if a spider returns an :ref:`item object <topics-items>` with the URLs
|
||||
field (``file_urls`` or ``image_urls``, for the Files or Images Pipeline
|
||||
respectively), the pipeline will put the results under the respective field
|
||||
(``files`` or ``images``).
|
||||
|
||||
If you prefer to use :class:`~.Item`, then define a custom item with the
|
||||
necessary fields, like in this example for Images Pipeline::
|
||||
When using :ref:`item types <item-types>` for which fields are defined beforehand,
|
||||
you must define both the URLs field and the results field. For example, when
|
||||
using the images pipeline, items must define both the ``image_urls`` and the
|
||||
``images`` field. For instance, using the :class:`~scrapy.item.Item` class::
|
||||
|
||||
import scrapy
|
||||
|
||||
class MyItem(scrapy.Item):
|
||||
|
||||
# ... other item fields ...
|
||||
image_urls = scrapy.Field()
|
||||
images = scrapy.Field()
|
||||
|
|
@ -445,8 +450,11 @@ See here the methods that you can override in your custom Files Pipeline:
|
|||
:meth:`~get_media_requests` method and return a Request for each
|
||||
file URL::
|
||||
|
||||
from itemadapter import ItemAdapter
|
||||
|
||||
def get_media_requests(self, item, info):
|
||||
for file_url in item['file_urls']:
|
||||
adapter = ItemAdapter(item)
|
||||
for file_url in adapter['file_urls']:
|
||||
yield scrapy.Request(file_url)
|
||||
|
||||
Those requests will be processed by the pipeline and, when they have finished
|
||||
|
|
@ -470,7 +478,11 @@ See here the methods that you can override in your custom Files Pipeline:
|
|||
|
||||
* ``checksum`` - a `MD5 hash`_ of the image contents
|
||||
|
||||
* ``status`` - the file status indication. It can be one of the following:
|
||||
* ``status`` - the file status indication.
|
||||
|
||||
.. versionadded:: 2.2
|
||||
|
||||
It can be one of the following:
|
||||
|
||||
* ``downloaded`` - file was downloaded.
|
||||
* ``uptodate`` - file was not downloaded, as it was downloaded recently,
|
||||
|
|
@ -509,13 +521,15 @@ See here the methods that you can override in your custom Files Pipeline:
|
|||
store the downloaded file paths (passed in results) in the ``file_paths``
|
||||
item field, and we drop the item if it doesn't contain any files::
|
||||
|
||||
from itemadapter import ItemAdapter
|
||||
from scrapy.exceptions import DropItem
|
||||
|
||||
def item_completed(self, results, item, info):
|
||||
file_paths = [x['path'] for ok, x in results if ok]
|
||||
if not file_paths:
|
||||
raise DropItem("Item contains no files")
|
||||
item['file_paths'] = file_paths
|
||||
adapter = ItemAdapter(item)
|
||||
adapter['file_paths'] = file_paths
|
||||
return item
|
||||
|
||||
By default, the :meth:`item_completed` method returns the item.
|
||||
|
|
@ -589,8 +603,9 @@ Here is a full example of the Images Pipeline whose methods are exemplified
|
|||
above::
|
||||
|
||||
import scrapy
|
||||
from scrapy.pipelines.images import ImagesPipeline
|
||||
from itemadapter import ItemAdapter
|
||||
from scrapy.exceptions import DropItem
|
||||
from scrapy.pipelines.images import ImagesPipeline
|
||||
|
||||
class MyImagesPipeline(ImagesPipeline):
|
||||
|
||||
|
|
@ -602,7 +617,8 @@ above::
|
|||
image_paths = [x['path'] for ok, x in results if ok]
|
||||
if not image_paths:
|
||||
raise DropItem("Item contains no images")
|
||||
item['image_paths'] = image_paths
|
||||
adapter = ItemAdapter(item)
|
||||
adapter['image_paths'] = image_paths
|
||||
return item
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -189,6 +189,10 @@ Request objects
|
|||
cloned using the ``copy()`` or ``replace()`` methods, and can also be
|
||||
accessed, in your spider, from the ``response.cb_kwargs`` attribute.
|
||||
|
||||
In case of a failure to process the request, this dict can be accessed as
|
||||
``failure.request.cb_kwargs`` in the request's errback. For more information,
|
||||
see :ref:`errback-cb_kwargs`.
|
||||
|
||||
.. method:: Request.copy()
|
||||
|
||||
Return a new Request which is a copy of this Request. See also:
|
||||
|
|
@ -312,6 +316,31 @@ errors if needed::
|
|||
request = failure.request
|
||||
self.logger.error('TimeoutError on %s', request.url)
|
||||
|
||||
.. _errback-cb_kwargs:
|
||||
|
||||
Accessing additional data in errback functions
|
||||
----------------------------------------------
|
||||
|
||||
In case of a failure to process the request, you may be interested in
|
||||
accessing arguments to the callback functions so you can process further
|
||||
based on the arguments in the errback. The following example shows how to
|
||||
achieve this by using ``Failure.request.cb_kwargs``::
|
||||
|
||||
def parse(self, response):
|
||||
request = scrapy.Request('http://www.example.com/index.html',
|
||||
callback=self.parse_page2,
|
||||
errback=self.errback_page2,
|
||||
cb_kwargs=dict(main_url=response.url))
|
||||
yield request
|
||||
|
||||
def parse_page2(self, response, main_url):
|
||||
pass
|
||||
|
||||
def errback_page2(self, failure):
|
||||
yield dict(
|
||||
main_url=failure.request.cb_kwargs['main_url'],
|
||||
)
|
||||
|
||||
.. _topics-request-meta:
|
||||
|
||||
Request.meta special keys
|
||||
|
|
|
|||
|
|
@ -236,8 +236,8 @@ CONCURRENT_ITEMS
|
|||
|
||||
Default: ``100``
|
||||
|
||||
Maximum number of concurrent items (per response) to process in parallel in the
|
||||
Item Processor (also known as the :ref:`Item Pipeline <topics-item-pipeline>`).
|
||||
Maximum number of concurrent items (per response) to process in parallel in
|
||||
:ref:`item pipelines <topics-item-pipeline>`.
|
||||
|
||||
.. setting:: CONCURRENT_REQUESTS
|
||||
|
||||
|
|
|
|||
|
|
@ -151,8 +151,8 @@ item_scraped
|
|||
|
||||
This signal supports returning deferreds from its handlers.
|
||||
|
||||
:param item: the item scraped
|
||||
:type item: dict or :class:`~scrapy.item.Item` object
|
||||
:param item: the scraped item
|
||||
:type item: :ref:`item object <item-types>`
|
||||
|
||||
:param spider: the spider which scraped the item
|
||||
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||
|
|
@ -172,7 +172,7 @@ item_dropped
|
|||
This signal supports returning deferreds from its handlers.
|
||||
|
||||
:param item: the item dropped from the :ref:`topics-item-pipeline`
|
||||
:type item: dict or :class:`~scrapy.item.Item` object
|
||||
:type item: :ref:`item object <item-types>`
|
||||
|
||||
:param spider: the spider which scraped the item
|
||||
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||
|
|
@ -196,8 +196,8 @@ item_error
|
|||
|
||||
This signal supports returning deferreds from its handlers.
|
||||
|
||||
:param item: the item dropped from the :ref:`topics-item-pipeline`
|
||||
:type item: dict or :class:`~scrapy.item.Item` object
|
||||
:param item: the item that caused the error in the :ref:`topics-item-pipeline`
|
||||
:type item: :ref:`item object <item-types>`
|
||||
|
||||
:param response: the response being processed when the exception was raised
|
||||
:type response: :class:`~scrapy.http.Response` object
|
||||
|
|
|
|||
|
|
@ -102,29 +102,28 @@ object gives you access, for example, to the :ref:`settings <topics-settings>`.
|
|||
it has processed the response.
|
||||
|
||||
:meth:`process_spider_output` must return an iterable of
|
||||
:class:`~scrapy.http.Request`, dict or :class:`~scrapy.item.Item`
|
||||
objects.
|
||||
:class:`~scrapy.http.Request` objects and :ref:`item object
|
||||
<topics-items>`.
|
||||
|
||||
:param response: the response which generated this output from the
|
||||
spider
|
||||
:type response: :class:`~scrapy.http.Response` object
|
||||
|
||||
:param result: the result returned by the spider
|
||||
:type result: an iterable of :class:`~scrapy.http.Request`, dict
|
||||
or :class:`~scrapy.item.Item` objects
|
||||
:type result: an iterable of :class:`~scrapy.http.Request` objects and
|
||||
:ref:`item object <topics-items>`
|
||||
|
||||
:param spider: the spider whose result is being processed
|
||||
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||
|
||||
|
||||
.. method:: process_spider_exception(response, exception, spider)
|
||||
|
||||
This method is called when a spider or :meth:`process_spider_output`
|
||||
method (from a previous spider middleware) raises an exception.
|
||||
|
||||
:meth:`process_spider_exception` should return either ``None`` or an
|
||||
iterable of :class:`~scrapy.http.Request`, dict or
|
||||
:class:`~scrapy.item.Item` objects.
|
||||
iterable of :class:`~scrapy.http.Request` objects and :ref:`item object
|
||||
<topics-items>`.
|
||||
|
||||
If it returns ``None``, Scrapy will continue processing this exception,
|
||||
executing any other :meth:`process_spider_exception` in the following
|
||||
|
|
|
|||
|
|
@ -23,8 +23,8 @@ For spiders, the scraping cycle goes through something like this:
|
|||
:attr:`~scrapy.spiders.Spider.parse` method as callback function for the
|
||||
Requests.
|
||||
|
||||
2. In the callback function, you parse the response (web page) and return either
|
||||
dicts with extracted data, :class:`~scrapy.item.Item` objects,
|
||||
2. In the callback function, you parse the response (web page) and return
|
||||
:ref:`item objects <topics-items>`,
|
||||
:class:`~scrapy.http.Request` objects, or an iterable of these objects.
|
||||
Those Requests will also contain a callback (maybe
|
||||
the same) and will then be downloaded by Scrapy and then their
|
||||
|
|
@ -179,8 +179,8 @@ scrapy.Spider
|
|||
the same requirements as the :class:`Spider` class.
|
||||
|
||||
This method, as well as any other Request callback, must return an
|
||||
iterable of :class:`~scrapy.http.Request` and/or
|
||||
dicts or :class:`~scrapy.item.Item` objects.
|
||||
iterable of :class:`~scrapy.http.Request` and/or :ref:`item objects
|
||||
<topics-items>`.
|
||||
|
||||
:param response: the response to parse
|
||||
:type response: :class:`~scrapy.http.Response`
|
||||
|
|
@ -234,7 +234,7 @@ Return multiple Requests and items from a single callback::
|
|||
yield scrapy.Request(response.urljoin(href), self.parse)
|
||||
|
||||
Instead of :attr:`~.start_urls` you can use :meth:`~.start_requests` directly;
|
||||
to give data more structure you can use :ref:`topics-items`::
|
||||
to give data more structure you can use :class:`~scrapy.item.Item` objects::
|
||||
|
||||
import scrapy
|
||||
from myproject.items import MyItem
|
||||
|
|
@ -364,7 +364,7 @@ CrawlSpider
|
|||
|
||||
This method is called for the start_urls responses. It allows to parse
|
||||
the initial responses and must return either an
|
||||
:class:`~scrapy.item.Item` object, a :class:`~scrapy.http.Request`
|
||||
:ref:`item object <topics-items>`, a :class:`~scrapy.http.Request`
|
||||
object, or an iterable containing any of them.
|
||||
|
||||
Crawling rules
|
||||
|
|
@ -383,7 +383,7 @@ Crawling rules
|
|||
object with that name will be used) to be called for each link extracted with
|
||||
the specified link extractor. This callback receives a :class:`~scrapy.http.Response`
|
||||
as its first argument and must return either a single instance or an iterable of
|
||||
:class:`~scrapy.item.Item`, ``dict`` and/or :class:`~scrapy.http.Request` objects
|
||||
:ref:`item objects <topics-items>` and/or :class:`~scrapy.http.Request` objects
|
||||
(or any subclass of them). As mentioned above, the received :class:`~scrapy.http.Response`
|
||||
object will contain the text of the link that produced the :class:`~scrapy.http.Request`
|
||||
in its ``meta`` dictionary (under the ``link_text`` key)
|
||||
|
|
@ -531,7 +531,7 @@ XMLFeedSpider
|
|||
(``itertag``). Receives the response and an
|
||||
:class:`~scrapy.selector.Selector` for each node. Overriding this
|
||||
method is mandatory. Otherwise, you spider won't work. This method
|
||||
must return either a :class:`~scrapy.item.Item` object, a
|
||||
must return an :ref:`item object <topics-items>`, a
|
||||
:class:`~scrapy.http.Request` object, or an iterable containing any of
|
||||
them.
|
||||
|
||||
|
|
@ -541,7 +541,7 @@ XMLFeedSpider
|
|||
spider, and it's intended to perform any last time processing required
|
||||
before returning the results to the framework core, for example setting the
|
||||
item IDs. It receives a list of results and the response which originated
|
||||
those results. It must return a list of results (Items or Requests).
|
||||
those results. It must return a list of results (items or requests).
|
||||
|
||||
|
||||
XMLFeedSpider example
|
||||
|
|
|
|||
|
|
@ -1 +1 @@
|
|||
2.1.0
|
||||
2.2.0
|
||||
|
|
|
|||
|
|
@ -28,8 +28,8 @@ twisted_version = (_txv.major, _txv.minor, _txv.micro)
|
|||
|
||||
|
||||
# Check minimum required Python version
|
||||
if sys.version_info < (3, 5):
|
||||
print("Scrapy %s requires Python 3.5" % __version__)
|
||||
if sys.version_info < (3, 5, 2):
|
||||
print("Scrapy %s requires Python 3.5.2" % __version__)
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -1,11 +1,11 @@
|
|||
import json
|
||||
import logging
|
||||
|
||||
from itemadapter import is_item, ItemAdapter
|
||||
from w3lib.url import is_url
|
||||
|
||||
from scrapy.commands import ScrapyCommand
|
||||
from scrapy.http import Request
|
||||
from scrapy.item import _BaseItem
|
||||
from scrapy.utils import display
|
||||
from scrapy.utils.conf import arglist_to_dict
|
||||
from scrapy.utils.spider import iterate_spider_output, spidercls_for_request
|
||||
|
|
@ -81,7 +81,7 @@ class Command(ScrapyCommand):
|
|||
items = self.items.get(lvl, [])
|
||||
|
||||
print("# Scraped Items ", "-" * 60)
|
||||
display.pprint([dict(x) for x in items], colorize=colour)
|
||||
display.pprint([ItemAdapter(x).asdict() for x in items], colorize=colour)
|
||||
|
||||
def print_requests(self, lvl=None, colour=True):
|
||||
if lvl is None:
|
||||
|
|
@ -117,7 +117,7 @@ class Command(ScrapyCommand):
|
|||
items, requests = [], []
|
||||
|
||||
for x in iterate_spider_output(callback(response, **cb_kwargs)):
|
||||
if isinstance(x, (_BaseItem, dict)):
|
||||
if is_item(x):
|
||||
items.append(x)
|
||||
elif isinstance(x, Request):
|
||||
requests.append(x)
|
||||
|
|
|
|||
|
|
@ -1,5 +1,6 @@
|
|||
import re
|
||||
import os
|
||||
import stat
|
||||
import string
|
||||
from importlib import import_module
|
||||
from os.path import join, exists, abspath
|
||||
|
|
@ -78,6 +79,29 @@ class Command(ScrapyCommand):
|
|||
else:
|
||||
copy2(srcname, dstname)
|
||||
copystat(src, dst)
|
||||
self._set_rw_permissions(dst)
|
||||
|
||||
def _set_rw_permissions(self, path):
|
||||
"""
|
||||
Sets permissions of a directory tree to +rw and +rwx for folders.
|
||||
This is necessary if the start template files come without write
|
||||
permissions.
|
||||
"""
|
||||
mode_rw = (stat.S_IRUSR
|
||||
| stat.S_IWUSR
|
||||
| stat.S_IRGRP
|
||||
| stat.S_IROTH)
|
||||
|
||||
mode_x = (stat.S_IXUSR
|
||||
| stat.S_IXGRP
|
||||
| stat.S_IXOTH)
|
||||
|
||||
os.chmod(path, mode_rw | mode_x)
|
||||
for root, dirs, files in os.walk(path):
|
||||
for dir in dirs:
|
||||
os.chmod(join(root, dir), mode_rw | mode_x)
|
||||
for file in files:
|
||||
os.chmod(join(root, file), mode_rw)
|
||||
|
||||
def run(self, args, opts):
|
||||
if len(args) not in (1, 2):
|
||||
|
|
|
|||
|
|
@ -1,10 +1,10 @@
|
|||
import json
|
||||
|
||||
from scrapy.item import _BaseItem
|
||||
from scrapy.http import Request
|
||||
from scrapy.exceptions import ContractFail
|
||||
from itemadapter import is_item, ItemAdapter
|
||||
|
||||
from scrapy.contracts import Contract
|
||||
from scrapy.exceptions import ContractFail
|
||||
from scrapy.http import Request
|
||||
|
||||
|
||||
# contracts
|
||||
|
|
@ -48,11 +48,11 @@ class ReturnsContract(Contract):
|
|||
"""
|
||||
|
||||
name = 'returns'
|
||||
objects = {
|
||||
'request': Request,
|
||||
'requests': Request,
|
||||
'item': (_BaseItem, dict),
|
||||
'items': (_BaseItem, dict),
|
||||
object_type_verifiers = {
|
||||
'request': lambda x: isinstance(x, Request),
|
||||
'requests': lambda x: isinstance(x, Request),
|
||||
'item': is_item,
|
||||
'items': is_item,
|
||||
}
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
|
|
@ -64,7 +64,7 @@ class ReturnsContract(Contract):
|
|||
% len(self.args)
|
||||
)
|
||||
self.obj_name = self.args[0] or None
|
||||
self.obj_type = self.objects[self.obj_name]
|
||||
self.obj_type_verifier = self.object_type_verifiers[self.obj_name]
|
||||
|
||||
try:
|
||||
self.min_bound = int(self.args[1])
|
||||
|
|
@ -79,7 +79,7 @@ class ReturnsContract(Contract):
|
|||
def post_process(self, output):
|
||||
occurrences = 0
|
||||
for x in output:
|
||||
if isinstance(x, self.obj_type):
|
||||
if self.obj_type_verifier(x):
|
||||
occurrences += 1
|
||||
|
||||
assertion = (self.min_bound <= occurrences <= self.max_bound)
|
||||
|
|
@ -103,8 +103,8 @@ class ScrapesContract(Contract):
|
|||
|
||||
def post_process(self, output):
|
||||
for x in output:
|
||||
if isinstance(x, (_BaseItem, dict)):
|
||||
missing = [arg for arg in self.args if arg not in x]
|
||||
if is_item(x):
|
||||
missing = [arg for arg in self.args if arg not in ItemAdapter(x)]
|
||||
if missing:
|
||||
raise ContractFail(
|
||||
"Missing fields: %s" % ", ".join(missing))
|
||||
missing_str = ", ".join(missing)
|
||||
raise ContractFail("Missing fields: %s" % missing_str)
|
||||
|
|
|
|||
|
|
@ -18,6 +18,7 @@ from scrapy.utils.defer import deferred_from_coro, _isasyncgen
|
|||
from scrapy.utils.misc import load_object
|
||||
from scrapy.utils.reactor import CallLaterOnce
|
||||
from scrapy.utils.log import logformatter_adapter, failure_to_exc_info
|
||||
from scrapy.spiders import WaitUntilQueueEmpty
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
|
@ -130,16 +131,14 @@ class ExecutionEngine:
|
|||
return
|
||||
self._waiting_for_request = True
|
||||
try:
|
||||
while not self._needs_backout(spider):
|
||||
if not self._next_request_from_scheduler(spider):
|
||||
break
|
||||
|
||||
if slot.start_requests and not self._needs_backout(spider):
|
||||
while not self._needs_backout(spider) and slot.start_requests:
|
||||
try:
|
||||
if _isasyncgen(slot.start_requests):
|
||||
request = yield deferred_from_coro(slot.start_requests.__anext__())
|
||||
else:
|
||||
request = next(slot.start_requests)
|
||||
if request == WaitUntilQueueEmpty:
|
||||
break
|
||||
except (StopIteration, StopAsyncIteration):
|
||||
slot.start_requests = None
|
||||
except Exception:
|
||||
|
|
@ -149,6 +148,10 @@ class ExecutionEngine:
|
|||
else:
|
||||
self.crawl(request, spider)
|
||||
|
||||
while not self._needs_backout(spider):
|
||||
if not self._next_request_from_scheduler(spider):
|
||||
break
|
||||
|
||||
if self.spider_is_idle(spider) and slot.close_if_idle:
|
||||
self._spider_idle(spider)
|
||||
finally:
|
||||
|
|
|
|||
|
|
@ -4,18 +4,18 @@ extracts information from them"""
|
|||
import logging
|
||||
from collections import deque
|
||||
|
||||
from twisted.python.failure import Failure
|
||||
from itemadapter import is_item
|
||||
from twisted.internet import defer
|
||||
from twisted.python.failure import Failure
|
||||
|
||||
from scrapy.utils.defer import defer_result, defer_succeed, parallel, iter_errback
|
||||
from scrapy.utils.spider import iterate_spider_output
|
||||
from scrapy.utils.misc import load_object, warn_on_generator_with_return_value
|
||||
from scrapy.utils.log import logformatter_adapter, failure_to_exc_info
|
||||
from scrapy.exceptions import CloseSpider, DropItem, IgnoreRequest
|
||||
from scrapy import signals
|
||||
from scrapy.http import Request, Response
|
||||
from scrapy.item import _BaseItem
|
||||
from scrapy.core.spidermw import SpiderMiddlewareManager
|
||||
from scrapy.exceptions import CloseSpider, DropItem, IgnoreRequest
|
||||
from scrapy.http import Request, Response
|
||||
from scrapy.utils.defer import defer_result, defer_succeed, iter_errback, parallel
|
||||
from scrapy.utils.log import failure_to_exc_info, logformatter_adapter
|
||||
from scrapy.utils.misc import load_object, warn_on_generator_with_return_value
|
||||
from scrapy.utils.spider import iterate_spider_output
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
|
@ -191,7 +191,7 @@ class Scraper:
|
|||
"""
|
||||
if isinstance(output, Request):
|
||||
self.crawler.engine.crawl(request=output, spider=spider)
|
||||
elif isinstance(output, (_BaseItem, dict)):
|
||||
elif is_item(output):
|
||||
self.slot.itemproc_size += 1
|
||||
dfd = self.itemproc.process_item(output, spider)
|
||||
dfd.addBoth(self._itemproc_finished, output, response, spider)
|
||||
|
|
@ -200,10 +200,11 @@ class Scraper:
|
|||
pass
|
||||
else:
|
||||
typename = type(output).__name__
|
||||
logger.error('Spider must return Request, BaseItem, dict or None, '
|
||||
'got %(typename)r in %(request)s',
|
||||
{'request': request, 'typename': typename},
|
||||
extra={'spider': spider})
|
||||
logger.error(
|
||||
'Spider must return request, item, or None, got %(typename)r in %(request)s',
|
||||
{'request': request, 'typename': typename},
|
||||
extra={'spider': spider},
|
||||
)
|
||||
|
||||
def _log_download_errors(self, spider_failure, download_failure, request, spider):
|
||||
"""Log and silence errors that come from the engine (typically download
|
||||
|
|
|
|||
|
|
@ -103,7 +103,7 @@ class Crawler:
|
|||
elif inspect.iscoroutinefunction(self.spider.start_requests):
|
||||
return deferred_from_coro(self.spider.start_requests())
|
||||
else:
|
||||
return iter(self.spider.start_requests())
|
||||
return iter(self.spider.start_requests_with_control())
|
||||
|
||||
def _create_spider(self, *args, **kwargs):
|
||||
return self.spidercls.from_crawler(self, *args, **kwargs)
|
||||
|
|
|
|||
|
|
@ -4,16 +4,18 @@ Item Exporters are used to export/serialize items into different formats.
|
|||
|
||||
import csv
|
||||
import io
|
||||
import pprint
|
||||
import marshal
|
||||
import warnings
|
||||
import pickle
|
||||
import pprint
|
||||
import warnings
|
||||
from xml.sax.saxutils import XMLGenerator
|
||||
|
||||
from scrapy.utils.serialize import ScrapyJSONEncoder
|
||||
from scrapy.utils.python import to_bytes, to_unicode, is_listlike
|
||||
from scrapy.item import _BaseItem
|
||||
from itemadapter import is_item, ItemAdapter
|
||||
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
from scrapy.item import _BaseItem
|
||||
from scrapy.utils.python import is_listlike, to_bytes, to_unicode
|
||||
from scrapy.utils.serialize import ScrapyJSONEncoder
|
||||
|
||||
|
||||
__all__ = ['BaseItemExporter', 'PprintItemExporter', 'PickleItemExporter',
|
||||
|
|
@ -56,11 +58,14 @@ class BaseItemExporter:
|
|||
"""Return the fields to export as an iterable of tuples
|
||||
(name, serialized_value)
|
||||
"""
|
||||
item = ItemAdapter(item)
|
||||
|
||||
if include_empty is None:
|
||||
include_empty = self.export_empty_fields
|
||||
|
||||
if self.fields_to_export is None:
|
||||
if include_empty and not isinstance(item, dict):
|
||||
field_iter = item.fields.keys()
|
||||
if include_empty:
|
||||
field_iter = item.field_names()
|
||||
else:
|
||||
field_iter = item.keys()
|
||||
else:
|
||||
|
|
@ -71,8 +76,8 @@ class BaseItemExporter:
|
|||
|
||||
for field_name in field_iter:
|
||||
if field_name in item:
|
||||
field = {} if isinstance(item, dict) else item.fields[field_name]
|
||||
value = self.serialize_field(field, field_name, item[field_name])
|
||||
field_meta = item.get_field_meta(field_name)
|
||||
value = self.serialize_field(field_meta, field_name, item[field_name])
|
||||
else:
|
||||
value = default_value
|
||||
|
||||
|
|
@ -297,6 +302,7 @@ class PythonItemExporter(BaseItemExporter):
|
|||
|
||||
.. _msgpack: https://pypi.org/project/msgpack/
|
||||
"""
|
||||
|
||||
def _configure(self, options, dont_fail=False):
|
||||
self.binary = options.pop('binary', True)
|
||||
super(PythonItemExporter, self)._configure(options, dont_fail)
|
||||
|
|
@ -314,22 +320,22 @@ class PythonItemExporter(BaseItemExporter):
|
|||
def _serialize_value(self, value):
|
||||
if isinstance(value, _BaseItem):
|
||||
return self.export_item(value)
|
||||
if isinstance(value, dict):
|
||||
return dict(self._serialize_dict(value))
|
||||
if is_listlike(value):
|
||||
elif is_item(value):
|
||||
return dict(self._serialize_item(value))
|
||||
elif is_listlike(value):
|
||||
return [self._serialize_value(v) for v in value]
|
||||
encode_func = to_bytes if self.binary else to_unicode
|
||||
if isinstance(value, (str, bytes)):
|
||||
return encode_func(value, encoding=self.encoding)
|
||||
return value
|
||||
|
||||
def _serialize_dict(self, value):
|
||||
for key, val in value.items():
|
||||
def _serialize_item(self, item):
|
||||
for key, value in ItemAdapter(item).items():
|
||||
key = to_bytes(key) if self.binary else key
|
||||
yield key, self._serialize_value(val)
|
||||
yield key, self._serialize_value(value)
|
||||
|
||||
def export_item(self, item):
|
||||
result = dict(self._get_serialized_fields(item))
|
||||
if self.binary:
|
||||
result = dict(self._serialize_dict(result))
|
||||
result = dict(self._serialize_item(result))
|
||||
return result
|
||||
|
|
|
|||
|
|
@ -270,18 +270,29 @@ class FeedExporter:
|
|||
if not slot.itemcount and not slot.store_empty:
|
||||
# We need to call slot.storage.store nonetheless to get the file
|
||||
# properly closed.
|
||||
return defer.maybeDeferred(slot.storage.store, slot.file)
|
||||
d = defer.maybeDeferred(slot.storage.store, slot.file)
|
||||
deferred_list.append(d)
|
||||
continue
|
||||
slot.finish_exporting()
|
||||
logfmt = "%s %%(format)s feed (%%(itemcount)d items) in: %%(uri)s"
|
||||
log_args = {'format': slot.format,
|
||||
'itemcount': slot.itemcount,
|
||||
'uri': slot.uri}
|
||||
d = defer.maybeDeferred(slot.storage.store, slot.file)
|
||||
d.addCallback(lambda _: logger.info(logfmt % "Stored", log_args,
|
||||
extra={'spider': spider}))
|
||||
d.addErrback(lambda f: logger.error(logfmt % "Error storing", log_args,
|
||||
exc_info=failure_to_exc_info(f),
|
||||
extra={'spider': spider}))
|
||||
|
||||
# Use `largs=log_args` to copy log_args into function's scope
|
||||
# instead of using `log_args` from the outer scope
|
||||
d.addCallback(
|
||||
lambda _, largs=log_args: logger.info(
|
||||
logfmt % "Stored", largs, extra={'spider': spider}
|
||||
)
|
||||
)
|
||||
d.addErrback(
|
||||
lambda f, largs=log_args: logger.error(
|
||||
logfmt % "Error storing", largs,
|
||||
exc_info=failure_to_exc_info(f), extra={'spider': spider}
|
||||
)
|
||||
)
|
||||
deferred_list.append(d)
|
||||
return defer.DeferredList(deferred_list) if deferred_list else None
|
||||
|
||||
|
|
|
|||
|
|
@ -74,6 +74,8 @@ class TextResponse(Response):
|
|||
|
||||
def json(self):
|
||||
"""
|
||||
.. versionadded:: 2.2
|
||||
|
||||
Deserialize a JSON document to a Python object.
|
||||
"""
|
||||
if self._cached_decoded_json is _NONE:
|
||||
|
|
|
|||
|
|
@ -36,7 +36,7 @@ class BaseItem(_BaseItem, metaclass=_BaseItemMeta):
|
|||
"""
|
||||
|
||||
def __new__(cls, *args, **kwargs):
|
||||
if issubclass(cls, BaseItem) and not (issubclass(cls, Item) or issubclass(cls, DictItem)):
|
||||
if issubclass(cls, BaseItem) and not issubclass(cls, (Item, DictItem)):
|
||||
warn('scrapy.item.BaseItem is deprecated, please use scrapy.item.Item instead',
|
||||
ScrapyDeprecationWarning, stacklevel=2)
|
||||
return super(BaseItem, cls).__new__(cls, *args, **kwargs)
|
||||
|
|
|
|||
|
|
@ -6,6 +6,8 @@ See documentation in docs/topics/loaders.rst
|
|||
from collections import defaultdict
|
||||
from contextlib import suppress
|
||||
|
||||
from itemadapter import ItemAdapter
|
||||
|
||||
from scrapy.item import Item
|
||||
from scrapy.loader.common import wrap_loader_context
|
||||
from scrapy.loader.processors import Identity
|
||||
|
|
@ -44,7 +46,7 @@ class ItemLoader:
|
|||
self._local_item = context['item'] = item
|
||||
self._local_values = defaultdict(list)
|
||||
# values from initial item
|
||||
for field_name, value in item.items():
|
||||
for field_name, value in ItemAdapter(item).items():
|
||||
self._values[field_name] += arg_to_iter(value)
|
||||
|
||||
@property
|
||||
|
|
@ -127,13 +129,12 @@ class ItemLoader:
|
|||
return value
|
||||
|
||||
def load_item(self):
|
||||
item = self.item
|
||||
adapter = ItemAdapter(self.item)
|
||||
for field_name in tuple(self._values):
|
||||
value = self.get_output_value(field_name)
|
||||
if value is not None:
|
||||
item[field_name] = value
|
||||
|
||||
return item
|
||||
adapter[field_name] = value
|
||||
return adapter.item
|
||||
|
||||
def get_output_value(self, field_name):
|
||||
proc = self.get_output_processor(field_name)
|
||||
|
|
@ -174,11 +175,8 @@ class ItemLoader:
|
|||
value, type(e).__name__, str(e)))
|
||||
|
||||
def _get_item_field_attr(self, field_name, key, default=None):
|
||||
if isinstance(self.item, Item):
|
||||
value = self.item.fields[field_name].get(key, default)
|
||||
else:
|
||||
value = default
|
||||
return value
|
||||
field_meta = ItemAdapter(self.item).get_field_meta(field_name)
|
||||
return field_meta.get(key, default)
|
||||
|
||||
def _check_selector_method(self):
|
||||
if self.selector is None:
|
||||
|
|
|
|||
|
|
@ -10,24 +10,26 @@ import mimetypes
|
|||
import os
|
||||
import time
|
||||
from collections import defaultdict
|
||||
from email.utils import parsedate_tz, mktime_tz
|
||||
from contextlib import suppress
|
||||
from email.utils import mktime_tz, parsedate_tz
|
||||
from ftplib import FTP
|
||||
from io import BytesIO
|
||||
from urllib.parse import urlparse
|
||||
|
||||
from itemadapter import ItemAdapter
|
||||
from twisted.internet import defer, threads
|
||||
|
||||
from scrapy.exceptions import IgnoreRequest, NotConfigured
|
||||
from scrapy.http import Request
|
||||
from scrapy.pipelines.media import MediaPipeline
|
||||
from scrapy.settings import Settings
|
||||
from scrapy.exceptions import NotConfigured, IgnoreRequest
|
||||
from scrapy.http import Request
|
||||
from scrapy.utils.misc import md5sum
|
||||
from scrapy.utils.log import failure_to_exc_info
|
||||
from scrapy.utils.python import to_bytes
|
||||
from scrapy.utils.request import referer_str
|
||||
from scrapy.utils.boto import is_botocore
|
||||
from scrapy.utils.datatypes import CaselessDict
|
||||
from scrapy.utils.ftp import ftp_store_file
|
||||
from scrapy.utils.log import failure_to_exc_info
|
||||
from scrapy.utils.misc import md5sum
|
||||
from scrapy.utils.python import to_bytes
|
||||
from scrapy.utils.request import referer_str
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
|
@ -517,7 +519,8 @@ class FilesPipeline(MediaPipeline):
|
|||
|
||||
# Overridable Interface
|
||||
def get_media_requests(self, item, info):
|
||||
return [Request(x) for x in item.get(self.files_urls_field, [])]
|
||||
urls = ItemAdapter(item).get(self.files_urls_field, [])
|
||||
return [Request(u) for u in urls]
|
||||
|
||||
def file_downloaded(self, response, request, info):
|
||||
path = self.file_path(request, response=response, info=info)
|
||||
|
|
@ -528,8 +531,8 @@ class FilesPipeline(MediaPipeline):
|
|||
return checksum
|
||||
|
||||
def item_completed(self, results, item, info):
|
||||
if isinstance(item, dict) or self.files_result_field in item.fields:
|
||||
item[self.files_result_field] = [x for ok, x in results if ok]
|
||||
with suppress(KeyError):
|
||||
ItemAdapter(item)[self.files_result_field] = [x for ok, x in results if ok]
|
||||
return item
|
||||
|
||||
def file_path(self, request, response=None, info=None):
|
||||
|
|
|
|||
|
|
@ -5,17 +5,19 @@ See documentation in topics/media-pipeline.rst
|
|||
"""
|
||||
import functools
|
||||
import hashlib
|
||||
from contextlib import suppress
|
||||
from io import BytesIO
|
||||
|
||||
from itemadapter import ItemAdapter
|
||||
from PIL import Image
|
||||
|
||||
from scrapy.exceptions import DropItem
|
||||
from scrapy.http import Request
|
||||
from scrapy.pipelines.files import FileException, FilesPipeline
|
||||
# TODO: from scrapy.pipelines.media import MediaPipeline
|
||||
from scrapy.settings import Settings
|
||||
from scrapy.utils.misc import md5sum
|
||||
from scrapy.utils.python import to_bytes
|
||||
from scrapy.http import Request
|
||||
from scrapy.settings import Settings
|
||||
from scrapy.exceptions import DropItem
|
||||
# TODO: from scrapy.pipelines.media import MediaPipeline
|
||||
from scrapy.pipelines.files import FileException, FilesPipeline
|
||||
|
||||
|
||||
class NoimagesDrop(DropItem):
|
||||
|
|
@ -157,11 +159,12 @@ class ImagesPipeline(FilesPipeline):
|
|||
return image, buf
|
||||
|
||||
def get_media_requests(self, item, info):
|
||||
return [Request(x) for x in item.get(self.images_urls_field, [])]
|
||||
urls = ItemAdapter(item).get(self.images_urls_field, [])
|
||||
return [Request(u) for u in urls]
|
||||
|
||||
def item_completed(self, results, item, info):
|
||||
if isinstance(item, dict) or self.images_result_field in item.fields:
|
||||
item[self.images_result_field] = [x for ok, x in results if ok]
|
||||
with suppress(KeyError):
|
||||
ItemAdapter(item)[self.images_result_field] = [x for ok, x in results if ok]
|
||||
return item
|
||||
|
||||
def file_path(self, request, response=None, info=None):
|
||||
|
|
|
|||
|
|
@ -6,6 +6,7 @@ See documentation in docs/topics/shell.rst
|
|||
import os
|
||||
import signal
|
||||
|
||||
from itemadapter import is_item
|
||||
from twisted.internet import threads, defer
|
||||
from twisted.python import threadable
|
||||
from w3lib.url import any_to_uri
|
||||
|
|
@ -13,20 +14,18 @@ from w3lib.url import any_to_uri
|
|||
from scrapy.crawler import Crawler
|
||||
from scrapy.exceptions import IgnoreRequest
|
||||
from scrapy.http import Request, Response
|
||||
from scrapy.item import _BaseItem
|
||||
from scrapy.settings import Settings
|
||||
from scrapy.spiders import Spider
|
||||
from scrapy.utils.console import start_python_console
|
||||
from scrapy.utils.conf import get_config
|
||||
from scrapy.utils.console import DEFAULT_PYTHON_SHELLS, start_python_console
|
||||
from scrapy.utils.datatypes import SequenceExclude
|
||||
from scrapy.utils.misc import load_object
|
||||
from scrapy.utils.response import open_in_browser
|
||||
from scrapy.utils.conf import get_config
|
||||
from scrapy.utils.console import DEFAULT_PYTHON_SHELLS
|
||||
|
||||
|
||||
class Shell:
|
||||
|
||||
relevant_classes = (Crawler, Spider, Request, Response, _BaseItem, Settings)
|
||||
relevant_classes = (Crawler, Spider, Request, Response, Settings)
|
||||
|
||||
def __init__(self, crawler, update_vars=None, code=None):
|
||||
self.crawler = crawler
|
||||
|
|
@ -154,7 +153,7 @@ class Shell:
|
|||
return "\n".join("[s] %s" % line for line in b)
|
||||
|
||||
def _is_relevant(self, value):
|
||||
return isinstance(value, self.relevant_classes)
|
||||
return isinstance(value, self.relevant_classes) or is_item(value)
|
||||
|
||||
|
||||
def inspect_response(response, spider):
|
||||
|
|
|
|||
|
|
@ -11,9 +11,14 @@ from scrapy.http import Request
|
|||
from scrapy.utils.trackref import object_ref
|
||||
from scrapy.utils.url import url_is_from_spider
|
||||
from scrapy.utils.deprecate import method_is_overridden
|
||||
from scrapy.utils.python import iflatten
|
||||
|
||||
|
||||
WaitUntilQueueEmpty = object()
|
||||
|
||||
|
||||
class Spider(object_ref):
|
||||
|
||||
"""Base class for scrapy spiders. All spiders must inherit from this
|
||||
class.
|
||||
"""
|
||||
|
|
@ -76,6 +81,11 @@ class Spider(object_ref):
|
|||
for url in self.start_urls:
|
||||
yield Request(url, dont_filter=True)
|
||||
|
||||
def start_requests_with_control(self):
|
||||
sig = WaitUntilQueueEmpty
|
||||
gen_ = ((r, sig) if r != sig else (r,) for r in self.start_requests())
|
||||
return iflatten(gen_)
|
||||
|
||||
def make_requests_from_url(self, url):
|
||||
""" This method is deprecated. """
|
||||
warnings.warn(
|
||||
|
|
|
|||
|
|
@ -31,7 +31,7 @@ class XMLFeedSpider(Spider):
|
|||
processing required before returning the results to the framework core,
|
||||
for example setting the item GUIDs. It receives a list of results and
|
||||
the response which originated that results. It must return a list of
|
||||
results (Items or Requests).
|
||||
results (items or requests).
|
||||
"""
|
||||
return results
|
||||
|
||||
|
|
|
|||
|
|
@ -5,6 +5,9 @@
|
|||
|
||||
from scrapy import signals
|
||||
|
||||
# useful for handling different item types with a single interface
|
||||
from itemadapter import is_item, ItemAdapter
|
||||
|
||||
|
||||
class ${ProjectName}SpiderMiddleware:
|
||||
# Not all methods need to be defined. If a method is not defined,
|
||||
|
|
@ -29,7 +32,7 @@ class ${ProjectName}SpiderMiddleware:
|
|||
# Called with the results returned from the Spider, after
|
||||
# it has processed the response.
|
||||
|
||||
# Must return an iterable of Request, dict or Item objects.
|
||||
# Must return an iterable of Request, or item objects.
|
||||
for i in result:
|
||||
yield i
|
||||
|
||||
|
|
@ -37,8 +40,7 @@ class ${ProjectName}SpiderMiddleware:
|
|||
# Called when a spider or process_spider_input() method
|
||||
# (from other spider middleware) raises an exception.
|
||||
|
||||
# Should return either None or an iterable of Request, dict
|
||||
# or Item objects.
|
||||
# Should return either None or an iterable of Request or item objects.
|
||||
pass
|
||||
|
||||
def process_start_requests(self, start_requests, spider):
|
||||
|
|
|
|||
|
|
@ -4,6 +4,10 @@
|
|||
# See: https://docs.scrapy.org/en/latest/topics/item-pipeline.html
|
||||
|
||||
|
||||
# useful for handling different item types with a single interface
|
||||
from itemadapter import ItemAdapter
|
||||
|
||||
|
||||
class ${ProjectName}Pipeline:
|
||||
def process_item(self, item, spider):
|
||||
return item
|
||||
|
|
|
|||
|
|
@ -17,8 +17,8 @@ class CurlParser(argparse.ArgumentParser):
|
|||
curl_parser = CurlParser()
|
||||
curl_parser.add_argument('url')
|
||||
curl_parser.add_argument('-H', '--header', dest='headers', action='append')
|
||||
curl_parser.add_argument('-X', '--request', dest='method', default='get')
|
||||
curl_parser.add_argument('-d', '--data', dest='data')
|
||||
curl_parser.add_argument('-X', '--request', dest='method')
|
||||
curl_parser.add_argument('-d', '--data', '--data-raw', dest='data')
|
||||
curl_parser.add_argument('-u', '--user', dest='auth')
|
||||
|
||||
|
||||
|
|
@ -66,7 +66,9 @@ def curl_to_request_kwargs(curl_command, ignore_unknown_options=True):
|
|||
if not parsed_url.scheme:
|
||||
url = 'http://' + url
|
||||
|
||||
result = {'method': parsed_args.method.upper(), 'url': url}
|
||||
method = parsed_args.method or 'GET'
|
||||
|
||||
result = {'method': method.upper(), 'url': url}
|
||||
|
||||
headers = []
|
||||
cookies = {}
|
||||
|
|
@ -90,5 +92,9 @@ def curl_to_request_kwargs(curl_command, ignore_unknown_options=True):
|
|||
result['cookies'] = cookies
|
||||
if parsed_args.data:
|
||||
result['body'] = parsed_args.data
|
||||
if not parsed_args.method:
|
||||
# if the "data" is specified but the "method" is not specified,
|
||||
# the default method is 'POST'
|
||||
result['method'] = 'POST'
|
||||
|
||||
return result
|
||||
|
|
|
|||
|
|
@ -138,8 +138,9 @@ def create_instance(objcls, settings, crawler, *args, **kwargs):
|
|||
|
||||
Raises ``ValueError`` if both ``settings`` and ``crawler`` are ``None``.
|
||||
|
||||
Raises ``TypeError`` if the resulting instance is ``None`` (e.g. if an
|
||||
extension has not been implemented correctly).
|
||||
.. versionchanged:: 2.2
|
||||
Raises ``TypeError`` if the resulting instance is ``None`` (e.g. if an
|
||||
extension has not been implemented correctly).
|
||||
"""
|
||||
if settings is None:
|
||||
if crawler is None:
|
||||
|
|
|
|||
|
|
@ -334,10 +334,10 @@ class MutableChain:
|
|||
"""
|
||||
|
||||
def __init__(self, *args):
|
||||
self.data = chain(*args)
|
||||
self.data = chain.from_iterable(args)
|
||||
|
||||
def extend(self, *iterables):
|
||||
self.data = chain(self.data, *iterables)
|
||||
self.data = chain(self.data, chain.from_iterable(iterables))
|
||||
|
||||
def __iter__(self):
|
||||
return self
|
||||
|
|
|
|||
|
|
@ -2,10 +2,10 @@ import json
|
|||
import datetime
|
||||
import decimal
|
||||
|
||||
from itemadapter import is_item, ItemAdapter
|
||||
from twisted.internet import defer
|
||||
|
||||
from scrapy.http import Request, Response
|
||||
from scrapy.item import _BaseItem
|
||||
|
||||
|
||||
class ScrapyJSONEncoder(json.JSONEncoder):
|
||||
|
|
@ -26,8 +26,8 @@ class ScrapyJSONEncoder(json.JSONEncoder):
|
|||
return str(o)
|
||||
elif isinstance(o, defer.Deferred):
|
||||
return str(o)
|
||||
elif isinstance(o, _BaseItem):
|
||||
return dict(o)
|
||||
elif is_item(o):
|
||||
return ItemAdapter(o).asdict()
|
||||
elif isinstance(o, Request):
|
||||
return "<%s %s %s>" % (type(o).__name__, o.method, o.url)
|
||||
elif isinstance(o, Response):
|
||||
|
|
|
|||
171
setup.cfg
171
setup.cfg
|
|
@ -3,3 +3,174 @@ doc_files = docs AUTHORS INSTALL LICENSE README.rst
|
|||
|
||||
[bdist_wheel]
|
||||
universal=1
|
||||
|
||||
[mypy]
|
||||
ignore_missing_imports = true
|
||||
follow_imports = skip
|
||||
|
||||
# FIXME: remove the following sections once the issues are solved
|
||||
|
||||
[mypy-scrapy]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-scrapy._monkeypatches]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-scrapy.commands]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-scrapy.commands.bench]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-scrapy.commands.parse]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-scrapy.downloadermiddlewares.httpproxy]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-scrapy.contracts]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-scrapy.core.spidermw]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-scrapy.interfaces]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-scrapy.item]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-scrapy.http.cookies]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-scrapy.mail]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-scrapy.pipelines.images]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-scrapy.settings.default_settings]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-scrapy.spidermiddlewares.referer]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-scrapy.utils.httpobj]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-scrapy.utils.request]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-scrapy.utils.response]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-scrapy.utils.spider]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-scrapy.utils.trackref]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.mocks.dummydbm]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.spiders]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_cmdline_crawl_with_pipeline.test_spider.spiders.exception]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_cmdline_crawl_with_pipeline.test_spider.spiders.normal]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_command_fetch]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_command_parse]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_command_shell]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_command_version]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_contracts]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_crawler]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_downloader_handlers]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_engine]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_exporters]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_http_request]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_linkextractors]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_loader]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_pipeline_crawl]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_pipeline_files]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_pipeline_images]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_pipelines]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_request_cb_kwargs]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_request_left]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_scheduler]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_signals]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_spiderloader.test_spiders.nested.spider4]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_spiderloader.test_spiders.spider1]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_spiderloader.test_spiders.spider2]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_spiderloader.test_spiders.spider3]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_spidermiddleware_httperror]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_spidermiddleware_output_chain]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_spidermiddleware_referer]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_utils_reqser]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_utils_serialize]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_utils_spider]
|
||||
ignore_errors = True
|
||||
|
||||
[mypy-tests.test_utils_url]
|
||||
ignore_errors = True
|
||||
|
|
|
|||
3
setup.py
3
setup.py
|
|
@ -66,7 +66,7 @@ setup(
|
|||
'Topic :: Software Development :: Libraries :: Application Frameworks',
|
||||
'Topic :: Software Development :: Libraries :: Python Modules',
|
||||
],
|
||||
python_requires='>=3.5',
|
||||
python_requires='>=3.5.2',
|
||||
install_requires=[
|
||||
'Twisted>=17.9.0',
|
||||
'cryptography>=2.0',
|
||||
|
|
@ -80,6 +80,7 @@ setup(
|
|||
'w3lib>=1.17.0',
|
||||
'zope.interface>=4.1.3',
|
||||
'protego>=0.1.15',
|
||||
'itemadapter>=0.1.0',
|
||||
],
|
||||
extras_require=extras_require,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -1,20 +1,20 @@
|
|||
-----BEGIN CERTIFICATE-----
|
||||
MIIDNzCCAh+gAwIBAgIJANWqWyPdTY8CMA0GCSqGSIb3DQEBCwUAMDIxCzAJBgNV
|
||||
BAYTAklFMQ8wDQYDVQQKDAZTY3JhcHkxEjAQBgNVBAMMCWxvY2FsaG9zdDAeFw0x
|
||||
NzA0MjcxNzQxNTdaFw0xODA0MjcxNzQxNTdaMDIxCzAJBgNVBAYTAklFMQ8wDQYD
|
||||
VQQKDAZTY3JhcHkxEjAQBgNVBAMMCWxvY2FsaG9zdDCCASIwDQYJKoZIhvcNAQEB
|
||||
BQADggEPADCCAQoCggEBAK1jcwlJ+bpr63lmK1mSk83nduF+27EPTU3RyteoPM2K
|
||||
o/RqZnr/mR29U6Pu42YuhLvBUu7rQxGi+rgkwno6lMFP4y5glxRygIlPsP4WQO3Y
|
||||
njmysWfYxQoIml2A+tiLewrMZocHI2cNgrO8Fd0u7KMiLlvUCN0pVyOwZ/ym9rPY
|
||||
ObfquG/xYTFzgYD/wy1n4AXE4ve3uZPfB3ZGtB3fUmuowg5KZ1L3uWpviyqr1qB/
|
||||
8NXcORLegAPsquLA05gnDPOuMs7dSMeKMphvpbSerRXLGxLIfWOZ0rs8oV96Re52
|
||||
gSEg/kIIS+ts37sJofcEnx9C4FkTR8zXin9eZhgCYs0CAwEAAaNQME4wHQYDVR0O
|
||||
BBYEFOoYbg0MvcnbTN0jxISsP2ctMbjpMB8GA1UdIwQYMBaAFOoYbg0MvcnbTN0j
|
||||
xISsP2ctMbjpMAwGA1UdEwQFMAMBAf8wDQYJKoZIhvcNAQELBQADggEBAF/JlzES
|
||||
9Z3Azaj60gvJHyPJsPSM4tUfnWoFfFrui3oPG5TJPxWqrLBsTEachUTKOd5+XR2i
|
||||
jxUuREMkcRjbc0jjsqhsxPvfgrUrbIvKjEFLfAPvvLvcQIMUJf09SEjaaMkUAYd+
|
||||
TJaxFn5kd9Q6HbkD/fEN+lKhNZI40IJvfu7u4emUj3uKy9zrw576/T8aDYUl/own
|
||||
tqqfXh/jN8wnKCQwma7gaPmMOMqBt6zCsrN9/eKnMBpdULkUtjJD4NDg03XUFLlM
|
||||
am/oQ+MnasCcctkaXKbTGx3WfBVmkGj4b3Au18CVZkRWN2QsMdBC8JLRTICKse8U
|
||||
Mjybr/hQK3mnVdE=
|
||||
MIIDRTCCAi2gAwIBAgIUGoISfeW3LwSWHC52ORXdZY9pNLswDQYJKoZIhvcNAQEL
|
||||
BQAwMjELMAkGA1UEBhMCSUUxDzANBgNVBAoMBlNjcmFweTESMBAGA1UEAwwJbG9j
|
||||
YWxob3N0MB4XDTIwMDYyODEyNTQxNVoXDTIxMDYyODEyNTQxNVowMjELMAkGA1UE
|
||||
BhMCSUUxDzANBgNVBAoMBlNjcmFweTESMBAGA1UEAwwJbG9jYWxob3N0MIIBIjAN
|
||||
BgkqhkiG9w0BAQEFAAOCAQ8AMIIBCgKCAQEAvCLxfTEQuIdf8JhiHrbVkGHYrNSK
|
||||
2XD2TCPaSIpJ2KKlFUrIz3A9tWlOfLnWabS5od89yOebhYj4DN/Qm2TViGg1mtWe
|
||||
pD1K2YWd1Af+hhAw5D+TpW2RH9TVhX7Ey5osWcl+0uy+RlKZE8qum72xi1vxWOmH
|
||||
wYw06iN8klQ3JfP2/eLRXBQjsh7WW0dbJ7yLvG6UFz1RbhFTtlxeIMenzNsHaMg7
|
||||
56Ru57/MMbaBwdBttXVzJDQ7imo8njuxDMszliC/QgIdBUBFzA2LB5qpr+v+laDN
|
||||
cN9t9Q9stsu446dFnRoofxJjMFW7lLu6h/lwP5r0kfeUkMDhXJ4mb6KwfwIDAQAB
|
||||
o1MwUTAdBgNVHQ4EFgQUVEdXn8ha2FA73zcy1Ia0FQMzMEYwHwYDVR0jBBgwFoAU
|
||||
VEdXn8ha2FA73zcy1Ia0FQMzMEYwDwYDVR0TAQH/BAUwAwEB/zANBgkqhkiG9w0B
|
||||
AQsFAAOCAQEAZpGBPsexMD+IwcMNIgc7FiaJsb8E30C9vWxgdnkpapi9zLJ4yiHQ
|
||||
VxkV9RTezUEADkaDj+2qFveamWTzJLnphgaaUpVeMcYACPhRVOYXidNrZyTmHIsX
|
||||
FwaTzAggW6CP7JxAcpxH0f9+NWFCZI36FihRdwuWyvrUl7rsXaexu0SOI/Ck0oWf
|
||||
2IW+jo67TSmcbte+J8wq77DX32mVLb/2nqpItH4T2Di+XjVBARACVOSdgdlo7lZE
|
||||
W8mSEXqP2BVx8JGG8X1znNLHcmjVj4EtkpH0wkYzpC4cvGkTsUcU7CU7ZyVUp+Bb
|
||||
dPMVxyRKWfAjRJc8o5Ot1mgHrx5coOtzAA==
|
||||
-----END CERTIFICATE-----
|
||||
|
|
|
|||
|
|
@ -1,28 +1,28 @@
|
|||
-----BEGIN PRIVATE KEY-----
|
||||
MIIEvQIBADANBgkqhkiG9w0BAQEFAASCBKcwggSjAgEAAoIBAQCtY3MJSfm6a+t5
|
||||
ZitZkpPN53bhftuxD01N0crXqDzNiqP0amZ6/5kdvVOj7uNmLoS7wVLu60MRovq4
|
||||
JMJ6OpTBT+MuYJcUcoCJT7D+FkDt2J45srFn2MUKCJpdgPrYi3sKzGaHByNnDYKz
|
||||
vBXdLuyjIi5b1AjdKVcjsGf8pvaz2Dm36rhv8WExc4GA/8MtZ+AFxOL3t7mT3wd2
|
||||
RrQd31JrqMIOSmdS97lqb4sqq9agf/DV3DkS3oAD7KriwNOYJwzzrjLO3UjHijKY
|
||||
b6W0nq0VyxsSyH1jmdK7PKFfekXudoEhIP5CCEvrbN+7CaH3BJ8fQuBZE0fM14p/
|
||||
XmYYAmLNAgMBAAECggEAQKY4GlqO1seugRFrUHaqzbdkSCf42kgOVtnGfCqqoSj0
|
||||
gQm7NFlhSglxykokV9E4hJlMxvDJjSXrvgVWziRRmtKiroQtUN5wtsIUCGlbxFNk
|
||||
i7bpFwNoVJlolTymS1+WfSxBfk9XD/GlrkaPEG2SpjD0gCDLPUtQxmncHARVMDDu
|
||||
Eysk3njGghsTF7XMh8ljTE3CqqNSx9BkeWQr6EYfXcgaQ2jp9E+FspB5+KWeO4ss
|
||||
ELVHgtwmYSRPAEuz4XHz87RLuakqafko6ftvh3upVQwm0VXuwM+lEUYZrzoU2JQ4
|
||||
hePKHRaWQC4tawV6FyVHK4X0MuKP4uESr7YHbJ03sQKBgQDV4CyQU6xccW6hMxlD
|
||||
7hvrGcPQEPg6M4rX2uqWpB6RCh6stZEydYeh5S+A6ltml/2csw9Bl8nZM6KbArZa
|
||||
EKrZcOn7JgFyPpiDHqgEIx+9XL/mnsKMSkBKTFcvucVgjIWE8GT7jfAqMkcSysWf
|
||||
uRyUvtNpshmRLcdNhEjrr3vcwwKBgQDPid6sxBVcoyvrYUsRRVpXATJ9tsmU93LG
|
||||
HMHDlXkZ2CMfEuA0xLK+B9iyHMhh8NwYFjcG5oeVyVjE8SbifX4Sg49hde8ykXSR
|
||||
UBSNt22/JaWgreL95LEC/y9q+G4osli7NwRW1x6tB5cN1mE0hZI8Z0ETvyr3DoWO
|
||||
j/dbdFYJLwKBgDjVLCJiCbA6+EHfuTwC3upXW2BD0iJtJdz8MFA9Zl32SXZtfRri
|
||||
fls38qqYHBekFeF493nfouSTwwbb7qb6PNwxFAwH6mR4W8Cj+dO3nayNI/VdhKcQ
|
||||
6AqWRKjK/bcNQEG2O69Y5VPhLl/BAEjUQNMJ7lXs3LxmZMqld1cht5FPAoGBAJbI
|
||||
xXbiU97lUmCGZKLcr4EtBoEdz6GiksnrVMAEFmM3jHTkIu9TxcWZL9BgZxn5g/8g
|
||||
DMS/styZ2BvmVWkS4gkTepXFuI8V7Qoyk2xPS7Yn5QkzrQroH89clhfy/R4mTZ9f
|
||||
npB1ZP0z2YSdMCyXqyKlpjtxlga/jzt/z6irgmLTAoGAPrmudajtSBq534Ql2lPM
|
||||
8U6baRSAMMzV7MXcR8F1CRewQiYOzlgsB8toELNtjg1IGPqmoiNDDKmkHs3R2mO6
|
||||
J45kDPLFe9DTyZLZj0pWWK6yRLc/BA/gGzKFpMkNcyzLlQjNPqY/9mrrYea4J9Cj
|
||||
Z+pMCFLbwAbFZ9Qb/NFlUv0=
|
||||
MIIEvgIBADANBgkqhkiG9w0BAQEFAASCBKgwggSkAgEAAoIBAQC8IvF9MRC4h1/w
|
||||
mGIettWQYdis1IrZcPZMI9pIiknYoqUVSsjPcD21aU58udZptLmh3z3I55uFiPgM
|
||||
39CbZNWIaDWa1Z6kPUrZhZ3UB/6GEDDkP5OlbZEf1NWFfsTLmixZyX7S7L5GUpkT
|
||||
yq6bvbGLW/FY6YfBjDTqI3ySVDcl8/b94tFcFCOyHtZbR1snvIu8bpQXPVFuEVO2
|
||||
XF4gx6fM2wdoyDvnpG7nv8wxtoHB0G21dXMkNDuKajyeO7EMyzOWIL9CAh0FQEXM
|
||||
DYsHmqmv6/6VoM1w3231D2y2y7jjp0WdGih/EmMwVbuUu7qH+XA/mvSR95SQwOFc
|
||||
niZvorB/AgMBAAECggEAHVpSVRb/pdqxNEeCH4qlHWa2uJhcpXpDYzPAzcqNpPgT
|
||||
S5QkaoD3j8NDVKBl/I4O3FuJNzwzfo0VLmUJFgWQbzzbCDJGExfhArkfG8K3ilEi
|
||||
X6ovrgK/PrklKzPRHncKbmPKnrwDH9OpQHZB8diRx81rhVTCModehh1NRUNQa2I1
|
||||
QzFC7uyXx3duoIsI5QXVeEGuwHZfqIY/z+9SscdVFL6elXTPFUzBzcmAqQgdgWKN
|
||||
HXgX22LE0rAu8NnRvOZZWt4/nOjvlCFCPTB11NgthmKlVnsx4H7gpQ2OPh4bZ+0W
|
||||
birVEtZ3E1jxoGvw1FzxyqqpGkcanRMa8QWzK4JwuQKBgQDrgclpkqZrgHB/TC1p
|
||||
hLvsdflGI2SGs+c/mYR3GEjf0kJtI88WL5fj1QezdkDyOpwxFvnLslswfzdtzvis
|
||||
vksGysV35vhMPQUcmWhvzA7Pdxdv4BZr+ckER0SAYBBxg9KYZyxewGb5XzB8Cz2o
|
||||
8V+YpwrMAOYGuXHTfafv4CKlTQKBgQDMgetvV9/E3HNtKsATiPIwT3e1MzyPXigq
|
||||
12NkHSZa6s4yqm/h/fSUn54sJbhx+OtRRhktOo0aB34tcogtrJyClvCPdRAP/4Qi
|
||||
M43FjKo2cWiubWvtWlOZU04bpClG324q420rK7dCA2stID/Fa0sMQgAAyPH8TGMo
|
||||
gbvyrk4W+wKBgQDMIOnYZTF0epaH8BponJFaqwMOhTzr+OGW4dTMebMotZG4EdK8
|
||||
kzIfW5XaOsSecKjTb+vCYGzkA1CjEEPBTwuu7nDstblAM5/Lozi/tmqb7sjUwrIM
|
||||
kyxmVfONJjb6fV07lioCUtiui5B15DRkzBqlMRyNqLW43GJKA19d7rN4/QKBgCzy
|
||||
kRBTu/bEjQn9T2H7w18i2CiXLkREaYeg91NVpMxutwsjspt0+YCA5H7He5ZxIycl
|
||||
xPrP15tU8kKC3bNMMMny6sRc8j7R5fSuaAZ3OCHnIx7TJdlw9NbKHGyu0/Ojv87l
|
||||
VWUbopd7sN6mK930CvaSuvVxNN5C27hXazuXW8ppAoGBANcWsenNKpCJgF0cNPHX
|
||||
abPaWfcs5FKMNz8gEdGk3B1z/KBpYz59smPwurYVCXaWE6iv99sDOP7CVneF02sV
|
||||
SqyNzVhcVSG788uB3CwnpEvm7ydoH89L5dvYekAHP8RJulhWCK45lXkHLiYGKvhv
|
||||
PWuPk5VX+qF78JhUhPO3nfnu
|
||||
-----END PRIVATE KEY-----
|
||||
|
|
|
|||
|
|
@ -0,0 +1,25 @@
|
|||
from scrapy.http import Request
|
||||
|
||||
|
||||
class RequestInOrderMiddleware:
|
||||
def process_spider_output(self, response, result, spider):
|
||||
return (self._preserve_in_order(r, spider) for r in result or ())
|
||||
|
||||
def process_start_requests(self, start_requests, spider):
|
||||
return (self._preserve_in_order(r, spider) for r in start_requests or ())
|
||||
|
||||
def _preserve_in_order(self, smth, spider):
|
||||
self.__preserve_in_order(smth, spider)
|
||||
return smth
|
||||
|
||||
def __preserve_in_order(self, smth, spider):
|
||||
|
||||
if not isinstance(smth, Request):
|
||||
return
|
||||
|
||||
request = smth
|
||||
|
||||
if getattr(spider, 'requests_in_order_of_scheduling', None) is None:
|
||||
setattr(spider, 'requests_in_order_of_scheduling', list())
|
||||
|
||||
spider.requests_in_order_of_scheduling.append(request)
|
||||
|
|
@ -1,4 +1,6 @@
|
|||
# Tests requirements
|
||||
attrs
|
||||
dataclasses; python_version == '3.6'
|
||||
jmespath
|
||||
mitmproxy; python_version >= '3.6'
|
||||
mitmproxy<4.0.0; python_version < '3.6'
|
||||
|
|
|
|||
|
|
@ -177,33 +177,29 @@ class ErrorSpider(FollowAllSpider):
|
|||
self.raise_exception()
|
||||
|
||||
|
||||
class YieldingRequestsSpider(FollowAllSpider):
|
||||
number_of_start_requests = 10
|
||||
|
||||
def start_requests(self):
|
||||
for s in range(self.number_of_start_requests):
|
||||
qargs = {'total': 10, 'seed': s}
|
||||
url = self.mockserver.url("/follow?%s") % urlencode(qargs, doseq=1)
|
||||
yield Request(url, meta={'seed': s})
|
||||
|
||||
|
||||
class BrokenStartRequestsSpider(FollowAllSpider):
|
||||
|
||||
fail_before_yield = False
|
||||
fail_yielding = False
|
||||
|
||||
def __init__(self, *a, **kw):
|
||||
super(BrokenStartRequestsSpider, self).__init__(*a, **kw)
|
||||
self.seedsseen = []
|
||||
fail_before_yield = True
|
||||
fail_yielding = True
|
||||
|
||||
def start_requests(self):
|
||||
if self.fail_before_yield:
|
||||
1 / 0
|
||||
|
||||
for s in range(100):
|
||||
qargs = {'total': 10, 'seed': s}
|
||||
url = self.mockserver.url("/follow?%s") % urlencode(qargs, doseq=1)
|
||||
yield Request(url, meta={'seed': s})
|
||||
for r in super().start_requests():
|
||||
yield r
|
||||
if self.fail_yielding:
|
||||
2 / 0
|
||||
|
||||
assert self.seedsseen, 'All start requests consumed before any download happened'
|
||||
|
||||
def parse(self, response):
|
||||
self.seedsseen.append(response.meta.get('seed'))
|
||||
for req in super(BrokenStartRequestsSpider, self).parse(response):
|
||||
yield req
|
||||
|
||||
|
||||
class SingleRequestSpider(MetaSpider):
|
||||
|
||||
|
|
|
|||
|
|
@ -34,6 +34,7 @@ from tests.spiders import (
|
|||
FollowAllSpider,
|
||||
SimpleSpider,
|
||||
SingleRequestSpider,
|
||||
YieldingRequestsSpider,
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -160,14 +161,69 @@ class CrawlTestCase(TestCase):
|
|||
self.assertIsNotNone(record.exc_info)
|
||||
self.assertIs(record.exc_info[0], ZeroDivisionError)
|
||||
|
||||
def is_eager(_, requests_in_order):
|
||||
"""
|
||||
All start requests(depth=0) are scheduled before
|
||||
any other requests(depth!=0)
|
||||
"""
|
||||
depths_in_order = [r.meta.get('depth', 0) for r in requests_in_order]
|
||||
order_of_start_requests = [
|
||||
o for o, d in enumerate(depths_in_order) if d == 0
|
||||
]
|
||||
order_of_other_requests = [
|
||||
o for o, d in enumerate(depths_in_order) if d != 0
|
||||
]
|
||||
last_start_request = max(order_of_start_requests)
|
||||
first_other_request = min(order_of_other_requests)
|
||||
return last_start_request < first_other_request
|
||||
|
||||
def number_of_start_requests_for_eagernees_testing(_):
|
||||
"""
|
||||
number_of_start_requests should be big enough, so scheduling such
|
||||
amount of requests takes longer than crawling first of them
|
||||
"""
|
||||
return 100
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_start_requests_eagerness(self):
|
||||
class EagerSpider(YieldingRequestsSpider):
|
||||
def start_requests_with_control(self):
|
||||
yield from self.start_requests()
|
||||
|
||||
settings = {
|
||||
"SPIDER_MIDDLEWARES": {
|
||||
"scrapy.spidermiddlewares.depth.DepthMiddleware": 0,
|
||||
"tests.middlewares.RequestInOrderMiddleware": 1,
|
||||
}
|
||||
}
|
||||
crawler = CrawlerRunner(settings).create_crawler(EagerSpider)
|
||||
yield crawler.crawl(
|
||||
mockserver=self.mockserver,
|
||||
number_of_start_requests=self.number_of_start_requests_for_eagernees_testing(),
|
||||
)
|
||||
self.assertTrue(
|
||||
self.is_eager(crawler.spider.requests_in_order_of_scheduling)
|
||||
)
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_start_requests_lazyness(self):
|
||||
settings = {"CONCURRENT_REQUESTS": 1}
|
||||
crawler = CrawlerRunner(settings).create_crawler(BrokenStartRequestsSpider)
|
||||
yield crawler.crawl(mockserver=self.mockserver)
|
||||
self.assertTrue(
|
||||
crawler.spider.seedsseen.index(None) < crawler.spider.seedsseen.index(99),
|
||||
crawler.spider.seedsseen)
|
||||
"""
|
||||
lazyness as a negation of eagerness
|
||||
"""
|
||||
settings = {
|
||||
"SPIDER_MIDDLEWARES": {
|
||||
"scrapy.spidermiddlewares.depth.DepthMiddleware": 0,
|
||||
"tests.middlewares.RequestInOrderMiddleware": 1,
|
||||
}
|
||||
}
|
||||
crawler = CrawlerRunner(settings).create_crawler(YieldingRequestsSpider)
|
||||
yield crawler.crawl(
|
||||
mockserver=self.mockserver,
|
||||
number_of_start_requests=self.number_of_start_requests_for_eagernees_testing(),
|
||||
)
|
||||
self.assertFalse(
|
||||
self.is_eager(crawler.spider.requests_in_order_of_scheduling)
|
||||
)
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_start_requests_dupes(self):
|
||||
|
|
|
|||
|
|
@ -16,10 +16,12 @@ import sys
|
|||
from collections import defaultdict
|
||||
from urllib.parse import urlparse
|
||||
|
||||
import attr
|
||||
from itemadapter import ItemAdapter
|
||||
from pydispatch import dispatcher
|
||||
from pytest import mark
|
||||
from testfixtures import LogCapture
|
||||
from twisted.internet import reactor, defer
|
||||
from twisted.internet import defer, reactor
|
||||
from twisted.trial import unittest
|
||||
from twisted.web import server, static, util
|
||||
|
||||
|
|
@ -33,7 +35,7 @@ from scrapy.spiders import Spider
|
|||
from scrapy.utils.signal import disconnect_all
|
||||
from scrapy.utils.test import get_crawler
|
||||
|
||||
from tests import tests_datadir, get_testdata
|
||||
from tests import get_testdata, tests_datadir
|
||||
|
||||
|
||||
class TestItem(Item):
|
||||
|
|
@ -42,6 +44,13 @@ class TestItem(Item):
|
|||
price = Field()
|
||||
|
||||
|
||||
@attr.s
|
||||
class AttrsItem:
|
||||
name = attr.ib(default="")
|
||||
url = attr.ib(default="")
|
||||
price = attr.ib(default=0)
|
||||
|
||||
|
||||
class TestSpider(Spider):
|
||||
name = "scrapytest.org"
|
||||
allowed_domains = ["scrapytest.org", "localhost"]
|
||||
|
|
@ -80,6 +89,27 @@ class DictItemsSpider(TestSpider):
|
|||
item_cls = dict
|
||||
|
||||
|
||||
class AttrsItemsSpider(TestSpider):
|
||||
item_class = AttrsItem
|
||||
|
||||
|
||||
try:
|
||||
from dataclasses import make_dataclass
|
||||
except ImportError:
|
||||
DataClassItemsSpider = None
|
||||
else:
|
||||
TestDataClass = make_dataclass("TestDataClass", [("name", str), ("url", str), ("price", int)])
|
||||
|
||||
class DataClassItemsSpider(DictItemsSpider):
|
||||
def parse_item(self, response):
|
||||
item = super().parse_item(response)
|
||||
return TestDataClass(
|
||||
name=item.get('name'),
|
||||
url=item.get('url'),
|
||||
price=item.get('price'),
|
||||
)
|
||||
|
||||
|
||||
class ItemZeroDivisionErrorSpider(TestSpider):
|
||||
custom_settings = {
|
||||
"ITEM_PIPELINES": {
|
||||
|
|
@ -210,7 +240,10 @@ class EngineTest(unittest.TestCase):
|
|||
|
||||
@defer.inlineCallbacks
|
||||
def test_crawler(self):
|
||||
for spider in TestSpider, DictItemsSpider:
|
||||
|
||||
for spider in (TestSpider, DictItemsSpider, AttrsItemsSpider, DataClassItemsSpider):
|
||||
if spider is None:
|
||||
continue
|
||||
self.run = CrawlerRun(spider)
|
||||
yield self.run.run()
|
||||
self._assert_visited_urls()
|
||||
|
|
@ -310,6 +343,7 @@ class EngineTest(unittest.TestCase):
|
|||
def _assert_scraped_items(self):
|
||||
self.assertEqual(2, len(self.run.itemresp))
|
||||
for item, response in self.run.itemresp:
|
||||
item = ItemAdapter(item)
|
||||
self.assertEqual(item['url'], response.url)
|
||||
if 'item1.html' in item['url']:
|
||||
self.assertEqual('Item 1 name', item['name'])
|
||||
|
|
|
|||
|
|
@ -7,6 +7,7 @@ import string
|
|||
import tempfile
|
||||
import warnings
|
||||
from io import BytesIO
|
||||
from logging import getLogger
|
||||
from pathlib import Path
|
||||
from string import ascii_letters, digits
|
||||
from unittest import mock
|
||||
|
|
@ -14,9 +15,11 @@ from urllib.parse import urljoin, urlparse, quote
|
|||
from urllib.request import pathname2url
|
||||
|
||||
import lxml.etree
|
||||
from testfixtures import LogCapture
|
||||
from twisted.internet import defer
|
||||
from twisted.trial import unittest
|
||||
from w3lib.url import path_to_file_uri
|
||||
from w3lib.url import file_uri_to_path, path_to_file_uri
|
||||
from zope.interface import implementer
|
||||
from zope.interface.verify import verifyObject
|
||||
|
||||
import scrapy
|
||||
|
|
@ -390,6 +393,46 @@ class FromCrawlerFileFeedStorage(FileFeedStorage, FromCrawlerMixin):
|
|||
pass
|
||||
|
||||
|
||||
class DummyBlockingFeedStorage(BlockingFeedStorage):
|
||||
|
||||
def __init__(self, uri):
|
||||
self.path = file_uri_to_path(uri)
|
||||
|
||||
def _store_in_thread(self, file):
|
||||
dirname = os.path.dirname(self.path)
|
||||
if dirname and not os.path.exists(dirname):
|
||||
os.makedirs(dirname)
|
||||
with open(self.path, 'ab') as output_file:
|
||||
output_file.write(file.read())
|
||||
|
||||
file.close()
|
||||
|
||||
|
||||
class FailingBlockingFeedStorage(DummyBlockingFeedStorage):
|
||||
|
||||
def _store_in_thread(self, file):
|
||||
raise OSError('Cannot store')
|
||||
|
||||
|
||||
@implementer(IFeedStorage)
|
||||
class LogOnStoreFileStorage:
|
||||
"""
|
||||
This storage logs inside `store` method.
|
||||
It can be used to make sure `store` method is invoked.
|
||||
"""
|
||||
|
||||
def __init__(self, uri):
|
||||
self.path = file_uri_to_path(uri)
|
||||
self.logger = getLogger()
|
||||
|
||||
def open(self, spider):
|
||||
return tempfile.NamedTemporaryFile(prefix='feed-')
|
||||
|
||||
def store(self, file):
|
||||
self.logger.info('Storage.store is called')
|
||||
file.close()
|
||||
|
||||
|
||||
class FeedExportTest(unittest.TestCase):
|
||||
|
||||
class MyItem(scrapy.Item):
|
||||
|
|
@ -426,11 +469,17 @@ class FeedExportTest(unittest.TestCase):
|
|||
yield runner.crawl(spider_cls)
|
||||
|
||||
for file_path, feed in FEEDS.items():
|
||||
if not os.path.exists(str(file_path)):
|
||||
continue
|
||||
|
||||
with open(str(file_path), 'rb') as f:
|
||||
content[feed['format']] = f.read()
|
||||
|
||||
finally:
|
||||
for file_path in FEEDS.keys():
|
||||
if not os.path.exists(str(file_path)):
|
||||
continue
|
||||
|
||||
os.remove(str(file_path))
|
||||
|
||||
return content
|
||||
|
|
@ -623,6 +672,25 @@ class FeedExportTest(unittest.TestCase):
|
|||
data = yield self.exported_no_data(settings)
|
||||
self.assertEqual(data[fmt], expctd)
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_export_no_items_multiple_feeds(self):
|
||||
""" Make sure that `storage.store` is called for every feed. """
|
||||
settings = {
|
||||
'FEEDS': {
|
||||
self._random_temp_filename(): {'format': 'json'},
|
||||
self._random_temp_filename(): {'format': 'xml'},
|
||||
self._random_temp_filename(): {'format': 'csv'},
|
||||
},
|
||||
'FEED_STORAGES': {'file': 'tests.test_feedexport.LogOnStoreFileStorage'},
|
||||
'FEED_STORE_EMPTY': False
|
||||
}
|
||||
|
||||
with LogCapture() as log:
|
||||
yield self.exported_no_data(settings)
|
||||
|
||||
print(log)
|
||||
self.assertEqual(str(log).count('Storage.store is called'), 3)
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_export_multiple_item_classes(self):
|
||||
|
||||
|
|
@ -978,3 +1046,45 @@ class FeedExportTest(unittest.TestCase):
|
|||
}
|
||||
data = yield self.exported_no_data(settings)
|
||||
self.assertEqual(data['csv'], b'')
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_multiple_feeds_success_logs_blocking_feed_storage(self):
|
||||
settings = {
|
||||
'FEEDS': {
|
||||
self._random_temp_filename(): {'format': 'json'},
|
||||
self._random_temp_filename(): {'format': 'xml'},
|
||||
self._random_temp_filename(): {'format': 'csv'},
|
||||
},
|
||||
'FEED_STORAGES': {'file': 'tests.test_feedexport.DummyBlockingFeedStorage'},
|
||||
}
|
||||
items = [
|
||||
{'foo': 'bar1', 'baz': ''},
|
||||
{'foo': 'bar2', 'baz': 'quux'},
|
||||
]
|
||||
with LogCapture() as log:
|
||||
yield self.exported_data(items, settings)
|
||||
|
||||
print(log)
|
||||
for fmt in ['json', 'xml', 'csv']:
|
||||
self.assertIn('Stored %s feed (2 items)' % fmt, str(log))
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_multiple_feeds_failing_logs_blocking_feed_storage(self):
|
||||
settings = {
|
||||
'FEEDS': {
|
||||
self._random_temp_filename(): {'format': 'json'},
|
||||
self._random_temp_filename(): {'format': 'xml'},
|
||||
self._random_temp_filename(): {'format': 'csv'},
|
||||
},
|
||||
'FEED_STORAGES': {'file': 'tests.test_feedexport.FailingBlockingFeedStorage'},
|
||||
}
|
||||
items = [
|
||||
{'foo': 'bar1', 'baz': ''},
|
||||
{'foo': 'bar2', 'baz': 'quux'},
|
||||
]
|
||||
with LogCapture() as log:
|
||||
yield self.exported_data(items, settings)
|
||||
|
||||
print(log)
|
||||
for fmt in ['json', 'xml', 'csv']:
|
||||
self.assertIn('Error storing %s feed (2 items)' % fmt, str(log))
|
||||
|
|
|
|||
|
|
@ -1,6 +1,9 @@
|
|||
from functools import partial
|
||||
import unittest
|
||||
|
||||
import attr
|
||||
from itemadapter import ItemAdapter
|
||||
|
||||
from scrapy.http import HtmlResponse
|
||||
from scrapy.item import Item, Field
|
||||
from scrapy.loader import ItemLoader
|
||||
|
|
@ -9,6 +12,13 @@ from scrapy.loader.processors import (Compose, Identity, Join,
|
|||
from scrapy.selector import Selector
|
||||
|
||||
|
||||
try:
|
||||
from dataclasses import make_dataclass, field as dataclass_field
|
||||
except ImportError:
|
||||
make_dataclass = None
|
||||
dataclass_field = None
|
||||
|
||||
|
||||
# test items
|
||||
class NameItem(Item):
|
||||
name = Field()
|
||||
|
|
@ -28,6 +38,11 @@ class TestNestedItem(Item):
|
|||
image = Field()
|
||||
|
||||
|
||||
@attr.s
|
||||
class AttrsNameItem:
|
||||
name = attr.ib(default="")
|
||||
|
||||
|
||||
# test item loaders
|
||||
class NameItemLoader(ItemLoader):
|
||||
default_item_class = TestItem
|
||||
|
|
@ -466,7 +481,7 @@ class InitializationTestMixin:
|
|||
il = ItemLoader(item=input_item)
|
||||
loaded_item = il.load_item()
|
||||
self.assertIsInstance(loaded_item, self.item_class)
|
||||
self.assertEqual(dict(loaded_item), {'name': ['foo']})
|
||||
self.assertEqual(ItemAdapter(loaded_item).asdict(), {'name': ['foo']})
|
||||
|
||||
def test_keep_list(self):
|
||||
"""Loaded item should contain values from the initial item"""
|
||||
|
|
@ -474,7 +489,7 @@ class InitializationTestMixin:
|
|||
il = ItemLoader(item=input_item)
|
||||
loaded_item = il.load_item()
|
||||
self.assertIsInstance(loaded_item, self.item_class)
|
||||
self.assertEqual(dict(loaded_item), {'name': ['foo', 'bar']})
|
||||
self.assertEqual(ItemAdapter(loaded_item).asdict(), {'name': ['foo', 'bar']})
|
||||
|
||||
def test_add_value_singlevalue_singlevalue(self):
|
||||
"""Values added after initialization should be appended"""
|
||||
|
|
@ -483,7 +498,7 @@ class InitializationTestMixin:
|
|||
il.add_value('name', 'bar')
|
||||
loaded_item = il.load_item()
|
||||
self.assertIsInstance(loaded_item, self.item_class)
|
||||
self.assertEqual(dict(loaded_item), {'name': ['foo', 'bar']})
|
||||
self.assertEqual(ItemAdapter(loaded_item).asdict(), {'name': ['foo', 'bar']})
|
||||
|
||||
def test_add_value_singlevalue_list(self):
|
||||
"""Values added after initialization should be appended"""
|
||||
|
|
@ -492,7 +507,7 @@ class InitializationTestMixin:
|
|||
il.add_value('name', ['item', 'loader'])
|
||||
loaded_item = il.load_item()
|
||||
self.assertIsInstance(loaded_item, self.item_class)
|
||||
self.assertEqual(dict(loaded_item), {'name': ['foo', 'item', 'loader']})
|
||||
self.assertEqual(ItemAdapter(loaded_item).asdict(), {'name': ['foo', 'item', 'loader']})
|
||||
|
||||
def test_add_value_list_singlevalue(self):
|
||||
"""Values added after initialization should be appended"""
|
||||
|
|
@ -501,7 +516,7 @@ class InitializationTestMixin:
|
|||
il.add_value('name', 'qwerty')
|
||||
loaded_item = il.load_item()
|
||||
self.assertIsInstance(loaded_item, self.item_class)
|
||||
self.assertEqual(dict(loaded_item), {'name': ['foo', 'bar', 'qwerty']})
|
||||
self.assertEqual(ItemAdapter(loaded_item).asdict(), {'name': ['foo', 'bar', 'qwerty']})
|
||||
|
||||
def test_add_value_list_list(self):
|
||||
"""Values added after initialization should be appended"""
|
||||
|
|
@ -510,7 +525,7 @@ class InitializationTestMixin:
|
|||
il.add_value('name', ['item', 'loader'])
|
||||
loaded_item = il.load_item()
|
||||
self.assertIsInstance(loaded_item, self.item_class)
|
||||
self.assertEqual(dict(loaded_item), {'name': ['foo', 'bar', 'item', 'loader']})
|
||||
self.assertEqual(ItemAdapter(loaded_item).asdict(), {'name': ['foo', 'bar', 'item', 'loader']})
|
||||
|
||||
def test_get_output_value_singlevalue(self):
|
||||
"""Getting output value must not remove value from item"""
|
||||
|
|
@ -519,7 +534,7 @@ class InitializationTestMixin:
|
|||
self.assertEqual(il.get_output_value('name'), ['foo'])
|
||||
loaded_item = il.load_item()
|
||||
self.assertIsInstance(loaded_item, self.item_class)
|
||||
self.assertEqual(loaded_item, dict({'name': ['foo']}))
|
||||
self.assertEqual(ItemAdapter(loaded_item).asdict(), dict({'name': ['foo']}))
|
||||
|
||||
def test_get_output_value_list(self):
|
||||
"""Getting output value must not remove value from item"""
|
||||
|
|
@ -528,7 +543,7 @@ class InitializationTestMixin:
|
|||
self.assertEqual(il.get_output_value('name'), ['foo', 'bar'])
|
||||
loaded_item = il.load_item()
|
||||
self.assertIsInstance(loaded_item, self.item_class)
|
||||
self.assertEqual(loaded_item, dict({'name': ['foo', 'bar']}))
|
||||
self.assertEqual(ItemAdapter(loaded_item).asdict(), dict({'name': ['foo', 'bar']}))
|
||||
|
||||
def test_values_single(self):
|
||||
"""Values from initial item must be added to loader._values"""
|
||||
|
|
@ -551,6 +566,22 @@ class InitializationFromItemTest(InitializationTestMixin, unittest.TestCase):
|
|||
item_class = NameItem
|
||||
|
||||
|
||||
class InitializationFromAttrsItemTest(InitializationTestMixin, unittest.TestCase):
|
||||
item_class = AttrsNameItem
|
||||
|
||||
|
||||
@unittest.skipIf(not make_dataclass, "dataclasses module is not available")
|
||||
class InitializationFromDataClassTest(InitializationTestMixin, unittest.TestCase):
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
super().__init__(*args, **kwargs)
|
||||
if make_dataclass:
|
||||
self.item_class = make_dataclass(
|
||||
"TestDataClass",
|
||||
[("name", list, dataclass_field(default_factory=list))],
|
||||
)
|
||||
|
||||
|
||||
class BaseNoInputReprocessingLoader(ItemLoader):
|
||||
title_in = MapCompose(str.upper)
|
||||
title_out = TakeFirst()
|
||||
|
|
|
|||
|
|
@ -2,22 +2,41 @@ import os
|
|||
import random
|
||||
import time
|
||||
from io import BytesIO
|
||||
from tempfile import mkdtemp
|
||||
from shutil import rmtree
|
||||
from unittest import mock
|
||||
from tempfile import mkdtemp
|
||||
from unittest import mock, skipIf
|
||||
from urllib.parse import urlparse
|
||||
|
||||
from twisted.trial import unittest
|
||||
import attr
|
||||
from itemadapter import ItemAdapter
|
||||
from twisted.internet import defer
|
||||
from twisted.trial import unittest
|
||||
|
||||
from scrapy.pipelines.files import FilesPipeline, FSFilesStore, S3FilesStore, GCSFilesStore, FTPFilesStore
|
||||
from scrapy.item import Item, Field
|
||||
from scrapy.http import Request, Response
|
||||
from scrapy.item import Field, Item
|
||||
from scrapy.pipelines.files import (
|
||||
FilesPipeline,
|
||||
FSFilesStore,
|
||||
FTPFilesStore,
|
||||
GCSFilesStore,
|
||||
S3FilesStore,
|
||||
)
|
||||
from scrapy.settings import Settings
|
||||
from scrapy.utils.test import assert_aws_environ, get_s3_content_and_delete
|
||||
from scrapy.utils.test import assert_gcs_environ, get_gcs_content_and_delete
|
||||
from scrapy.utils.test import get_ftp_content_and_delete
|
||||
from scrapy.utils.boto import is_botocore
|
||||
from scrapy.utils.test import (
|
||||
assert_aws_environ,
|
||||
assert_gcs_environ,
|
||||
get_ftp_content_and_delete,
|
||||
get_gcs_content_and_delete,
|
||||
get_s3_content_and_delete,
|
||||
)
|
||||
|
||||
|
||||
try:
|
||||
from dataclasses import make_dataclass, field as dataclass_field
|
||||
except ImportError:
|
||||
make_dataclass = None
|
||||
dataclass_field = None
|
||||
|
||||
|
||||
def _mocked_download_func(request, info):
|
||||
|
|
@ -143,43 +162,88 @@ class FilesPipelineTestCase(unittest.TestCase):
|
|||
p.stop()
|
||||
|
||||
|
||||
class FilesPipelineTestCaseFields(unittest.TestCase):
|
||||
class FilesPipelineTestCaseFieldsMixin:
|
||||
|
||||
def test_item_fields_default(self):
|
||||
class TestItem(Item):
|
||||
name = Field()
|
||||
file_urls = Field()
|
||||
files = Field()
|
||||
|
||||
for cls in TestItem, dict:
|
||||
url = 'http://www.example.com/files/1.txt'
|
||||
item = cls({'name': 'item1', 'file_urls': [url]})
|
||||
pipeline = FilesPipeline.from_settings(Settings({'FILES_STORE': 's3://example/files/'}))
|
||||
requests = list(pipeline.get_media_requests(item, None))
|
||||
self.assertEqual(requests[0].url, url)
|
||||
results = [(True, {'url': url})]
|
||||
pipeline.item_completed(results, item, None)
|
||||
self.assertEqual(item['files'], [results[0][1]])
|
||||
url = 'http://www.example.com/files/1.txt'
|
||||
item = self.item_class(name='item1', file_urls=[url])
|
||||
pipeline = FilesPipeline.from_settings(Settings({'FILES_STORE': 's3://example/files/'}))
|
||||
requests = list(pipeline.get_media_requests(item, None))
|
||||
self.assertEqual(requests[0].url, url)
|
||||
results = [(True, {'url': url})]
|
||||
item = pipeline.item_completed(results, item, None)
|
||||
files = ItemAdapter(item).get("files")
|
||||
self.assertEqual(files, [results[0][1]])
|
||||
self.assertIsInstance(item, self.item_class)
|
||||
|
||||
def test_item_fields_override_settings(self):
|
||||
class TestItem(Item):
|
||||
name = Field()
|
||||
files = Field()
|
||||
stored_file = Field()
|
||||
url = 'http://www.example.com/files/1.txt'
|
||||
item = self.item_class(name='item1', custom_file_urls=[url])
|
||||
pipeline = FilesPipeline.from_settings(Settings({
|
||||
'FILES_STORE': 's3://example/files/',
|
||||
'FILES_URLS_FIELD': 'custom_file_urls',
|
||||
'FILES_RESULT_FIELD': 'custom_files'
|
||||
}))
|
||||
requests = list(pipeline.get_media_requests(item, None))
|
||||
self.assertEqual(requests[0].url, url)
|
||||
results = [(True, {'url': url})]
|
||||
item = pipeline.item_completed(results, item, None)
|
||||
custom_files = ItemAdapter(item).get("custom_files")
|
||||
self.assertEqual(custom_files, [results[0][1]])
|
||||
self.assertIsInstance(item, self.item_class)
|
||||
|
||||
for cls in TestItem, dict:
|
||||
url = 'http://www.example.com/files/1.txt'
|
||||
item = cls({'name': 'item1', 'files': [url]})
|
||||
pipeline = FilesPipeline.from_settings(Settings({
|
||||
'FILES_STORE': 's3://example/files/',
|
||||
'FILES_URLS_FIELD': 'files',
|
||||
'FILES_RESULT_FIELD': 'stored_file'
|
||||
}))
|
||||
requests = list(pipeline.get_media_requests(item, None))
|
||||
self.assertEqual(requests[0].url, url)
|
||||
results = [(True, {'url': url})]
|
||||
pipeline.item_completed(results, item, None)
|
||||
self.assertEqual(item['stored_file'], [results[0][1]])
|
||||
|
||||
class FilesPipelineTestCaseFieldsDict(FilesPipelineTestCaseFieldsMixin, unittest.TestCase):
|
||||
item_class = dict
|
||||
|
||||
|
||||
class FilesPipelineTestItem(Item):
|
||||
name = Field()
|
||||
# default fields
|
||||
file_urls = Field()
|
||||
files = Field()
|
||||
# overridden fields
|
||||
custom_file_urls = Field()
|
||||
custom_files = Field()
|
||||
|
||||
|
||||
class FilesPipelineTestCaseFieldsItem(FilesPipelineTestCaseFieldsMixin, unittest.TestCase):
|
||||
item_class = FilesPipelineTestItem
|
||||
|
||||
|
||||
@skipIf(not make_dataclass, "dataclasses module is not available")
|
||||
class FilesPipelineTestCaseFieldsDataClass(FilesPipelineTestCaseFieldsMixin, unittest.TestCase):
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
super().__init__(*args, **kwargs)
|
||||
if make_dataclass:
|
||||
self.item_class = make_dataclass(
|
||||
"FilesPipelineTestDataClass",
|
||||
[
|
||||
("name", str),
|
||||
# default fields
|
||||
("file_urls", list, dataclass_field(default_factory=list)),
|
||||
("files", list, dataclass_field(default_factory=list)),
|
||||
# overridden fields
|
||||
("custom_file_urls", list, dataclass_field(default_factory=list)),
|
||||
("custom_files", list, dataclass_field(default_factory=list)),
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
@attr.s
|
||||
class FilesPipelineTestAttrsItem:
|
||||
name = attr.ib(default="")
|
||||
# default fields
|
||||
file_urls = attr.ib(default=lambda: [])
|
||||
files = attr.ib(default=lambda: [])
|
||||
# overridden fields
|
||||
custom_file_urls = attr.ib(default=lambda: [])
|
||||
custom_files = attr.ib(default=lambda: [])
|
||||
|
||||
|
||||
class FilesPipelineTestCaseFieldsAttrsItem(FilesPipelineTestCaseFieldsMixin, unittest.TestCase):
|
||||
item_class = FilesPipelineTestAttrsItem
|
||||
|
||||
|
||||
class FilesPipelineTestCaseCustomSettings(unittest.TestCase):
|
||||
|
|
|
|||
|
|
@ -1,17 +1,28 @@
|
|||
import io
|
||||
import hashlib
|
||||
import io
|
||||
import random
|
||||
from tempfile import mkdtemp
|
||||
from shutil import rmtree
|
||||
from tempfile import mkdtemp
|
||||
from unittest import skipIf
|
||||
|
||||
import attr
|
||||
from itemadapter import ItemAdapter
|
||||
from twisted.trial import unittest
|
||||
|
||||
from scrapy.item import Item, Field
|
||||
from scrapy.http import Request, Response
|
||||
from scrapy.settings import Settings
|
||||
from scrapy.item import Field, Item
|
||||
from scrapy.pipelines.images import ImagesPipeline
|
||||
from scrapy.settings import Settings
|
||||
from scrapy.utils.python import to_bytes
|
||||
|
||||
|
||||
try:
|
||||
from dataclasses import make_dataclass, field as dataclass_field
|
||||
except ImportError:
|
||||
make_dataclass = None
|
||||
dataclass_field = None
|
||||
|
||||
|
||||
skip = False
|
||||
try:
|
||||
from PIL import Image
|
||||
|
|
@ -124,43 +135,89 @@ class DeprecatedImagesPipeline(ImagesPipeline):
|
|||
return 'thumbsup/%s/%s.jpg' % (thumb_id, thumb_guid)
|
||||
|
||||
|
||||
class ImagesPipelineTestCaseFields(unittest.TestCase):
|
||||
class ImagesPipelineTestCaseFieldsMixin:
|
||||
|
||||
def test_item_fields_default(self):
|
||||
class TestItem(Item):
|
||||
name = Field()
|
||||
image_urls = Field()
|
||||
images = Field()
|
||||
|
||||
for cls in TestItem, dict:
|
||||
url = 'http://www.example.com/images/1.jpg'
|
||||
item = cls({'name': 'item1', 'image_urls': [url]})
|
||||
pipeline = ImagesPipeline.from_settings(Settings({'IMAGES_STORE': 's3://example/images/'}))
|
||||
requests = list(pipeline.get_media_requests(item, None))
|
||||
self.assertEqual(requests[0].url, url)
|
||||
results = [(True, {'url': url})]
|
||||
pipeline.item_completed(results, item, None)
|
||||
self.assertEqual(item['images'], [results[0][1]])
|
||||
url = 'http://www.example.com/images/1.jpg'
|
||||
item = self.item_class(name='item1', image_urls=[url])
|
||||
pipeline = ImagesPipeline.from_settings(Settings({'IMAGES_STORE': 's3://example/images/'}))
|
||||
requests = list(pipeline.get_media_requests(item, None))
|
||||
self.assertEqual(requests[0].url, url)
|
||||
results = [(True, {'url': url})]
|
||||
item = pipeline.item_completed(results, item, None)
|
||||
images = ItemAdapter(item).get("images")
|
||||
self.assertEqual(images, [results[0][1]])
|
||||
self.assertIsInstance(item, self.item_class)
|
||||
|
||||
def test_item_fields_override_settings(self):
|
||||
class TestItem(Item):
|
||||
name = Field()
|
||||
image = Field()
|
||||
stored_image = Field()
|
||||
url = 'http://www.example.com/images/1.jpg'
|
||||
item = self.item_class(name='item1', custom_image_urls=[url])
|
||||
pipeline = ImagesPipeline.from_settings(Settings({
|
||||
'IMAGES_STORE': 's3://example/images/',
|
||||
'IMAGES_URLS_FIELD': 'custom_image_urls',
|
||||
'IMAGES_RESULT_FIELD': 'custom_images'
|
||||
}))
|
||||
requests = list(pipeline.get_media_requests(item, None))
|
||||
self.assertEqual(requests[0].url, url)
|
||||
results = [(True, {'url': url})]
|
||||
item = pipeline.item_completed(results, item, None)
|
||||
custom_images = ItemAdapter(item).get("custom_images")
|
||||
self.assertEqual(custom_images, [results[0][1]])
|
||||
self.assertIsInstance(item, self.item_class)
|
||||
|
||||
for cls in TestItem, dict:
|
||||
url = 'http://www.example.com/images/1.jpg'
|
||||
item = cls({'name': 'item1', 'image': [url]})
|
||||
pipeline = ImagesPipeline.from_settings(Settings({
|
||||
'IMAGES_STORE': 's3://example/images/',
|
||||
'IMAGES_URLS_FIELD': 'image',
|
||||
'IMAGES_RESULT_FIELD': 'stored_image'
|
||||
}))
|
||||
requests = list(pipeline.get_media_requests(item, None))
|
||||
self.assertEqual(requests[0].url, url)
|
||||
results = [(True, {'url': url})]
|
||||
pipeline.item_completed(results, item, None)
|
||||
self.assertEqual(item['stored_image'], [results[0][1]])
|
||||
|
||||
class ImagesPipelineTestCaseFieldsDict(ImagesPipelineTestCaseFieldsMixin, unittest.TestCase):
|
||||
item_class = dict
|
||||
|
||||
|
||||
class ImagesPipelineTestItem(Item):
|
||||
name = Field()
|
||||
# default fields
|
||||
image_urls = Field()
|
||||
images = Field()
|
||||
# overridden fields
|
||||
custom_image_urls = Field()
|
||||
custom_images = Field()
|
||||
|
||||
|
||||
class ImagesPipelineTestCaseFieldsItem(ImagesPipelineTestCaseFieldsMixin, unittest.TestCase):
|
||||
item_class = ImagesPipelineTestItem
|
||||
|
||||
|
||||
@skipIf(not make_dataclass, "dataclasses module is not available")
|
||||
class ImagesPipelineTestCaseFieldsDataClass(ImagesPipelineTestCaseFieldsMixin, unittest.TestCase):
|
||||
item_class = None
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
super().__init__(*args, **kwargs)
|
||||
if make_dataclass:
|
||||
self.item_class = make_dataclass(
|
||||
"FilesPipelineTestDataClass",
|
||||
[
|
||||
("name", str),
|
||||
# default fields
|
||||
("image_urls", list, dataclass_field(default_factory=list)),
|
||||
("images", list, dataclass_field(default_factory=list)),
|
||||
# overridden fields
|
||||
("custom_image_urls", list, dataclass_field(default_factory=list)),
|
||||
("custom_images", list, dataclass_field(default_factory=list)),
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
@attr.s
|
||||
class ImagesPipelineTestAttrsItem:
|
||||
name = attr.ib(default="")
|
||||
# default fields
|
||||
image_urls = attr.ib(default=lambda: [])
|
||||
images = attr.ib(default=lambda: [])
|
||||
# overridden fields
|
||||
custom_image_urls = attr.ib(default=lambda: [])
|
||||
custom_images = attr.ib(default=lambda: [])
|
||||
|
||||
|
||||
class ImagesPipelineTestCaseFieldsAttrsItem(ImagesPipelineTestCaseFieldsMixin, unittest.TestCase):
|
||||
item_class = ImagesPipelineTestAttrsItem
|
||||
|
||||
|
||||
class ImagesPipelineTestCaseCustomSettings(unittest.TestCase):
|
||||
|
|
|
|||
|
|
@ -59,7 +59,7 @@ class KeywordArgumentsSpider(MockServerSpider):
|
|||
|
||||
checks = list()
|
||||
|
||||
def start_requests(self):
|
||||
def start_requests_with_control(self):
|
||||
data = {'key': 'value', 'number': 123}
|
||||
yield Request(self.mockserver.url('/first'), self.parse_first, cb_kwargs=data)
|
||||
yield Request(self.mockserver.url('/general_with'), self.parse_general, cb_kwargs=data)
|
||||
|
|
|
|||
|
|
@ -296,7 +296,7 @@ class StartUrlsSpider(Spider):
|
|||
|
||||
def __init__(self, start_urls):
|
||||
self.start_urls = start_urls
|
||||
super(StartUrlsSpider, self).__init__(start_urls)
|
||||
super(StartUrlsSpider, self).__init__(name='StartUrlsSpider')
|
||||
|
||||
def parse(self, response):
|
||||
pass
|
||||
|
|
|
|||
|
|
@ -141,6 +141,31 @@ class CurlToRequestKwargsTest(unittest.TestCase):
|
|||
}
|
||||
self._test_command(curl_command, expected_result)
|
||||
|
||||
def test_post_data_raw(self):
|
||||
curl_command = (
|
||||
"curl 'https://www.example.org/' --data-raw 'excerptLength=200&ena"
|
||||
"bleDidYouMean=true&sortCriteria=ffirstz32xnamez32x201740686%20asc"
|
||||
"ending&queryFunctions=%5B%5D&rankingFunctions=%5B%5D'"
|
||||
)
|
||||
expected_result = {
|
||||
"method": "POST",
|
||||
"url": "https://www.example.org/",
|
||||
"body": (
|
||||
"excerptLength=200&enableDidYouMean=true&sortCriteria=ffirstz3"
|
||||
"2xnamez32x201740686%20ascending&queryFunctions=%5B%5D&ranking"
|
||||
"Functions=%5B%5D")
|
||||
}
|
||||
self._test_command(curl_command, expected_result)
|
||||
|
||||
def test_explicit_get_with_data(self):
|
||||
curl_command = 'curl httpbin.org/anything -X GET --data asdf'
|
||||
expected_result = {
|
||||
"method": "GET",
|
||||
"url": "http://httpbin.org/anything",
|
||||
"body": "asdf"
|
||||
}
|
||||
self._test_command(curl_command, expected_result)
|
||||
|
||||
def test_patch(self):
|
||||
curl_command = (
|
||||
'curl "https://example.com/api/fake" -u "username:password" -H "Ac'
|
||||
|
|
|
|||
|
|
@ -1,18 +1,25 @@
|
|||
import datetime
|
||||
import json
|
||||
import unittest
|
||||
import datetime
|
||||
from decimal import Decimal
|
||||
|
||||
import attr
|
||||
from twisted.internet import defer
|
||||
|
||||
from scrapy.utils.serialize import ScrapyJSONEncoder
|
||||
from scrapy.http import Request, Response
|
||||
from scrapy.utils.serialize import ScrapyJSONEncoder
|
||||
|
||||
|
||||
try:
|
||||
from dataclasses import make_dataclass
|
||||
except ImportError:
|
||||
make_dataclass = None
|
||||
|
||||
|
||||
class JsonEncoderTestCase(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
self.encoder = ScrapyJSONEncoder()
|
||||
self.encoder = ScrapyJSONEncoder(sort_keys=True)
|
||||
|
||||
def test_encode_decode(self):
|
||||
dt = datetime.datetime(2010, 1, 2, 10, 11, 12)
|
||||
|
|
@ -31,7 +38,8 @@ class JsonEncoderTestCase(unittest.TestCase):
|
|||
for input, output in [('foo', 'foo'), (d, ds), (t, ts), (dt, dts),
|
||||
(dec, decs), (['foo', d], ['foo', ds]), (s, ss),
|
||||
(dt_set, dt_sets)]:
|
||||
self.assertEqual(self.encoder.encode(input), json.dumps(output))
|
||||
self.assertEqual(self.encoder.encode(input),
|
||||
json.dumps(output, sort_keys=True))
|
||||
|
||||
def test_encode_deferred(self):
|
||||
self.assertIn('Deferred', self.encoder.encode(defer.Deferred()))
|
||||
|
|
@ -47,3 +55,30 @@ class JsonEncoderTestCase(unittest.TestCase):
|
|||
rs = self.encoder.encode(r)
|
||||
self.assertIn(r.url, rs)
|
||||
self.assertIn(str(r.status), rs)
|
||||
|
||||
@unittest.skipIf(not make_dataclass, "No dataclass support")
|
||||
def test_encode_dataclass_item(self):
|
||||
TestDataClass = make_dataclass(
|
||||
"TestDataClass",
|
||||
[("name", str), ("url", str), ("price", int)],
|
||||
)
|
||||
item = TestDataClass(name="Product", url="http://product.org", price=1)
|
||||
encoded = self.encoder.encode(item)
|
||||
self.assertEqual(
|
||||
encoded,
|
||||
'{"name": "Product", "price": 1, "url": "http://product.org"}'
|
||||
)
|
||||
|
||||
def test_encode_attrs_item(self):
|
||||
@attr.s
|
||||
class AttrsItem:
|
||||
name = attr.ib(type=str)
|
||||
url = attr.ib(type=str)
|
||||
price = attr.ib(type=int)
|
||||
|
||||
item = AttrsItem(name="Product", url="http://product.org", price=1)
|
||||
encoded = self.encoder.encode(item)
|
||||
self.assertEqual(
|
||||
encoded,
|
||||
'{"name": "Product", "price": 1, "url": "http://product.org"}'
|
||||
)
|
||||
|
|
|
|||
10
tox.ini
10
tox.ini
|
|
@ -23,6 +23,13 @@ passenv =
|
|||
commands =
|
||||
py.test --cov=scrapy --cov-report= {posargs:--durations=10 docs scrapy tests}
|
||||
|
||||
[testenv:typing]
|
||||
basepython = python3
|
||||
deps =
|
||||
mypy==0.780
|
||||
commands =
|
||||
mypy {posargs: scrapy tests}
|
||||
|
||||
[testenv:security]
|
||||
basepython = python3
|
||||
deps =
|
||||
|
|
@ -37,7 +44,7 @@ deps =
|
|||
pytest-flake8
|
||||
commands =
|
||||
py.test --flake8 {posargs:docs scrapy tests}
|
||||
|
||||
|
||||
[testenv:pylint]
|
||||
basepython = python3
|
||||
deps =
|
||||
|
|
@ -62,6 +69,7 @@ deps =
|
|||
-ctests/constraints.txt
|
||||
cryptography==2.0
|
||||
cssselect==0.9.1
|
||||
itemadapter==0.1.0
|
||||
lxml==3.5.0
|
||||
parsel==1.5.0
|
||||
Protego==0.1.15
|
||||
|
|
|
|||
Loading…
Reference in New Issue