mirror of https://github.com/scrapy/scrapy.git
Merge remote-tracking branch 'origin/master' into asyncio-startrequests-asyncgen
This commit is contained in:
commit
351a400cf8
|
|
@ -1,4 +1,5 @@
|
|||
version: 2
|
||||
formats: all
|
||||
sphinx:
|
||||
configuration: docs/conf.py
|
||||
fail_on_warning: true
|
||||
|
|
|
|||
|
|
@ -57,3 +57,12 @@ There is a way to recreate the doc automatically when you make changes, you
|
|||
need to install watchdog (``pip install watchdog``) and then use::
|
||||
|
||||
make watch
|
||||
|
||||
Alternative method using tox
|
||||
----------------------------
|
||||
|
||||
To compile the documentation to HTML run the following command::
|
||||
|
||||
tox -e docs
|
||||
|
||||
Documentation will be generated (in HTML format) inside the ``.tox/docs/tmp/html`` dir.
|
||||
|
|
|
|||
|
|
@ -100,6 +100,9 @@ exclude_trees = ['.build']
|
|||
# The name of the Pygments (syntax highlighting) style to use.
|
||||
pygments_style = 'sphinx'
|
||||
|
||||
# List of Sphinx warnings that will not be raised
|
||||
suppress_warnings = ['epub.unknown_project_files']
|
||||
|
||||
|
||||
# Options for HTML output
|
||||
# -----------------------
|
||||
|
|
@ -300,3 +303,4 @@ hoverxref_role_types = {
|
|||
"mod": "tooltip",
|
||||
"ref": "tooltip",
|
||||
}
|
||||
hoverxref_roles = ['command', 'reqmeta', 'setting', 'signal']
|
||||
|
|
|
|||
13
docs/faq.rst
13
docs/faq.rst
|
|
@ -371,6 +371,19 @@ Twisted reactor is :class:`twisted.internet.selectreactor.SelectReactor`. Switch
|
|||
different reactor is possible by using the :setting:`TWISTED_REACTOR` setting.
|
||||
|
||||
|
||||
.. _faq-stop-response-download:
|
||||
|
||||
How can I cancel the download of a given response?
|
||||
--------------------------------------------------
|
||||
|
||||
In some situations, it might be useful to stop the download of a certain response.
|
||||
For instance, if you only need the first part of a large response and you would like
|
||||
to save resources by avoiding the download of the whole body.
|
||||
In that case, you could attach a handler to the :class:`~scrapy.signals.bytes_received`
|
||||
signal and raise a :exc:`~scrapy.exceptions.StopDownload` exception. Please refer to
|
||||
the :ref:`topics-stop-response-download` topic for additional information and examples.
|
||||
|
||||
|
||||
.. _has been reported: https://github.com/scrapy/scrapy/issues/2905
|
||||
.. _user agents: https://en.wikipedia.org/wiki/User_agent
|
||||
.. _LIFO: https://en.wikipedia.org/wiki/Stack_(abstract_data_type)
|
||||
|
|
|
|||
|
|
@ -91,7 +91,7 @@ how you :ref:`configure the downloader middlewares
|
|||
provided while constructing the crawler, and it is created after the
|
||||
arguments given in the :meth:`crawl` method.
|
||||
|
||||
.. method:: crawl(\*args, \**kwargs)
|
||||
.. method:: crawl(*args, **kwargs)
|
||||
|
||||
Starts the crawler by instantiating its spider class with the given
|
||||
``args`` and ``kwargs`` arguments, while setting the execution engine in
|
||||
|
|
|
|||
|
|
@ -78,7 +78,7 @@ override three methods:
|
|||
|
||||
.. module:: scrapy.contracts
|
||||
|
||||
.. class:: Contract(method, \*args)
|
||||
.. class:: Contract(method, *args)
|
||||
|
||||
:param method: callback function to which the contract is associated
|
||||
:type method: function
|
||||
|
|
|
|||
|
|
@ -202,6 +202,11 @@ CookiesMiddleware
|
|||
sends them back on subsequent requests (from that spider), just like web
|
||||
browsers do.
|
||||
|
||||
.. caution:: When non-UTF8 encoded byte sequences are passed to a
|
||||
:class:`~scrapy.http.Request`, the ``CookiesMiddleware`` will log
|
||||
a warning. Refer to :ref:`topics-logging-advanced-customization`
|
||||
to customize the logging behaviour.
|
||||
|
||||
The following settings can be used to configure the cookie middleware:
|
||||
|
||||
* :setting:`COOKIES_ENABLED`
|
||||
|
|
|
|||
|
|
@ -14,13 +14,6 @@ Built-in Exceptions reference
|
|||
|
||||
Here's a list of all exceptions included in Scrapy and their usage.
|
||||
|
||||
DropItem
|
||||
--------
|
||||
|
||||
.. exception:: DropItem
|
||||
|
||||
The exception that must be raised by item pipeline stages to stop processing an
|
||||
Item. For more information see :ref:`topics-item-pipeline`.
|
||||
|
||||
CloseSpider
|
||||
-----------
|
||||
|
|
@ -47,6 +40,14 @@ DontCloseSpider
|
|||
This exception can be raised in a :signal:`spider_idle` signal handler to
|
||||
prevent the spider from being closed.
|
||||
|
||||
DropItem
|
||||
--------
|
||||
|
||||
.. exception:: DropItem
|
||||
|
||||
The exception that must be raised by item pipeline stages to stop processing an
|
||||
Item. For more information see :ref:`topics-item-pipeline`.
|
||||
|
||||
IgnoreRequest
|
||||
-------------
|
||||
|
||||
|
|
@ -77,3 +78,37 @@ NotSupported
|
|||
|
||||
This exception is raised to indicate an unsupported feature.
|
||||
|
||||
StopDownload
|
||||
-------------
|
||||
|
||||
.. versionadded:: 2.2
|
||||
|
||||
.. exception:: StopDownload(fail=True)
|
||||
|
||||
Raised from a :class:`~scrapy.signals.bytes_received` signal handler to
|
||||
indicate that no further bytes should be downloaded for a response.
|
||||
|
||||
The ``fail`` boolean parameter controls which method will handle the resulting
|
||||
response:
|
||||
|
||||
* If ``fail=True`` (default), the request errback is called. The response object is
|
||||
available as the ``response`` attribute of the ``StopDownload`` exception,
|
||||
which is in turn stored as the ``value`` attribute of the received
|
||||
:class:`~twisted.python.failure.Failure` object. This means that in an errback
|
||||
defined as ``def errback(self, failure)``, the response can be accessed though
|
||||
``failure.value.response``.
|
||||
|
||||
* If ``fail=False``, the request callback is called instead.
|
||||
|
||||
In both cases, the response could have its body truncated: the body contains
|
||||
all bytes received up until the exception is raised, including the bytes
|
||||
received in the signal handler that raises the exception. Also, the response
|
||||
object is marked with ``"download_stopped"`` in its :attr:`Response.flags`
|
||||
attribute.
|
||||
|
||||
.. note:: ``fail`` is a keyword-only parameter, i.e. raising
|
||||
``StopDownload(False)`` or ``StopDownload(True)`` will raise
|
||||
a :class:`TypeError`.
|
||||
|
||||
See the documentation for the :class:`~scrapy.signals.bytes_received` signal
|
||||
and the :ref:`topics-stop-response-download` topic for additional information and examples.
|
||||
|
|
|
|||
|
|
@ -236,7 +236,7 @@ PythonItemExporter
|
|||
XmlItemExporter
|
||||
---------------
|
||||
|
||||
.. class:: XmlItemExporter(file, item_element='item', root_element='items', \**kwargs)
|
||||
.. class:: XmlItemExporter(file, item_element='item', root_element='items', **kwargs)
|
||||
|
||||
Exports Items in XML format to the specified file object.
|
||||
|
||||
|
|
@ -290,7 +290,7 @@ XmlItemExporter
|
|||
CsvItemExporter
|
||||
---------------
|
||||
|
||||
.. class:: CsvItemExporter(file, include_headers_line=True, join_multivalued=',', \**kwargs)
|
||||
.. class:: CsvItemExporter(file, include_headers_line=True, join_multivalued=',', **kwargs)
|
||||
|
||||
Exports Items in CSV format to the given file-like object. If the
|
||||
:attr:`fields_to_export` attribute is set, it will be used to define the
|
||||
|
|
@ -323,7 +323,7 @@ CsvItemExporter
|
|||
PickleItemExporter
|
||||
------------------
|
||||
|
||||
.. class:: PickleItemExporter(file, protocol=0, \**kwargs)
|
||||
.. class:: PickleItemExporter(file, protocol=0, **kwargs)
|
||||
|
||||
Exports Items in pickle format to the given file-like object.
|
||||
|
||||
|
|
@ -343,7 +343,7 @@ PickleItemExporter
|
|||
PprintItemExporter
|
||||
------------------
|
||||
|
||||
.. class:: PprintItemExporter(file, \**kwargs)
|
||||
.. class:: PprintItemExporter(file, **kwargs)
|
||||
|
||||
Exports Items in pretty print format to the specified file object.
|
||||
|
||||
|
|
@ -363,7 +363,7 @@ PprintItemExporter
|
|||
JsonItemExporter
|
||||
----------------
|
||||
|
||||
.. class:: JsonItemExporter(file, \**kwargs)
|
||||
.. class:: JsonItemExporter(file, **kwargs)
|
||||
|
||||
Exports Items in JSON format to the specified file-like object, writing all
|
||||
objects as a list of objects. The additional ``__init__`` method arguments are
|
||||
|
|
@ -392,7 +392,7 @@ JsonItemExporter
|
|||
JsonLinesItemExporter
|
||||
---------------------
|
||||
|
||||
.. class:: JsonLinesItemExporter(file, \**kwargs)
|
||||
.. class:: JsonLinesItemExporter(file, **kwargs)
|
||||
|
||||
Exports Items in JSON format to the specified file-like object, writing one
|
||||
JSON-encoded item per line. The additional ``__init__`` method arguments are passed
|
||||
|
|
|
|||
|
|
@ -167,11 +167,13 @@ method and how to clean up the resources properly.::
|
|||
Take screenshot of item
|
||||
-----------------------
|
||||
|
||||
This example demonstrates how to return a
|
||||
:class:`~twisted.internet.defer.Deferred` from the :meth:`process_item` method.
|
||||
It uses Splash_ to render screenshot of item url. Pipeline
|
||||
makes request to locally running instance of Splash_. After request is downloaded,
|
||||
it saves the screenshot to a file and adds filename to the item.
|
||||
This example demonstrates how to use :doc:`coroutine syntax <coroutines>` in
|
||||
the :meth:`process_item` method.
|
||||
|
||||
This item pipeline makes a request to a locally-running instance of Splash_ to
|
||||
render a screenshot of the item URL. After the request response is downloaded,
|
||||
the item pipeline saves the screenshot to a file and adds the filename to the
|
||||
item.
|
||||
|
||||
::
|
||||
|
||||
|
|
|
|||
|
|
@ -273,7 +273,7 @@ There are several ways to modify Item Loader context values:
|
|||
ItemLoader objects
|
||||
==================
|
||||
|
||||
.. class:: ItemLoader([item, selector, response], \**kwargs)
|
||||
.. class:: ItemLoader([item, selector, response], **kwargs)
|
||||
|
||||
Return a new Item Loader for populating the given Item. If no item is
|
||||
given, one is instantiated automatically using the class in
|
||||
|
|
@ -303,7 +303,7 @@ ItemLoader objects
|
|||
|
||||
:class:`ItemLoader` instances have the following methods:
|
||||
|
||||
.. method:: get_value(value, \*processors, \**kwargs)
|
||||
.. method:: get_value(value, *processors, **kwargs)
|
||||
|
||||
Process the given ``value`` by the given ``processors`` and keyword
|
||||
arguments.
|
||||
|
|
@ -321,7 +321,7 @@ ItemLoader objects
|
|||
>>> loader.get_value(u'name: foo', TakeFirst(), unicode.upper, re='name: (.+)')
|
||||
'FOO`
|
||||
|
||||
.. method:: add_value(field_name, value, \*processors, \**kwargs)
|
||||
.. method:: add_value(field_name, value, *processors, **kwargs)
|
||||
|
||||
Process and then add the given ``value`` for the given field.
|
||||
|
||||
|
|
@ -343,11 +343,11 @@ ItemLoader objects
|
|||
loader.add_value('name', u'name: foo', TakeFirst(), re='name: (.+)')
|
||||
loader.add_value(None, {'name': u'foo', 'sex': u'male'})
|
||||
|
||||
.. method:: replace_value(field_name, value, \*processors, \**kwargs)
|
||||
.. method:: replace_value(field_name, value, *processors, **kwargs)
|
||||
|
||||
Similar to :meth:`add_value` but replaces the collected data with the
|
||||
new value instead of adding it.
|
||||
.. method:: get_xpath(xpath, \*processors, \**kwargs)
|
||||
.. method:: get_xpath(xpath, *processors, **kwargs)
|
||||
|
||||
Similar to :meth:`ItemLoader.get_value` but receives an XPath instead of a
|
||||
value, which is used to extract a list of unicode strings from the
|
||||
|
|
@ -367,7 +367,7 @@ ItemLoader objects
|
|||
# HTML snippet: <p id="price">the price is $1200</p>
|
||||
loader.get_xpath('//p[@id="price"]', TakeFirst(), re='the price is (.*)')
|
||||
|
||||
.. method:: add_xpath(field_name, xpath, \*processors, \**kwargs)
|
||||
.. method:: add_xpath(field_name, xpath, *processors, **kwargs)
|
||||
|
||||
Similar to :meth:`ItemLoader.add_value` but receives an XPath instead of a
|
||||
value, which is used to extract a list of unicode strings from the
|
||||
|
|
@ -385,12 +385,12 @@ ItemLoader objects
|
|||
# HTML snippet: <p id="price">the price is $1200</p>
|
||||
loader.add_xpath('price', '//p[@id="price"]', re='the price is (.*)')
|
||||
|
||||
.. method:: replace_xpath(field_name, xpath, \*processors, \**kwargs)
|
||||
.. method:: replace_xpath(field_name, xpath, *processors, **kwargs)
|
||||
|
||||
Similar to :meth:`add_xpath` but replaces collected data instead of
|
||||
adding it.
|
||||
|
||||
.. method:: get_css(css, \*processors, \**kwargs)
|
||||
.. method:: get_css(css, *processors, **kwargs)
|
||||
|
||||
Similar to :meth:`ItemLoader.get_value` but receives a CSS selector
|
||||
instead of a value, which is used to extract a list of unicode strings
|
||||
|
|
@ -410,7 +410,7 @@ ItemLoader objects
|
|||
# HTML snippet: <p id="price">the price is $1200</p>
|
||||
loader.get_css('p#price', TakeFirst(), re='the price is (.*)')
|
||||
|
||||
.. method:: add_css(field_name, css, \*processors, \**kwargs)
|
||||
.. method:: add_css(field_name, css, *processors, **kwargs)
|
||||
|
||||
Similar to :meth:`ItemLoader.add_value` but receives a CSS selector
|
||||
instead of a value, which is used to extract a list of unicode strings
|
||||
|
|
@ -428,7 +428,7 @@ ItemLoader objects
|
|||
# HTML snippet: <p id="price">the price is $1200</p>
|
||||
loader.add_css('price', 'p#price', re='the price is (.*)')
|
||||
|
||||
.. method:: replace_css(field_name, css, \*processors, \**kwargs)
|
||||
.. method:: replace_css(field_name, css, *processors, **kwargs)
|
||||
|
||||
Similar to :meth:`add_css` but replaces collected data instead of
|
||||
adding it.
|
||||
|
|
@ -678,7 +678,7 @@ Here is a list of all built-in processors:
|
|||
>>> proc(['one', 'two', 'three'])
|
||||
'one<br>two<br>three'
|
||||
|
||||
.. class:: Compose(\*functions, \**default_loader_context)
|
||||
.. class:: Compose(*functions, **default_loader_context)
|
||||
|
||||
A processor which is constructed from the composition of the given
|
||||
functions. This means that each input value of this processor is passed to
|
||||
|
|
@ -706,7 +706,7 @@ Here is a list of all built-in processors:
|
|||
active Loader context accessible through the :meth:`ItemLoader.context`
|
||||
attribute.
|
||||
|
||||
.. class:: MapCompose(\*functions, \**default_loader_context)
|
||||
.. class:: MapCompose(*functions, **default_loader_context)
|
||||
|
||||
A processor which is constructed from the composition of the given
|
||||
functions, similar to the :class:`Compose` processor. The difference with
|
||||
|
|
|
|||
|
|
@ -202,6 +202,9 @@ A custom log format can be set for different actions by extending
|
|||
.. autoclass:: scrapy.logformatter.LogFormatter
|
||||
:members:
|
||||
|
||||
|
||||
.. _topics-logging-advanced-customization:
|
||||
|
||||
Advanced customization
|
||||
----------------------
|
||||
|
||||
|
|
@ -262,7 +265,6 @@ scrapy.utils.log module
|
|||
This is an example on how to redirect ``INFO`` or higher messages to a file::
|
||||
|
||||
import logging
|
||||
from scrapy.utils.log import configure_logging
|
||||
|
||||
logging.basicConfig(
|
||||
filename='log.txt',
|
||||
|
|
|
|||
|
|
@ -385,6 +385,51 @@ The meta key is used set retry times per request. When initialized, the
|
|||
:reqmeta:`max_retry_times` meta key takes higher precedence over the
|
||||
:setting:`RETRY_TIMES` setting.
|
||||
|
||||
|
||||
.. _topics-stop-response-download:
|
||||
|
||||
Stopping the download of a Response
|
||||
===================================
|
||||
|
||||
Raising a :exc:`~scrapy.exceptions.StopDownload` exception from a
|
||||
:class:`~scrapy.signals.bytes_received` signal handler will stop the
|
||||
download of a given response. See the following example::
|
||||
|
||||
import scrapy
|
||||
|
||||
|
||||
class StopSpider(scrapy.Spider):
|
||||
name = "stop"
|
||||
start_urls = ["https://docs.scrapy.org/en/latest/"]
|
||||
|
||||
@classmethod
|
||||
def from_crawler(cls, crawler):
|
||||
spider = super().from_crawler(crawler)
|
||||
crawler.signals.connect(spider.on_bytes_received, signal=scrapy.signals.bytes_received)
|
||||
return spider
|
||||
|
||||
def parse(self, response):
|
||||
# 'last_chars' show that the full response was not downloaded
|
||||
yield {"len": len(response.text), "last_chars": response.text[-40:]}
|
||||
|
||||
def on_bytes_received(self, data, request, spider):
|
||||
raise scrapy.exceptions.StopDownload(fail=False)
|
||||
|
||||
which produces the following output::
|
||||
|
||||
2020-05-19 17:26:12 [scrapy.core.engine] INFO: Spider opened
|
||||
2020-05-19 17:26:12 [scrapy.extensions.logstats] INFO: Crawled 0 pages (at 0 pages/min), scraped 0 items (at 0 items/min)
|
||||
2020-05-19 17:26:13 [scrapy.core.downloader.handlers.http11] DEBUG: Download stopped for <GET https://docs.scrapy.org/en/latest/> from signal handler StopSpider.on_bytes_received
|
||||
2020-05-19 17:26:13 [scrapy.core.engine] DEBUG: Crawled (200) <GET https://docs.scrapy.org/en/latest/> (referer: None) ['download_stopped']
|
||||
2020-05-19 17:26:13 [scrapy.core.scraper] DEBUG: Scraped from <200 https://docs.scrapy.org/en/latest/>
|
||||
{'len': 279, 'last_chars': 'dth, initial-scale=1.0">\n \n <title>Scr'}
|
||||
2020-05-19 17:26:13 [scrapy.core.engine] INFO: Closing spider (finished)
|
||||
|
||||
By default, resulting responses are handled by their corresponding errbacks. To
|
||||
call their callback instead, like in this example, pass ``fail=False`` to the
|
||||
:exc:`~scrapy.exceptions.StopDownload` exception.
|
||||
|
||||
|
||||
.. _topics-request-response-ref-request-subclasses:
|
||||
|
||||
Request subclasses
|
||||
|
|
@ -716,9 +761,9 @@ Response objects
|
|||
.. versionadded:: 2.1.0
|
||||
|
||||
The IP address of the server from which the Response originated.
|
||||
|
||||
|
||||
This attribute is currently only populated by the HTTP 1.1 download
|
||||
handler, i.e. for ``http(s)`` responses. For other handlers,
|
||||
handler, i.e. for ``http(s)`` responses. For other handlers,
|
||||
:attr:`ip_address` is always ``None``.
|
||||
|
||||
.. method:: Response.copy()
|
||||
|
|
@ -834,6 +879,11 @@ TextResponse objects
|
|||
|
||||
.. automethod:: TextResponse.follow_all
|
||||
|
||||
.. automethod:: TextResponse.json()
|
||||
|
||||
Returns a Python object from deserialized JSON document.
|
||||
The result is cached after the first call.
|
||||
|
||||
|
||||
HtmlResponse objects
|
||||
--------------------
|
||||
|
|
|
|||
|
|
@ -373,6 +373,8 @@ request_left_downloader
|
|||
bytes_received
|
||||
~~~~~~~~~~~~~~
|
||||
|
||||
.. versionadded:: 2.2
|
||||
|
||||
.. signal:: bytes_received
|
||||
.. function:: bytes_received(data, request, spider)
|
||||
|
||||
|
|
@ -385,14 +387,19 @@ bytes_received
|
|||
This signal does not support returning deferreds from its handlers.
|
||||
|
||||
:param data: the data received by the download handler
|
||||
:type spider: :class:`bytes` object
|
||||
:type data: :class:`bytes` object
|
||||
|
||||
:param request: the request that generated the response
|
||||
:param request: the request that generated the download
|
||||
:type request: :class:`~scrapy.http.Request` object
|
||||
|
||||
:param spider: the spider associated with the response
|
||||
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||
|
||||
.. note:: Handlers of this signal can stop the download of a response while it
|
||||
is in progress by raising the :exc:`~scrapy.exceptions.StopDownload`
|
||||
exception. Please refer to the :ref:`topics-stop-response-download` topic
|
||||
for additional information and examples.
|
||||
|
||||
Response signals
|
||||
----------------
|
||||
|
||||
|
|
|
|||
|
|
@ -121,7 +121,7 @@ scrapy.Spider
|
|||
send log messages through it as described on
|
||||
:ref:`topics-logging-from-spiders`.
|
||||
|
||||
.. method:: from_crawler(crawler, \*args, \**kwargs)
|
||||
.. method:: from_crawler(crawler, *args, **kwargs)
|
||||
|
||||
This is the class method used by Scrapy to create your spiders.
|
||||
|
||||
|
|
|
|||
|
|
@ -27,18 +27,16 @@ flake8-ignore =
|
|||
# Exclude files that are meant to provide top-level imports
|
||||
# E402: Module level import not at top of file
|
||||
# F401: Module imported but unused
|
||||
scrapy/__init__.py E402
|
||||
scrapy/core/downloader/handlers/http.py F401
|
||||
scrapy/http/__init__.py F401
|
||||
scrapy/linkextractors/__init__.py E402 F401
|
||||
scrapy/selector/__init__.py F401
|
||||
scrapy/spiders/__init__.py E402 F401
|
||||
|
||||
# Issues pending a review:
|
||||
scrapy/__init__.py E402
|
||||
scrapy/selector/__init__.py F403
|
||||
scrapy/spiders/__init__.py E402
|
||||
scrapy/utils/http.py F403
|
||||
scrapy/utils/markup.py F403
|
||||
scrapy/utils/multipart.py F403
|
||||
scrapy/utils/url.py F403 F405
|
||||
tests/test_loader.py E741
|
||||
tests/test_webclient.py E402
|
||||
|
|
|
|||
|
|
@ -2,33 +2,11 @@
|
|||
Scrapy - a web crawling and web scraping framework written for Python
|
||||
"""
|
||||
|
||||
__all__ = ['__version__', 'version_info', 'twisted_version',
|
||||
'Spider', 'Request', 'FormRequest', 'Selector', 'Item', 'Field']
|
||||
|
||||
# Scrapy version
|
||||
import pkgutil
|
||||
__version__ = pkgutil.get_data(__package__, 'VERSION').decode('ascii').strip()
|
||||
version_info = tuple(int(v) if v.isdigit() else v
|
||||
for v in __version__.split('.'))
|
||||
del pkgutil
|
||||
|
||||
# Check minimum required Python version
|
||||
import sys
|
||||
if sys.version_info < (3, 5):
|
||||
print("Scrapy %s requires Python 3.5" % __version__)
|
||||
sys.exit(1)
|
||||
|
||||
# Ignore noisy twisted deprecation warnings
|
||||
import warnings
|
||||
warnings.filterwarnings('ignore', category=DeprecationWarning, module='twisted')
|
||||
del warnings
|
||||
|
||||
# Apply monkey patches to fix issues in external libraries
|
||||
from scrapy import _monkeypatches
|
||||
del _monkeypatches
|
||||
|
||||
from twisted import version as _txv
|
||||
twisted_version = (_txv.major, _txv.minor, _txv.micro)
|
||||
|
||||
# Declare top-level shortcuts
|
||||
from scrapy.spiders import Spider
|
||||
|
|
@ -36,4 +14,29 @@ from scrapy.http import Request, FormRequest
|
|||
from scrapy.selector import Selector
|
||||
from scrapy.item import Item, Field
|
||||
|
||||
|
||||
__all__ = [
|
||||
'__version__', 'version_info', 'twisted_version', 'Spider',
|
||||
'Request', 'FormRequest', 'Selector', 'Item', 'Field',
|
||||
]
|
||||
|
||||
|
||||
# Scrapy and Twisted versions
|
||||
__version__ = pkgutil.get_data(__package__, 'VERSION').decode('ascii').strip()
|
||||
version_info = tuple(int(v) if v.isdigit() else v for v in __version__.split('.'))
|
||||
twisted_version = (_txv.major, _txv.minor, _txv.micro)
|
||||
|
||||
|
||||
# Check minimum required Python version
|
||||
if sys.version_info < (3, 5):
|
||||
print("Scrapy %s requires Python 3.5" % __version__)
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
# Ignore noisy twisted deprecation warnings
|
||||
warnings.filterwarnings('ignore', category=DeprecationWarning, module='twisted')
|
||||
|
||||
|
||||
del pkgutil
|
||||
del sys
|
||||
del warnings
|
||||
|
|
|
|||
|
|
@ -1,11 +0,0 @@
|
|||
import copyreg
|
||||
|
||||
|
||||
# Undo what Twisted's perspective broker adds to pickle register
|
||||
# to prevent bugs like Twisted#7989 while serializing requests
|
||||
import twisted.persisted.styles # NOQA
|
||||
# Remove only entries with twisted serializers for non-twisted types.
|
||||
for k, v in frozenset(copyreg.dispatch_table.items()):
|
||||
if not str(getattr(k, '__module__', '')).startswith('twisted') \
|
||||
and str(getattr(v, '__module__', '')).startswith('twisted'):
|
||||
copyreg.dispatch_table.pop(k)
|
||||
|
|
@ -5,7 +5,7 @@ import os
|
|||
from optparse import OptionGroup
|
||||
from twisted.python import failure
|
||||
|
||||
from scrapy.utils.conf import arglist_to_dict
|
||||
from scrapy.utils.conf import arglist_to_dict, feed_process_params_from_cli
|
||||
from scrapy.exceptions import UsageError
|
||||
|
||||
|
||||
|
|
@ -104,3 +104,27 @@ class ScrapyCommand:
|
|||
Entry point for running commands
|
||||
"""
|
||||
raise NotImplementedError
|
||||
|
||||
|
||||
class BaseRunSpiderCommand(ScrapyCommand):
|
||||
"""
|
||||
Common class used to share functionality between the crawl and runspider commands
|
||||
"""
|
||||
def add_options(self, parser):
|
||||
ScrapyCommand.add_options(self, parser)
|
||||
parser.add_option("-a", dest="spargs", action="append", default=[], metavar="NAME=VALUE",
|
||||
help="set spider argument (may be repeated)")
|
||||
parser.add_option("-o", "--output", metavar="FILE", action="append",
|
||||
help="dump scraped items into FILE (use - for stdout)")
|
||||
parser.add_option("-t", "--output-format", metavar="FORMAT",
|
||||
help="format to use for dumping items with -o")
|
||||
|
||||
def process_options(self, args, opts):
|
||||
ScrapyCommand.process_options(self, args, opts)
|
||||
try:
|
||||
opts.spargs = arglist_to_dict(opts.spargs)
|
||||
except ValueError:
|
||||
raise UsageError("Invalid -a value, use -a NAME=VALUE", print_help=False)
|
||||
if opts.output:
|
||||
feeds = feed_process_params_from_cli(self.settings, opts.output, opts.output_format)
|
||||
self.settings.set('FEEDS', feeds, priority='cmdline')
|
||||
|
|
|
|||
|
|
@ -1,9 +1,8 @@
|
|||
from scrapy.commands import ScrapyCommand
|
||||
from scrapy.utils.conf import arglist_to_dict, feed_process_params_from_cli
|
||||
from scrapy.commands import BaseRunSpiderCommand
|
||||
from scrapy.exceptions import UsageError
|
||||
|
||||
|
||||
class Command(ScrapyCommand):
|
||||
class Command(BaseRunSpiderCommand):
|
||||
|
||||
requires_project = True
|
||||
|
||||
|
|
@ -13,25 +12,6 @@ class Command(ScrapyCommand):
|
|||
def short_desc(self):
|
||||
return "Run a spider"
|
||||
|
||||
def add_options(self, parser):
|
||||
ScrapyCommand.add_options(self, parser)
|
||||
parser.add_option("-a", dest="spargs", action="append", default=[], metavar="NAME=VALUE",
|
||||
help="set spider argument (may be repeated)")
|
||||
parser.add_option("-o", "--output", metavar="FILE", action="append",
|
||||
help="dump scraped items into FILE (use - for stdout)")
|
||||
parser.add_option("-t", "--output-format", metavar="FORMAT",
|
||||
help="format to use for dumping items with -o")
|
||||
|
||||
def process_options(self, args, opts):
|
||||
ScrapyCommand.process_options(self, args, opts)
|
||||
try:
|
||||
opts.spargs = arglist_to_dict(opts.spargs)
|
||||
except ValueError:
|
||||
raise UsageError("Invalid -a value, use -a NAME=VALUE", print_help=False)
|
||||
if opts.output:
|
||||
feeds = feed_process_params_from_cli(self.settings, opts.output, opts.output_format)
|
||||
self.settings.set('FEEDS', feeds, priority='cmdline')
|
||||
|
||||
def run(self, args, opts):
|
||||
if len(args) < 1:
|
||||
raise UsageError()
|
||||
|
|
|
|||
|
|
@ -3,9 +3,8 @@ import os
|
|||
from importlib import import_module
|
||||
|
||||
from scrapy.utils.spider import iter_spider_classes
|
||||
from scrapy.commands import ScrapyCommand
|
||||
from scrapy.exceptions import UsageError
|
||||
from scrapy.utils.conf import arglist_to_dict, feed_process_params_from_cli
|
||||
from scrapy.commands import BaseRunSpiderCommand
|
||||
|
||||
|
||||
def _import_file(filepath):
|
||||
|
|
@ -24,7 +23,7 @@ def _import_file(filepath):
|
|||
return module
|
||||
|
||||
|
||||
class Command(ScrapyCommand):
|
||||
class Command(BaseRunSpiderCommand):
|
||||
|
||||
requires_project = False
|
||||
default_settings = {'SPIDER_LOADER_WARN_ONLY': True}
|
||||
|
|
@ -38,25 +37,6 @@ class Command(ScrapyCommand):
|
|||
def long_desc(self):
|
||||
return "Run the spider defined in the given file"
|
||||
|
||||
def add_options(self, parser):
|
||||
ScrapyCommand.add_options(self, parser)
|
||||
parser.add_option("-a", dest="spargs", action="append", default=[], metavar="NAME=VALUE",
|
||||
help="set spider argument (may be repeated)")
|
||||
parser.add_option("-o", "--output", metavar="FILE", action="append",
|
||||
help="dump scraped items into FILE (use - for stdout)")
|
||||
parser.add_option("-t", "--output-format", metavar="FORMAT",
|
||||
help="format to use for dumping items with -o")
|
||||
|
||||
def process_options(self, args, opts):
|
||||
ScrapyCommand.process_options(self, args, opts)
|
||||
try:
|
||||
opts.spargs = arglist_to_dict(opts.spargs)
|
||||
except ValueError:
|
||||
raise UsageError("Invalid -a value, use -a NAME=VALUE", print_help=False)
|
||||
if opts.output:
|
||||
feeds = feed_process_params_from_cli(self.settings, opts.output, opts.output_format)
|
||||
self.settings.set('FEEDS', feeds, priority='cmdline')
|
||||
|
||||
def run(self, args, opts):
|
||||
if len(args) != 1:
|
||||
raise UsageError()
|
||||
|
|
|
|||
|
|
@ -12,6 +12,7 @@ from urllib.parse import urldefrag
|
|||
from twisted.internet import defer, protocol, ssl
|
||||
from twisted.internet.endpoints import TCP4ClientEndpoint
|
||||
from twisted.internet.error import TimeoutError
|
||||
from twisted.python.failure import Failure
|
||||
from twisted.web.client import Agent, HTTPConnectionPool, ResponseDone, ResponseFailed, URI
|
||||
from twisted.web.http import _DataLoss, PotentialDataLoss
|
||||
from twisted.web.http_headers import Headers as TxHeaders
|
||||
|
|
@ -21,7 +22,7 @@ from zope.interface import implementer
|
|||
from scrapy import signals
|
||||
from scrapy.core.downloader.tls import openssl_methods
|
||||
from scrapy.core.downloader.webclient import _parse
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning, StopDownload
|
||||
from scrapy.http import Headers
|
||||
from scrapy.responsetypes import responsetypes
|
||||
from scrapy.utils.misc import create_instance, load_object
|
||||
|
|
@ -431,7 +432,7 @@ class ScrapyAgent:
|
|||
def _cb_bodydone(self, result, request, url):
|
||||
headers = Headers(result["txresponse"].headers.getAllRawHeaders())
|
||||
respcls = responsetypes.from_args(headers=headers, url=url, body=result["body"])
|
||||
return respcls(
|
||||
response = respcls(
|
||||
url=url,
|
||||
status=int(result["txresponse"].code),
|
||||
headers=headers,
|
||||
|
|
@ -440,6 +441,10 @@ class ScrapyAgent:
|
|||
certificate=result["certificate"],
|
||||
ip_address=result["ip_address"],
|
||||
)
|
||||
if result.get("failure"):
|
||||
result["failure"].value.response = response
|
||||
return result["failure"]
|
||||
return response
|
||||
|
||||
|
||||
@implementer(IBodyProducer)
|
||||
|
|
@ -477,6 +482,16 @@ class _ResponseReader(protocol.Protocol):
|
|||
self._ip_address = None
|
||||
self._crawler = crawler
|
||||
|
||||
def _finish_response(self, flags=None, failure=None):
|
||||
self._finished.callback({
|
||||
"txresponse": self._txresponse,
|
||||
"body": self._bodybuf.getvalue(),
|
||||
"flags": flags,
|
||||
"certificate": self._certificate,
|
||||
"ip_address": self._ip_address,
|
||||
"failure": failure,
|
||||
})
|
||||
|
||||
def connectionMade(self):
|
||||
if self._certificate is None:
|
||||
with suppress(AttributeError):
|
||||
|
|
@ -493,12 +508,19 @@ class _ResponseReader(protocol.Protocol):
|
|||
self._bodybuf.write(bodyBytes)
|
||||
self._bytes_received += len(bodyBytes)
|
||||
|
||||
self._crawler.signals.send_catch_log(
|
||||
bytes_received_result = self._crawler.signals.send_catch_log(
|
||||
signal=signals.bytes_received,
|
||||
data=bodyBytes,
|
||||
request=self._request,
|
||||
spider=self._crawler.spider,
|
||||
)
|
||||
for handler, result in bytes_received_result:
|
||||
if isinstance(result, Failure) and isinstance(result.value, StopDownload):
|
||||
logger.debug("Download stopped for %(request)s from signal handler %(handler)s",
|
||||
{"request": self._request, "handler": handler.__qualname__})
|
||||
self.transport._producer.loseConnection()
|
||||
failure = result if result.value.fail else None
|
||||
self._finish_response(flags=["download_stopped"], failure=failure)
|
||||
|
||||
if self._maxsize and self._bytes_received > self._maxsize:
|
||||
logger.error("Received (%(bytes)s) bytes larger than download "
|
||||
|
|
@ -521,36 +543,17 @@ class _ResponseReader(protocol.Protocol):
|
|||
if self._finished.called:
|
||||
return
|
||||
|
||||
body = self._bodybuf.getvalue()
|
||||
if reason.check(ResponseDone):
|
||||
self._finished.callback({
|
||||
"txresponse": self._txresponse,
|
||||
"body": body,
|
||||
"flags": None,
|
||||
"certificate": self._certificate,
|
||||
"ip_address": self._ip_address,
|
||||
})
|
||||
self._finish_response()
|
||||
return
|
||||
|
||||
if reason.check(PotentialDataLoss):
|
||||
self._finished.callback({
|
||||
"txresponse": self._txresponse,
|
||||
"body": body,
|
||||
"flags": ["partial"],
|
||||
"certificate": self._certificate,
|
||||
"ip_address": self._ip_address,
|
||||
})
|
||||
self._finish_response(flags=["partial"])
|
||||
return
|
||||
|
||||
if reason.check(ResponseFailed) and any(r.check(_DataLoss) for r in reason.value.reasons):
|
||||
if not self._fail_on_dataloss:
|
||||
self._finished.callback({
|
||||
"txresponse": self._txresponse,
|
||||
"body": body,
|
||||
"flags": ["dataloss"],
|
||||
"certificate": self._certificate,
|
||||
"ip_address": self._ip_address,
|
||||
})
|
||||
self._finish_response(flags=["dataloss"])
|
||||
return
|
||||
|
||||
elif not self._fail_on_dataloss_warned:
|
||||
|
|
|
|||
|
|
@ -29,8 +29,7 @@ class CookiesMiddleware:
|
|||
|
||||
cookiejarkey = request.meta.get("cookiejar")
|
||||
jar = self.jars[cookiejarkey]
|
||||
cookies = self._get_request_cookies(jar, request)
|
||||
for cookie in cookies:
|
||||
for cookie in self._get_request_cookies(jar, request):
|
||||
jar.set_cookie_if_ok(cookie, request)
|
||||
|
||||
# set Cookie header
|
||||
|
|
@ -68,28 +67,65 @@ class CookiesMiddleware:
|
|||
msg = "Received cookies from: {}\n{}".format(response, cookies)
|
||||
logger.debug(msg, extra={'spider': spider})
|
||||
|
||||
def _format_cookie(self, cookie):
|
||||
# build cookie string
|
||||
cookie_str = '%s=%s' % (cookie['name'], cookie['value'])
|
||||
|
||||
if cookie.get('path', None):
|
||||
cookie_str += '; Path=%s' % cookie['path']
|
||||
if cookie.get('domain', None):
|
||||
cookie_str += '; Domain=%s' % cookie['domain']
|
||||
def _format_cookie(self, cookie, request):
|
||||
"""
|
||||
Given a dict consisting of cookie components, return its string representation.
|
||||
Decode from bytes if necessary.
|
||||
"""
|
||||
decoded = {}
|
||||
for key in ("name", "value", "path", "domain"):
|
||||
if not cookie.get(key):
|
||||
if key in ("name", "value"):
|
||||
msg = "Invalid cookie found in request {}: {} ('{}' is missing)"
|
||||
logger.warning(msg.format(request, cookie, key))
|
||||
return
|
||||
continue
|
||||
if isinstance(cookie[key], str):
|
||||
decoded[key] = cookie[key]
|
||||
else:
|
||||
try:
|
||||
decoded[key] = cookie[key].decode("utf8")
|
||||
except UnicodeDecodeError:
|
||||
logger.warning("Non UTF-8 encoded cookie found in request %s: %s",
|
||||
request, cookie)
|
||||
decoded[key] = cookie[key].decode("latin1", errors="replace")
|
||||
|
||||
cookie_str = "{}={}".format(decoded.pop("name"), decoded.pop("value"))
|
||||
for key, value in decoded.items(): # path, domain
|
||||
cookie_str += "; {}={}".format(key.capitalize(), value)
|
||||
return cookie_str
|
||||
|
||||
def _get_request_cookies(self, jar, request):
|
||||
if isinstance(request.cookies, dict):
|
||||
cookie_list = [
|
||||
{'name': k, 'value': v}
|
||||
for k, v in request.cookies.items()
|
||||
]
|
||||
else:
|
||||
cookie_list = request.cookies
|
||||
"""
|
||||
Extract cookies from a Request. Values from the `Request.cookies` attribute
|
||||
take precedence over values from the `Cookie` request header.
|
||||
"""
|
||||
def get_cookies_from_header(jar, request):
|
||||
cookie_header = request.headers.get("Cookie")
|
||||
if not cookie_header:
|
||||
return []
|
||||
cookie_gen_bytes = (s.strip() for s in cookie_header.split(b";"))
|
||||
cookie_list_unicode = []
|
||||
for cookie_bytes in cookie_gen_bytes:
|
||||
try:
|
||||
cookie_unicode = cookie_bytes.decode("utf8")
|
||||
except UnicodeDecodeError:
|
||||
logger.warning("Non UTF-8 encoded cookie found in request %s: %s",
|
||||
request, cookie_bytes)
|
||||
cookie_unicode = cookie_bytes.decode("latin1", errors="replace")
|
||||
cookie_list_unicode.append(cookie_unicode)
|
||||
response = Response(request.url, headers={"Set-Cookie": cookie_list_unicode})
|
||||
return jar.make_cookies(response, request)
|
||||
|
||||
cookies = [self._format_cookie(x) for x in cookie_list]
|
||||
headers = {'Set-Cookie': cookies}
|
||||
response = Response(request.url, headers=headers)
|
||||
def get_cookies_from_attribute(jar, request):
|
||||
if not request.cookies:
|
||||
return []
|
||||
elif isinstance(request.cookies, dict):
|
||||
cookies = ({"name": k, "value": v} for k, v in request.cookies.items())
|
||||
else:
|
||||
cookies = request.cookies
|
||||
formatted = filter(None, (self._format_cookie(c, request) for c in cookies))
|
||||
response = Response(request.url, headers={"Set-Cookie": formatted})
|
||||
return jar.make_cookies(response, request)
|
||||
|
||||
return jar.make_cookies(response, request)
|
||||
return get_cookies_from_header(jar, request) + get_cookies_from_attribute(jar, request)
|
||||
|
|
|
|||
|
|
@ -41,6 +41,18 @@ class CloseSpider(Exception):
|
|||
self.reason = reason
|
||||
|
||||
|
||||
class StopDownload(Exception):
|
||||
"""
|
||||
Stop the download of the body for a given response.
|
||||
The 'fail' boolean parameter indicates whether or not the resulting partial response
|
||||
should be handled by the request errback. Note that 'fail' is a keyword-only argument.
|
||||
"""
|
||||
|
||||
def __init__(self, *, fail=True):
|
||||
super().__init__()
|
||||
self.fail = fail
|
||||
|
||||
|
||||
# Items
|
||||
|
||||
|
||||
|
|
@ -59,6 +71,7 @@ class NotSupported(Exception):
|
|||
|
||||
class UsageError(Exception):
|
||||
"""To indicate a command-line usage error"""
|
||||
|
||||
def __init__(self, *a, **kw):
|
||||
self.print_help = kw.pop('print_help', True)
|
||||
super(UsageError, self).__init__(*a, **kw)
|
||||
|
|
|
|||
|
|
@ -5,6 +5,7 @@ discovering (through HTTP headers) to base Response class.
|
|||
See documentation in docs/topics/request-response.rst
|
||||
"""
|
||||
|
||||
import json
|
||||
import warnings
|
||||
from contextlib import suppress
|
||||
from typing import Generator
|
||||
|
|
@ -21,10 +22,13 @@ from scrapy.http.response import Response
|
|||
from scrapy.utils.python import memoizemethod_noargs, to_unicode
|
||||
from scrapy.utils.response import get_base_url
|
||||
|
||||
_NONE = object()
|
||||
|
||||
|
||||
class TextResponse(Response):
|
||||
|
||||
_DEFAULT_ENCODING = 'ascii'
|
||||
_cached_decoded_json = _NONE
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
self._encoding = kwargs.pop('encoding', None)
|
||||
|
|
@ -68,6 +72,14 @@ class TextResponse(Response):
|
|||
ScrapyDeprecationWarning, stacklevel=2)
|
||||
return self.text
|
||||
|
||||
def json(self):
|
||||
"""
|
||||
Deserialize a JSON document to a Python object.
|
||||
"""
|
||||
if self._cached_decoded_json is _NONE:
|
||||
self._cached_decoded_json = json.loads(self.text)
|
||||
return self._cached_decoded_json
|
||||
|
||||
@property
|
||||
def text(self):
|
||||
""" Body as unicode """
|
||||
|
|
|
|||
|
|
@ -133,4 +133,4 @@ class FilteringLinkExtractor:
|
|||
|
||||
|
||||
# Top-level imports
|
||||
from scrapy.linkextractors.lxmlhtml import LxmlLinkExtractor as LinkExtractor # noqa: F401
|
||||
from scrapy.linkextractors.lxmlhtml import LxmlLinkExtractor as LinkExtractor
|
||||
|
|
|
|||
|
|
@ -1,4 +1,6 @@
|
|||
"""
|
||||
Selectors
|
||||
"""
|
||||
from scrapy.selector.unified import * # noqa: F401
|
||||
|
||||
# top-level imports
|
||||
from scrapy.selector.unified import Selector, SelectorList
|
||||
|
|
|
|||
|
|
@ -110,6 +110,6 @@ class Spider(object_ref):
|
|||
|
||||
|
||||
# Top-level imports
|
||||
from scrapy.spiders.crawl import CrawlSpider, Rule # noqa: F401
|
||||
from scrapy.spiders.feed import XMLFeedSpider, CSVFeedSpider # noqa: F401
|
||||
from scrapy.spiders.sitemap import SitemapSpider # noqa: F401
|
||||
from scrapy.spiders.crawl import CrawlSpider, Rule
|
||||
from scrapy.spiders.feed import XMLFeedSpider, CSVFeedSpider
|
||||
from scrapy.spiders.sitemap import SitemapSpider
|
||||
|
|
|
|||
|
|
@ -105,8 +105,8 @@ class LocalWeakReferencedCache(weakref.WeakKeyDictionary):
|
|||
def __getitem__(self, key):
|
||||
try:
|
||||
return super(LocalWeakReferencedCache, self).__getitem__(key)
|
||||
except TypeError:
|
||||
return None # key is not weak-referenceable, it's not cached
|
||||
except (TypeError, KeyError):
|
||||
return None # key is either not weak-referenceable or not cached
|
||||
|
||||
|
||||
class SequenceExclude:
|
||||
|
|
|
|||
|
|
@ -5,13 +5,14 @@ import logging
|
|||
from twisted.internet.defer import DeferredList, Deferred
|
||||
from twisted.python.failure import Failure
|
||||
|
||||
from pydispatch.dispatcher import Any, Anonymous, liveReceivers, \
|
||||
getAllReceivers, disconnect
|
||||
from pydispatch.dispatcher import Anonymous, Any, disconnect, getAllReceivers, liveReceivers
|
||||
from pydispatch.robustapply import robustApply
|
||||
|
||||
from scrapy.exceptions import StopDownload
|
||||
from scrapy.utils.defer import maybeDeferred_coro
|
||||
from scrapy.utils.log import failure_to_exc_info
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
|
|
@ -23,7 +24,7 @@ def send_catch_log(signal=Any, sender=Anonymous, *arguments, **named):
|
|||
"""Like pydispatcher.robust.sendRobust but it also logs errors and returns
|
||||
Failures instead of exceptions.
|
||||
"""
|
||||
dont_log = named.pop('dont_log', _IgnoredException)
|
||||
dont_log = (named.pop('dont_log', _IgnoredException), StopDownload)
|
||||
spider = named.get('spider', None)
|
||||
responses = []
|
||||
for receiver in liveReceivers(getAllReceivers(sender, signal)):
|
||||
|
|
|
|||
|
|
@ -1,5 +1,5 @@
|
|||
import logging
|
||||
import inspect
|
||||
import logging
|
||||
|
||||
from scrapy.spiders import Spider
|
||||
from scrapy.utils.defer import deferred_from_coro, _isasyncgen
|
||||
|
|
@ -18,7 +18,11 @@ def iterate_spider_output(result):
|
|||
d = deferred_from_coro(collect_asyncgen(result))
|
||||
d.addCallback(iterate_spider_output)
|
||||
return d
|
||||
return arg_to_iter(deferred_from_coro(result))
|
||||
elif inspect.iscoroutine(result):
|
||||
d = deferred_from_coro(result)
|
||||
d.addCallback(iterate_spider_output)
|
||||
return d
|
||||
return arg_to_iter(result)
|
||||
|
||||
|
||||
def iter_spider_classes(module):
|
||||
|
|
|
|||
|
|
@ -2,7 +2,8 @@
|
|||
jmespath
|
||||
mitmproxy; python_version >= '3.6'
|
||||
mitmproxy<4.0.0; python_version < '3.6'
|
||||
pytest < 5.4
|
||||
# https://github.com/pytest-dev/pytest-twisted/issues/93
|
||||
pytest != 5.4, != 5.4.1
|
||||
pytest-cov
|
||||
pytest-twisted >= 1.11
|
||||
pytest-xdist
|
||||
|
|
|
|||
|
|
@ -7,6 +7,8 @@ from urllib.parse import urlencode
|
|||
|
||||
from twisted.internet import defer
|
||||
|
||||
from scrapy import signals
|
||||
from scrapy.exceptions import StopDownload
|
||||
from scrapy.http import Request
|
||||
from scrapy.item import Item
|
||||
from scrapy.linkextractors import LinkExtractor
|
||||
|
|
@ -117,6 +119,17 @@ class AsyncDefAsyncioReturnSpider(SimpleSpider):
|
|||
return [{'id': 1}, {'id': 2}]
|
||||
|
||||
|
||||
class AsyncDefAsyncioReturnSingleElementSpider(SimpleSpider):
|
||||
|
||||
name = "asyncdef_asyncio_return_single_element"
|
||||
|
||||
async def parse(self, response):
|
||||
await asyncio.sleep(0.1)
|
||||
status = await get_from_asyncio_queue(response.status)
|
||||
self.logger.info("Got response %d" % status)
|
||||
return {"foo": 42}
|
||||
|
||||
|
||||
class AsyncDefAsyncioReqsReturnSpider(SimpleSpider):
|
||||
|
||||
name = 'asyncdef_asyncio_reqs_return'
|
||||
|
|
@ -267,3 +280,36 @@ class CrawlSpiderWithErrback(MockServerSpider, CrawlSpider):
|
|||
|
||||
def errback(self, failure):
|
||||
self.logger.info('[errback] status %i', failure.value.response.status)
|
||||
|
||||
|
||||
class BytesReceivedCallbackSpider(MetaSpider):
|
||||
|
||||
full_response_length = 2**18
|
||||
|
||||
@classmethod
|
||||
def from_crawler(cls, crawler, *args, **kwargs):
|
||||
spider = super().from_crawler(crawler, *args, **kwargs)
|
||||
crawler.signals.connect(spider.bytes_received, signals.bytes_received)
|
||||
return spider
|
||||
|
||||
def start_requests(self):
|
||||
body = b"a" * self.full_response_length
|
||||
url = self.mockserver.url("/alpayload")
|
||||
yield Request(url, method="POST", body=body, errback=self.errback)
|
||||
|
||||
def parse(self, response):
|
||||
self.meta["response"] = response
|
||||
|
||||
def errback(self, failure):
|
||||
self.meta["failure"] = failure
|
||||
|
||||
def bytes_received(self, data, request, spider):
|
||||
self.meta["bytes_received"] = data
|
||||
raise StopDownload(fail=False)
|
||||
|
||||
|
||||
class BytesReceivedErrbackSpider(BytesReceivedCallbackSpider):
|
||||
|
||||
def bytes_received(self, data, request, spider):
|
||||
self.meta["bytes_received"] = data
|
||||
raise StopDownload(fail=True)
|
||||
|
|
|
|||
|
|
@ -9,17 +9,32 @@ from pytest import mark
|
|||
from testfixtures import LogCapture
|
||||
from twisted.internet import defer
|
||||
from twisted.internet.ssl import Certificate
|
||||
from twisted.python.failure import Failure
|
||||
from twisted.trial.unittest import TestCase
|
||||
|
||||
from scrapy import signals
|
||||
from scrapy.crawler import CrawlerRunner
|
||||
from scrapy.exceptions import StopDownload
|
||||
from scrapy.http import Request
|
||||
from scrapy.http.response import Response
|
||||
from scrapy.utils.python import to_unicode
|
||||
from tests.mockserver import MockServer
|
||||
from tests.spiders import (FollowAllSpider, DelaySpider, SimpleSpider, BrokenStartRequestsSpider,
|
||||
SingleRequestSpider, DuplicateStartRequestsSpider, CrawlSpiderWithErrback,
|
||||
AsyncDefSpider, AsyncDefAsyncioSpider, AsyncDefAsyncioReturnSpider,
|
||||
AsyncDefAsyncioReqsReturnSpider)
|
||||
from tests.spiders import (
|
||||
AsyncDefAsyncioReqsReturnSpider,
|
||||
AsyncDefAsyncioReturnSingleElementSpider,
|
||||
AsyncDefAsyncioReturnSpider,
|
||||
AsyncDefAsyncioSpider,
|
||||
AsyncDefSpider,
|
||||
BrokenStartRequestsSpider,
|
||||
BytesReceivedCallbackSpider,
|
||||
BytesReceivedErrbackSpider,
|
||||
CrawlSpiderWithErrback,
|
||||
DelaySpider,
|
||||
DuplicateStartRequestsSpider,
|
||||
FollowAllSpider,
|
||||
SimpleSpider,
|
||||
SingleRequestSpider,
|
||||
)
|
||||
|
||||
|
||||
class CrawlTestCase(TestCase):
|
||||
|
|
@ -350,6 +365,21 @@ with multiples lines
|
|||
self.assertIn({'id': 1}, items)
|
||||
self.assertIn({'id': 2}, items)
|
||||
|
||||
@mark.only_asyncio()
|
||||
@defer.inlineCallbacks
|
||||
def test_async_def_asyncio_parse_items_single_element(self):
|
||||
items = []
|
||||
|
||||
def _on_item_scraped(item):
|
||||
items.append(item)
|
||||
|
||||
crawler = self.runner.create_crawler(AsyncDefAsyncioReturnSingleElementSpider)
|
||||
crawler.signals.connect(_on_item_scraped, signals.item_scraped)
|
||||
with LogCapture() as log:
|
||||
yield crawler.crawl(self.mockserver.url("/status?n=200"), mockserver=self.mockserver)
|
||||
self.assertIn("Got response 200", str(log))
|
||||
self.assertIn({"foo": 42}, items)
|
||||
|
||||
@mark.skipif(sys.version_info < (3, 6), reason="Async generators require Python 3.6 or higher")
|
||||
@mark.only_asyncio()
|
||||
@defer.inlineCallbacks
|
||||
|
|
@ -457,3 +487,27 @@ with multiples lines
|
|||
ip_address = crawler.spider.meta['responses'][0].ip_address
|
||||
self.assertIsInstance(ip_address, IPv4Address)
|
||||
self.assertEqual(str(ip_address), gethostbyname(expected_netloc))
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_stop_download_callback(self):
|
||||
crawler = self.runner.create_crawler(BytesReceivedCallbackSpider)
|
||||
yield crawler.crawl(mockserver=self.mockserver)
|
||||
self.assertIsNone(crawler.spider.meta.get("failure"))
|
||||
self.assertIsInstance(crawler.spider.meta["response"], Response)
|
||||
self.assertEqual(crawler.spider.meta["response"].body, crawler.spider.meta.get("bytes_received"))
|
||||
self.assertLess(len(crawler.spider.meta["response"].body), crawler.spider.full_response_length)
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_stop_download_errback(self):
|
||||
crawler = self.runner.create_crawler(BytesReceivedErrbackSpider)
|
||||
yield crawler.crawl(mockserver=self.mockserver)
|
||||
self.assertIsNone(crawler.spider.meta.get("response"))
|
||||
self.assertIsInstance(crawler.spider.meta["failure"], Failure)
|
||||
self.assertIsInstance(crawler.spider.meta["failure"].value, StopDownload)
|
||||
self.assertIsInstance(crawler.spider.meta["failure"].value.response, Response)
|
||||
self.assertEqual(
|
||||
crawler.spider.meta["failure"].value.response.body,
|
||||
crawler.spider.meta.get("bytes_received"))
|
||||
self.assertLess(
|
||||
len(crawler.spider.meta["failure"].value.response.body),
|
||||
crawler.spider.full_response_length)
|
||||
|
|
|
|||
|
|
@ -1,20 +1,21 @@
|
|||
import re
|
||||
import logging
|
||||
from unittest import TestCase
|
||||
from testfixtures import LogCapture
|
||||
from unittest import TestCase
|
||||
|
||||
from scrapy.downloadermiddlewares.cookies import CookiesMiddleware
|
||||
from scrapy.downloadermiddlewares.defaultheaders import DefaultHeadersMiddleware
|
||||
from scrapy.exceptions import NotConfigured
|
||||
from scrapy.http import Response, Request
|
||||
from scrapy.spiders import Spider
|
||||
from scrapy.utils.python import to_bytes
|
||||
from scrapy.utils.test import get_crawler
|
||||
from scrapy.exceptions import NotConfigured
|
||||
from scrapy.downloadermiddlewares.cookies import CookiesMiddleware
|
||||
|
||||
|
||||
class CookiesMiddlewareTest(TestCase):
|
||||
|
||||
def assertCookieValEqual(self, first, second, msg=None):
|
||||
def split_cookies(cookies):
|
||||
return sorted(re.split(r";\s*", cookies.decode("latin1")))
|
||||
return sorted([s.strip() for s in to_bytes(cookies).split(b";")])
|
||||
return self.assertEqual(split_cookies(first), split_cookies(second), msg=msg)
|
||||
|
||||
def setUp(self):
|
||||
|
|
@ -61,12 +62,13 @@ class CookiesMiddlewareTest(TestCase):
|
|||
def test_setting_enabled_cookies_debug(self):
|
||||
crawler = get_crawler(settings_dict={'COOKIES_DEBUG': True})
|
||||
mw = CookiesMiddleware.from_crawler(crawler)
|
||||
with LogCapture('scrapy.downloadermiddlewares.cookies',
|
||||
propagate=False,
|
||||
level=logging.DEBUG) as log:
|
||||
with LogCapture(
|
||||
'scrapy.downloadermiddlewares.cookies',
|
||||
propagate=False,
|
||||
level=logging.DEBUG,
|
||||
) as log:
|
||||
req = Request('http://scrapytest.org/')
|
||||
res = Response('http://scrapytest.org/',
|
||||
headers={'Set-Cookie': 'C1=value1; path=/'})
|
||||
res = Response('http://scrapytest.org/', headers={'Set-Cookie': 'C1=value1; path=/'})
|
||||
mw.process_response(req, res, crawler.spider)
|
||||
req2 = Request('http://scrapytest.org/sub1/')
|
||||
mw.process_request(req2, crawler.spider)
|
||||
|
|
@ -85,12 +87,13 @@ class CookiesMiddlewareTest(TestCase):
|
|||
def test_setting_disabled_cookies_debug(self):
|
||||
crawler = get_crawler(settings_dict={'COOKIES_DEBUG': False})
|
||||
mw = CookiesMiddleware.from_crawler(crawler)
|
||||
with LogCapture('scrapy.downloadermiddlewares.cookies',
|
||||
propagate=False,
|
||||
level=logging.DEBUG) as log:
|
||||
with LogCapture(
|
||||
'scrapy.downloadermiddlewares.cookies',
|
||||
propagate=False,
|
||||
level=logging.DEBUG,
|
||||
) as log:
|
||||
req = Request('http://scrapytest.org/')
|
||||
res = Response('http://scrapytest.org/',
|
||||
headers={'Set-Cookie': 'C1=value1; path=/'})
|
||||
res = Response('http://scrapytest.org/', headers={'Set-Cookie': 'C1=value1; path=/'})
|
||||
mw.process_response(req, res, crawler.spider)
|
||||
req2 = Request('http://scrapytest.org/sub1/')
|
||||
mw.process_request(req2, crawler.spider)
|
||||
|
|
@ -102,8 +105,7 @@ class CookiesMiddlewareTest(TestCase):
|
|||
assert self.mw.process_request(req, self.spider) is None
|
||||
assert 'Cookie' not in req.headers
|
||||
|
||||
headers = {'Set-Cookie': b'C1=in\xa3valid; path=/',
|
||||
'Other': b'ignore\xa3me'}
|
||||
headers = {'Set-Cookie': b'C1=in\xa3valid; path=/', 'Other': b'ignore\xa3me'}
|
||||
res = Response('http://scrapytest.org/', headers=headers)
|
||||
assert self.mw.process_response(req, res, self.spider) is res
|
||||
|
||||
|
|
@ -124,7 +126,10 @@ class CookiesMiddlewareTest(TestCase):
|
|||
assert 'Cookie' not in req.headers
|
||||
|
||||
# check that returned cookies are not merged back to jar
|
||||
res = Response('http://scrapytest.org/dontmerge', headers={'Set-Cookie': 'dont=mergeme; path=/'})
|
||||
res = Response(
|
||||
'http://scrapytest.org/dontmerge',
|
||||
headers={'Set-Cookie': 'dont=mergeme; path=/'},
|
||||
)
|
||||
assert self.mw.process_response(req, res, self.spider) is res
|
||||
|
||||
# check that cookies are merged back
|
||||
|
|
@ -179,7 +184,11 @@ class CookiesMiddlewareTest(TestCase):
|
|||
self.assertCookieValEqual(req2.headers.get('Cookie'), b"C1=value1; galleta=salada")
|
||||
|
||||
def test_cookiejar_key(self):
|
||||
req = Request('http://scrapytest.org/', cookies={'galleta': 'salada'}, meta={'cookiejar': "store1"})
|
||||
req = Request(
|
||||
'http://scrapytest.org/',
|
||||
cookies={'galleta': 'salada'},
|
||||
meta={'cookiejar': "store1"},
|
||||
)
|
||||
assert self.mw.process_request(req, self.spider) is None
|
||||
self.assertEqual(req.headers.get('Cookie'), b'galleta=salada')
|
||||
|
||||
|
|
@ -191,7 +200,11 @@ class CookiesMiddlewareTest(TestCase):
|
|||
assert self.mw.process_request(req2, self.spider) is None
|
||||
self.assertCookieValEqual(req2.headers.get('Cookie'), b'C1=value1; galleta=salada')
|
||||
|
||||
req3 = Request('http://scrapytest.org/', cookies={'galleta': 'dulce'}, meta={'cookiejar': "store2"})
|
||||
req3 = Request(
|
||||
'http://scrapytest.org/',
|
||||
cookies={'galleta': 'dulce'},
|
||||
meta={'cookiejar': "store2"},
|
||||
)
|
||||
assert self.mw.process_request(req3, self.spider) is None
|
||||
self.assertEqual(req3.headers.get('Cookie'), b'galleta=dulce')
|
||||
|
||||
|
|
@ -229,3 +242,95 @@ class CookiesMiddlewareTest(TestCase):
|
|||
assert self.mw.process_request(request, self.spider) is None
|
||||
self.assertIn('Cookie', request.headers)
|
||||
self.assertEqual(b'currencyCookie=USD', request.headers['Cookie'])
|
||||
|
||||
def test_keep_cookie_from_default_request_headers_middleware(self):
|
||||
DEFAULT_REQUEST_HEADERS = dict(Cookie='default=value; asdf=qwerty')
|
||||
mw_default_headers = DefaultHeadersMiddleware(DEFAULT_REQUEST_HEADERS.items())
|
||||
# overwrite with values from 'cookies' request argument
|
||||
req1 = Request('http://example.org', cookies={'default': 'something'})
|
||||
assert mw_default_headers.process_request(req1, self.spider) is None
|
||||
assert self.mw.process_request(req1, self.spider) is None
|
||||
self.assertCookieValEqual(req1.headers['Cookie'], b'default=something; asdf=qwerty')
|
||||
# keep both
|
||||
req2 = Request('http://example.com', cookies={'a': 'b'})
|
||||
assert mw_default_headers.process_request(req2, self.spider) is None
|
||||
assert self.mw.process_request(req2, self.spider) is None
|
||||
self.assertCookieValEqual(req2.headers['Cookie'], b'default=value; a=b; asdf=qwerty')
|
||||
|
||||
def test_keep_cookie_header(self):
|
||||
# keep only cookies from 'Cookie' request header
|
||||
req1 = Request('http://scrapytest.org', headers={'Cookie': 'a=b; c=d'})
|
||||
assert self.mw.process_request(req1, self.spider) is None
|
||||
self.assertCookieValEqual(req1.headers['Cookie'], 'a=b; c=d')
|
||||
# keep cookies from both 'Cookie' request header and 'cookies' keyword
|
||||
req2 = Request('http://scrapytest.org', headers={'Cookie': 'a=b; c=d'}, cookies={'e': 'f'})
|
||||
assert self.mw.process_request(req2, self.spider) is None
|
||||
self.assertCookieValEqual(req2.headers['Cookie'], 'a=b; c=d; e=f')
|
||||
# overwrite values from 'Cookie' request header with 'cookies' keyword
|
||||
req3 = Request(
|
||||
'http://scrapytest.org',
|
||||
headers={'Cookie': 'a=b; c=d'},
|
||||
cookies={'a': 'new', 'e': 'f'},
|
||||
)
|
||||
assert self.mw.process_request(req3, self.spider) is None
|
||||
self.assertCookieValEqual(req3.headers['Cookie'], 'a=new; c=d; e=f')
|
||||
|
||||
def test_request_cookies_encoding(self):
|
||||
# 1) UTF8-encoded bytes
|
||||
req1 = Request('http://example.org', cookies={'a': u'á'.encode('utf8')})
|
||||
assert self.mw.process_request(req1, self.spider) is None
|
||||
self.assertCookieValEqual(req1.headers['Cookie'], b'a=\xc3\xa1')
|
||||
|
||||
# 2) Non UTF8-encoded bytes
|
||||
req2 = Request('http://example.org', cookies={'a': u'á'.encode('latin1')})
|
||||
assert self.mw.process_request(req2, self.spider) is None
|
||||
self.assertCookieValEqual(req2.headers['Cookie'], b'a=\xc3\xa1')
|
||||
|
||||
# 3) Unicode string
|
||||
req3 = Request('http://example.org', cookies={'a': u'á'})
|
||||
assert self.mw.process_request(req3, self.spider) is None
|
||||
self.assertCookieValEqual(req3.headers['Cookie'], b'a=\xc3\xa1')
|
||||
|
||||
def test_request_headers_cookie_encoding(self):
|
||||
# 1) UTF8-encoded bytes
|
||||
req1 = Request('http://example.org', headers={'Cookie': u'a=á'.encode('utf8')})
|
||||
assert self.mw.process_request(req1, self.spider) is None
|
||||
self.assertCookieValEqual(req1.headers['Cookie'], b'a=\xc3\xa1')
|
||||
|
||||
# 2) Non UTF8-encoded bytes
|
||||
req2 = Request('http://example.org', headers={'Cookie': u'a=á'.encode('latin1')})
|
||||
assert self.mw.process_request(req2, self.spider) is None
|
||||
self.assertCookieValEqual(req2.headers['Cookie'], b'a=\xc3\xa1')
|
||||
|
||||
# 3) Unicode string
|
||||
req3 = Request('http://example.org', headers={'Cookie': u'a=á'})
|
||||
assert self.mw.process_request(req3, self.spider) is None
|
||||
self.assertCookieValEqual(req3.headers['Cookie'], b'a=\xc3\xa1')
|
||||
|
||||
def test_invalid_cookies(self):
|
||||
"""
|
||||
Invalid cookies are logged as warnings and discarded
|
||||
"""
|
||||
with LogCapture(
|
||||
'scrapy.downloadermiddlewares.cookies',
|
||||
propagate=False,
|
||||
level=logging.INFO,
|
||||
) as lc:
|
||||
cookies1 = [{'value': 'bar'}, {'name': 'key', 'value': 'value1'}]
|
||||
req1 = Request('http://example.org/1', cookies=cookies1)
|
||||
assert self.mw.process_request(req1, self.spider) is None
|
||||
cookies2 = [{'name': 'foo'}, {'name': 'key', 'value': 'value2'}]
|
||||
req2 = Request('http://example.org/2', cookies=cookies2)
|
||||
assert self.mw.process_request(req2, self.spider) is None
|
||||
lc.check(
|
||||
("scrapy.downloadermiddlewares.cookies",
|
||||
"WARNING",
|
||||
"Invalid cookie found in request <GET http://example.org/1>:"
|
||||
" {'value': 'bar'} ('name' is missing)"),
|
||||
("scrapy.downloadermiddlewares.cookies",
|
||||
"WARNING",
|
||||
"Invalid cookie found in request <GET http://example.org/2>:"
|
||||
" {'name': 'foo'} ('value' is missing)"),
|
||||
)
|
||||
self.assertCookieValEqual(req1.headers['Cookie'], 'key=value1')
|
||||
self.assertCookieValEqual(req2.headers['Cookie'], 'key=value2')
|
||||
|
|
|
|||
|
|
@ -16,14 +16,16 @@ import sys
|
|||
from collections import defaultdict
|
||||
from urllib.parse import urlparse
|
||||
|
||||
from pydispatch import dispatcher
|
||||
from pytest import mark
|
||||
from testfixtures import LogCapture
|
||||
from twisted.internet import reactor, defer
|
||||
from twisted.trial import unittest
|
||||
from twisted.web import server, static, util
|
||||
from pydispatch import dispatcher
|
||||
|
||||
from scrapy import signals
|
||||
from scrapy.core.engine import ExecutionEngine
|
||||
from scrapy.exceptions import StopDownload
|
||||
from scrapy.http import Request
|
||||
from scrapy.item import Item, Field
|
||||
from scrapy.linkextractors import LinkExtractor
|
||||
|
|
@ -96,7 +98,7 @@ def start_test_site(debug=False):
|
|||
r = static.File(root_dir)
|
||||
r.putChild(b"redirect", util.Redirect(b"/redirected"))
|
||||
r.putChild(b"redirected", static.Data(b"Redirected here", "text/plain"))
|
||||
numbers = [str(x).encode("utf8") for x in range(2**14)]
|
||||
numbers = [str(x).encode("utf8") for x in range(2**18)]
|
||||
r.putChild(b"numbers", static.Data(b"".join(numbers), "text/plain"))
|
||||
|
||||
port = reactor.listenTCP(0, server.Site(r), interface="127.0.0.1")
|
||||
|
|
@ -194,6 +196,16 @@ class CrawlerRun:
|
|||
self.signals_caught[sig] = signalargs
|
||||
|
||||
|
||||
class StopDownloadCrawlerRun(CrawlerRun):
|
||||
"""
|
||||
Make sure raising the StopDownload exception stops the download of the response body
|
||||
"""
|
||||
|
||||
def bytes_received(self, data, request, spider):
|
||||
super().bytes_received(data, request, spider)
|
||||
raise StopDownload(fail=False)
|
||||
|
||||
|
||||
class EngineTest(unittest.TestCase):
|
||||
|
||||
@defer.inlineCallbacks
|
||||
|
|
@ -345,7 +357,7 @@ class EngineTest(unittest.TestCase):
|
|||
# signal was fired multiple times
|
||||
self.assertTrue(len(data) > 1)
|
||||
# bytes were received in order
|
||||
numbers = [str(x).encode("utf8") for x in range(2**14)]
|
||||
numbers = [str(x).encode("utf8") for x in range(2**18)]
|
||||
self.assertEqual(joined_data, b"".join(numbers))
|
||||
|
||||
def _assert_signals_caught(self):
|
||||
|
|
@ -386,6 +398,45 @@ class EngineTest(unittest.TestCase):
|
|||
self.assertEqual(len(e.open_spiders), 0)
|
||||
|
||||
|
||||
class StopDownloadEngineTest(EngineTest):
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_crawler(self):
|
||||
for spider in TestSpider, DictItemsSpider:
|
||||
self.run = StopDownloadCrawlerRun(spider)
|
||||
with LogCapture() as log:
|
||||
yield self.run.run()
|
||||
log.check_present(("scrapy.core.downloader.handlers.http11",
|
||||
"DEBUG",
|
||||
"Download stopped for <GET http://localhost:{}/redirected> from signal handler"
|
||||
" StopDownloadCrawlerRun.bytes_received".format(self.run.portno)))
|
||||
log.check_present(("scrapy.core.downloader.handlers.http11",
|
||||
"DEBUG",
|
||||
"Download stopped for <GET http://localhost:{}/> from signal handler"
|
||||
" StopDownloadCrawlerRun.bytes_received".format(self.run.portno)))
|
||||
log.check_present(("scrapy.core.downloader.handlers.http11",
|
||||
"DEBUG",
|
||||
"Download stopped for <GET http://localhost:{}/numbers> from signal handler"
|
||||
" StopDownloadCrawlerRun.bytes_received".format(self.run.portno)))
|
||||
self._assert_visited_urls()
|
||||
self._assert_scheduled_requests(urls_to_visit=9)
|
||||
self._assert_downloaded_responses()
|
||||
self._assert_signals_caught()
|
||||
self._assert_bytes_received()
|
||||
|
||||
def _assert_bytes_received(self):
|
||||
self.assertEqual(9, len(self.run.bytes))
|
||||
for request, data in self.run.bytes.items():
|
||||
joined_data = b"".join(data)
|
||||
self.assertTrue(len(data) == 1) # signal was fired only once
|
||||
if self.run.getpath(request.url) == "/numbers":
|
||||
# Received bytes are not the complete response. The exact amount depends
|
||||
# on the buffer size, which can vary, so we only check that the amount
|
||||
# of received bytes is strictly less than the full response.
|
||||
numbers = [str(x).encode("utf8") for x in range(2**18)]
|
||||
self.assertTrue(len(joined_data) < len(b"".join(numbers)))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
if len(sys.argv) > 1 and sys.argv[1] == 'runserver':
|
||||
start_test_site(debug=True)
|
||||
|
|
|
|||
|
|
@ -1,4 +1,5 @@
|
|||
import unittest
|
||||
from unittest import mock
|
||||
from warnings import catch_warnings
|
||||
|
||||
from w3lib.encoding import resolve_encoding
|
||||
|
|
@ -685,6 +686,26 @@ class TextResponseTest(BaseResponseTest):
|
|||
self.assertEqual(len(warnings), 1)
|
||||
self.assertEqual(warnings[0].category, ScrapyDeprecationWarning)
|
||||
|
||||
def test_json_response(self):
|
||||
json_body = b"""{"ip": "109.187.217.200"}"""
|
||||
json_response = self.response_class("http://www.example.com", body=json_body)
|
||||
self.assertEqual(json_response.json(), {'ip': '109.187.217.200'})
|
||||
|
||||
text_body = b"""<html><body>text</body></html>"""
|
||||
text_response = self.response_class("http://www.example.com", body=text_body)
|
||||
with self.assertRaises(ValueError):
|
||||
text_response.json()
|
||||
|
||||
def test_cache_json_response(self):
|
||||
json_valid_bodies = [b"""{"ip": "109.187.217.200"}""", b"""null"""]
|
||||
for json_body in json_valid_bodies:
|
||||
json_response = self.response_class("http://www.example.com", body=json_body)
|
||||
|
||||
with mock.patch('json.loads') as mock_json:
|
||||
for _ in range(2):
|
||||
json_response.json()
|
||||
mock_json.assert_called_once_with(json_body.decode())
|
||||
|
||||
|
||||
class HtmlResponseTest(TextResponseTest):
|
||||
|
||||
|
|
|
|||
|
|
@ -271,6 +271,7 @@ class LocalWeakReferencedCacheTest(unittest.TestCase):
|
|||
self.assertNotIn(r1, cache)
|
||||
self.assertIn(r2, cache)
|
||||
self.assertIn(r3, cache)
|
||||
self.assertEqual(cache[r1], None)
|
||||
self.assertEqual(cache[r2], 2)
|
||||
self.assertEqual(cache[r3], 3)
|
||||
del r2
|
||||
|
|
|
|||
Loading…
Reference in New Issue