Merge remote-tracking branch 'origin/master' into asyncio-startrequests-asyncgen

This commit is contained in:
Andrey Rakhmatullin 2020-06-03 17:00:59 +05:00
commit 351a400cf8
38 changed files with 658 additions and 206 deletions

View File

@ -1,4 +1,5 @@
version: 2
formats: all
sphinx:
configuration: docs/conf.py
fail_on_warning: true

View File

@ -57,3 +57,12 @@ There is a way to recreate the doc automatically when you make changes, you
need to install watchdog (``pip install watchdog``) and then use::
make watch
Alternative method using tox
----------------------------
To compile the documentation to HTML run the following command::
tox -e docs
Documentation will be generated (in HTML format) inside the ``.tox/docs/tmp/html`` dir.

View File

@ -100,6 +100,9 @@ exclude_trees = ['.build']
# The name of the Pygments (syntax highlighting) style to use.
pygments_style = 'sphinx'
# List of Sphinx warnings that will not be raised
suppress_warnings = ['epub.unknown_project_files']
# Options for HTML output
# -----------------------
@ -300,3 +303,4 @@ hoverxref_role_types = {
"mod": "tooltip",
"ref": "tooltip",
}
hoverxref_roles = ['command', 'reqmeta', 'setting', 'signal']

View File

@ -371,6 +371,19 @@ Twisted reactor is :class:`twisted.internet.selectreactor.SelectReactor`. Switch
different reactor is possible by using the :setting:`TWISTED_REACTOR` setting.
.. _faq-stop-response-download:
How can I cancel the download of a given response?
--------------------------------------------------
In some situations, it might be useful to stop the download of a certain response.
For instance, if you only need the first part of a large response and you would like
to save resources by avoiding the download of the whole body.
In that case, you could attach a handler to the :class:`~scrapy.signals.bytes_received`
signal and raise a :exc:`~scrapy.exceptions.StopDownload` exception. Please refer to
the :ref:`topics-stop-response-download` topic for additional information and examples.
.. _has been reported: https://github.com/scrapy/scrapy/issues/2905
.. _user agents: https://en.wikipedia.org/wiki/User_agent
.. _LIFO: https://en.wikipedia.org/wiki/Stack_(abstract_data_type)

View File

@ -91,7 +91,7 @@ how you :ref:`configure the downloader middlewares
provided while constructing the crawler, and it is created after the
arguments given in the :meth:`crawl` method.
.. method:: crawl(\*args, \**kwargs)
.. method:: crawl(*args, **kwargs)
Starts the crawler by instantiating its spider class with the given
``args`` and ``kwargs`` arguments, while setting the execution engine in

View File

@ -78,7 +78,7 @@ override three methods:
.. module:: scrapy.contracts
.. class:: Contract(method, \*args)
.. class:: Contract(method, *args)
:param method: callback function to which the contract is associated
:type method: function

View File

@ -202,6 +202,11 @@ CookiesMiddleware
sends them back on subsequent requests (from that spider), just like web
browsers do.
.. caution:: When non-UTF8 encoded byte sequences are passed to a
:class:`~scrapy.http.Request`, the ``CookiesMiddleware`` will log
a warning. Refer to :ref:`topics-logging-advanced-customization`
to customize the logging behaviour.
The following settings can be used to configure the cookie middleware:
* :setting:`COOKIES_ENABLED`

View File

@ -14,13 +14,6 @@ Built-in Exceptions reference
Here's a list of all exceptions included in Scrapy and their usage.
DropItem
--------
.. exception:: DropItem
The exception that must be raised by item pipeline stages to stop processing an
Item. For more information see :ref:`topics-item-pipeline`.
CloseSpider
-----------
@ -47,6 +40,14 @@ DontCloseSpider
This exception can be raised in a :signal:`spider_idle` signal handler to
prevent the spider from being closed.
DropItem
--------
.. exception:: DropItem
The exception that must be raised by item pipeline stages to stop processing an
Item. For more information see :ref:`topics-item-pipeline`.
IgnoreRequest
-------------
@ -77,3 +78,37 @@ NotSupported
This exception is raised to indicate an unsupported feature.
StopDownload
-------------
.. versionadded:: 2.2
.. exception:: StopDownload(fail=True)
Raised from a :class:`~scrapy.signals.bytes_received` signal handler to
indicate that no further bytes should be downloaded for a response.
The ``fail`` boolean parameter controls which method will handle the resulting
response:
* If ``fail=True`` (default), the request errback is called. The response object is
available as the ``response`` attribute of the ``StopDownload`` exception,
which is in turn stored as the ``value`` attribute of the received
:class:`~twisted.python.failure.Failure` object. This means that in an errback
defined as ``def errback(self, failure)``, the response can be accessed though
``failure.value.response``.
* If ``fail=False``, the request callback is called instead.
In both cases, the response could have its body truncated: the body contains
all bytes received up until the exception is raised, including the bytes
received in the signal handler that raises the exception. Also, the response
object is marked with ``"download_stopped"`` in its :attr:`Response.flags`
attribute.
.. note:: ``fail`` is a keyword-only parameter, i.e. raising
``StopDownload(False)`` or ``StopDownload(True)`` will raise
a :class:`TypeError`.
See the documentation for the :class:`~scrapy.signals.bytes_received` signal
and the :ref:`topics-stop-response-download` topic for additional information and examples.

View File

@ -236,7 +236,7 @@ PythonItemExporter
XmlItemExporter
---------------
.. class:: XmlItemExporter(file, item_element='item', root_element='items', \**kwargs)
.. class:: XmlItemExporter(file, item_element='item', root_element='items', **kwargs)
Exports Items in XML format to the specified file object.
@ -290,7 +290,7 @@ XmlItemExporter
CsvItemExporter
---------------
.. class:: CsvItemExporter(file, include_headers_line=True, join_multivalued=',', \**kwargs)
.. class:: CsvItemExporter(file, include_headers_line=True, join_multivalued=',', **kwargs)
Exports Items in CSV format to the given file-like object. If the
:attr:`fields_to_export` attribute is set, it will be used to define the
@ -323,7 +323,7 @@ CsvItemExporter
PickleItemExporter
------------------
.. class:: PickleItemExporter(file, protocol=0, \**kwargs)
.. class:: PickleItemExporter(file, protocol=0, **kwargs)
Exports Items in pickle format to the given file-like object.
@ -343,7 +343,7 @@ PickleItemExporter
PprintItemExporter
------------------
.. class:: PprintItemExporter(file, \**kwargs)
.. class:: PprintItemExporter(file, **kwargs)
Exports Items in pretty print format to the specified file object.
@ -363,7 +363,7 @@ PprintItemExporter
JsonItemExporter
----------------
.. class:: JsonItemExporter(file, \**kwargs)
.. class:: JsonItemExporter(file, **kwargs)
Exports Items in JSON format to the specified file-like object, writing all
objects as a list of objects. The additional ``__init__`` method arguments are
@ -392,7 +392,7 @@ JsonItemExporter
JsonLinesItemExporter
---------------------
.. class:: JsonLinesItemExporter(file, \**kwargs)
.. class:: JsonLinesItemExporter(file, **kwargs)
Exports Items in JSON format to the specified file-like object, writing one
JSON-encoded item per line. The additional ``__init__`` method arguments are passed

View File

@ -167,11 +167,13 @@ method and how to clean up the resources properly.::
Take screenshot of item
-----------------------
This example demonstrates how to return a
:class:`~twisted.internet.defer.Deferred` from the :meth:`process_item` method.
It uses Splash_ to render screenshot of item url. Pipeline
makes request to locally running instance of Splash_. After request is downloaded,
it saves the screenshot to a file and adds filename to the item.
This example demonstrates how to use :doc:`coroutine syntax <coroutines>` in
the :meth:`process_item` method.
This item pipeline makes a request to a locally-running instance of Splash_ to
render a screenshot of the item URL. After the request response is downloaded,
the item pipeline saves the screenshot to a file and adds the filename to the
item.
::

View File

@ -273,7 +273,7 @@ There are several ways to modify Item Loader context values:
ItemLoader objects
==================
.. class:: ItemLoader([item, selector, response], \**kwargs)
.. class:: ItemLoader([item, selector, response], **kwargs)
Return a new Item Loader for populating the given Item. If no item is
given, one is instantiated automatically using the class in
@ -303,7 +303,7 @@ ItemLoader objects
:class:`ItemLoader` instances have the following methods:
.. method:: get_value(value, \*processors, \**kwargs)
.. method:: get_value(value, *processors, **kwargs)
Process the given ``value`` by the given ``processors`` and keyword
arguments.
@ -321,7 +321,7 @@ ItemLoader objects
>>> loader.get_value(u'name: foo', TakeFirst(), unicode.upper, re='name: (.+)')
'FOO`
.. method:: add_value(field_name, value, \*processors, \**kwargs)
.. method:: add_value(field_name, value, *processors, **kwargs)
Process and then add the given ``value`` for the given field.
@ -343,11 +343,11 @@ ItemLoader objects
loader.add_value('name', u'name: foo', TakeFirst(), re='name: (.+)')
loader.add_value(None, {'name': u'foo', 'sex': u'male'})
.. method:: replace_value(field_name, value, \*processors, \**kwargs)
.. method:: replace_value(field_name, value, *processors, **kwargs)
Similar to :meth:`add_value` but replaces the collected data with the
new value instead of adding it.
.. method:: get_xpath(xpath, \*processors, \**kwargs)
.. method:: get_xpath(xpath, *processors, **kwargs)
Similar to :meth:`ItemLoader.get_value` but receives an XPath instead of a
value, which is used to extract a list of unicode strings from the
@ -367,7 +367,7 @@ ItemLoader objects
# HTML snippet: <p id="price">the price is $1200</p>
loader.get_xpath('//p[@id="price"]', TakeFirst(), re='the price is (.*)')
.. method:: add_xpath(field_name, xpath, \*processors, \**kwargs)
.. method:: add_xpath(field_name, xpath, *processors, **kwargs)
Similar to :meth:`ItemLoader.add_value` but receives an XPath instead of a
value, which is used to extract a list of unicode strings from the
@ -385,12 +385,12 @@ ItemLoader objects
# HTML snippet: <p id="price">the price is $1200</p>
loader.add_xpath('price', '//p[@id="price"]', re='the price is (.*)')
.. method:: replace_xpath(field_name, xpath, \*processors, \**kwargs)
.. method:: replace_xpath(field_name, xpath, *processors, **kwargs)
Similar to :meth:`add_xpath` but replaces collected data instead of
adding it.
.. method:: get_css(css, \*processors, \**kwargs)
.. method:: get_css(css, *processors, **kwargs)
Similar to :meth:`ItemLoader.get_value` but receives a CSS selector
instead of a value, which is used to extract a list of unicode strings
@ -410,7 +410,7 @@ ItemLoader objects
# HTML snippet: <p id="price">the price is $1200</p>
loader.get_css('p#price', TakeFirst(), re='the price is (.*)')
.. method:: add_css(field_name, css, \*processors, \**kwargs)
.. method:: add_css(field_name, css, *processors, **kwargs)
Similar to :meth:`ItemLoader.add_value` but receives a CSS selector
instead of a value, which is used to extract a list of unicode strings
@ -428,7 +428,7 @@ ItemLoader objects
# HTML snippet: <p id="price">the price is $1200</p>
loader.add_css('price', 'p#price', re='the price is (.*)')
.. method:: replace_css(field_name, css, \*processors, \**kwargs)
.. method:: replace_css(field_name, css, *processors, **kwargs)
Similar to :meth:`add_css` but replaces collected data instead of
adding it.
@ -678,7 +678,7 @@ Here is a list of all built-in processors:
>>> proc(['one', 'two', 'three'])
'one<br>two<br>three'
.. class:: Compose(\*functions, \**default_loader_context)
.. class:: Compose(*functions, **default_loader_context)
A processor which is constructed from the composition of the given
functions. This means that each input value of this processor is passed to
@ -706,7 +706,7 @@ Here is a list of all built-in processors:
active Loader context accessible through the :meth:`ItemLoader.context`
attribute.
.. class:: MapCompose(\*functions, \**default_loader_context)
.. class:: MapCompose(*functions, **default_loader_context)
A processor which is constructed from the composition of the given
functions, similar to the :class:`Compose` processor. The difference with

View File

@ -202,6 +202,9 @@ A custom log format can be set for different actions by extending
.. autoclass:: scrapy.logformatter.LogFormatter
:members:
.. _topics-logging-advanced-customization:
Advanced customization
----------------------
@ -262,7 +265,6 @@ scrapy.utils.log module
This is an example on how to redirect ``INFO`` or higher messages to a file::
import logging
from scrapy.utils.log import configure_logging
logging.basicConfig(
filename='log.txt',

View File

@ -385,6 +385,51 @@ The meta key is used set retry times per request. When initialized, the
:reqmeta:`max_retry_times` meta key takes higher precedence over the
:setting:`RETRY_TIMES` setting.
.. _topics-stop-response-download:
Stopping the download of a Response
===================================
Raising a :exc:`~scrapy.exceptions.StopDownload` exception from a
:class:`~scrapy.signals.bytes_received` signal handler will stop the
download of a given response. See the following example::
import scrapy
class StopSpider(scrapy.Spider):
name = "stop"
start_urls = ["https://docs.scrapy.org/en/latest/"]
@classmethod
def from_crawler(cls, crawler):
spider = super().from_crawler(crawler)
crawler.signals.connect(spider.on_bytes_received, signal=scrapy.signals.bytes_received)
return spider
def parse(self, response):
# 'last_chars' show that the full response was not downloaded
yield {"len": len(response.text), "last_chars": response.text[-40:]}
def on_bytes_received(self, data, request, spider):
raise scrapy.exceptions.StopDownload(fail=False)
which produces the following output::
2020-05-19 17:26:12 [scrapy.core.engine] INFO: Spider opened
2020-05-19 17:26:12 [scrapy.extensions.logstats] INFO: Crawled 0 pages (at 0 pages/min), scraped 0 items (at 0 items/min)
2020-05-19 17:26:13 [scrapy.core.downloader.handlers.http11] DEBUG: Download stopped for <GET https://docs.scrapy.org/en/latest/> from signal handler StopSpider.on_bytes_received
2020-05-19 17:26:13 [scrapy.core.engine] DEBUG: Crawled (200) <GET https://docs.scrapy.org/en/latest/> (referer: None) ['download_stopped']
2020-05-19 17:26:13 [scrapy.core.scraper] DEBUG: Scraped from <200 https://docs.scrapy.org/en/latest/>
{'len': 279, 'last_chars': 'dth, initial-scale=1.0">\n \n <title>Scr'}
2020-05-19 17:26:13 [scrapy.core.engine] INFO: Closing spider (finished)
By default, resulting responses are handled by their corresponding errbacks. To
call their callback instead, like in this example, pass ``fail=False`` to the
:exc:`~scrapy.exceptions.StopDownload` exception.
.. _topics-request-response-ref-request-subclasses:
Request subclasses
@ -716,9 +761,9 @@ Response objects
.. versionadded:: 2.1.0
The IP address of the server from which the Response originated.
This attribute is currently only populated by the HTTP 1.1 download
handler, i.e. for ``http(s)`` responses. For other handlers,
handler, i.e. for ``http(s)`` responses. For other handlers,
:attr:`ip_address` is always ``None``.
.. method:: Response.copy()
@ -834,6 +879,11 @@ TextResponse objects
.. automethod:: TextResponse.follow_all
.. automethod:: TextResponse.json()
Returns a Python object from deserialized JSON document.
The result is cached after the first call.
HtmlResponse objects
--------------------

View File

@ -373,6 +373,8 @@ request_left_downloader
bytes_received
~~~~~~~~~~~~~~
.. versionadded:: 2.2
.. signal:: bytes_received
.. function:: bytes_received(data, request, spider)
@ -385,14 +387,19 @@ bytes_received
This signal does not support returning deferreds from its handlers.
:param data: the data received by the download handler
:type spider: :class:`bytes` object
:type data: :class:`bytes` object
:param request: the request that generated the response
:param request: the request that generated the download
:type request: :class:`~scrapy.http.Request` object
:param spider: the spider associated with the response
:type spider: :class:`~scrapy.spiders.Spider` object
.. note:: Handlers of this signal can stop the download of a response while it
is in progress by raising the :exc:`~scrapy.exceptions.StopDownload`
exception. Please refer to the :ref:`topics-stop-response-download` topic
for additional information and examples.
Response signals
----------------

View File

@ -121,7 +121,7 @@ scrapy.Spider
send log messages through it as described on
:ref:`topics-logging-from-spiders`.
.. method:: from_crawler(crawler, \*args, \**kwargs)
.. method:: from_crawler(crawler, *args, **kwargs)
This is the class method used by Scrapy to create your spiders.

View File

@ -27,18 +27,16 @@ flake8-ignore =
# Exclude files that are meant to provide top-level imports
# E402: Module level import not at top of file
# F401: Module imported but unused
scrapy/__init__.py E402
scrapy/core/downloader/handlers/http.py F401
scrapy/http/__init__.py F401
scrapy/linkextractors/__init__.py E402 F401
scrapy/selector/__init__.py F401
scrapy/spiders/__init__.py E402 F401
# Issues pending a review:
scrapy/__init__.py E402
scrapy/selector/__init__.py F403
scrapy/spiders/__init__.py E402
scrapy/utils/http.py F403
scrapy/utils/markup.py F403
scrapy/utils/multipart.py F403
scrapy/utils/url.py F403 F405
tests/test_loader.py E741
tests/test_webclient.py E402

View File

@ -2,33 +2,11 @@
Scrapy - a web crawling and web scraping framework written for Python
"""
__all__ = ['__version__', 'version_info', 'twisted_version',
'Spider', 'Request', 'FormRequest', 'Selector', 'Item', 'Field']
# Scrapy version
import pkgutil
__version__ = pkgutil.get_data(__package__, 'VERSION').decode('ascii').strip()
version_info = tuple(int(v) if v.isdigit() else v
for v in __version__.split('.'))
del pkgutil
# Check minimum required Python version
import sys
if sys.version_info < (3, 5):
print("Scrapy %s requires Python 3.5" % __version__)
sys.exit(1)
# Ignore noisy twisted deprecation warnings
import warnings
warnings.filterwarnings('ignore', category=DeprecationWarning, module='twisted')
del warnings
# Apply monkey patches to fix issues in external libraries
from scrapy import _monkeypatches
del _monkeypatches
from twisted import version as _txv
twisted_version = (_txv.major, _txv.minor, _txv.micro)
# Declare top-level shortcuts
from scrapy.spiders import Spider
@ -36,4 +14,29 @@ from scrapy.http import Request, FormRequest
from scrapy.selector import Selector
from scrapy.item import Item, Field
__all__ = [
'__version__', 'version_info', 'twisted_version', 'Spider',
'Request', 'FormRequest', 'Selector', 'Item', 'Field',
]
# Scrapy and Twisted versions
__version__ = pkgutil.get_data(__package__, 'VERSION').decode('ascii').strip()
version_info = tuple(int(v) if v.isdigit() else v for v in __version__.split('.'))
twisted_version = (_txv.major, _txv.minor, _txv.micro)
# Check minimum required Python version
if sys.version_info < (3, 5):
print("Scrapy %s requires Python 3.5" % __version__)
sys.exit(1)
# Ignore noisy twisted deprecation warnings
warnings.filterwarnings('ignore', category=DeprecationWarning, module='twisted')
del pkgutil
del sys
del warnings

View File

@ -1,11 +0,0 @@
import copyreg
# Undo what Twisted's perspective broker adds to pickle register
# to prevent bugs like Twisted#7989 while serializing requests
import twisted.persisted.styles # NOQA
# Remove only entries with twisted serializers for non-twisted types.
for k, v in frozenset(copyreg.dispatch_table.items()):
if not str(getattr(k, '__module__', '')).startswith('twisted') \
and str(getattr(v, '__module__', '')).startswith('twisted'):
copyreg.dispatch_table.pop(k)

View File

@ -5,7 +5,7 @@ import os
from optparse import OptionGroup
from twisted.python import failure
from scrapy.utils.conf import arglist_to_dict
from scrapy.utils.conf import arglist_to_dict, feed_process_params_from_cli
from scrapy.exceptions import UsageError
@ -104,3 +104,27 @@ class ScrapyCommand:
Entry point for running commands
"""
raise NotImplementedError
class BaseRunSpiderCommand(ScrapyCommand):
"""
Common class used to share functionality between the crawl and runspider commands
"""
def add_options(self, parser):
ScrapyCommand.add_options(self, parser)
parser.add_option("-a", dest="spargs", action="append", default=[], metavar="NAME=VALUE",
help="set spider argument (may be repeated)")
parser.add_option("-o", "--output", metavar="FILE", action="append",
help="dump scraped items into FILE (use - for stdout)")
parser.add_option("-t", "--output-format", metavar="FORMAT",
help="format to use for dumping items with -o")
def process_options(self, args, opts):
ScrapyCommand.process_options(self, args, opts)
try:
opts.spargs = arglist_to_dict(opts.spargs)
except ValueError:
raise UsageError("Invalid -a value, use -a NAME=VALUE", print_help=False)
if opts.output:
feeds = feed_process_params_from_cli(self.settings, opts.output, opts.output_format)
self.settings.set('FEEDS', feeds, priority='cmdline')

View File

@ -1,9 +1,8 @@
from scrapy.commands import ScrapyCommand
from scrapy.utils.conf import arglist_to_dict, feed_process_params_from_cli
from scrapy.commands import BaseRunSpiderCommand
from scrapy.exceptions import UsageError
class Command(ScrapyCommand):
class Command(BaseRunSpiderCommand):
requires_project = True
@ -13,25 +12,6 @@ class Command(ScrapyCommand):
def short_desc(self):
return "Run a spider"
def add_options(self, parser):
ScrapyCommand.add_options(self, parser)
parser.add_option("-a", dest="spargs", action="append", default=[], metavar="NAME=VALUE",
help="set spider argument (may be repeated)")
parser.add_option("-o", "--output", metavar="FILE", action="append",
help="dump scraped items into FILE (use - for stdout)")
parser.add_option("-t", "--output-format", metavar="FORMAT",
help="format to use for dumping items with -o")
def process_options(self, args, opts):
ScrapyCommand.process_options(self, args, opts)
try:
opts.spargs = arglist_to_dict(opts.spargs)
except ValueError:
raise UsageError("Invalid -a value, use -a NAME=VALUE", print_help=False)
if opts.output:
feeds = feed_process_params_from_cli(self.settings, opts.output, opts.output_format)
self.settings.set('FEEDS', feeds, priority='cmdline')
def run(self, args, opts):
if len(args) < 1:
raise UsageError()

View File

@ -3,9 +3,8 @@ import os
from importlib import import_module
from scrapy.utils.spider import iter_spider_classes
from scrapy.commands import ScrapyCommand
from scrapy.exceptions import UsageError
from scrapy.utils.conf import arglist_to_dict, feed_process_params_from_cli
from scrapy.commands import BaseRunSpiderCommand
def _import_file(filepath):
@ -24,7 +23,7 @@ def _import_file(filepath):
return module
class Command(ScrapyCommand):
class Command(BaseRunSpiderCommand):
requires_project = False
default_settings = {'SPIDER_LOADER_WARN_ONLY': True}
@ -38,25 +37,6 @@ class Command(ScrapyCommand):
def long_desc(self):
return "Run the spider defined in the given file"
def add_options(self, parser):
ScrapyCommand.add_options(self, parser)
parser.add_option("-a", dest="spargs", action="append", default=[], metavar="NAME=VALUE",
help="set spider argument (may be repeated)")
parser.add_option("-o", "--output", metavar="FILE", action="append",
help="dump scraped items into FILE (use - for stdout)")
parser.add_option("-t", "--output-format", metavar="FORMAT",
help="format to use for dumping items with -o")
def process_options(self, args, opts):
ScrapyCommand.process_options(self, args, opts)
try:
opts.spargs = arglist_to_dict(opts.spargs)
except ValueError:
raise UsageError("Invalid -a value, use -a NAME=VALUE", print_help=False)
if opts.output:
feeds = feed_process_params_from_cli(self.settings, opts.output, opts.output_format)
self.settings.set('FEEDS', feeds, priority='cmdline')
def run(self, args, opts):
if len(args) != 1:
raise UsageError()

View File

@ -12,6 +12,7 @@ from urllib.parse import urldefrag
from twisted.internet import defer, protocol, ssl
from twisted.internet.endpoints import TCP4ClientEndpoint
from twisted.internet.error import TimeoutError
from twisted.python.failure import Failure
from twisted.web.client import Agent, HTTPConnectionPool, ResponseDone, ResponseFailed, URI
from twisted.web.http import _DataLoss, PotentialDataLoss
from twisted.web.http_headers import Headers as TxHeaders
@ -21,7 +22,7 @@ from zope.interface import implementer
from scrapy import signals
from scrapy.core.downloader.tls import openssl_methods
from scrapy.core.downloader.webclient import _parse
from scrapy.exceptions import ScrapyDeprecationWarning
from scrapy.exceptions import ScrapyDeprecationWarning, StopDownload
from scrapy.http import Headers
from scrapy.responsetypes import responsetypes
from scrapy.utils.misc import create_instance, load_object
@ -431,7 +432,7 @@ class ScrapyAgent:
def _cb_bodydone(self, result, request, url):
headers = Headers(result["txresponse"].headers.getAllRawHeaders())
respcls = responsetypes.from_args(headers=headers, url=url, body=result["body"])
return respcls(
response = respcls(
url=url,
status=int(result["txresponse"].code),
headers=headers,
@ -440,6 +441,10 @@ class ScrapyAgent:
certificate=result["certificate"],
ip_address=result["ip_address"],
)
if result.get("failure"):
result["failure"].value.response = response
return result["failure"]
return response
@implementer(IBodyProducer)
@ -477,6 +482,16 @@ class _ResponseReader(protocol.Protocol):
self._ip_address = None
self._crawler = crawler
def _finish_response(self, flags=None, failure=None):
self._finished.callback({
"txresponse": self._txresponse,
"body": self._bodybuf.getvalue(),
"flags": flags,
"certificate": self._certificate,
"ip_address": self._ip_address,
"failure": failure,
})
def connectionMade(self):
if self._certificate is None:
with suppress(AttributeError):
@ -493,12 +508,19 @@ class _ResponseReader(protocol.Protocol):
self._bodybuf.write(bodyBytes)
self._bytes_received += len(bodyBytes)
self._crawler.signals.send_catch_log(
bytes_received_result = self._crawler.signals.send_catch_log(
signal=signals.bytes_received,
data=bodyBytes,
request=self._request,
spider=self._crawler.spider,
)
for handler, result in bytes_received_result:
if isinstance(result, Failure) and isinstance(result.value, StopDownload):
logger.debug("Download stopped for %(request)s from signal handler %(handler)s",
{"request": self._request, "handler": handler.__qualname__})
self.transport._producer.loseConnection()
failure = result if result.value.fail else None
self._finish_response(flags=["download_stopped"], failure=failure)
if self._maxsize and self._bytes_received > self._maxsize:
logger.error("Received (%(bytes)s) bytes larger than download "
@ -521,36 +543,17 @@ class _ResponseReader(protocol.Protocol):
if self._finished.called:
return
body = self._bodybuf.getvalue()
if reason.check(ResponseDone):
self._finished.callback({
"txresponse": self._txresponse,
"body": body,
"flags": None,
"certificate": self._certificate,
"ip_address": self._ip_address,
})
self._finish_response()
return
if reason.check(PotentialDataLoss):
self._finished.callback({
"txresponse": self._txresponse,
"body": body,
"flags": ["partial"],
"certificate": self._certificate,
"ip_address": self._ip_address,
})
self._finish_response(flags=["partial"])
return
if reason.check(ResponseFailed) and any(r.check(_DataLoss) for r in reason.value.reasons):
if not self._fail_on_dataloss:
self._finished.callback({
"txresponse": self._txresponse,
"body": body,
"flags": ["dataloss"],
"certificate": self._certificate,
"ip_address": self._ip_address,
})
self._finish_response(flags=["dataloss"])
return
elif not self._fail_on_dataloss_warned:

View File

@ -29,8 +29,7 @@ class CookiesMiddleware:
cookiejarkey = request.meta.get("cookiejar")
jar = self.jars[cookiejarkey]
cookies = self._get_request_cookies(jar, request)
for cookie in cookies:
for cookie in self._get_request_cookies(jar, request):
jar.set_cookie_if_ok(cookie, request)
# set Cookie header
@ -68,28 +67,65 @@ class CookiesMiddleware:
msg = "Received cookies from: {}\n{}".format(response, cookies)
logger.debug(msg, extra={'spider': spider})
def _format_cookie(self, cookie):
# build cookie string
cookie_str = '%s=%s' % (cookie['name'], cookie['value'])
if cookie.get('path', None):
cookie_str += '; Path=%s' % cookie['path']
if cookie.get('domain', None):
cookie_str += '; Domain=%s' % cookie['domain']
def _format_cookie(self, cookie, request):
"""
Given a dict consisting of cookie components, return its string representation.
Decode from bytes if necessary.
"""
decoded = {}
for key in ("name", "value", "path", "domain"):
if not cookie.get(key):
if key in ("name", "value"):
msg = "Invalid cookie found in request {}: {} ('{}' is missing)"
logger.warning(msg.format(request, cookie, key))
return
continue
if isinstance(cookie[key], str):
decoded[key] = cookie[key]
else:
try:
decoded[key] = cookie[key].decode("utf8")
except UnicodeDecodeError:
logger.warning("Non UTF-8 encoded cookie found in request %s: %s",
request, cookie)
decoded[key] = cookie[key].decode("latin1", errors="replace")
cookie_str = "{}={}".format(decoded.pop("name"), decoded.pop("value"))
for key, value in decoded.items(): # path, domain
cookie_str += "; {}={}".format(key.capitalize(), value)
return cookie_str
def _get_request_cookies(self, jar, request):
if isinstance(request.cookies, dict):
cookie_list = [
{'name': k, 'value': v}
for k, v in request.cookies.items()
]
else:
cookie_list = request.cookies
"""
Extract cookies from a Request. Values from the `Request.cookies` attribute
take precedence over values from the `Cookie` request header.
"""
def get_cookies_from_header(jar, request):
cookie_header = request.headers.get("Cookie")
if not cookie_header:
return []
cookie_gen_bytes = (s.strip() for s in cookie_header.split(b";"))
cookie_list_unicode = []
for cookie_bytes in cookie_gen_bytes:
try:
cookie_unicode = cookie_bytes.decode("utf8")
except UnicodeDecodeError:
logger.warning("Non UTF-8 encoded cookie found in request %s: %s",
request, cookie_bytes)
cookie_unicode = cookie_bytes.decode("latin1", errors="replace")
cookie_list_unicode.append(cookie_unicode)
response = Response(request.url, headers={"Set-Cookie": cookie_list_unicode})
return jar.make_cookies(response, request)
cookies = [self._format_cookie(x) for x in cookie_list]
headers = {'Set-Cookie': cookies}
response = Response(request.url, headers=headers)
def get_cookies_from_attribute(jar, request):
if not request.cookies:
return []
elif isinstance(request.cookies, dict):
cookies = ({"name": k, "value": v} for k, v in request.cookies.items())
else:
cookies = request.cookies
formatted = filter(None, (self._format_cookie(c, request) for c in cookies))
response = Response(request.url, headers={"Set-Cookie": formatted})
return jar.make_cookies(response, request)
return jar.make_cookies(response, request)
return get_cookies_from_header(jar, request) + get_cookies_from_attribute(jar, request)

View File

@ -41,6 +41,18 @@ class CloseSpider(Exception):
self.reason = reason
class StopDownload(Exception):
"""
Stop the download of the body for a given response.
The 'fail' boolean parameter indicates whether or not the resulting partial response
should be handled by the request errback. Note that 'fail' is a keyword-only argument.
"""
def __init__(self, *, fail=True):
super().__init__()
self.fail = fail
# Items
@ -59,6 +71,7 @@ class NotSupported(Exception):
class UsageError(Exception):
"""To indicate a command-line usage error"""
def __init__(self, *a, **kw):
self.print_help = kw.pop('print_help', True)
super(UsageError, self).__init__(*a, **kw)

View File

@ -5,6 +5,7 @@ discovering (through HTTP headers) to base Response class.
See documentation in docs/topics/request-response.rst
"""
import json
import warnings
from contextlib import suppress
from typing import Generator
@ -21,10 +22,13 @@ from scrapy.http.response import Response
from scrapy.utils.python import memoizemethod_noargs, to_unicode
from scrapy.utils.response import get_base_url
_NONE = object()
class TextResponse(Response):
_DEFAULT_ENCODING = 'ascii'
_cached_decoded_json = _NONE
def __init__(self, *args, **kwargs):
self._encoding = kwargs.pop('encoding', None)
@ -68,6 +72,14 @@ class TextResponse(Response):
ScrapyDeprecationWarning, stacklevel=2)
return self.text
def json(self):
"""
Deserialize a JSON document to a Python object.
"""
if self._cached_decoded_json is _NONE:
self._cached_decoded_json = json.loads(self.text)
return self._cached_decoded_json
@property
def text(self):
""" Body as unicode """

View File

@ -133,4 +133,4 @@ class FilteringLinkExtractor:
# Top-level imports
from scrapy.linkextractors.lxmlhtml import LxmlLinkExtractor as LinkExtractor # noqa: F401
from scrapy.linkextractors.lxmlhtml import LxmlLinkExtractor as LinkExtractor

View File

@ -1,4 +1,6 @@
"""
Selectors
"""
from scrapy.selector.unified import * # noqa: F401
# top-level imports
from scrapy.selector.unified import Selector, SelectorList

View File

@ -110,6 +110,6 @@ class Spider(object_ref):
# Top-level imports
from scrapy.spiders.crawl import CrawlSpider, Rule # noqa: F401
from scrapy.spiders.feed import XMLFeedSpider, CSVFeedSpider # noqa: F401
from scrapy.spiders.sitemap import SitemapSpider # noqa: F401
from scrapy.spiders.crawl import CrawlSpider, Rule
from scrapy.spiders.feed import XMLFeedSpider, CSVFeedSpider
from scrapy.spiders.sitemap import SitemapSpider

View File

@ -105,8 +105,8 @@ class LocalWeakReferencedCache(weakref.WeakKeyDictionary):
def __getitem__(self, key):
try:
return super(LocalWeakReferencedCache, self).__getitem__(key)
except TypeError:
return None # key is not weak-referenceable, it's not cached
except (TypeError, KeyError):
return None # key is either not weak-referenceable or not cached
class SequenceExclude:

View File

@ -5,13 +5,14 @@ import logging
from twisted.internet.defer import DeferredList, Deferred
from twisted.python.failure import Failure
from pydispatch.dispatcher import Any, Anonymous, liveReceivers, \
getAllReceivers, disconnect
from pydispatch.dispatcher import Anonymous, Any, disconnect, getAllReceivers, liveReceivers
from pydispatch.robustapply import robustApply
from scrapy.exceptions import StopDownload
from scrapy.utils.defer import maybeDeferred_coro
from scrapy.utils.log import failure_to_exc_info
logger = logging.getLogger(__name__)
@ -23,7 +24,7 @@ def send_catch_log(signal=Any, sender=Anonymous, *arguments, **named):
"""Like pydispatcher.robust.sendRobust but it also logs errors and returns
Failures instead of exceptions.
"""
dont_log = named.pop('dont_log', _IgnoredException)
dont_log = (named.pop('dont_log', _IgnoredException), StopDownload)
spider = named.get('spider', None)
responses = []
for receiver in liveReceivers(getAllReceivers(sender, signal)):

View File

@ -1,5 +1,5 @@
import logging
import inspect
import logging
from scrapy.spiders import Spider
from scrapy.utils.defer import deferred_from_coro, _isasyncgen
@ -18,7 +18,11 @@ def iterate_spider_output(result):
d = deferred_from_coro(collect_asyncgen(result))
d.addCallback(iterate_spider_output)
return d
return arg_to_iter(deferred_from_coro(result))
elif inspect.iscoroutine(result):
d = deferred_from_coro(result)
d.addCallback(iterate_spider_output)
return d
return arg_to_iter(result)
def iter_spider_classes(module):

View File

@ -2,7 +2,8 @@
jmespath
mitmproxy; python_version >= '3.6'
mitmproxy<4.0.0; python_version < '3.6'
pytest < 5.4
# https://github.com/pytest-dev/pytest-twisted/issues/93
pytest != 5.4, != 5.4.1
pytest-cov
pytest-twisted >= 1.11
pytest-xdist

View File

@ -7,6 +7,8 @@ from urllib.parse import urlencode
from twisted.internet import defer
from scrapy import signals
from scrapy.exceptions import StopDownload
from scrapy.http import Request
from scrapy.item import Item
from scrapy.linkextractors import LinkExtractor
@ -117,6 +119,17 @@ class AsyncDefAsyncioReturnSpider(SimpleSpider):
return [{'id': 1}, {'id': 2}]
class AsyncDefAsyncioReturnSingleElementSpider(SimpleSpider):
name = "asyncdef_asyncio_return_single_element"
async def parse(self, response):
await asyncio.sleep(0.1)
status = await get_from_asyncio_queue(response.status)
self.logger.info("Got response %d" % status)
return {"foo": 42}
class AsyncDefAsyncioReqsReturnSpider(SimpleSpider):
name = 'asyncdef_asyncio_reqs_return'
@ -267,3 +280,36 @@ class CrawlSpiderWithErrback(MockServerSpider, CrawlSpider):
def errback(self, failure):
self.logger.info('[errback] status %i', failure.value.response.status)
class BytesReceivedCallbackSpider(MetaSpider):
full_response_length = 2**18
@classmethod
def from_crawler(cls, crawler, *args, **kwargs):
spider = super().from_crawler(crawler, *args, **kwargs)
crawler.signals.connect(spider.bytes_received, signals.bytes_received)
return spider
def start_requests(self):
body = b"a" * self.full_response_length
url = self.mockserver.url("/alpayload")
yield Request(url, method="POST", body=body, errback=self.errback)
def parse(self, response):
self.meta["response"] = response
def errback(self, failure):
self.meta["failure"] = failure
def bytes_received(self, data, request, spider):
self.meta["bytes_received"] = data
raise StopDownload(fail=False)
class BytesReceivedErrbackSpider(BytesReceivedCallbackSpider):
def bytes_received(self, data, request, spider):
self.meta["bytes_received"] = data
raise StopDownload(fail=True)

View File

@ -9,17 +9,32 @@ from pytest import mark
from testfixtures import LogCapture
from twisted.internet import defer
from twisted.internet.ssl import Certificate
from twisted.python.failure import Failure
from twisted.trial.unittest import TestCase
from scrapy import signals
from scrapy.crawler import CrawlerRunner
from scrapy.exceptions import StopDownload
from scrapy.http import Request
from scrapy.http.response import Response
from scrapy.utils.python import to_unicode
from tests.mockserver import MockServer
from tests.spiders import (FollowAllSpider, DelaySpider, SimpleSpider, BrokenStartRequestsSpider,
SingleRequestSpider, DuplicateStartRequestsSpider, CrawlSpiderWithErrback,
AsyncDefSpider, AsyncDefAsyncioSpider, AsyncDefAsyncioReturnSpider,
AsyncDefAsyncioReqsReturnSpider)
from tests.spiders import (
AsyncDefAsyncioReqsReturnSpider,
AsyncDefAsyncioReturnSingleElementSpider,
AsyncDefAsyncioReturnSpider,
AsyncDefAsyncioSpider,
AsyncDefSpider,
BrokenStartRequestsSpider,
BytesReceivedCallbackSpider,
BytesReceivedErrbackSpider,
CrawlSpiderWithErrback,
DelaySpider,
DuplicateStartRequestsSpider,
FollowAllSpider,
SimpleSpider,
SingleRequestSpider,
)
class CrawlTestCase(TestCase):
@ -350,6 +365,21 @@ with multiples lines
self.assertIn({'id': 1}, items)
self.assertIn({'id': 2}, items)
@mark.only_asyncio()
@defer.inlineCallbacks
def test_async_def_asyncio_parse_items_single_element(self):
items = []
def _on_item_scraped(item):
items.append(item)
crawler = self.runner.create_crawler(AsyncDefAsyncioReturnSingleElementSpider)
crawler.signals.connect(_on_item_scraped, signals.item_scraped)
with LogCapture() as log:
yield crawler.crawl(self.mockserver.url("/status?n=200"), mockserver=self.mockserver)
self.assertIn("Got response 200", str(log))
self.assertIn({"foo": 42}, items)
@mark.skipif(sys.version_info < (3, 6), reason="Async generators require Python 3.6 or higher")
@mark.only_asyncio()
@defer.inlineCallbacks
@ -457,3 +487,27 @@ with multiples lines
ip_address = crawler.spider.meta['responses'][0].ip_address
self.assertIsInstance(ip_address, IPv4Address)
self.assertEqual(str(ip_address), gethostbyname(expected_netloc))
@defer.inlineCallbacks
def test_stop_download_callback(self):
crawler = self.runner.create_crawler(BytesReceivedCallbackSpider)
yield crawler.crawl(mockserver=self.mockserver)
self.assertIsNone(crawler.spider.meta.get("failure"))
self.assertIsInstance(crawler.spider.meta["response"], Response)
self.assertEqual(crawler.spider.meta["response"].body, crawler.spider.meta.get("bytes_received"))
self.assertLess(len(crawler.spider.meta["response"].body), crawler.spider.full_response_length)
@defer.inlineCallbacks
def test_stop_download_errback(self):
crawler = self.runner.create_crawler(BytesReceivedErrbackSpider)
yield crawler.crawl(mockserver=self.mockserver)
self.assertIsNone(crawler.spider.meta.get("response"))
self.assertIsInstance(crawler.spider.meta["failure"], Failure)
self.assertIsInstance(crawler.spider.meta["failure"].value, StopDownload)
self.assertIsInstance(crawler.spider.meta["failure"].value.response, Response)
self.assertEqual(
crawler.spider.meta["failure"].value.response.body,
crawler.spider.meta.get("bytes_received"))
self.assertLess(
len(crawler.spider.meta["failure"].value.response.body),
crawler.spider.full_response_length)

View File

@ -1,20 +1,21 @@
import re
import logging
from unittest import TestCase
from testfixtures import LogCapture
from unittest import TestCase
from scrapy.downloadermiddlewares.cookies import CookiesMiddleware
from scrapy.downloadermiddlewares.defaultheaders import DefaultHeadersMiddleware
from scrapy.exceptions import NotConfigured
from scrapy.http import Response, Request
from scrapy.spiders import Spider
from scrapy.utils.python import to_bytes
from scrapy.utils.test import get_crawler
from scrapy.exceptions import NotConfigured
from scrapy.downloadermiddlewares.cookies import CookiesMiddleware
class CookiesMiddlewareTest(TestCase):
def assertCookieValEqual(self, first, second, msg=None):
def split_cookies(cookies):
return sorted(re.split(r";\s*", cookies.decode("latin1")))
return sorted([s.strip() for s in to_bytes(cookies).split(b";")])
return self.assertEqual(split_cookies(first), split_cookies(second), msg=msg)
def setUp(self):
@ -61,12 +62,13 @@ class CookiesMiddlewareTest(TestCase):
def test_setting_enabled_cookies_debug(self):
crawler = get_crawler(settings_dict={'COOKIES_DEBUG': True})
mw = CookiesMiddleware.from_crawler(crawler)
with LogCapture('scrapy.downloadermiddlewares.cookies',
propagate=False,
level=logging.DEBUG) as log:
with LogCapture(
'scrapy.downloadermiddlewares.cookies',
propagate=False,
level=logging.DEBUG,
) as log:
req = Request('http://scrapytest.org/')
res = Response('http://scrapytest.org/',
headers={'Set-Cookie': 'C1=value1; path=/'})
res = Response('http://scrapytest.org/', headers={'Set-Cookie': 'C1=value1; path=/'})
mw.process_response(req, res, crawler.spider)
req2 = Request('http://scrapytest.org/sub1/')
mw.process_request(req2, crawler.spider)
@ -85,12 +87,13 @@ class CookiesMiddlewareTest(TestCase):
def test_setting_disabled_cookies_debug(self):
crawler = get_crawler(settings_dict={'COOKIES_DEBUG': False})
mw = CookiesMiddleware.from_crawler(crawler)
with LogCapture('scrapy.downloadermiddlewares.cookies',
propagate=False,
level=logging.DEBUG) as log:
with LogCapture(
'scrapy.downloadermiddlewares.cookies',
propagate=False,
level=logging.DEBUG,
) as log:
req = Request('http://scrapytest.org/')
res = Response('http://scrapytest.org/',
headers={'Set-Cookie': 'C1=value1; path=/'})
res = Response('http://scrapytest.org/', headers={'Set-Cookie': 'C1=value1; path=/'})
mw.process_response(req, res, crawler.spider)
req2 = Request('http://scrapytest.org/sub1/')
mw.process_request(req2, crawler.spider)
@ -102,8 +105,7 @@ class CookiesMiddlewareTest(TestCase):
assert self.mw.process_request(req, self.spider) is None
assert 'Cookie' not in req.headers
headers = {'Set-Cookie': b'C1=in\xa3valid; path=/',
'Other': b'ignore\xa3me'}
headers = {'Set-Cookie': b'C1=in\xa3valid; path=/', 'Other': b'ignore\xa3me'}
res = Response('http://scrapytest.org/', headers=headers)
assert self.mw.process_response(req, res, self.spider) is res
@ -124,7 +126,10 @@ class CookiesMiddlewareTest(TestCase):
assert 'Cookie' not in req.headers
# check that returned cookies are not merged back to jar
res = Response('http://scrapytest.org/dontmerge', headers={'Set-Cookie': 'dont=mergeme; path=/'})
res = Response(
'http://scrapytest.org/dontmerge',
headers={'Set-Cookie': 'dont=mergeme; path=/'},
)
assert self.mw.process_response(req, res, self.spider) is res
# check that cookies are merged back
@ -179,7 +184,11 @@ class CookiesMiddlewareTest(TestCase):
self.assertCookieValEqual(req2.headers.get('Cookie'), b"C1=value1; galleta=salada")
def test_cookiejar_key(self):
req = Request('http://scrapytest.org/', cookies={'galleta': 'salada'}, meta={'cookiejar': "store1"})
req = Request(
'http://scrapytest.org/',
cookies={'galleta': 'salada'},
meta={'cookiejar': "store1"},
)
assert self.mw.process_request(req, self.spider) is None
self.assertEqual(req.headers.get('Cookie'), b'galleta=salada')
@ -191,7 +200,11 @@ class CookiesMiddlewareTest(TestCase):
assert self.mw.process_request(req2, self.spider) is None
self.assertCookieValEqual(req2.headers.get('Cookie'), b'C1=value1; galleta=salada')
req3 = Request('http://scrapytest.org/', cookies={'galleta': 'dulce'}, meta={'cookiejar': "store2"})
req3 = Request(
'http://scrapytest.org/',
cookies={'galleta': 'dulce'},
meta={'cookiejar': "store2"},
)
assert self.mw.process_request(req3, self.spider) is None
self.assertEqual(req3.headers.get('Cookie'), b'galleta=dulce')
@ -229,3 +242,95 @@ class CookiesMiddlewareTest(TestCase):
assert self.mw.process_request(request, self.spider) is None
self.assertIn('Cookie', request.headers)
self.assertEqual(b'currencyCookie=USD', request.headers['Cookie'])
def test_keep_cookie_from_default_request_headers_middleware(self):
DEFAULT_REQUEST_HEADERS = dict(Cookie='default=value; asdf=qwerty')
mw_default_headers = DefaultHeadersMiddleware(DEFAULT_REQUEST_HEADERS.items())
# overwrite with values from 'cookies' request argument
req1 = Request('http://example.org', cookies={'default': 'something'})
assert mw_default_headers.process_request(req1, self.spider) is None
assert self.mw.process_request(req1, self.spider) is None
self.assertCookieValEqual(req1.headers['Cookie'], b'default=something; asdf=qwerty')
# keep both
req2 = Request('http://example.com', cookies={'a': 'b'})
assert mw_default_headers.process_request(req2, self.spider) is None
assert self.mw.process_request(req2, self.spider) is None
self.assertCookieValEqual(req2.headers['Cookie'], b'default=value; a=b; asdf=qwerty')
def test_keep_cookie_header(self):
# keep only cookies from 'Cookie' request header
req1 = Request('http://scrapytest.org', headers={'Cookie': 'a=b; c=d'})
assert self.mw.process_request(req1, self.spider) is None
self.assertCookieValEqual(req1.headers['Cookie'], 'a=b; c=d')
# keep cookies from both 'Cookie' request header and 'cookies' keyword
req2 = Request('http://scrapytest.org', headers={'Cookie': 'a=b; c=d'}, cookies={'e': 'f'})
assert self.mw.process_request(req2, self.spider) is None
self.assertCookieValEqual(req2.headers['Cookie'], 'a=b; c=d; e=f')
# overwrite values from 'Cookie' request header with 'cookies' keyword
req3 = Request(
'http://scrapytest.org',
headers={'Cookie': 'a=b; c=d'},
cookies={'a': 'new', 'e': 'f'},
)
assert self.mw.process_request(req3, self.spider) is None
self.assertCookieValEqual(req3.headers['Cookie'], 'a=new; c=d; e=f')
def test_request_cookies_encoding(self):
# 1) UTF8-encoded bytes
req1 = Request('http://example.org', cookies={'a': u'á'.encode('utf8')})
assert self.mw.process_request(req1, self.spider) is None
self.assertCookieValEqual(req1.headers['Cookie'], b'a=\xc3\xa1')
# 2) Non UTF8-encoded bytes
req2 = Request('http://example.org', cookies={'a': u'á'.encode('latin1')})
assert self.mw.process_request(req2, self.spider) is None
self.assertCookieValEqual(req2.headers['Cookie'], b'a=\xc3\xa1')
# 3) Unicode string
req3 = Request('http://example.org', cookies={'a': u'á'})
assert self.mw.process_request(req3, self.spider) is None
self.assertCookieValEqual(req3.headers['Cookie'], b'a=\xc3\xa1')
def test_request_headers_cookie_encoding(self):
# 1) UTF8-encoded bytes
req1 = Request('http://example.org', headers={'Cookie': u'a=á'.encode('utf8')})
assert self.mw.process_request(req1, self.spider) is None
self.assertCookieValEqual(req1.headers['Cookie'], b'a=\xc3\xa1')
# 2) Non UTF8-encoded bytes
req2 = Request('http://example.org', headers={'Cookie': u'a=á'.encode('latin1')})
assert self.mw.process_request(req2, self.spider) is None
self.assertCookieValEqual(req2.headers['Cookie'], b'a=\xc3\xa1')
# 3) Unicode string
req3 = Request('http://example.org', headers={'Cookie': u'a=á'})
assert self.mw.process_request(req3, self.spider) is None
self.assertCookieValEqual(req3.headers['Cookie'], b'a=\xc3\xa1')
def test_invalid_cookies(self):
"""
Invalid cookies are logged as warnings and discarded
"""
with LogCapture(
'scrapy.downloadermiddlewares.cookies',
propagate=False,
level=logging.INFO,
) as lc:
cookies1 = [{'value': 'bar'}, {'name': 'key', 'value': 'value1'}]
req1 = Request('http://example.org/1', cookies=cookies1)
assert self.mw.process_request(req1, self.spider) is None
cookies2 = [{'name': 'foo'}, {'name': 'key', 'value': 'value2'}]
req2 = Request('http://example.org/2', cookies=cookies2)
assert self.mw.process_request(req2, self.spider) is None
lc.check(
("scrapy.downloadermiddlewares.cookies",
"WARNING",
"Invalid cookie found in request <GET http://example.org/1>:"
" {'value': 'bar'} ('name' is missing)"),
("scrapy.downloadermiddlewares.cookies",
"WARNING",
"Invalid cookie found in request <GET http://example.org/2>:"
" {'name': 'foo'} ('value' is missing)"),
)
self.assertCookieValEqual(req1.headers['Cookie'], 'key=value1')
self.assertCookieValEqual(req2.headers['Cookie'], 'key=value2')

View File

@ -16,14 +16,16 @@ import sys
from collections import defaultdict
from urllib.parse import urlparse
from pydispatch import dispatcher
from pytest import mark
from testfixtures import LogCapture
from twisted.internet import reactor, defer
from twisted.trial import unittest
from twisted.web import server, static, util
from pydispatch import dispatcher
from scrapy import signals
from scrapy.core.engine import ExecutionEngine
from scrapy.exceptions import StopDownload
from scrapy.http import Request
from scrapy.item import Item, Field
from scrapy.linkextractors import LinkExtractor
@ -96,7 +98,7 @@ def start_test_site(debug=False):
r = static.File(root_dir)
r.putChild(b"redirect", util.Redirect(b"/redirected"))
r.putChild(b"redirected", static.Data(b"Redirected here", "text/plain"))
numbers = [str(x).encode("utf8") for x in range(2**14)]
numbers = [str(x).encode("utf8") for x in range(2**18)]
r.putChild(b"numbers", static.Data(b"".join(numbers), "text/plain"))
port = reactor.listenTCP(0, server.Site(r), interface="127.0.0.1")
@ -194,6 +196,16 @@ class CrawlerRun:
self.signals_caught[sig] = signalargs
class StopDownloadCrawlerRun(CrawlerRun):
"""
Make sure raising the StopDownload exception stops the download of the response body
"""
def bytes_received(self, data, request, spider):
super().bytes_received(data, request, spider)
raise StopDownload(fail=False)
class EngineTest(unittest.TestCase):
@defer.inlineCallbacks
@ -345,7 +357,7 @@ class EngineTest(unittest.TestCase):
# signal was fired multiple times
self.assertTrue(len(data) > 1)
# bytes were received in order
numbers = [str(x).encode("utf8") for x in range(2**14)]
numbers = [str(x).encode("utf8") for x in range(2**18)]
self.assertEqual(joined_data, b"".join(numbers))
def _assert_signals_caught(self):
@ -386,6 +398,45 @@ class EngineTest(unittest.TestCase):
self.assertEqual(len(e.open_spiders), 0)
class StopDownloadEngineTest(EngineTest):
@defer.inlineCallbacks
def test_crawler(self):
for spider in TestSpider, DictItemsSpider:
self.run = StopDownloadCrawlerRun(spider)
with LogCapture() as log:
yield self.run.run()
log.check_present(("scrapy.core.downloader.handlers.http11",
"DEBUG",
"Download stopped for <GET http://localhost:{}/redirected> from signal handler"
" StopDownloadCrawlerRun.bytes_received".format(self.run.portno)))
log.check_present(("scrapy.core.downloader.handlers.http11",
"DEBUG",
"Download stopped for <GET http://localhost:{}/> from signal handler"
" StopDownloadCrawlerRun.bytes_received".format(self.run.portno)))
log.check_present(("scrapy.core.downloader.handlers.http11",
"DEBUG",
"Download stopped for <GET http://localhost:{}/numbers> from signal handler"
" StopDownloadCrawlerRun.bytes_received".format(self.run.portno)))
self._assert_visited_urls()
self._assert_scheduled_requests(urls_to_visit=9)
self._assert_downloaded_responses()
self._assert_signals_caught()
self._assert_bytes_received()
def _assert_bytes_received(self):
self.assertEqual(9, len(self.run.bytes))
for request, data in self.run.bytes.items():
joined_data = b"".join(data)
self.assertTrue(len(data) == 1) # signal was fired only once
if self.run.getpath(request.url) == "/numbers":
# Received bytes are not the complete response. The exact amount depends
# on the buffer size, which can vary, so we only check that the amount
# of received bytes is strictly less than the full response.
numbers = [str(x).encode("utf8") for x in range(2**18)]
self.assertTrue(len(joined_data) < len(b"".join(numbers)))
if __name__ == "__main__":
if len(sys.argv) > 1 and sys.argv[1] == 'runserver':
start_test_site(debug=True)

View File

@ -1,4 +1,5 @@
import unittest
from unittest import mock
from warnings import catch_warnings
from w3lib.encoding import resolve_encoding
@ -685,6 +686,26 @@ class TextResponseTest(BaseResponseTest):
self.assertEqual(len(warnings), 1)
self.assertEqual(warnings[0].category, ScrapyDeprecationWarning)
def test_json_response(self):
json_body = b"""{"ip": "109.187.217.200"}"""
json_response = self.response_class("http://www.example.com", body=json_body)
self.assertEqual(json_response.json(), {'ip': '109.187.217.200'})
text_body = b"""<html><body>text</body></html>"""
text_response = self.response_class("http://www.example.com", body=text_body)
with self.assertRaises(ValueError):
text_response.json()
def test_cache_json_response(self):
json_valid_bodies = [b"""{"ip": "109.187.217.200"}""", b"""null"""]
for json_body in json_valid_bodies:
json_response = self.response_class("http://www.example.com", body=json_body)
with mock.patch('json.loads') as mock_json:
for _ in range(2):
json_response.json()
mock_json.assert_called_once_with(json_body.decode())
class HtmlResponseTest(TextResponseTest):

View File

@ -271,6 +271,7 @@ class LocalWeakReferencedCacheTest(unittest.TestCase):
self.assertNotIn(r1, cache)
self.assertIn(r2, cache)
self.assertIn(r3, cache)
self.assertEqual(cache[r1], None)
self.assertEqual(cache[r2], 2)
self.assertEqual(cache[r3], 3)
del r2