diff --git a/docs/faq.rst b/docs/faq.rst
index 0b6bd6a86..c06cb945b 100644
--- a/docs/faq.rst
+++ b/docs/faq.rst
@@ -371,6 +371,19 @@ Twisted reactor is :class:`twisted.internet.selectreactor.SelectReactor`. Switch
different reactor is possible by using the :setting:`TWISTED_REACTOR` setting.
+.. _faq-stop-response-download:
+
+How can I cancel the download of a given response?
+--------------------------------------------------
+
+In some situations, it might be useful to stop the download of a certain response.
+For instance, if you only need the first part of a large response and you would like
+to save resources by avoiding the download of the whole body.
+In that case, you could attach a handler to the :class:`~scrapy.signals.bytes_received`
+signal and raise a :exc:`~scrapy.exceptions.StopDownload` exception. Please refer to
+the :ref:`topics-stop-response-download` topic for additional information and examples.
+
+
.. _has been reported: https://github.com/scrapy/scrapy/issues/2905
.. _user agents: https://en.wikipedia.org/wiki/User_agent
.. _LIFO: https://en.wikipedia.org/wiki/Stack_(abstract_data_type)
diff --git a/docs/topics/exceptions.rst b/docs/topics/exceptions.rst
index 09cb8ed66..583a50ab8 100644
--- a/docs/topics/exceptions.rst
+++ b/docs/topics/exceptions.rst
@@ -14,13 +14,6 @@ Built-in Exceptions reference
Here's a list of all exceptions included in Scrapy and their usage.
-DropItem
---------
-
-.. exception:: DropItem
-
-The exception that must be raised by item pipeline stages to stop processing an
-Item. For more information see :ref:`topics-item-pipeline`.
CloseSpider
-----------
@@ -47,6 +40,14 @@ DontCloseSpider
This exception can be raised in a :signal:`spider_idle` signal handler to
prevent the spider from being closed.
+DropItem
+--------
+
+.. exception:: DropItem
+
+The exception that must be raised by item pipeline stages to stop processing an
+Item. For more information see :ref:`topics-item-pipeline`.
+
IgnoreRequest
-------------
@@ -77,3 +78,37 @@ NotSupported
This exception is raised to indicate an unsupported feature.
+StopDownload
+-------------
+
+.. versionadded:: 2.2
+
+.. exception:: StopDownload(fail=True)
+
+Raised from a :class:`~scrapy.signals.bytes_received` signal handler to
+indicate that no further bytes should be downloaded for a response.
+
+The ``fail`` boolean parameter controls which method will handle the resulting
+response:
+
+* If ``fail=True`` (default), the request errback is called. The response object is
+ available as the ``response`` attribute of the ``StopDownload`` exception,
+ which is in turn stored as the ``value`` attribute of the received
+ :class:`~twisted.python.failure.Failure` object. This means that in an errback
+ defined as ``def errback(self, failure)``, the response can be accessed though
+ ``failure.value.response``.
+
+* If ``fail=False``, the request callback is called instead.
+
+In both cases, the response could have its body truncated: the body contains
+all bytes received up until the exception is raised, including the bytes
+received in the signal handler that raises the exception. Also, the response
+object is marked with ``"download_stopped"`` in its :attr:`Response.flags`
+attribute.
+
+.. note:: ``fail`` is a keyword-only parameter, i.e. raising
+ ``StopDownload(False)`` or ``StopDownload(True)`` will raise
+ a :class:`TypeError`.
+
+See the documentation for the :class:`~scrapy.signals.bytes_received` signal
+and the :ref:`topics-stop-response-download` topic for additional information and examples.
diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst
index 15a83f453..ae25ff7e4 100644
--- a/docs/topics/request-response.rst
+++ b/docs/topics/request-response.rst
@@ -385,6 +385,51 @@ The meta key is used set retry times per request. When initialized, the
:reqmeta:`max_retry_times` meta key takes higher precedence over the
:setting:`RETRY_TIMES` setting.
+
+.. _topics-stop-response-download:
+
+Stopping the download of a Response
+===================================
+
+Raising a :exc:`~scrapy.exceptions.StopDownload` exception from a
+:class:`~scrapy.signals.bytes_received` signal handler will stop the
+download of a given response. See the following example::
+
+ import scrapy
+
+
+ class StopSpider(scrapy.Spider):
+ name = "stop"
+ start_urls = ["https://docs.scrapy.org/en/latest/"]
+
+ @classmethod
+ def from_crawler(cls, crawler):
+ spider = super().from_crawler(crawler)
+ crawler.signals.connect(spider.on_bytes_received, signal=scrapy.signals.bytes_received)
+ return spider
+
+ def parse(self, response):
+ # 'last_chars' show that the full response was not downloaded
+ yield {"len": len(response.text), "last_chars": response.text[-40:]}
+
+ def on_bytes_received(self, data, request, spider):
+ raise scrapy.exceptions.StopDownload(fail=False)
+
+which produces the following output::
+
+ 2020-05-19 17:26:12 [scrapy.core.engine] INFO: Spider opened
+ 2020-05-19 17:26:12 [scrapy.extensions.logstats] INFO: Crawled 0 pages (at 0 pages/min), scraped 0 items (at 0 items/min)
+ 2020-05-19 17:26:13 [scrapy.core.downloader.handlers.http11] DEBUG: Download stopped for from signal handler StopSpider.on_bytes_received
+ 2020-05-19 17:26:13 [scrapy.core.engine] DEBUG: Crawled (200) (referer: None) ['download_stopped']
+ 2020-05-19 17:26:13 [scrapy.core.scraper] DEBUG: Scraped from <200 https://docs.scrapy.org/en/latest/>
+ {'len': 279, 'last_chars': 'dth, initial-scale=1.0">\n \n
Scr'}
+ 2020-05-19 17:26:13 [scrapy.core.engine] INFO: Closing spider (finished)
+
+By default, resulting responses are handled by their corresponding errbacks. To
+call their callback instead, like in this example, pass ``fail=False`` to the
+:exc:`~scrapy.exceptions.StopDownload` exception.
+
+
.. _topics-request-response-ref-request-subclasses:
Request subclasses
@@ -716,9 +761,9 @@ Response objects
.. versionadded:: 2.1.0
The IP address of the server from which the Response originated.
-
+
This attribute is currently only populated by the HTTP 1.1 download
- handler, i.e. for ``http(s)`` responses. For other handlers,
+ handler, i.e. for ``http(s)`` responses. For other handlers,
:attr:`ip_address` is always ``None``.
.. method:: Response.copy()
diff --git a/docs/topics/signals.rst b/docs/topics/signals.rst
index 7fe63a7b0..fe4fb0834 100644
--- a/docs/topics/signals.rst
+++ b/docs/topics/signals.rst
@@ -373,6 +373,8 @@ request_left_downloader
bytes_received
~~~~~~~~~~~~~~
+.. versionadded:: 2.2
+
.. signal:: bytes_received
.. function:: bytes_received(data, request, spider)
@@ -385,14 +387,19 @@ bytes_received
This signal does not support returning deferreds from its handlers.
:param data: the data received by the download handler
- :type spider: :class:`bytes` object
+ :type data: :class:`bytes` object
- :param request: the request that generated the response
+ :param request: the request that generated the download
:type request: :class:`~scrapy.http.Request` object
:param spider: the spider associated with the response
:type spider: :class:`~scrapy.spiders.Spider` object
+.. note:: Handlers of this signal can stop the download of a response while it
+ is in progress by raising the :exc:`~scrapy.exceptions.StopDownload`
+ exception. Please refer to the :ref:`topics-stop-response-download` topic
+ for additional information and examples.
+
Response signals
----------------
diff --git a/scrapy/core/downloader/handlers/http11.py b/scrapy/core/downloader/handlers/http11.py
index c21491f52..22c9ac520 100644
--- a/scrapy/core/downloader/handlers/http11.py
+++ b/scrapy/core/downloader/handlers/http11.py
@@ -12,6 +12,7 @@ from urllib.parse import urldefrag
from twisted.internet import defer, protocol, ssl
from twisted.internet.endpoints import TCP4ClientEndpoint
from twisted.internet.error import TimeoutError
+from twisted.python.failure import Failure
from twisted.web.client import Agent, HTTPConnectionPool, ResponseDone, ResponseFailed, URI
from twisted.web.http import _DataLoss, PotentialDataLoss
from twisted.web.http_headers import Headers as TxHeaders
@@ -21,7 +22,7 @@ from zope.interface import implementer
from scrapy import signals
from scrapy.core.downloader.tls import openssl_methods
from scrapy.core.downloader.webclient import _parse
-from scrapy.exceptions import ScrapyDeprecationWarning
+from scrapy.exceptions import ScrapyDeprecationWarning, StopDownload
from scrapy.http import Headers
from scrapy.responsetypes import responsetypes
from scrapy.utils.misc import create_instance, load_object
@@ -431,7 +432,7 @@ class ScrapyAgent:
def _cb_bodydone(self, result, request, url):
headers = Headers(result["txresponse"].headers.getAllRawHeaders())
respcls = responsetypes.from_args(headers=headers, url=url, body=result["body"])
- return respcls(
+ response = respcls(
url=url,
status=int(result["txresponse"].code),
headers=headers,
@@ -440,6 +441,10 @@ class ScrapyAgent:
certificate=result["certificate"],
ip_address=result["ip_address"],
)
+ if result.get("failure"):
+ result["failure"].value.response = response
+ return result["failure"]
+ return response
@implementer(IBodyProducer)
@@ -477,6 +482,16 @@ class _ResponseReader(protocol.Protocol):
self._ip_address = None
self._crawler = crawler
+ def _finish_response(self, flags=None, failure=None):
+ self._finished.callback({
+ "txresponse": self._txresponse,
+ "body": self._bodybuf.getvalue(),
+ "flags": flags,
+ "certificate": self._certificate,
+ "ip_address": self._ip_address,
+ "failure": failure,
+ })
+
def connectionMade(self):
if self._certificate is None:
with suppress(AttributeError):
@@ -493,12 +508,19 @@ class _ResponseReader(protocol.Protocol):
self._bodybuf.write(bodyBytes)
self._bytes_received += len(bodyBytes)
- self._crawler.signals.send_catch_log(
+ bytes_received_result = self._crawler.signals.send_catch_log(
signal=signals.bytes_received,
data=bodyBytes,
request=self._request,
spider=self._crawler.spider,
)
+ for handler, result in bytes_received_result:
+ if isinstance(result, Failure) and isinstance(result.value, StopDownload):
+ logger.debug("Download stopped for %(request)s from signal handler %(handler)s",
+ {"request": self._request, "handler": handler.__qualname__})
+ self.transport._producer.loseConnection()
+ failure = result if result.value.fail else None
+ self._finish_response(flags=["download_stopped"], failure=failure)
if self._maxsize and self._bytes_received > self._maxsize:
logger.error("Received (%(bytes)s) bytes larger than download "
@@ -521,36 +543,17 @@ class _ResponseReader(protocol.Protocol):
if self._finished.called:
return
- body = self._bodybuf.getvalue()
if reason.check(ResponseDone):
- self._finished.callback({
- "txresponse": self._txresponse,
- "body": body,
- "flags": None,
- "certificate": self._certificate,
- "ip_address": self._ip_address,
- })
+ self._finish_response()
return
if reason.check(PotentialDataLoss):
- self._finished.callback({
- "txresponse": self._txresponse,
- "body": body,
- "flags": ["partial"],
- "certificate": self._certificate,
- "ip_address": self._ip_address,
- })
+ self._finish_response(flags=["partial"])
return
if reason.check(ResponseFailed) and any(r.check(_DataLoss) for r in reason.value.reasons):
if not self._fail_on_dataloss:
- self._finished.callback({
- "txresponse": self._txresponse,
- "body": body,
- "flags": ["dataloss"],
- "certificate": self._certificate,
- "ip_address": self._ip_address,
- })
+ self._finish_response(flags=["dataloss"])
return
elif not self._fail_on_dataloss_warned:
diff --git a/scrapy/exceptions.py b/scrapy/exceptions.py
index 7c4bb3d00..45f152321 100644
--- a/scrapy/exceptions.py
+++ b/scrapy/exceptions.py
@@ -41,6 +41,18 @@ class CloseSpider(Exception):
self.reason = reason
+class StopDownload(Exception):
+ """
+ Stop the download of the body for a given response.
+ The 'fail' boolean parameter indicates whether or not the resulting partial response
+ should be handled by the request errback. Note that 'fail' is a keyword-only argument.
+ """
+
+ def __init__(self, *, fail=True):
+ super().__init__()
+ self.fail = fail
+
+
# Items
@@ -59,6 +71,7 @@ class NotSupported(Exception):
class UsageError(Exception):
"""To indicate a command-line usage error"""
+
def __init__(self, *a, **kw):
self.print_help = kw.pop('print_help', True)
super(UsageError, self).__init__(*a, **kw)
diff --git a/scrapy/utils/signal.py b/scrapy/utils/signal.py
index a311e9257..115707182 100644
--- a/scrapy/utils/signal.py
+++ b/scrapy/utils/signal.py
@@ -5,13 +5,14 @@ import logging
from twisted.internet.defer import DeferredList, Deferred
from twisted.python.failure import Failure
-from pydispatch.dispatcher import Any, Anonymous, liveReceivers, \
- getAllReceivers, disconnect
+from pydispatch.dispatcher import Anonymous, Any, disconnect, getAllReceivers, liveReceivers
from pydispatch.robustapply import robustApply
+from scrapy.exceptions import StopDownload
from scrapy.utils.defer import maybeDeferred_coro
from scrapy.utils.log import failure_to_exc_info
+
logger = logging.getLogger(__name__)
@@ -23,7 +24,7 @@ def send_catch_log(signal=Any, sender=Anonymous, *arguments, **named):
"""Like pydispatcher.robust.sendRobust but it also logs errors and returns
Failures instead of exceptions.
"""
- dont_log = named.pop('dont_log', _IgnoredException)
+ dont_log = (named.pop('dont_log', _IgnoredException), StopDownload)
spider = named.get('spider', None)
responses = []
for receiver in liveReceivers(getAllReceivers(sender, signal)):
diff --git a/tests/spiders.py b/tests/spiders.py
index 33d5d02e1..05078cc04 100644
--- a/tests/spiders.py
+++ b/tests/spiders.py
@@ -7,6 +7,8 @@ from urllib.parse import urlencode
from twisted.internet import defer
+from scrapy import signals
+from scrapy.exceptions import StopDownload
from scrapy.http import Request
from scrapy.item import Item
from scrapy.linkextractors import LinkExtractor
@@ -267,3 +269,36 @@ class CrawlSpiderWithErrback(MockServerSpider, CrawlSpider):
def errback(self, failure):
self.logger.info('[errback] status %i', failure.value.response.status)
+
+
+class BytesReceivedCallbackSpider(MetaSpider):
+
+ full_response_length = 2**18
+
+ @classmethod
+ def from_crawler(cls, crawler, *args, **kwargs):
+ spider = super().from_crawler(crawler, *args, **kwargs)
+ crawler.signals.connect(spider.bytes_received, signals.bytes_received)
+ return spider
+
+ def start_requests(self):
+ body = b"a" * self.full_response_length
+ url = self.mockserver.url("/alpayload")
+ yield Request(url, method="POST", body=body, errback=self.errback)
+
+ def parse(self, response):
+ self.meta["response"] = response
+
+ def errback(self, failure):
+ self.meta["failure"] = failure
+
+ def bytes_received(self, data, request, spider):
+ self.meta["bytes_received"] = data
+ raise StopDownload(fail=False)
+
+
+class BytesReceivedErrbackSpider(BytesReceivedCallbackSpider):
+
+ def bytes_received(self, data, request, spider):
+ self.meta["bytes_received"] = data
+ raise StopDownload(fail=True)
diff --git a/tests/test_crawl.py b/tests/test_crawl.py
index 84f80d103..0115b8fb9 100644
--- a/tests/test_crawl.py
+++ b/tests/test_crawl.py
@@ -9,17 +9,31 @@ from pytest import mark
from testfixtures import LogCapture
from twisted.internet import defer
from twisted.internet.ssl import Certificate
+from twisted.python.failure import Failure
from twisted.trial.unittest import TestCase
from scrapy import signals
from scrapy.crawler import CrawlerRunner
+from scrapy.exceptions import StopDownload
from scrapy.http import Request
+from scrapy.http.response import Response
from scrapy.utils.python import to_unicode
from tests.mockserver import MockServer
-from tests.spiders import (FollowAllSpider, DelaySpider, SimpleSpider, BrokenStartRequestsSpider,
- SingleRequestSpider, DuplicateStartRequestsSpider, CrawlSpiderWithErrback,
- AsyncDefSpider, AsyncDefAsyncioSpider, AsyncDefAsyncioReturnSpider,
- AsyncDefAsyncioReqsReturnSpider)
+from tests.spiders import (
+ AsyncDefAsyncioReqsReturnSpider,
+ AsyncDefAsyncioReturnSpider,
+ AsyncDefAsyncioSpider,
+ AsyncDefSpider,
+ BrokenStartRequestsSpider,
+ BytesReceivedCallbackSpider,
+ BytesReceivedErrbackSpider,
+ CrawlSpiderWithErrback,
+ DelaySpider,
+ DuplicateStartRequestsSpider,
+ FollowAllSpider,
+ SimpleSpider,
+ SingleRequestSpider,
+)
class CrawlTestCase(TestCase):
@@ -457,3 +471,27 @@ with multiples lines
ip_address = crawler.spider.meta['responses'][0].ip_address
self.assertIsInstance(ip_address, IPv4Address)
self.assertEqual(str(ip_address), gethostbyname(expected_netloc))
+
+ @defer.inlineCallbacks
+ def test_stop_download_callback(self):
+ crawler = self.runner.create_crawler(BytesReceivedCallbackSpider)
+ yield crawler.crawl(mockserver=self.mockserver)
+ self.assertIsNone(crawler.spider.meta.get("failure"))
+ self.assertIsInstance(crawler.spider.meta["response"], Response)
+ self.assertEqual(crawler.spider.meta["response"].body, crawler.spider.meta.get("bytes_received"))
+ self.assertLess(len(crawler.spider.meta["response"].body), crawler.spider.full_response_length)
+
+ @defer.inlineCallbacks
+ def test_stop_download_errback(self):
+ crawler = self.runner.create_crawler(BytesReceivedErrbackSpider)
+ yield crawler.crawl(mockserver=self.mockserver)
+ self.assertIsNone(crawler.spider.meta.get("response"))
+ self.assertIsInstance(crawler.spider.meta["failure"], Failure)
+ self.assertIsInstance(crawler.spider.meta["failure"].value, StopDownload)
+ self.assertIsInstance(crawler.spider.meta["failure"].value.response, Response)
+ self.assertEqual(
+ crawler.spider.meta["failure"].value.response.body,
+ crawler.spider.meta.get("bytes_received"))
+ self.assertLess(
+ len(crawler.spider.meta["failure"].value.response.body),
+ crawler.spider.full_response_length)
diff --git a/tests/test_engine.py b/tests/test_engine.py
index d781665dc..6696ee52e 100644
--- a/tests/test_engine.py
+++ b/tests/test_engine.py
@@ -16,13 +16,15 @@ import sys
from collections import defaultdict
from urllib.parse import urlparse
+from pydispatch import dispatcher
+from testfixtures import LogCapture
from twisted.internet import reactor, defer
from twisted.trial import unittest
from twisted.web import server, static, util
-from pydispatch import dispatcher
from scrapy import signals
from scrapy.core.engine import ExecutionEngine
+from scrapy.exceptions import StopDownload
from scrapy.http import Request
from scrapy.item import Item, Field
from scrapy.linkextractors import LinkExtractor
@@ -90,7 +92,7 @@ def start_test_site(debug=False):
r = static.File(root_dir)
r.putChild(b"redirect", util.Redirect(b"/redirected"))
r.putChild(b"redirected", static.Data(b"Redirected here", "text/plain"))
- numbers = [str(x).encode("utf8") for x in range(2**14)]
+ numbers = [str(x).encode("utf8") for x in range(2**18)]
r.putChild(b"numbers", static.Data(b"".join(numbers), "text/plain"))
port = reactor.listenTCP(0, server.Site(r), interface="127.0.0.1")
@@ -188,6 +190,16 @@ class CrawlerRun:
self.signals_caught[sig] = signalargs
+class StopDownloadCrawlerRun(CrawlerRun):
+ """
+ Make sure raising the StopDownload exception stops the download of the response body
+ """
+
+ def bytes_received(self, data, request, spider):
+ super().bytes_received(data, request, spider)
+ raise StopDownload(fail=False)
+
+
class EngineTest(unittest.TestCase):
@defer.inlineCallbacks
@@ -316,7 +328,7 @@ class EngineTest(unittest.TestCase):
# signal was fired multiple times
self.assertTrue(len(data) > 1)
# bytes were received in order
- numbers = [str(x).encode("utf8") for x in range(2**14)]
+ numbers = [str(x).encode("utf8") for x in range(2**18)]
self.assertEqual(joined_data, b"".join(numbers))
def _assert_signals_caught(self):
@@ -357,6 +369,45 @@ class EngineTest(unittest.TestCase):
self.assertEqual(len(e.open_spiders), 0)
+class StopDownloadEngineTest(EngineTest):
+
+ @defer.inlineCallbacks
+ def test_crawler(self):
+ for spider in TestSpider, DictItemsSpider:
+ self.run = StopDownloadCrawlerRun(spider)
+ with LogCapture() as log:
+ yield self.run.run()
+ log.check_present(("scrapy.core.downloader.handlers.http11",
+ "DEBUG",
+ "Download stopped for from signal handler"
+ " StopDownloadCrawlerRun.bytes_received".format(self.run.portno)))
+ log.check_present(("scrapy.core.downloader.handlers.http11",
+ "DEBUG",
+ "Download stopped for from signal handler"
+ " StopDownloadCrawlerRun.bytes_received".format(self.run.portno)))
+ log.check_present(("scrapy.core.downloader.handlers.http11",
+ "DEBUG",
+ "Download stopped for from signal handler"
+ " StopDownloadCrawlerRun.bytes_received".format(self.run.portno)))
+ self._assert_visited_urls()
+ self._assert_scheduled_requests(urls_to_visit=9)
+ self._assert_downloaded_responses()
+ self._assert_signals_caught()
+ self._assert_bytes_received()
+
+ def _assert_bytes_received(self):
+ self.assertEqual(9, len(self.run.bytes))
+ for request, data in self.run.bytes.items():
+ joined_data = b"".join(data)
+ self.assertTrue(len(data) == 1) # signal was fired only once
+ if self.run.getpath(request.url) == "/numbers":
+ # Received bytes are not the complete response. The exact amount depends
+ # on the buffer size, which can vary, so we only check that the amount
+ # of received bytes is strictly less than the full response.
+ numbers = [str(x).encode("utf8") for x in range(2**18)]
+ self.assertTrue(len(joined_data) < len(b"".join(numbers)))
+
+
if __name__ == "__main__":
if len(sys.argv) > 1 and sys.argv[1] == 'runserver':
start_test_site(debug=True)