diff --git a/docs/ref/signals.rst b/docs/ref/signals.rst index 1dc4d8686..d6f60c4c2 100644 --- a/docs/ref/signals.rst +++ b/docs/ref/signals.rst @@ -23,131 +23,181 @@ Here's a list of signals used in Scrapy and their meaning, in alphabetical order. .. signal:: domain_closed -.. function:: domain_closed(domain, spider, status) +.. function:: domain_closed(domain, spider, reason) -Sent right after a spider/domain has been closed. + Sent right after a spider/domain has been closed. -``domain`` is a string which contains the domain of the spider which has been closed -``spider`` is the spider which has been closed -``status`` is a string which can have two values: ``'finished'`` if the domain -has finished successfully, or ``'cancelled'`` if the domain was cancelled (for -example, by hitting Ctrl-C, by calling the engine ``stop()`` method or by -explicitly closing the domain). + :param domain: a string which contains the domain of the spider which has + been closed + :type domain: str + + :param spider: the spider which has been closed + :type spider: :class:`~scrapy.spider.BaseSpider` object + + :param reason: a string which describes the reason why the domain was closed. If + it was closed because the domain has completed scraping, it the reason + is ``'finished'``. Otherwise, if the domain was manually closed by + calling the ``close_domain`` engine method, then the reason is the one + passed in the ``reason`` argument of that method (which defaults to + ``'cancelled'``). If the engine was shutdown (for example, by hitting + Ctrl-C to stop it) the reason will be ``'shutdown'``. + :type reason: str .. signal:: domain_open .. function:: domain_open(domain, spider) -Sent right before a spider has been opened for crawling. + Sent right before a spider has been opened for crawling. -``domain`` is a string which contains the domain of the spider which is about -to be opened -``spider`` is the spider which is about to be opened + :param domain: a string which contains the domain of the spider which is about + to be opened + :type domain: str + + :param spider: the spider which is about to be opened + :type spider: :class:`~scrapy.spider.BaseSpider` object .. signal:: domain_opened .. function:: domain_opened(domain, spider) -Sent right after a spider has been opened for crawling. + Sent right after a spider has been opened for crawling. -``domain`` is a string with the domain of the spider which has been opened -``spider`` is the spider which has been opened + :param domain: a string with the domain of the spider which has been opened + :type domain: str + + :param spider: the spider which has been opened + :type spider: :class:`~scrapy.spider.BaseSpider` object .. signal:: domain_idle .. function:: domain_idle(domain, spider) -Sent when a domain has no further: - * requests waiting to be downloaded - * requests scheduled - * items being processed in the item pipeline + Sent when a domain has gone idle, which means the spider has no further: + * requests waiting to be downloaded + * requests scheduled + * items being processed in the item pipeline -``domain`` is a string with the domain of the spider which has gone idle -``spider`` is the spider which has gone idle + :param domain: is a string with the domain of the spider which has gone idle + :type domain: str -If any handler of this signals raises a :exception:`DontCloseDomain` the domain -won't be closed at this time and will wait until another idle signal is sent. -Otherwise (if no handler raises :exception:`DontCloseDomain`) the domain will -be closed immediately after all handlers of ``domain_idle`` have finished, and -a :signal:`domain_closed` will thus be sent. + :param spider: the spider which has gone idle + :type spider: :class:`~scrapy.spider.BaseSpider` object + + If any handler of this signal handlers raises a + :exception:`DontCloseDomain` the domain won't be closed this time and will + wait until another idle signal is sent. Otherwise (if no handler raises + :exception:`DontCloseDomain`) the domain will be closed immediately after + all handlers of ``domain_idle`` have finished, and a + :signal:`domain_closed` will thus be sent. .. signal:: engine_started .. function:: engine_started() -Sent when the Scrapy engine is started (for example, when a crawling -process has started). + Sent when the Scrapy engine is started (for example, when a crawling + process has started). .. signal:: engine_stopped .. function:: engine_stopped() -Sent when the Scrapy engine is stopped (for example, when a crawling -process has finished). - -.. signal:: request_received -.. function:: request_received(request, spider, response) - -Sent when the engine receives a :class:`~scrapy.http.Request` from a spider. - -``request`` is the :class:`~scrapy.http.Request` received -``spider`` is the spider which generated the request -``response`` is the :class:`~scrapy.http.Response` fed to the spider which -generated the request - -.. signal:: request_uploaded -.. function:: request_uploaded(request, spider) - -Sent right after the download has sent a :class:`~scrapy.http.Request`. - -``request`` is the :class:`~scrapy.http.Request` uploaded/sent -``spider`` is the spider which generated the request - -.. signal:: response_received -.. function:: response_received(response, spider) - -``response`` is the :class:`~scrapy.http.Response` received -``spider`` is the spider for which the response is intended - -Sent when the engine receives a new :class:`~scrapy.http.Response` from the -downloader. - -.. signal:: response_downloaded -.. function:: response_downloaded(response, spider) - -Sent by the downloader right after a ``HTTPResponse`` is downloaded. - -``response`` is the ``HTTPResponse`` downloaded -``spider`` is the spider for which the response is intended + Sent when the Scrapy engine is stopped (for example, when a crawling + process has finished). .. signal:: item_scraped .. function:: item_scraped(item, spider, response) -Sent when the engine receives a new scraped item from the spider, and right -before the item is sent to the :ref:`topics-item-pipeline`. + Sent when the engine receives a new scraped item from the spider, and right + before the item is sent to the :ref:`topics-item-pipeline`. -``item`` is the item scraped -``spider`` is the spider which scraped the item -``response`` is the :class:`~scrapy.http.Response` from which the item was -scraped + :param item: is the item scraped + :type item: :class:`~scrapy.item.ScrapedItem` object + + :param spider: the spider which scraped the item + :type spider: :class:`~scrapy.spider.BaseSpider` object + + :param response: the response from which the item was scraped + :type response: :class:`~scrapy.http.Response` object .. signal:: item_passed .. function:: item_passed(item, spider, response, pipe_output) -Sent after an item has passed al the :ref:`topics-item-pipeline` stages without -being dropped. + Sent after an item has passed al the :ref:`topics-item-pipeline` stages without + being dropped. -``item`` is the item which passed the pipeline -``spider`` is the spider which scraped the item -``response`` is the :class:`~scrapy.http.Response` from which the item was scraped -``pipe_output`` is the output of the item pipeline. Typically, this points to -the same ``item`` object, unless some pipeline stage created a new item. + :param item: the item which passed the pipeline + :type item: :class:`~scrapy.item.ScrapedItem` object + + :param spider: the spider which scraped the item + :type spider: :class:`~scrapy.spider.BaseSpider` object + + :param response: the response from which the item was scraped + :type response: :class:`~scrapy.http.Response` object + + :param pipe_output: the output of the item pipeline. This is typically the + same :class:`~scrapy.item.ScrapedItem` object received in the ``item`` + parameter, unless some pipeline stage created a new item. .. signal:: item_dropped .. function:: item_dropped(item, spider, response, exception) -Sent after an item has dropped from the :ref:`topics-item-pipeline` when some stage -raised a :exception:`DropItem` exception. + Sent after an item has dropped from the :ref:`topics-item-pipeline` when some stage + raised a :exception:`DropItem` exception. -``item`` is the item dropped from the :ref:`topics-item-pipeline` -``spider`` is the spider which scraped the item -``response`` is the :class:`~scrapy.http.Response` from which the item was scraped -``exception`` is the (:exception:`DropItem` child) exception that caused the -item to be dropped + :param item: the item dropped from the :ref:`topics-item-pipeline` + :type item: :class:`~scrapy.item.ScrapedItem` object + :param spider: the spider which scraped the item + :type spider: :class:`~scrapy.spider.BaseSpider` object + + :param response: the response from which the item was scraped + :type response: :class:`~scrapy.http.Response` object + + :param exception: the exception (which must be a :exception:`DropItem` + subclass) which caused the item to be dropped + :type exception: :exception:`DropItem` exception + +.. signal:: request_received +.. function:: request_received(request, spider, response) + + Sent when the engine receives a :class:`~scrapy.http.Request` from a spider. + + :param request: the request received + :type request: :class:`~scrapy.http.Request` object + + :param spider: the spider which generated the request + :type spider: :class:`~scrapy.spider.BaseSpider` object + + :param response: the :class:`~scrapy.http.Response` fed to the spider which + generated the request later + :type response: :class:`~scrapy.http.Response` object + +.. signal:: request_uploaded +.. function:: request_uploaded(request, spider) + + Sent right after the download has sent a :class:`~scrapy.http.Request`. + + :param request: the request uploaded/sent + :type request: :class:`~scrapy.http.Request` object + + :param spider: the spider which generated the request + :type spider: :class:`~scrapy.spider.BaseSpider` object + +.. signal:: response_received +.. function:: response_received(response, spider) + + :param response: the response received + :type response: :class:`~scrapy.http.Response` object + + :param spider: the spider for which the response is intended + :type spider: :class:`~scrapy.spider.BaseSpider` object + + Sent when the engine receives a new :class:`~scrapy.http.Response` from the + downloader. + +.. signal:: response_downloaded +.. function:: response_downloaded(response, spider) + + Sent by the downloader right after a ``HTTPResponse`` is downloaded. + + :param response: the response downloaded + :type response: :class:`~scrapy.http.Response` object + + :param spider: the spider for which the response is intended + :type spider: :class:`~scrapy.spider.BaseSpider` object diff --git a/scrapy/contrib/itemsampler.py b/scrapy/contrib/itemsampler.py index d7c1f61ae..e0774be80 100644 --- a/scrapy/contrib/itemsampler.py +++ b/scrapy/contrib/itemsampler.py @@ -68,8 +68,8 @@ class ItemSamplerPipeline(object): if self.empty_domains: log.msg("No products sampled for: %s" % " ".join(self.empty_domains), level=log.WARNING) - def domain_closed(self, domain, spider, status): - if status == 'finished' and not stats.getpath("%s/items_sampled" % domain): + def domain_closed(self, domain, spider, reason): + if reason == 'finished' and not stats.getpath("%s/items_sampled" % domain): self.empty_domains.add(domain) self.domains_count += 1 log.msg("Sampled %d domains so far (%d empty)" % (self.domains_count, len(self.empty_domains)), level=log.INFO) diff --git a/scrapy/core/downloader/dnscache.py b/scrapy/core/downloader/dnscache.py index 2bc8acb75..1a8418391 100644 --- a/scrapy/core/downloader/dnscache.py +++ b/scrapy/core/downloader/dnscache.py @@ -47,6 +47,6 @@ class DNSCache(object): self._cache[host] = ips return ips[0] - def domain_closed(self, domain, spider, status): + def domain_closed(self, domain, spider, reason): if domain in self._cache: del self._cache[domain] diff --git a/scrapy/core/engine.py b/scrapy/core/engine.py index f360a8dcf..a466e1d45 100644 --- a/scrapy/core/engine.py +++ b/scrapy/core/engine.py @@ -47,7 +47,7 @@ class ExecutionEngine(object): def __init__(self): self.configured = False self.keep_alive = False - self.cancelled = set() # domains in cancelation state + self.closing = {} # domains being closed self.debug_mode = settings.getbool('ENGINE_DEBUG') self.tasks = [] self.ports = [] @@ -135,7 +135,7 @@ class ExecutionEngine(object): self.running = False for domain in self.open_domains: spider = spiders.fromdomain(domain) - signals.send_catch_log(signal=signals.domain_closed, sender=self.__class__, domain=domain, spider=spider, status='cancelled') + signals.send_catch_log(signal=signals.domain_closed, sender=self.__class__, domain=domain, spider=spider, reason='shutdown') for tsk, _, _ in self.tasks: # stop looping calls if tsk.running: tsk.stop() @@ -186,7 +186,7 @@ class ExecutionEngine(object): return # backout enqueing downloads if domain needs it - if domain in self.cancelled or self.downloader.needs_backout(domain): + if domain in self.closing or self.downloader.needs_backout(domain): return # Next pending request from scheduler @@ -235,7 +235,7 @@ class ExecutionEngine(object): signals.send_catch_log(signal=signals.item_passed, sender=self.__class__, item=item, spider=spider, response=response, pipe_output=pipe_result) self.next_request(spider) - if domain in self.cancelled: + if domain in self.closing: return elif isinstance(output, ScrapedItem): log.msg("Scraped %s in <%s>" % (output, request.url), log.INFO, domain=domain) @@ -346,7 +346,7 @@ class ExecutionEngine(object): spider = spider or spiders.fromdomain(domain) self.next_request(spider) - self.cancelled.discard(domain) + self.closing.pop(domain, None) self.downloader.open_domain(domain) self.pipeline.open_domain(domain) self._scraping[domain] = set() @@ -379,11 +379,11 @@ class ExecutionEngine(object): if self.is_idle() and not self.keep_alive: self.stop() - def close_domain(self, domain): + def close_domain(self, domain, reason='cancelled'): """Close (cancel) domain and clear all its outstanding requests""" - if domain not in self.cancelled: - log.msg("Closing domain", domain=domain) - self.cancelled.add(domain) + if domain not in self.closing: + log.msg("Closing domain (%s)" % reason, domain=domain) + self.closing[domain] = reason self._close_domain(domain) def _close_domain(self, domain): @@ -400,10 +400,10 @@ class ExecutionEngine(object): self.scheduler.close_domain(domain) self.pipeline.close_domain(domain) del self._scraping[domain] - status = 'cancelled' if domain in self.cancelled else 'finished' - signals.send_catch_log(signal=signals.domain_closed, sender=self.__class__, domain=domain, spider=spider, status=status) - log.msg("Domain closed (%s)" % status, domain=domain) - self.cancelled.discard(domain) + reason = self.closing.get(domain, 'finished') + signals.send_catch_log(signal=signals.domain_closed, sender=self.__class__, domain=domain, spider=spider, reason=reason) + log.msg("Domain closed (%s)" % reason, domain=domain) + self.closing.pop(domain, None) self._mainloop() def getstatus(self): @@ -426,7 +426,7 @@ class ExecutionEngine(object): ] domain_tests = [ "self.domain_is_idle(domain)", - "domain in self.cancelled", + "self.closing.get(domain)", "self.scheduler.domain_has_pending_requests(domain)", "len(self.scheduler.pending_requests[domain])", "len(self.downloader.sites[domain].queue)", diff --git a/scrapy/core/signals.py b/scrapy/core/signals.py index b8178853d..341c07fa3 100644 --- a/scrapy/core/signals.py +++ b/scrapy/core/signals.py @@ -30,8 +30,7 @@ domain_opened = object() domain_idle = object() # After a domain has been closed -# args: domain, spider, status -# status is a string and its possible values are: "finished" or "cancelled" +# args: domain, spider, reason domain_closed = object() # New request received from spiders diff --git a/scrapy/stats/corestats.py b/scrapy/stats/corestats.py index 079d8dab8..88b0ef2e8 100644 --- a/scrapy/stats/corestats.py +++ b/scrapy/stats/corestats.py @@ -32,10 +32,10 @@ class CoreStats(object): stats.setpath('%s/envinfo' % domain, stats.getpath('_envinfo')) stats.incpath('_global/domain_count/opened') - def stats_domain_closing(self, domain, spider, status): + def stats_domain_closing(self, domain, spider, reason): stats.setpath('%s/finish_time' % domain, datetime.datetime.now()) - stats.setpath('%s/finish_status' % domain, 'OK' if status == 'finished' else status) - stats.incpath('_global/domain_count/%s' % status) + stats.setpath('%s/finish_status' % domain, 'OK' if reason == 'finished' else reason) + stats.incpath('_global/domain_count/%s' % reason) def item_scraped(self, item, spider): stats.incpath('%s/item_scraped_count' % spider.domain_name) diff --git a/scrapy/stats/statscollector.py b/scrapy/stats/statscollector.py index 2ff1f3e37..e9f21483b 100644 --- a/scrapy/stats/statscollector.py +++ b/scrapy/stats/statscollector.py @@ -66,12 +66,12 @@ class StatsCollector(dict): def _domain_open(self, domain, spider): dispatcher.send(signal=self.domain_open, sender=self.__class__, domain=domain, spider=spider) - def _domain_closed(self, domain, spider, status): - dispatcher.send(signal=self.domain_closing, sender=self.__class__, domain=domain, spider=spider, status=status) + def _domain_closed(self, domain, spider, reason): + dispatcher.send(signal=self.domain_closing, sender=self.__class__, domain=domain, spider=spider, reason=reason) if self.debug: log.msg(pprint.pformat(self[domain]), domain=domain, level=log.DEBUG) if self.db: self.db.put(domain, self[domain]) if self.cleanup: del self[domain] - dispatcher.send(signal=self.domain_closed, sender=self.__class__, domain=domain, spider=spider, status=status) + dispatcher.send(signal=self.domain_closed, sender=self.__class__, domain=domain, spider=spider, reason=reason) diff --git a/scrapy/tests/test_engine.py b/scrapy/tests/test_engine.py index dc9374c09..e28e7143a 100644 --- a/scrapy/tests/test_engine.py +++ b/scrapy/tests/test_engine.py @@ -186,7 +186,7 @@ class EngineTest(unittest.TestCase): session.signals_catched[signals.domain_opened]) self.assertEqual({'domain': session.domain, 'spider': session.spider}, session.signals_catched[signals.domain_idle]) - self.assertEqual({'domain': session.domain, 'spider': session.spider, 'status': 'finished'}, + self.assertEqual({'domain': session.domain, 'spider': session.spider, 'reason': 'finished'}, session.signals_catched[signals.domain_closed]) if __name__ == "__main__":