improved RequestLimitMiddleware to conform to better programming standards

--HG--
extra : convert_revision : svn%3Ab85faa78-f9eb-468e-a121-7cced6da292c%40343
This commit is contained in:
Pablo Hoffman 2008-10-27 01:24:19 +00:00
parent 23524ccc86
commit 968a55eaf3
1 changed files with 29 additions and 24 deletions

View File

@ -1,37 +1,42 @@
"""
RequestLimitMiddleware: Limits the scheduler request queue from the point of
view of the spider. That is, if the scheduler queue contains an equal or
greater ammount of requests than the specified limit, the new requests
(generated by the spider) will be ignored.
RequestLimitMiddleware: Limits the scheduler request queue size. When spiders
try to schedule more than the allowed amount of requests the new requests
(returned by the spider) will be dropped.
The limit is setted from the spider attribute "requests_queue_size". If not
found, from the scrapy setting "REQUESTS_QUEUE_SIZE". If not found, no limit
will be applied. If given a value of 0, no limit will be applied.
The limit can be set using the spider attribue `requests_queue_size` or the
setting "REQUESTS_QUEUE_SIZE". If not specified (or 0), no limit will be
applied.
"""
from scrapy.core.engine import scrapyengine
from scrapy.core.exceptions import NotConfigured
from scrapy.conf import settings
from scrapy.http import Request
from scrapy import log
class RequestLimitMiddleware(object):
#_last_queue_size = 0
def __init__(self):
self.max_queue_size = settings.getint("REQUESTS_QUEUE_SIZE")
if not self.max_queue_size:
raise NotConfigured
def process_result(self, response, result, spider):
requests = []
other = []
[requests.append(r) if isinstance(r, Request) else other.append(r) for r in result]
items = []
for r in result:
if isinstance(r, Request):
requests.append(r)
else:
items.append(r)
max_pending = spider.requests_queue_size if hasattr(spider,"requests_queue_size") else settings.getint("REQUESTS_QUEUE_SIZE")
if not max_pending:
accepted = requests
else:
free_slots = max_pending - len(scrapyengine.scheduler.pending_requests[spider.domain_name])
accepted = requests[:free_slots]
dropped = set(requests) - set(accepted)
if dropped:
for r in dropped:
log.msg("Ignoring link (max schedule queue size reached): %s " % r.url, level=log.WARNING, domain=spider.domain_name)
#actual_size = len(scrapyengine.scheduler.pending_requests[spider.domain_name])
#log.msg("queue size: %d (%+d)" % (actual_size, actual_size - self._last_queue_size) )
#self._last_queue_size = actual_size
return accepted + other
max_pending = getattr(spider, 'requests_queue_size', self.max_queue_size)
if max_pending:
pending_count = len(scrapyengine.scheduler.pending_requests.get(spider.domain_name, []))
free_slots = max_pending - pending_count
dropped_count = len(requests) - free_slots
if dropped_count > 0:
requests = requests[:free_slots]
log.msg("Dropping %d request(s) because the maximum schedule size (%d) has been exceeded" % \
(dropped_count, max_pending), level=log.WARNING, domain=spider.domain_name)
return requests + items