add an undocumented setting for lookup body size

This commit is contained in:
Mikhail Korobov 2013-12-19 00:23:38 +06:00
parent a87b3bd1c8
commit 71c59e1c94
1 changed files with 7 additions and 6 deletions

View File

@ -13,17 +13,18 @@ class AjaxCrawlableMiddleware(object):
For more info see https://developers.google.com/webmasters/ajax-crawling/docs/getting-started.
"""
# XXX: Google parses at least first 100k bytes; scrapy's redirect
# middleware parses first 4k. 4k turns out to be insufficient
# for this middleware, and parsing 100k could be slow.
_lookup_bytes = 32768
enabled_setting = 'AJAXCRAWLABLE_ENABLED'
def __init__(self, settings):
if not settings.getbool(self.enabled_setting):
raise NotConfigured
# XXX: Google parses at least first 100k bytes; scrapy's redirect
# middleware parses first 4k. 4k turns out to be insufficient
# for this middleware, and parsing 100k could be slow.
# We use something in between (32K) by default.
self.lookup_bytes = settings.getint('AJAXCRAWLABLE_MAXSIZE', 32768)
@classmethod
def from_crawler(cls, crawler):
return cls(crawler.settings)
@ -57,7 +58,7 @@ class AjaxCrawlableMiddleware(object):
Return True if a page without hash fragment could be "AJAX crawlable"
according to https://developers.google.com/webmasters/ajax-crawling/docs/getting-started.
"""
body = response.body_as_unicode()[:self._lookup_bytes]
body = response.body_as_unicode()[:self.lookup_bytes]
return _has_ajaxcrawlable_meta(body)