Added --nofollow option switch to crawl command, and its functionality into CrawlSpiders

--HG--
extra : convert_revision : svn%3Ab85faa78-f9eb-468e-a121-7cced6da292c%40388
This commit is contained in:
elpolilla 2008-11-20 12:10:36 +00:00
parent 90fe6ef1b8
commit 99c4c06380
2 changed files with 16 additions and 4 deletions

View File

@ -18,6 +18,7 @@ class Command(ScrapyCommand):
parser.add_option("--restrict", dest="restrict", action="store_true", help="restrict crawling only to the given urls")
parser.add_option("--record", dest="record", help="use FILE for recording session (see replay command)", metavar="FILE")
parser.add_option("--record-dir", dest="recorddir", help="use DIR for recording (instead of file)", metavar="DIR")
parser.add_option("-n", "--nofollow", dest="nofollow", action="store_true", help="don't follow links (for use with URLs only)")
def process_options(self, args, opts):
ScrapyCommand.process_options(self, args, opts)
@ -30,6 +31,9 @@ class Command(ScrapyCommand):
if opts.restrict:
settings.overrides['RESTRICT_TO_URLS'] = args
if opts.nofollow:
settings.overrides['FOLLOW_LINKS'] = False
if opts.record or opts.recorddir:
# self.replay is used for preventing Replay signals handler from
# disconnecting since pydispatcher uses weak references

View File

@ -3,6 +3,7 @@ This module contains some basic spiders for scraping websites (CrawlSpider)
and XML feeds (XMLFeedSpider).
"""
from scrapy.conf import settings
from scrapy.http import Request, Response, ResponseBody
from scrapy.spider import BaseSpider
from scrapy.item import ScrapedItem
@ -15,6 +16,7 @@ class CrawlSpider(BaseSpider):
"""
This class works as a base class for spiders that crawl over websites
"""
def __init__(self):
super(BaseSpider, self).__init__()
@ -29,7 +31,10 @@ class CrawlSpider(BaseSpider):
def parse(self, response):
"""This function is called by the core for all the start_urls. Do not
override this function, override parse_start_url instead."""
return self._parse_wrapper(response, self.parse_start_url)
if response.url in self.start_urls:
return self._parse_wrapper(response, self.parse_start_url)
else:
return self.parse_url(response)
def parse_start_url(self, response):
"""Callback function for processing start_urls. It must return a list
@ -52,8 +57,10 @@ class CrawlSpider(BaseSpider):
return res
def _parse_wrapper(self, response, callback):
res = self._links_to_follow(response)
res += callback(response) if callback else ()
res = []
if settings.getbool('FOLLOW_LINKS', True):
res.extend(self._links_to_follow(response))
res.extend(callback(response) if callback else ())
for entry in res:
if isinstance(entry, ScrapedItem):
self.set_guid(entry)
@ -68,7 +75,8 @@ class CrawlSpider(BaseSpider):
ret = []
for name in extractor_names:
extractor = getattr(self, name)
ret.extend(self._links_to_follow(response))
if settings.getbool('FOLLOW_LINKS', True):
ret.extend(self._links_to_follow(response))
callback_name = 'parse_%s' % name[6:]
if hasattr(self, callback_name):
if extractor.match(response.url):