diff --git a/scrapy/trunk/scrapy/contrib/adaptors/extraction.py b/scrapy/trunk/scrapy/contrib/adaptors/extraction.py index 0da6f24d7..70c4c109c 100644 --- a/scrapy/trunk/scrapy/contrib/adaptors/extraction.py +++ b/scrapy/trunk/scrapy/contrib/adaptors/extraction.py @@ -43,9 +43,9 @@ def extract_unquoted(locations): class ExtractImages(object): """ - This adaptor receives either an XPathSelector containing - the desired locations for finding urls, or a list of relative - links to be resolved. + This adaptor may receive either an XPathSelector containing + the desired locations for finding urls, a list of relative + links to be resolved, or simply a link (relative or not). Input: XPathSelector, XPathSelectorList, iterable Output: list of unicodes @@ -79,6 +79,9 @@ class ExtractImages(object): if not self.base_url: raise AttributeError('You must specify either a response or a base_url to the ExtractImages adaptor.') + if isinstance(locations, basestring): + locations = [locations] + rel_links = [] for location in flatten(locations): if isinstance(location, (XPathSelector, XPathSelectorList)): diff --git a/scrapy/trunk/scrapy/contrib/spiders.py b/scrapy/trunk/scrapy/contrib/spiders.py index 8a8404e81..266c671b2 100644 --- a/scrapy/trunk/scrapy/contrib/spiders.py +++ b/scrapy/trunk/scrapy/contrib/spiders.py @@ -11,16 +11,7 @@ from scrapy.core.exceptions import UsageError from scrapy.utils.iterators import xmliter, csviter from scrapy.utils.misc import hash_values -class BasicSpider(BaseSpider): - """ - This class is basically a BaseSpider with support for GUID generating - """ - gen_guid_attribs = [] - - def set_guid(self, item): - item.guid = hash_values(self.domain_name, *[str(getattr(item, aname) or '') for aname in self.gen_guid_attribs]) - -class CrawlSpider(BasicSpider): +class CrawlSpider(BaseSpider): """ This class works as a base class for spiders that crawl over websites """ @@ -86,7 +77,15 @@ class CrawlSpider(BasicSpider): self.set_guid(entry) return ret -class XMLFeedSpider(BasicSpider): + def set_guid(self, item): + """ + This method is called whenever the spider returns items, for each item. + It should set the 'guid' attribute to the given item with a string that + identifies the item uniquely. + """ + raise NotConfigured('You must define set_guid method in order to scrape items.') + +class XMLFeedSpider(BaseSpider): """ This class intends to be the base class for spiders that scrape from XML feeds.