From 8b37282752a27104da23dfaa931046dc2b16d6e6 Mon Sep 17 00:00:00 2001 From: elpolilla Date: Mon, 1 Dec 2008 02:59:34 +0000 Subject: [PATCH] Changed CrawlSpider's way of setting crawling rules --HG-- rename : scrapy/trunk/scrapy/contrib/spiders2/crawl.py => scrapy/trunk/scrapy/contrib/spiders/crawl.py extra : convert_revision : svn%3Ab85faa78-f9eb-468e-a121-7cced6da292c%40450 --- .../templates/spider_crawl.tmpl | 21 ++- .../scrapy/contrib/adaptors/extraction.py | 47 +++--- scrapy/trunk/scrapy/contrib/spiders.py | 159 ------------------ .../trunk/scrapy/contrib/spiders/__init__.py | 2 + .../contrib/{spiders2 => spiders}/crawl.py | 0 scrapy/trunk/scrapy/contrib/spiders/feed.py | 87 ++++++++++ .../trunk/scrapy/contrib/spiders2/__init__.py | 2 - scrapy/trunk/scrapy/contrib/spiders2/feed.py | 79 --------- 8 files changed, 127 insertions(+), 270 deletions(-) delete mode 100644 scrapy/trunk/scrapy/contrib/spiders.py create mode 100644 scrapy/trunk/scrapy/contrib/spiders/__init__.py rename scrapy/trunk/scrapy/contrib/{spiders2 => spiders}/crawl.py (100%) create mode 100644 scrapy/trunk/scrapy/contrib/spiders/feed.py delete mode 100644 scrapy/trunk/scrapy/contrib/spiders2/__init__.py delete mode 100644 scrapy/trunk/scrapy/contrib/spiders2/feed.py diff --git a/scrapy/trunk/scrapy/conf/project_template/templates/spider_crawl.tmpl b/scrapy/trunk/scrapy/conf/project_template/templates/spider_crawl.tmpl index 44db54daf..95b63d98e 100644 --- a/scrapy/trunk/scrapy/conf/project_template/templates/spider_crawl.tmpl +++ b/scrapy/trunk/scrapy/conf/project_template/templates/spider_crawl.tmpl @@ -2,24 +2,23 @@ import re from scrapy.xpath import HtmlXPathSelector +from scrapy.item import ScrapedItem from scrapy.link.extractors import RegexLinkExtractor -from scrapy.contrib.spiders import CrawlSpider -from scrapy.utils.url import url_query_parameter +from scrapy.contrib.spiders import CrawlSpider, Rule class $classname(CrawlSpider): domain_name = "$site" start_urls = ['http://www.$site/'] - links_category = RegexLinkExtractor(allow=(re.compile(r'ProductList'), )) - links_product = RegexLinkExtractor(allow=(re.compile(r'ProductDetail'), )) + rules = ( + Rule(RegexLinkExtractor(allow=(r'Items/', ), 'parse_items', follow=True) + ) - def parse_product(self, response): + def parse_item(self, response): #xs = HtmlXPathSelector(response) - #p = self.create_product(response) - #p.attribute('site_id', url_query_parameter(response.url,'productid')) - #p.attribute('site_id', xs.x("//input[@name='productid']/@value")) - #p.attribute('site_id', xs.x("")) - #p.attribute('name', xs.x("")) - #p.attribute('description', xs.x("")) + #i = ScrapedItem() + #i.attribute('site_id', xs.x("//input[@id="sid"]/@value")) + #i.attribute('name', xs.x("//div[@id='name']")) + #i.attribute('description', xs.x("//div[@id='description']")) SPIDER = $classname() diff --git a/scrapy/trunk/scrapy/contrib/adaptors/extraction.py b/scrapy/trunk/scrapy/contrib/adaptors/extraction.py index d2afbf1d2..b626234f2 100644 --- a/scrapy/trunk/scrapy/contrib/adaptors/extraction.py +++ b/scrapy/trunk/scrapy/contrib/adaptors/extraction.py @@ -4,7 +4,9 @@ Adaptors related with extraction of data import urlparse import re +from scrapy import log from scrapy.http import Response +from scrapy.utils.url import is_url from scrapy.utils.python import flatten from scrapy.xpath.selector import XPathSelector, XPathSelectorList @@ -64,26 +66,29 @@ class ExtractImages(object): self.base_url = base_url if not self.base_url: - raise AttributeError('You must specify either a response or a base_url to the ExtractImages adaptor.') + log.msg('No base URL was found for ExtractImages adaptor, will only extract absolute URLs', log.WARNING) def extract_from_xpath(self, selector): ret = [] if selector.xmlNode.type == 'element': - if selector.xmlNode.name == 'a': - children = selector.x('child::*') - if len(children) > 1: + if selector.xmlNode.name == 'a': + children = selector.x('child::*') + if len(children) > 1: + ret.extend(selector.x('.//@href')) + ret.extend(selector.x('.//@src')) + elif len(children) == 1 and children[0].xmlNode.name == 'img': + ret.extend(children.x('@src')) + else: + ret.extend(selector.x('@href')) + elif selector.xmlNode.name == 'img': + ret.extend(selector.x('@src')) + elif selector.xmlNode.name == 'text': + ret.extend(selector) + else: ret.extend(selector.x('.//@href')) ret.extend(selector.x('.//@src')) - elif len(children) == 1 and children[0].xmlNode.name == 'img': - ret.extend(children.x('@src')) - else: - ret.extend(selector.x('@href')) - elif selector.xmlNode.name == 'img': - ret.extend(selector.x('@src')) - else: - ret.extend(selector.x('.//@href')) - ret.extend(selector.x('.//@src')) - elif selector.xmlNode.type == 'attribute' and selector.xmlNode.name in ['href', 'src']: + ret.extend(selector.x('.//text()')) + else: ret.append(selector) return ret @@ -92,11 +97,15 @@ class ExtractImages(object): if isinstance(locations, basestring): locations = [locations] - rel_links = [] + rel_urls = [] for location in flatten(locations): if isinstance(location, (XPathSelector, XPathSelectorList)): - rel_links.extend(self.extract_from_xpath(location)) + rel_urls.extend(self.extract_from_xpath(location)) else: - rel_links.append(location) - rel_links = extract(rel_links) - return [urlparse.urljoin(self.base_url, link) for link in rel_links] + rel_urls.append(location) + rel_urls = extract(rel_urls) + + if self.base_url: + return [urlparse.urljoin(self.base_url, url) for url in rel_urls] + else: + return filter(rel_urls, is_url) diff --git a/scrapy/trunk/scrapy/contrib/spiders.py b/scrapy/trunk/scrapy/contrib/spiders.py deleted file mode 100644 index 722306db7..000000000 --- a/scrapy/trunk/scrapy/contrib/spiders.py +++ /dev/null @@ -1,159 +0,0 @@ -""" -This module contains some basic spiders for scraping websites (CrawlSpider) -and XML feeds (XMLFeedSpider). -""" - -from scrapy.conf import settings -from scrapy.http import Request -from scrapy.spider import BaseSpider -from scrapy.item import ScrapedItem -from scrapy.xpath.selector import XmlXPathSelector -from scrapy.core.exceptions import NotConfigured -from scrapy.utils.iterators import xmliter, csviter - - -def _set_guid(spider, item): - """ - This method is called whenever the spider returns items, for each item. - It should set the 'guid' attribute to the given item with a string that - identifies the item uniquely. - """ - raise NotConfigured('You must define a set_guid method in order to scrape items.') - -class CrawlSpider(BaseSpider): - """ - This class works as a base class for spiders that crawl over websites - """ - set_guid = _set_guid - - def __init__(self): - super(CrawlSpider, self).__init__() - - self._links_callback = [] - for attr in dir(self): - if attr.startswith('links_'): - suffix = attr.split('_', 1)[1] - value = getattr(self, attr) - callback = getattr(self, 'parse_%s' % suffix, None) - self._links_callback.append((value, callback)) - - def parse(self, response): - """This function is called by the core for all the start_urls. Do not - override this function, override parse_start_url instead.""" - if response.url in self.start_urls: - return self._parse_wrapper(response, self.parse_start_url) - else: - return self.parse_url(response) - - def parse_start_url(self, response): - """Callback function for processing start_urls. It must return a list - of ScrapedItems and/or Requests.""" - return [] - - def _links_to_follow(self, response): - res = [] - links_to_follow = {} - for lx, callback in self._links_callback: - links = lx.extract_urls(response) - links = self.post_extract_links(links) if hasattr(self, 'post_extract_links') else links - for link in links: - links_to_follow[link.url] = (callback, link.text) - - for url, (callback, link_text) in links_to_follow.iteritems(): - request = Request(url=url, link_text=link_text) - request.append_callback(self._parse_wrapper, callback) - res.append(request) - return res - - def _parse_wrapper(self, response, callback): - res = [] - if settings.getbool('CRAWLSPIDER_FOLLOW_LINKS', True): - res.extend(self._links_to_follow(response)) - res.extend(callback(response) if callback else ()) - for entry in res: - if isinstance(entry, ScrapedItem): - self.set_guid(entry) - return res - - def parse_url(self, response): - """ - This method is called whenever you run scrapy with the 'parse' command - over an URL. - """ - extractor_names = [attrname for attrname in dir(self) if attrname.startswith('links_')] - ret = [] - for name in extractor_names: - extractor = getattr(self, name) - if settings.getbool('CRAWLSPIDER_FOLLOW_LINKS', True): - ret.extend(self._links_to_follow(response)) - callback_name = 'parse_%s' % name[6:] - if hasattr(self, callback_name): - if extractor.match(response.url): - ret.extend(getattr(self, callback_name)(response)) - for entry in ret: - if isinstance(entry, ScrapedItem): - self.set_guid(entry) - return ret - -class XMLFeedSpider(BaseSpider): - """ - This class intends to be the base class for spiders that scrape - from XML feeds. - - You can choose whether to parse the file using the iternodes tool, - or not using it (which just splits the tags using xpath) - """ - set_guid = _set_guid - iternodes = True - itertag = 'item' - - def parse_item_wrapper(self, response, xSel): - ret = self.parse_item(response, xSel) - if isinstance(ret, ScrapedItem): - self.set_guid(ret) - return ret - - def parse(self, response): - if not hasattr(self, 'parse_item'): - raise NotConfigured('You must define parse_item method in order to scrape this XML feed') - - if self.iternodes: - nodes = xmliter(response, self.itertag) - else: - nodes = XmlXPathSelector(response).x('//%s' % self.itertag) - - return (self.parse_item_wrapper(response, xSel) for xSel in nodes) - -class CSVFeedSpider(BaseSpider): - """ - Spider for parsing CSV feeds. - It receives a CSV file in a response; iterates through each of its rows, - and calls parse_row with a dict containing each field's data. - - You can set some options regarding the CSV file, such as the delimiter - and the file's headers. - """ - set_guid = _set_guid - delimiter = None # When this is None, python's csv module's default delimiter is used - headers = None - - def adapt_feed(self, response): - """You can override this function in order to make any changes you want - to into the feed before parsing it. This function may return either a - response or a string. """ - - return response - - def parse_row_wrapper(self, response, row): - ret = self.parse_row(response, row) - if isinstance(ret, ScrapedItem): - self.set_guid(ret) - return ret - - def parse(self, response): - if not hasattr(self, 'parse_row'): - raise NotConfigured('You must define parse_row method in order to scrape this CSV feed') - - feed = self.adapt_feed(response) - return (self.parse_row_wrapper(feed, row) for row in csviter(response, self.delimiter, self.headers)) - diff --git a/scrapy/trunk/scrapy/contrib/spiders/__init__.py b/scrapy/trunk/scrapy/contrib/spiders/__init__.py new file mode 100644 index 000000000..6a310dd44 --- /dev/null +++ b/scrapy/trunk/scrapy/contrib/spiders/__init__.py @@ -0,0 +1,2 @@ +from scrapy.contrib.spiders.crawl import CrawlSpider, Rule +from scrapy.contrib.spiders.feed import XMLFeedSpider, CSVFeedSpider diff --git a/scrapy/trunk/scrapy/contrib/spiders2/crawl.py b/scrapy/trunk/scrapy/contrib/spiders/crawl.py similarity index 100% rename from scrapy/trunk/scrapy/contrib/spiders2/crawl.py rename to scrapy/trunk/scrapy/contrib/spiders/crawl.py diff --git a/scrapy/trunk/scrapy/contrib/spiders/feed.py b/scrapy/trunk/scrapy/contrib/spiders/feed.py new file mode 100644 index 000000000..d9ea1ccc1 --- /dev/null +++ b/scrapy/trunk/scrapy/contrib/spiders/feed.py @@ -0,0 +1,87 @@ +# -*- coding: utf8 -*- +from scrapy.spider import BaseSpider +from scrapy.item import ScrapedItem +from scrapy.http import Request +from scrapy.utils.iterators import xmliter, csviter +from scrapy.xpath.selector import XmlXPathSelector +from scrapy.core.exceptions import UsageError, NotConfigured + +class XMLFeedSpider(BaseSpider): + """ + This class intends to be the base class for spiders that scrape + from XML feeds. + + You can choose whether to parse the file using the iternodes tool, + or not using it (which just splits the tags using xpath) + """ + iternodes = True + itertag = 'item' + + def process_results(self, results, response): + """This overridable method is called for each result (item or request) + returned by the spider, and it's intended to perform any last time + processing required before returning the results to the framework core, + for example setting the item GUIDs. It receives a list of results and + the response which originated that results. It must return a list + of results (Items or Requests).""" + return results + + def parse_nodes(self, response, nodes): + for xSel in nodes: + ret = self.parse_item(response, xSel) + if isinstance(ret, (ScrapedItem, Request)): + ret = [ret] + if not isinstance(ret, (list, tuple)): + raise UsageError('You cannot return an "%s" object from a spider' % type(ret).__name__) + for result_item in self.process_results(ret, response): + yield result_item + + def parse(self, response): + if not hasattr(self, 'parse_item'): + raise NotConfigured('You must define parse_item method in order to scrape this XML feed') + + if self.iternodes: + nodes = xmliter(response, self.itertag) + else: + nodes = XmlXPathSelector(response).x('//%s' % self.itertag) + + return self.parse_nodes(response, nodes) + +class CSVFeedSpider(BaseSpider): + """ + Spider for parsing CSV feeds. + It receives a CSV file in a response; iterates through each of its rows, + and calls parse_row with a dict containing each field's data. + + You can set some options regarding the CSV file, such as the delimiter + and the file's headers. + """ + delimiter = None # When this is None, python's csv module's default delimiter is used + headers = None + + def process_results(self, results, response): + """This method has the same purpose as the one in XMLFeedSpider""" + return results + + def adapt_response(self, response): + """You can override this function in order to make any changes you want + to into the feed before parsing it. This function must return a response.""" + return response + + def parse_rows(self, response): + for row in csviter(response, self.delimiter, self.headers): + ret = self.parse_row(response, row) + if isinstance(ret, (ScrapedItem, Request)): + ret = [ret] + if not isinstance(ret, (list, tuple)): + raise UsageError('You cannot return an "%s" object from a spider' % type(ret).__name__) + for result_item in self.process_results(ret, response): + yield result_item + + def parse(self, response): + if not hasattr(self, 'parse_row'): + raise NotConfigured('You must define parse_row method in order to scrape this CSV feed') + + response = self.adapt_response(response) + return self.parse_rows(response) + diff --git a/scrapy/trunk/scrapy/contrib/spiders2/__init__.py b/scrapy/trunk/scrapy/contrib/spiders2/__init__.py deleted file mode 100644 index 1b7107f45..000000000 --- a/scrapy/trunk/scrapy/contrib/spiders2/__init__.py +++ /dev/null @@ -1,2 +0,0 @@ -from scrapy.contrib.spiders2.crawl import CrawlSpider, Rule -from scrapy.contrib.spiders2.feed import XMLFeedSpider, CSVFeedSpider diff --git a/scrapy/trunk/scrapy/contrib/spiders2/feed.py b/scrapy/trunk/scrapy/contrib/spiders2/feed.py deleted file mode 100644 index 9ebefb2e0..000000000 --- a/scrapy/trunk/scrapy/contrib/spiders2/feed.py +++ /dev/null @@ -1,79 +0,0 @@ -from scrapy.spider import BaseSpider -from scrapy.item import ScrapedItem -from scrapy.utils.iterators import xmliter, csviter -from scrapy.xpath.selector import XmlXPathSelector -from scrapy.core.exceptions import NotConfigured - -class XMLFeedSpider(BaseSpider): - """ - This class intends to be the base class for spiders that scrape - from XML feeds. - - You can choose whether to parse the file using the iternodes tool, - or not using it (which just splits the tags using xpath) - """ - iternodes = True - itertag = 'item' - - def item_scraped(self, response, item): - """ - This method is called for each item returned by the spider, and it's intended - to do anything that it's needed before returning the item to the core, specially - setting its GUID. - It receives and returns an item - """ - return item - - def parse_item_wrapper(self, response, xSel): - ret = self.parse_item(response, xSel) - if isinstance(ret, ScrapedItem): - self.scraped_item(response, ret) - return ret - - def parse(self, response): - if not hasattr(self, 'parse_item'): - raise NotConfigured('You must define parse_item method in order to scrape this XML feed') - - if self.iternodes: - nodes = xmliter(response, self.itertag) - else: - nodes = XmlXPathSelector(response).x('//%s' % self.itertag) - - return (self.parse_item_wrapper(response, xSel) for xSel in nodes) - -class CSVFeedSpider(BaseSpider): - """ - Spider for parsing CSV feeds. - It receives a CSV file in a response; iterates through each of its rows, - and calls parse_row with a dict containing each field's data. - - You can set some options regarding the CSV file, such as the delimiter - and the file's headers. - """ - delimiter = None # When this is None, python's csv module's default delimiter is used - headers = None - - def scraped_item(self, response, item): - """This method has the same purpose as the one in XMLFeedSpider""" - return item - - def adapt_feed(self, response): - """You can override this function in order to make any changes you want - to into the feed before parsing it. This function may return either a - response or a string. """ - - return response - - def parse_row_wrapper(self, response, row): - ret = self.parse_row(response, row) - if isinstance(ret, ScrapedItem): - self.scraped_item(response, ret) - return ret - - def parse(self, response): - if not hasattr(self, 'parse_row'): - raise NotConfigured('You must define parse_row method in order to scrape this CSV feed') - - feed = self.adapt_feed(response) - return (self.parse_row_wrapper(feed, row) for row in csviter(response, self.delimiter, self.headers)) -