Changed CrawlSpider's way of setting crawling rules

--HG--
rename : scrapy/trunk/scrapy/contrib/spiders2/crawl.py => scrapy/trunk/scrapy/contrib/spiders/crawl.py
extra : convert_revision : svn%3Ab85faa78-f9eb-468e-a121-7cced6da292c%40450
This commit is contained in:
elpolilla 2008-12-01 02:59:34 +00:00
parent 7f05b923ab
commit 8b37282752
8 changed files with 127 additions and 270 deletions

View File

@ -2,24 +2,23 @@
import re
from scrapy.xpath import HtmlXPathSelector
from scrapy.item import ScrapedItem
from scrapy.link.extractors import RegexLinkExtractor
from scrapy.contrib.spiders import CrawlSpider
from scrapy.utils.url import url_query_parameter
from scrapy.contrib.spiders import CrawlSpider, Rule
class $classname(CrawlSpider):
domain_name = "$site"
start_urls = ['http://www.$site/']
links_category = RegexLinkExtractor(allow=(re.compile(r'ProductList'), ))
links_product = RegexLinkExtractor(allow=(re.compile(r'ProductDetail'), ))
rules = (
Rule(RegexLinkExtractor(allow=(r'Items/', ), 'parse_items', follow=True)
)
def parse_product(self, response):
def parse_item(self, response):
#xs = HtmlXPathSelector(response)
#p = self.create_product(response)
#p.attribute('site_id', url_query_parameter(response.url,'productid'))
#p.attribute('site_id', xs.x("//input[@name='productid']/@value"))
#p.attribute('site_id', xs.x(""))
#p.attribute('name', xs.x(""))
#p.attribute('description', xs.x(""))
#i = ScrapedItem()
#i.attribute('site_id', xs.x("//input[@id="sid"]/@value"))
#i.attribute('name', xs.x("//div[@id='name']"))
#i.attribute('description', xs.x("//div[@id='description']"))
SPIDER = $classname()

View File

@ -4,7 +4,9 @@ Adaptors related with extraction of data
import urlparse
import re
from scrapy import log
from scrapy.http import Response
from scrapy.utils.url import is_url
from scrapy.utils.python import flatten
from scrapy.xpath.selector import XPathSelector, XPathSelectorList
@ -64,26 +66,29 @@ class ExtractImages(object):
self.base_url = base_url
if not self.base_url:
raise AttributeError('You must specify either a response or a base_url to the ExtractImages adaptor.')
log.msg('No base URL was found for ExtractImages adaptor, will only extract absolute URLs', log.WARNING)
def extract_from_xpath(self, selector):
ret = []
if selector.xmlNode.type == 'element':
if selector.xmlNode.name == 'a':
children = selector.x('child::*')
if len(children) > 1:
if selector.xmlNode.name == 'a':
children = selector.x('child::*')
if len(children) > 1:
ret.extend(selector.x('.//@href'))
ret.extend(selector.x('.//@src'))
elif len(children) == 1 and children[0].xmlNode.name == 'img':
ret.extend(children.x('@src'))
else:
ret.extend(selector.x('@href'))
elif selector.xmlNode.name == 'img':
ret.extend(selector.x('@src'))
elif selector.xmlNode.name == 'text':
ret.extend(selector)
else:
ret.extend(selector.x('.//@href'))
ret.extend(selector.x('.//@src'))
elif len(children) == 1 and children[0].xmlNode.name == 'img':
ret.extend(children.x('@src'))
else:
ret.extend(selector.x('@href'))
elif selector.xmlNode.name == 'img':
ret.extend(selector.x('@src'))
else:
ret.extend(selector.x('.//@href'))
ret.extend(selector.x('.//@src'))
elif selector.xmlNode.type == 'attribute' and selector.xmlNode.name in ['href', 'src']:
ret.extend(selector.x('.//text()'))
else:
ret.append(selector)
return ret
@ -92,11 +97,15 @@ class ExtractImages(object):
if isinstance(locations, basestring):
locations = [locations]
rel_links = []
rel_urls = []
for location in flatten(locations):
if isinstance(location, (XPathSelector, XPathSelectorList)):
rel_links.extend(self.extract_from_xpath(location))
rel_urls.extend(self.extract_from_xpath(location))
else:
rel_links.append(location)
rel_links = extract(rel_links)
return [urlparse.urljoin(self.base_url, link) for link in rel_links]
rel_urls.append(location)
rel_urls = extract(rel_urls)
if self.base_url:
return [urlparse.urljoin(self.base_url, url) for url in rel_urls]
else:
return filter(rel_urls, is_url)

View File

@ -1,159 +0,0 @@
"""
This module contains some basic spiders for scraping websites (CrawlSpider)
and XML feeds (XMLFeedSpider).
"""
from scrapy.conf import settings
from scrapy.http import Request
from scrapy.spider import BaseSpider
from scrapy.item import ScrapedItem
from scrapy.xpath.selector import XmlXPathSelector
from scrapy.core.exceptions import NotConfigured
from scrapy.utils.iterators import xmliter, csviter
def _set_guid(spider, item):
"""
This method is called whenever the spider returns items, for each item.
It should set the 'guid' attribute to the given item with a string that
identifies the item uniquely.
"""
raise NotConfigured('You must define a set_guid method in order to scrape items.')
class CrawlSpider(BaseSpider):
"""
This class works as a base class for spiders that crawl over websites
"""
set_guid = _set_guid
def __init__(self):
super(CrawlSpider, self).__init__()
self._links_callback = []
for attr in dir(self):
if attr.startswith('links_'):
suffix = attr.split('_', 1)[1]
value = getattr(self, attr)
callback = getattr(self, 'parse_%s' % suffix, None)
self._links_callback.append((value, callback))
def parse(self, response):
"""This function is called by the core for all the start_urls. Do not
override this function, override parse_start_url instead."""
if response.url in self.start_urls:
return self._parse_wrapper(response, self.parse_start_url)
else:
return self.parse_url(response)
def parse_start_url(self, response):
"""Callback function for processing start_urls. It must return a list
of ScrapedItems and/or Requests."""
return []
def _links_to_follow(self, response):
res = []
links_to_follow = {}
for lx, callback in self._links_callback:
links = lx.extract_urls(response)
links = self.post_extract_links(links) if hasattr(self, 'post_extract_links') else links
for link in links:
links_to_follow[link.url] = (callback, link.text)
for url, (callback, link_text) in links_to_follow.iteritems():
request = Request(url=url, link_text=link_text)
request.append_callback(self._parse_wrapper, callback)
res.append(request)
return res
def _parse_wrapper(self, response, callback):
res = []
if settings.getbool('CRAWLSPIDER_FOLLOW_LINKS', True):
res.extend(self._links_to_follow(response))
res.extend(callback(response) if callback else ())
for entry in res:
if isinstance(entry, ScrapedItem):
self.set_guid(entry)
return res
def parse_url(self, response):
"""
This method is called whenever you run scrapy with the 'parse' command
over an URL.
"""
extractor_names = [attrname for attrname in dir(self) if attrname.startswith('links_')]
ret = []
for name in extractor_names:
extractor = getattr(self, name)
if settings.getbool('CRAWLSPIDER_FOLLOW_LINKS', True):
ret.extend(self._links_to_follow(response))
callback_name = 'parse_%s' % name[6:]
if hasattr(self, callback_name):
if extractor.match(response.url):
ret.extend(getattr(self, callback_name)(response))
for entry in ret:
if isinstance(entry, ScrapedItem):
self.set_guid(entry)
return ret
class XMLFeedSpider(BaseSpider):
"""
This class intends to be the base class for spiders that scrape
from XML feeds.
You can choose whether to parse the file using the iternodes tool,
or not using it (which just splits the tags using xpath)
"""
set_guid = _set_guid
iternodes = True
itertag = 'item'
def parse_item_wrapper(self, response, xSel):
ret = self.parse_item(response, xSel)
if isinstance(ret, ScrapedItem):
self.set_guid(ret)
return ret
def parse(self, response):
if not hasattr(self, 'parse_item'):
raise NotConfigured('You must define parse_item method in order to scrape this XML feed')
if self.iternodes:
nodes = xmliter(response, self.itertag)
else:
nodes = XmlXPathSelector(response).x('//%s' % self.itertag)
return (self.parse_item_wrapper(response, xSel) for xSel in nodes)
class CSVFeedSpider(BaseSpider):
"""
Spider for parsing CSV feeds.
It receives a CSV file in a response; iterates through each of its rows,
and calls parse_row with a dict containing each field's data.
You can set some options regarding the CSV file, such as the delimiter
and the file's headers.
"""
set_guid = _set_guid
delimiter = None # When this is None, python's csv module's default delimiter is used
headers = None
def adapt_feed(self, response):
"""You can override this function in order to make any changes you want
to into the feed before parsing it. This function may return either a
response or a string. """
return response
def parse_row_wrapper(self, response, row):
ret = self.parse_row(response, row)
if isinstance(ret, ScrapedItem):
self.set_guid(ret)
return ret
def parse(self, response):
if not hasattr(self, 'parse_row'):
raise NotConfigured('You must define parse_row method in order to scrape this CSV feed')
feed = self.adapt_feed(response)
return (self.parse_row_wrapper(feed, row) for row in csviter(response, self.delimiter, self.headers))

View File

@ -0,0 +1,2 @@
from scrapy.contrib.spiders.crawl import CrawlSpider, Rule
from scrapy.contrib.spiders.feed import XMLFeedSpider, CSVFeedSpider

View File

@ -0,0 +1,87 @@
# -*- coding: utf8 -*-
from scrapy.spider import BaseSpider
from scrapy.item import ScrapedItem
from scrapy.http import Request
from scrapy.utils.iterators import xmliter, csviter
from scrapy.xpath.selector import XmlXPathSelector
from scrapy.core.exceptions import UsageError, NotConfigured
class XMLFeedSpider(BaseSpider):
"""
This class intends to be the base class for spiders that scrape
from XML feeds.
You can choose whether to parse the file using the iternodes tool,
or not using it (which just splits the tags using xpath)
"""
iternodes = True
itertag = 'item'
def process_results(self, results, response):
"""This overridable method is called for each result (item or request)
returned by the spider, and it's intended to perform any last time
processing required before returning the results to the framework core,
for example setting the item GUIDs. It receives a list of results and
the response which originated that results. It must return a list
of results (Items or Requests)."""
return results
def parse_nodes(self, response, nodes):
for xSel in nodes:
ret = self.parse_item(response, xSel)
if isinstance(ret, (ScrapedItem, Request)):
ret = [ret]
if not isinstance(ret, (list, tuple)):
raise UsageError('You cannot return an "%s" object from a spider' % type(ret).__name__)
for result_item in self.process_results(ret, response):
yield result_item
def parse(self, response):
if not hasattr(self, 'parse_item'):
raise NotConfigured('You must define parse_item method in order to scrape this XML feed')
if self.iternodes:
nodes = xmliter(response, self.itertag)
else:
nodes = XmlXPathSelector(response).x('//%s' % self.itertag)
return self.parse_nodes(response, nodes)
class CSVFeedSpider(BaseSpider):
"""
Spider for parsing CSV feeds.
It receives a CSV file in a response; iterates through each of its rows,
and calls parse_row with a dict containing each field's data.
You can set some options regarding the CSV file, such as the delimiter
and the file's headers.
"""
delimiter = None # When this is None, python's csv module's default delimiter is used
headers = None
def process_results(self, results, response):
"""This method has the same purpose as the one in XMLFeedSpider"""
return results
def adapt_response(self, response):
"""You can override this function in order to make any changes you want
to into the feed before parsing it. This function must return a response."""
return response
def parse_rows(self, response):
for row in csviter(response, self.delimiter, self.headers):
ret = self.parse_row(response, row)
if isinstance(ret, (ScrapedItem, Request)):
ret = [ret]
if not isinstance(ret, (list, tuple)):
raise UsageError('You cannot return an "%s" object from a spider' % type(ret).__name__)
for result_item in self.process_results(ret, response):
yield result_item
def parse(self, response):
if not hasattr(self, 'parse_row'):
raise NotConfigured('You must define parse_row method in order to scrape this CSV feed')
response = self.adapt_response(response)
return self.parse_rows(response)

View File

@ -1,2 +0,0 @@
from scrapy.contrib.spiders2.crawl import CrawlSpider, Rule
from scrapy.contrib.spiders2.feed import XMLFeedSpider, CSVFeedSpider

View File

@ -1,79 +0,0 @@
from scrapy.spider import BaseSpider
from scrapy.item import ScrapedItem
from scrapy.utils.iterators import xmliter, csviter
from scrapy.xpath.selector import XmlXPathSelector
from scrapy.core.exceptions import NotConfigured
class XMLFeedSpider(BaseSpider):
"""
This class intends to be the base class for spiders that scrape
from XML feeds.
You can choose whether to parse the file using the iternodes tool,
or not using it (which just splits the tags using xpath)
"""
iternodes = True
itertag = 'item'
def item_scraped(self, response, item):
"""
This method is called for each item returned by the spider, and it's intended
to do anything that it's needed before returning the item to the core, specially
setting its GUID.
It receives and returns an item
"""
return item
def parse_item_wrapper(self, response, xSel):
ret = self.parse_item(response, xSel)
if isinstance(ret, ScrapedItem):
self.scraped_item(response, ret)
return ret
def parse(self, response):
if not hasattr(self, 'parse_item'):
raise NotConfigured('You must define parse_item method in order to scrape this XML feed')
if self.iternodes:
nodes = xmliter(response, self.itertag)
else:
nodes = XmlXPathSelector(response).x('//%s' % self.itertag)
return (self.parse_item_wrapper(response, xSel) for xSel in nodes)
class CSVFeedSpider(BaseSpider):
"""
Spider for parsing CSV feeds.
It receives a CSV file in a response; iterates through each of its rows,
and calls parse_row with a dict containing each field's data.
You can set some options regarding the CSV file, such as the delimiter
and the file's headers.
"""
delimiter = None # When this is None, python's csv module's default delimiter is used
headers = None
def scraped_item(self, response, item):
"""This method has the same purpose as the one in XMLFeedSpider"""
return item
def adapt_feed(self, response):
"""You can override this function in order to make any changes you want
to into the feed before parsing it. This function may return either a
response or a string. """
return response
def parse_row_wrapper(self, response, row):
ret = self.parse_row(response, row)
if isinstance(ret, ScrapedItem):
self.scraped_item(response, ret)
return ret
def parse(self, response):
if not hasattr(self, 'parse_row'):
raise NotConfigured('You must define parse_row method in order to scrape this CSV feed')
feed = self.adapt_feed(response)
return (self.parse_row_wrapper(feed, row) for row in csviter(response, self.delimiter, self.headers))