mirror of https://github.com/scrapy/scrapy.git
Changed CrawlSpider's way of setting crawling rules
--HG-- rename : scrapy/trunk/scrapy/contrib/spiders2/crawl.py => scrapy/trunk/scrapy/contrib/spiders/crawl.py extra : convert_revision : svn%3Ab85faa78-f9eb-468e-a121-7cced6da292c%40450
This commit is contained in:
parent
7f05b923ab
commit
8b37282752
|
|
@ -2,24 +2,23 @@
|
|||
import re
|
||||
|
||||
from scrapy.xpath import HtmlXPathSelector
|
||||
from scrapy.item import ScrapedItem
|
||||
from scrapy.link.extractors import RegexLinkExtractor
|
||||
from scrapy.contrib.spiders import CrawlSpider
|
||||
from scrapy.utils.url import url_query_parameter
|
||||
from scrapy.contrib.spiders import CrawlSpider, Rule
|
||||
|
||||
class $classname(CrawlSpider):
|
||||
domain_name = "$site"
|
||||
start_urls = ['http://www.$site/']
|
||||
|
||||
links_category = RegexLinkExtractor(allow=(re.compile(r'ProductList'), ))
|
||||
links_product = RegexLinkExtractor(allow=(re.compile(r'ProductDetail'), ))
|
||||
rules = (
|
||||
Rule(RegexLinkExtractor(allow=(r'Items/', ), 'parse_items', follow=True)
|
||||
)
|
||||
|
||||
def parse_product(self, response):
|
||||
def parse_item(self, response):
|
||||
#xs = HtmlXPathSelector(response)
|
||||
#p = self.create_product(response)
|
||||
#p.attribute('site_id', url_query_parameter(response.url,'productid'))
|
||||
#p.attribute('site_id', xs.x("//input[@name='productid']/@value"))
|
||||
#p.attribute('site_id', xs.x(""))
|
||||
#p.attribute('name', xs.x(""))
|
||||
#p.attribute('description', xs.x(""))
|
||||
#i = ScrapedItem()
|
||||
#i.attribute('site_id', xs.x("//input[@id="sid"]/@value"))
|
||||
#i.attribute('name', xs.x("//div[@id='name']"))
|
||||
#i.attribute('description', xs.x("//div[@id='description']"))
|
||||
|
||||
SPIDER = $classname()
|
||||
|
|
|
|||
|
|
@ -4,7 +4,9 @@ Adaptors related with extraction of data
|
|||
|
||||
import urlparse
|
||||
import re
|
||||
from scrapy import log
|
||||
from scrapy.http import Response
|
||||
from scrapy.utils.url import is_url
|
||||
from scrapy.utils.python import flatten
|
||||
from scrapy.xpath.selector import XPathSelector, XPathSelectorList
|
||||
|
||||
|
|
@ -64,26 +66,29 @@ class ExtractImages(object):
|
|||
self.base_url = base_url
|
||||
|
||||
if not self.base_url:
|
||||
raise AttributeError('You must specify either a response or a base_url to the ExtractImages adaptor.')
|
||||
log.msg('No base URL was found for ExtractImages adaptor, will only extract absolute URLs', log.WARNING)
|
||||
|
||||
def extract_from_xpath(self, selector):
|
||||
ret = []
|
||||
if selector.xmlNode.type == 'element':
|
||||
if selector.xmlNode.name == 'a':
|
||||
children = selector.x('child::*')
|
||||
if len(children) > 1:
|
||||
if selector.xmlNode.name == 'a':
|
||||
children = selector.x('child::*')
|
||||
if len(children) > 1:
|
||||
ret.extend(selector.x('.//@href'))
|
||||
ret.extend(selector.x('.//@src'))
|
||||
elif len(children) == 1 and children[0].xmlNode.name == 'img':
|
||||
ret.extend(children.x('@src'))
|
||||
else:
|
||||
ret.extend(selector.x('@href'))
|
||||
elif selector.xmlNode.name == 'img':
|
||||
ret.extend(selector.x('@src'))
|
||||
elif selector.xmlNode.name == 'text':
|
||||
ret.extend(selector)
|
||||
else:
|
||||
ret.extend(selector.x('.//@href'))
|
||||
ret.extend(selector.x('.//@src'))
|
||||
elif len(children) == 1 and children[0].xmlNode.name == 'img':
|
||||
ret.extend(children.x('@src'))
|
||||
else:
|
||||
ret.extend(selector.x('@href'))
|
||||
elif selector.xmlNode.name == 'img':
|
||||
ret.extend(selector.x('@src'))
|
||||
else:
|
||||
ret.extend(selector.x('.//@href'))
|
||||
ret.extend(selector.x('.//@src'))
|
||||
elif selector.xmlNode.type == 'attribute' and selector.xmlNode.name in ['href', 'src']:
|
||||
ret.extend(selector.x('.//text()'))
|
||||
else:
|
||||
ret.append(selector)
|
||||
|
||||
return ret
|
||||
|
|
@ -92,11 +97,15 @@ class ExtractImages(object):
|
|||
if isinstance(locations, basestring):
|
||||
locations = [locations]
|
||||
|
||||
rel_links = []
|
||||
rel_urls = []
|
||||
for location in flatten(locations):
|
||||
if isinstance(location, (XPathSelector, XPathSelectorList)):
|
||||
rel_links.extend(self.extract_from_xpath(location))
|
||||
rel_urls.extend(self.extract_from_xpath(location))
|
||||
else:
|
||||
rel_links.append(location)
|
||||
rel_links = extract(rel_links)
|
||||
return [urlparse.urljoin(self.base_url, link) for link in rel_links]
|
||||
rel_urls.append(location)
|
||||
rel_urls = extract(rel_urls)
|
||||
|
||||
if self.base_url:
|
||||
return [urlparse.urljoin(self.base_url, url) for url in rel_urls]
|
||||
else:
|
||||
return filter(rel_urls, is_url)
|
||||
|
|
|
|||
|
|
@ -1,159 +0,0 @@
|
|||
"""
|
||||
This module contains some basic spiders for scraping websites (CrawlSpider)
|
||||
and XML feeds (XMLFeedSpider).
|
||||
"""
|
||||
|
||||
from scrapy.conf import settings
|
||||
from scrapy.http import Request
|
||||
from scrapy.spider import BaseSpider
|
||||
from scrapy.item import ScrapedItem
|
||||
from scrapy.xpath.selector import XmlXPathSelector
|
||||
from scrapy.core.exceptions import NotConfigured
|
||||
from scrapy.utils.iterators import xmliter, csviter
|
||||
|
||||
|
||||
def _set_guid(spider, item):
|
||||
"""
|
||||
This method is called whenever the spider returns items, for each item.
|
||||
It should set the 'guid' attribute to the given item with a string that
|
||||
identifies the item uniquely.
|
||||
"""
|
||||
raise NotConfigured('You must define a set_guid method in order to scrape items.')
|
||||
|
||||
class CrawlSpider(BaseSpider):
|
||||
"""
|
||||
This class works as a base class for spiders that crawl over websites
|
||||
"""
|
||||
set_guid = _set_guid
|
||||
|
||||
def __init__(self):
|
||||
super(CrawlSpider, self).__init__()
|
||||
|
||||
self._links_callback = []
|
||||
for attr in dir(self):
|
||||
if attr.startswith('links_'):
|
||||
suffix = attr.split('_', 1)[1]
|
||||
value = getattr(self, attr)
|
||||
callback = getattr(self, 'parse_%s' % suffix, None)
|
||||
self._links_callback.append((value, callback))
|
||||
|
||||
def parse(self, response):
|
||||
"""This function is called by the core for all the start_urls. Do not
|
||||
override this function, override parse_start_url instead."""
|
||||
if response.url in self.start_urls:
|
||||
return self._parse_wrapper(response, self.parse_start_url)
|
||||
else:
|
||||
return self.parse_url(response)
|
||||
|
||||
def parse_start_url(self, response):
|
||||
"""Callback function for processing start_urls. It must return a list
|
||||
of ScrapedItems and/or Requests."""
|
||||
return []
|
||||
|
||||
def _links_to_follow(self, response):
|
||||
res = []
|
||||
links_to_follow = {}
|
||||
for lx, callback in self._links_callback:
|
||||
links = lx.extract_urls(response)
|
||||
links = self.post_extract_links(links) if hasattr(self, 'post_extract_links') else links
|
||||
for link in links:
|
||||
links_to_follow[link.url] = (callback, link.text)
|
||||
|
||||
for url, (callback, link_text) in links_to_follow.iteritems():
|
||||
request = Request(url=url, link_text=link_text)
|
||||
request.append_callback(self._parse_wrapper, callback)
|
||||
res.append(request)
|
||||
return res
|
||||
|
||||
def _parse_wrapper(self, response, callback):
|
||||
res = []
|
||||
if settings.getbool('CRAWLSPIDER_FOLLOW_LINKS', True):
|
||||
res.extend(self._links_to_follow(response))
|
||||
res.extend(callback(response) if callback else ())
|
||||
for entry in res:
|
||||
if isinstance(entry, ScrapedItem):
|
||||
self.set_guid(entry)
|
||||
return res
|
||||
|
||||
def parse_url(self, response):
|
||||
"""
|
||||
This method is called whenever you run scrapy with the 'parse' command
|
||||
over an URL.
|
||||
"""
|
||||
extractor_names = [attrname for attrname in dir(self) if attrname.startswith('links_')]
|
||||
ret = []
|
||||
for name in extractor_names:
|
||||
extractor = getattr(self, name)
|
||||
if settings.getbool('CRAWLSPIDER_FOLLOW_LINKS', True):
|
||||
ret.extend(self._links_to_follow(response))
|
||||
callback_name = 'parse_%s' % name[6:]
|
||||
if hasattr(self, callback_name):
|
||||
if extractor.match(response.url):
|
||||
ret.extend(getattr(self, callback_name)(response))
|
||||
for entry in ret:
|
||||
if isinstance(entry, ScrapedItem):
|
||||
self.set_guid(entry)
|
||||
return ret
|
||||
|
||||
class XMLFeedSpider(BaseSpider):
|
||||
"""
|
||||
This class intends to be the base class for spiders that scrape
|
||||
from XML feeds.
|
||||
|
||||
You can choose whether to parse the file using the iternodes tool,
|
||||
or not using it (which just splits the tags using xpath)
|
||||
"""
|
||||
set_guid = _set_guid
|
||||
iternodes = True
|
||||
itertag = 'item'
|
||||
|
||||
def parse_item_wrapper(self, response, xSel):
|
||||
ret = self.parse_item(response, xSel)
|
||||
if isinstance(ret, ScrapedItem):
|
||||
self.set_guid(ret)
|
||||
return ret
|
||||
|
||||
def parse(self, response):
|
||||
if not hasattr(self, 'parse_item'):
|
||||
raise NotConfigured('You must define parse_item method in order to scrape this XML feed')
|
||||
|
||||
if self.iternodes:
|
||||
nodes = xmliter(response, self.itertag)
|
||||
else:
|
||||
nodes = XmlXPathSelector(response).x('//%s' % self.itertag)
|
||||
|
||||
return (self.parse_item_wrapper(response, xSel) for xSel in nodes)
|
||||
|
||||
class CSVFeedSpider(BaseSpider):
|
||||
"""
|
||||
Spider for parsing CSV feeds.
|
||||
It receives a CSV file in a response; iterates through each of its rows,
|
||||
and calls parse_row with a dict containing each field's data.
|
||||
|
||||
You can set some options regarding the CSV file, such as the delimiter
|
||||
and the file's headers.
|
||||
"""
|
||||
set_guid = _set_guid
|
||||
delimiter = None # When this is None, python's csv module's default delimiter is used
|
||||
headers = None
|
||||
|
||||
def adapt_feed(self, response):
|
||||
"""You can override this function in order to make any changes you want
|
||||
to into the feed before parsing it. This function may return either a
|
||||
response or a string. """
|
||||
|
||||
return response
|
||||
|
||||
def parse_row_wrapper(self, response, row):
|
||||
ret = self.parse_row(response, row)
|
||||
if isinstance(ret, ScrapedItem):
|
||||
self.set_guid(ret)
|
||||
return ret
|
||||
|
||||
def parse(self, response):
|
||||
if not hasattr(self, 'parse_row'):
|
||||
raise NotConfigured('You must define parse_row method in order to scrape this CSV feed')
|
||||
|
||||
feed = self.adapt_feed(response)
|
||||
return (self.parse_row_wrapper(feed, row) for row in csviter(response, self.delimiter, self.headers))
|
||||
|
||||
|
|
@ -0,0 +1,2 @@
|
|||
from scrapy.contrib.spiders.crawl import CrawlSpider, Rule
|
||||
from scrapy.contrib.spiders.feed import XMLFeedSpider, CSVFeedSpider
|
||||
|
|
@ -0,0 +1,87 @@
|
|||
# -*- coding: utf8 -*-
|
||||
from scrapy.spider import BaseSpider
|
||||
from scrapy.item import ScrapedItem
|
||||
from scrapy.http import Request
|
||||
from scrapy.utils.iterators import xmliter, csviter
|
||||
from scrapy.xpath.selector import XmlXPathSelector
|
||||
from scrapy.core.exceptions import UsageError, NotConfigured
|
||||
|
||||
class XMLFeedSpider(BaseSpider):
|
||||
"""
|
||||
This class intends to be the base class for spiders that scrape
|
||||
from XML feeds.
|
||||
|
||||
You can choose whether to parse the file using the iternodes tool,
|
||||
or not using it (which just splits the tags using xpath)
|
||||
"""
|
||||
iternodes = True
|
||||
itertag = 'item'
|
||||
|
||||
def process_results(self, results, response):
|
||||
"""This overridable method is called for each result (item or request)
|
||||
returned by the spider, and it's intended to perform any last time
|
||||
processing required before returning the results to the framework core,
|
||||
for example setting the item GUIDs. It receives a list of results and
|
||||
the response which originated that results. It must return a list
|
||||
of results (Items or Requests)."""
|
||||
return results
|
||||
|
||||
def parse_nodes(self, response, nodes):
|
||||
for xSel in nodes:
|
||||
ret = self.parse_item(response, xSel)
|
||||
if isinstance(ret, (ScrapedItem, Request)):
|
||||
ret = [ret]
|
||||
if not isinstance(ret, (list, tuple)):
|
||||
raise UsageError('You cannot return an "%s" object from a spider' % type(ret).__name__)
|
||||
for result_item in self.process_results(ret, response):
|
||||
yield result_item
|
||||
|
||||
def parse(self, response):
|
||||
if not hasattr(self, 'parse_item'):
|
||||
raise NotConfigured('You must define parse_item method in order to scrape this XML feed')
|
||||
|
||||
if self.iternodes:
|
||||
nodes = xmliter(response, self.itertag)
|
||||
else:
|
||||
nodes = XmlXPathSelector(response).x('//%s' % self.itertag)
|
||||
|
||||
return self.parse_nodes(response, nodes)
|
||||
|
||||
class CSVFeedSpider(BaseSpider):
|
||||
"""
|
||||
Spider for parsing CSV feeds.
|
||||
It receives a CSV file in a response; iterates through each of its rows,
|
||||
and calls parse_row with a dict containing each field's data.
|
||||
|
||||
You can set some options regarding the CSV file, such as the delimiter
|
||||
and the file's headers.
|
||||
"""
|
||||
delimiter = None # When this is None, python's csv module's default delimiter is used
|
||||
headers = None
|
||||
|
||||
def process_results(self, results, response):
|
||||
"""This method has the same purpose as the one in XMLFeedSpider"""
|
||||
return results
|
||||
|
||||
def adapt_response(self, response):
|
||||
"""You can override this function in order to make any changes you want
|
||||
to into the feed before parsing it. This function must return a response."""
|
||||
return response
|
||||
|
||||
def parse_rows(self, response):
|
||||
for row in csviter(response, self.delimiter, self.headers):
|
||||
ret = self.parse_row(response, row)
|
||||
if isinstance(ret, (ScrapedItem, Request)):
|
||||
ret = [ret]
|
||||
if not isinstance(ret, (list, tuple)):
|
||||
raise UsageError('You cannot return an "%s" object from a spider' % type(ret).__name__)
|
||||
for result_item in self.process_results(ret, response):
|
||||
yield result_item
|
||||
|
||||
def parse(self, response):
|
||||
if not hasattr(self, 'parse_row'):
|
||||
raise NotConfigured('You must define parse_row method in order to scrape this CSV feed')
|
||||
|
||||
response = self.adapt_response(response)
|
||||
return self.parse_rows(response)
|
||||
|
||||
|
|
@ -1,2 +0,0 @@
|
|||
from scrapy.contrib.spiders2.crawl import CrawlSpider, Rule
|
||||
from scrapy.contrib.spiders2.feed import XMLFeedSpider, CSVFeedSpider
|
||||
|
|
@ -1,79 +0,0 @@
|
|||
from scrapy.spider import BaseSpider
|
||||
from scrapy.item import ScrapedItem
|
||||
from scrapy.utils.iterators import xmliter, csviter
|
||||
from scrapy.xpath.selector import XmlXPathSelector
|
||||
from scrapy.core.exceptions import NotConfigured
|
||||
|
||||
class XMLFeedSpider(BaseSpider):
|
||||
"""
|
||||
This class intends to be the base class for spiders that scrape
|
||||
from XML feeds.
|
||||
|
||||
You can choose whether to parse the file using the iternodes tool,
|
||||
or not using it (which just splits the tags using xpath)
|
||||
"""
|
||||
iternodes = True
|
||||
itertag = 'item'
|
||||
|
||||
def item_scraped(self, response, item):
|
||||
"""
|
||||
This method is called for each item returned by the spider, and it's intended
|
||||
to do anything that it's needed before returning the item to the core, specially
|
||||
setting its GUID.
|
||||
It receives and returns an item
|
||||
"""
|
||||
return item
|
||||
|
||||
def parse_item_wrapper(self, response, xSel):
|
||||
ret = self.parse_item(response, xSel)
|
||||
if isinstance(ret, ScrapedItem):
|
||||
self.scraped_item(response, ret)
|
||||
return ret
|
||||
|
||||
def parse(self, response):
|
||||
if not hasattr(self, 'parse_item'):
|
||||
raise NotConfigured('You must define parse_item method in order to scrape this XML feed')
|
||||
|
||||
if self.iternodes:
|
||||
nodes = xmliter(response, self.itertag)
|
||||
else:
|
||||
nodes = XmlXPathSelector(response).x('//%s' % self.itertag)
|
||||
|
||||
return (self.parse_item_wrapper(response, xSel) for xSel in nodes)
|
||||
|
||||
class CSVFeedSpider(BaseSpider):
|
||||
"""
|
||||
Spider for parsing CSV feeds.
|
||||
It receives a CSV file in a response; iterates through each of its rows,
|
||||
and calls parse_row with a dict containing each field's data.
|
||||
|
||||
You can set some options regarding the CSV file, such as the delimiter
|
||||
and the file's headers.
|
||||
"""
|
||||
delimiter = None # When this is None, python's csv module's default delimiter is used
|
||||
headers = None
|
||||
|
||||
def scraped_item(self, response, item):
|
||||
"""This method has the same purpose as the one in XMLFeedSpider"""
|
||||
return item
|
||||
|
||||
def adapt_feed(self, response):
|
||||
"""You can override this function in order to make any changes you want
|
||||
to into the feed before parsing it. This function may return either a
|
||||
response or a string. """
|
||||
|
||||
return response
|
||||
|
||||
def parse_row_wrapper(self, response, row):
|
||||
ret = self.parse_row(response, row)
|
||||
if isinstance(ret, ScrapedItem):
|
||||
self.scraped_item(response, ret)
|
||||
return ret
|
||||
|
||||
def parse(self, response):
|
||||
if not hasattr(self, 'parse_row'):
|
||||
raise NotConfigured('You must define parse_row method in order to scrape this CSV feed')
|
||||
|
||||
feed = self.adapt_feed(response)
|
||||
return (self.parse_row_wrapper(feed, row) for row in csviter(response, self.delimiter, self.headers))
|
||||
|
||||
Loading…
Reference in New Issue