mirror of https://github.com/scrapy/scrapy.git
added start_requests method to BaseSpider, made start_urls empty by default instead of a required attribute in every spider
--HG-- extra : convert_revision : svn%3Ab85faa78-f9eb-468e-a121-7cced6da292c%40553
This commit is contained in:
parent
5b159d0f04
commit
9557e24203
|
|
@ -108,10 +108,6 @@ class ExecutionManager(object):
|
|||
sites.add(a)
|
||||
|
||||
perdomain = {}
|
||||
def _add(domain, request):
|
||||
if domain not in perdomain:
|
||||
perdomain[domain] = []
|
||||
perdomain[domain] += [request]
|
||||
|
||||
# sites
|
||||
for domain in sites:
|
||||
|
|
@ -119,15 +115,15 @@ class ExecutionManager(object):
|
|||
if not spider:
|
||||
log.msg('Could not find spider for %s' % domain, log.ERROR)
|
||||
continue
|
||||
for url in self._start_urls(spider):
|
||||
request = Request(url, callback=spider.parse, dont_filter=True)
|
||||
_add(domain, request)
|
||||
reqs = spider.start_requests()
|
||||
perdomain.setdefault(domain, []).extend(reqs)
|
||||
|
||||
# urls
|
||||
for url in urls:
|
||||
spider = spiders.fromurl(url)
|
||||
if spider:
|
||||
request = Request(url=url, callback=spider.parse, dont_filter=True)
|
||||
_add(spider.domain_name, request)
|
||||
reqs = spider.start_requests([url])
|
||||
perdomain.setdefault(domain, []).extend(reqs)
|
||||
else:
|
||||
log.msg('Could not find spider for <%s>' % url, log.ERROR)
|
||||
|
||||
|
|
@ -140,10 +136,7 @@ class ExecutionManager(object):
|
|||
if not spider:
|
||||
log.msg('Could not find spider for %s' % request, log.ERROR)
|
||||
continue
|
||||
_add(spider.domain_name, request)
|
||||
perdomain.setdefault(domain, []).append(request)
|
||||
return perdomain
|
||||
|
||||
def _start_urls(self, spider):
|
||||
return spider.start_urls if hasattr(spider.start_urls, '__iter__') else [spider.start_urls]
|
||||
|
||||
scrapymanager = ExecutionManager()
|
||||
|
|
|
|||
|
|
@ -5,13 +5,9 @@ from zope.interface import Interface, Attribute, invariant, implements
|
|||
from twisted.plugin import IPlugin
|
||||
|
||||
from scrapy import log
|
||||
from scrapy.http import Request
|
||||
from scrapy.core.exceptions import UsageError
|
||||
|
||||
def _valid_start_urls(obj):
|
||||
"""Check the start urls specified are valid"""
|
||||
if not obj.start_urls:
|
||||
raise UsageError("A start url is required")
|
||||
|
||||
def _valid_domain_name(obj):
|
||||
"""Check the domain name specified is valid"""
|
||||
if not obj.domain_name:
|
||||
|
|
@ -28,10 +24,6 @@ def _valid_download_delay(obj):
|
|||
class ISpider(Interface, IPlugin) :
|
||||
"""Interface to be implemented by site-specific web spiders"""
|
||||
|
||||
start_urls = Attribute(
|
||||
"""A sequence of URLs to retrieve to initiate the spider for this
|
||||
site. A single URL may also be provided here.""")
|
||||
|
||||
domain_name = Attribute(
|
||||
"""The domain name of the site to be scraped.""")
|
||||
|
||||
|
|
@ -43,37 +35,9 @@ class ISpider(Interface, IPlugin) :
|
|||
user_agent = Attribute(
|
||||
"""Optional User-Agent to use for this domain""")
|
||||
|
||||
invariant(_valid_start_urls)
|
||||
invariant(_valid_domain_name)
|
||||
invariant(_valid_download_delay)
|
||||
|
||||
def parse(self, pagedata) :
|
||||
"""This is first called with the data corresponding to start_url. It
|
||||
must return a (possibly empty) sequence where each element is either:
|
||||
* A Request object for further processing.
|
||||
* An object that extends ScrapedItem (defined in scrapeditem module)
|
||||
* or None (this will be ignored)
|
||||
|
||||
When a Request object is returned, the Request is scheduled, then
|
||||
downloaded and finally its results is handled to the Request callback.
|
||||
That callback must behave the same way as this function.
|
||||
|
||||
When a ScrapedItem is returned, it is passed to the transformation pipeline
|
||||
and finally the destination systems are updated.
|
||||
|
||||
The simplest way to use this is to have a method in your class for each
|
||||
page type. So each function knows the layout and how to extract data
|
||||
for a single page (or set of similar pages). A typical class might work
|
||||
like:
|
||||
* parse() parses the landing page and returns Requests
|
||||
for a category() function to parse category pages.
|
||||
* category() parses the category pages and returns links and callbacks
|
||||
for a item() function to parse item pages.
|
||||
* item() parses the item details page and returns objects that
|
||||
extend ScrapedItem
|
||||
"""
|
||||
pass
|
||||
|
||||
def init_domain(self):
|
||||
"""This is first called to initialize domain specific quirks, like
|
||||
session cookies or login stuff
|
||||
|
|
@ -83,9 +47,13 @@ class ISpider(Interface, IPlugin) :
|
|||
|
||||
class BaseSpider(object):
|
||||
"""Base class for scrapy spiders. All spiders must inherit from this
|
||||
class."""
|
||||
|
||||
class.
|
||||
|
||||
"""
|
||||
|
||||
implements(ISpider)
|
||||
|
||||
start_urls = []
|
||||
domain_name = None
|
||||
extra_domain_names = []
|
||||
|
||||
|
|
@ -94,3 +62,37 @@ class BaseSpider(object):
|
|||
method to send log messages from your spider
|
||||
"""
|
||||
log.msg(message, domain=self.domain_name, level=level)
|
||||
|
||||
def start_requests(self, urls=None):
|
||||
"""Return the requests to crawl when this spider is opened for
|
||||
scraping. urls contain the urls passed from command line (if any),
|
||||
otherwise None if the enrie domain was requested for scraping.
|
||||
|
||||
This function must return a list of Requests to be crawled, based on
|
||||
the given urls. The Requests must include a callback function which
|
||||
must return a list of:
|
||||
* Request's for further crawling
|
||||
* ScrapedItem's for processing
|
||||
* Both
|
||||
Or None (which will be treated the same way as an empty list)
|
||||
|
||||
When a Request object is returned, the Request is scheduled, then
|
||||
downloaded and finally its results is handled to the Request callback.
|
||||
|
||||
When a ScrapedItem is returned, it is passed to the item pipeline.
|
||||
|
||||
Unless this method is overrided, the start_urls attribute will be used
|
||||
to create the initial requests (when urls is None).
|
||||
"""
|
||||
if urls is None:
|
||||
urls = self.start_urls
|
||||
requests = [Request(url, callback=self.parse, dont_filter=True) for url in urls]
|
||||
return requests
|
||||
|
||||
return []
|
||||
|
||||
def parse(self, response):
|
||||
"""This is the default callback function used to parse the start
|
||||
requests, although it can be overrided in descendant spiders.
|
||||
"""
|
||||
pass
|
||||
|
|
|
|||
Loading…
Reference in New Issue