diff --git a/AUTHORS b/AUTHORS index a0fbe722f..1392aa71f 100644 --- a/AUTHORS +++ b/AUTHORS @@ -1,28 +1,25 @@ Scrapy was brought to life by Shane Evans while hacking a scraping framework prototype for Mydeco (mydeco.com). It soon became maintained, extended and -improved by Insophia (insophia.com), with the sponsorship of By Design (the -company behind Mydeco). +improved by Insophia (insophia.com), with the initial sponsorship of Mydeco to +bootstrap the project. -Here is the list of the primary authors & contributors, along with their user -name (in Scrapy trac/subversion). Emails are intentionally left out to avoid -spam. +Here is the list of the primary authors & contributors: - * Pablo Hoffman (pablo) - * Daniel Graña (daniel) - * Martin Olveyra (olveyra) - * Gabriel García (elpolilla) - * Michael Cetrulo (samus_) - * Artem Bogomyagkov (artem) - * Damian Canabal (calarval) - * Andres Moreira (andres) - * Ismael Carnales (ismael) - * Matías Aguirre (omab) - * German Hoffman (german) - * Anibal Pacheco (anibal) + * Pablo Hoffman + * Daniel Graña + * Martin Olveyra + * Gabriel García + * Michael Cetrulo + * Artem Bogomyagkov + * Damian Canabal + * Andres Moreira + * Ismael Carnales + * Matías Aguirre + * German Hoffmann + * Anibal Pacheco * Bruno Deferrari * Shane Evans - -And here is the list of people who have helped to put the Scrapy homepage live: - - * Ezequiel Rivero (ezequiel) + * Ezequiel Rivero + * Patrick Mezard + * Rolando Espinoza diff --git a/docs/experimental/crawlspider-v2.rst b/docs/experimental/crawlspider-v2.rst new file mode 100644 index 000000000..a1ea09c48 --- /dev/null +++ b/docs/experimental/crawlspider-v2.rst @@ -0,0 +1,128 @@ +.. _topics-crawlspider-v2: + +============== +CrawlSpider v2 +============== + +Introduction +============ + +TODO: introduction + +Rules Matching +============== + +TODO: describe purpose of rules + +Request Extractors & Processors +=============================== + +TODO: describe purpose of extractors & processors + +Examples +======== + +TODO: plenty of examples + + +.. module:: scrapy.contrib_exp.crawlspider.spider + :synopsis: CrawlSpider + + +Reference +========= + +CrawlSpider +----------- + +TODO: describe crawlspider + +.. class:: CrawlSpider + + TODO: describe class + + +.. module:: scrapy.contrib_exp.crawlspider.rules + :synopsis: Rules + +Rules +----- + +TODO: describe spider rules + +.. class:: Rule + + TODO: describe Rules class + + +.. module:: scrapy.contrib_exp.crawlspider.reqext + :synopsis: Request Extractors + +Request Extractors +------------------ + +TODO: describe extractors purpose + +.. class:: BaseSgmlRequestExtractor + + TODO: describe base extractor + +.. class:: SgmlRequestExtractor + + TODO: describe sgml extractor + +.. class:: XPathRequestExtractor + + TODO: describe xpath request extractor + + +.. module:: scrapy.contrib_exp.crawlspider.reqproc + :synopsis: Request Processors + +Request Processors +------------------ + +TODO: describe request processors + +.. class:: Canonicalize + + TODO: describe proc + +.. class:: Unique + + TODO: describe unique + +.. class:: FilterDomain + + TODO: describe filter domain + +.. class:: FilterUrl + + TODO: describe filter url + + +.. module:: scrapy.contrib_exp.crawlspider.matchers + :synopsis: Matchers + +Request/Response Matchers +------------------------- + +TODO: describe matchers + +.. class:: BaseMatcher + + TODO: describe base matcher + +.. class:: UrlMatcher + + TODO: describe url matcher + +.. class:: UrlRegexMatcher + + TODO: describe UrlListMatcher + +.. class:: UrlListMatcher + + TODO: describe url list matcher + + diff --git a/docs/experimental/index.rst b/docs/experimental/index.rst index 47f4ee4ac..f63ceddd9 100644 --- a/docs/experimental/index.rst +++ b/docs/experimental/index.rst @@ -21,3 +21,4 @@ it's properly merged) . Use at your own risk. djangoitems scheduler-middleware + crawlspider-v2 diff --git a/docs/intro/tutorial.rst b/docs/intro/tutorial.rst index 3fb965110..6d7c3c64a 100644 --- a/docs/intro/tutorial.rst +++ b/docs/intro/tutorial.rst @@ -420,8 +420,8 @@ scraped so far, the code for our Spider should be like this:: Now doing a crawl on the dmoz.org domain yields ``DmozItem``'s:: - [dmoz.org] DEBUG: Scraped DmozItem({'title': [u'Text Processing in Python'], 'link': [u'http://gnosis.cx/TPiP/'], 'desc': [u' - By David Mertz; Addison Wesley. Book in progress, full text, ASCII format. Asks for feedback. [author website, Gnosis Software, Inc.]\n']}) in - [dmoz.org] DEBUG: Scraped DmozItem({'title': [u'XML Processing with Python'], 'link': [u'http://www.informit.com/store/product.aspx?isbn=0130211192'], 'desc': [u' - By Sean McGrath; Prentice Hall PTR, 2000, ISBN 0130211192, has CD-ROM. Methods to build XML applications fast, Python tutorial, DOM and SAX, new Pyxie open source XML processing library. [Prentice Hall PTR]\n']}) in + [dmoz.org] DEBUG: Scraped DmozItem(desc=[u' - By David Mertz; Addison Wesley. Book in progress, full text, ASCII format. Asks for feedback. [author website, Gnosis Software, Inc.]\n'], link=[u'http://gnosis.cx/TPiP/'], title=[u'Text Processing in Python']) in + [dmoz.org] DEBUG: Scraped DmozItem(desc=[u' - By Sean McGrath; Prentice Hall PTR, 2000, ISBN 0130211192, has CD-ROM. Methods to build XML applications fast, Python tutorial, DOM and SAX, new Pyxie open source XML processing library. [Prentice Hall PTR]\n'], link=[u'http://www.informit.com/store/product.aspx?isbn=0130211192'], title=[u'XML Processing with Python']) in Storing the data (using an Item Pipeline) diff --git a/docs/topics/item-pipeline.rst b/docs/topics/item-pipeline.rst index a72b285c3..019c9c32b 100644 --- a/docs/topics/item-pipeline.rst +++ b/docs/topics/item-pipeline.rst @@ -98,10 +98,10 @@ spider returns multiples items with the same id:: del self.duplicates[spider] def process_item(self, spider, item): - if item.id in self.duplicates[spider]: + if item['id'] in self.duplicates[spider]: raise DropItem("Duplicate item found: %s" % item) else: - self.duplicates[spider].add(item.id) + self.duplicates[spider].add(item['id']) return item Built-in Item Pipelines reference @@ -178,7 +178,7 @@ on the respective Item Exporter to get more info. the first line. This format requires you to specify the fields to export using the :setting:`EXPORT_FIELDS` setting. -* ``jsonlines``: uses a :class:`~jsonlines.JsonLinesItemExporter` +* ``json``: uses a :class:`~jsonlines.JsonLinesItemExporter` * ``pickle``: uses a :class:`PickleItemExporter` diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index 3472b1937..9f4dcdc46 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -466,12 +466,14 @@ TextResponse objects .. attribute:: TextResponse.encoding - A string with the encoding of this response. The encoding is resolved in the - following order: + A string with the encoding of this response. The encoding is resolved by + trying the following mechanisms, in order: 1. the encoding passed in the constructor `encoding` argument - 2. the encoding declared in the Content-Type HTTP header + 2. the encoding declared in the Content-Type HTTP header. If this + encoding is not valid (ie. unknown), it is ignored and the next + resolution mechanism is tried. 3. the encoding declared in the response body. The TextResponse class doesn't provide any special functionality for this. However, the diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index 4b3dbc78f..4058eec3e 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -418,6 +418,15 @@ supported. Example:: DOWNLOAD_DELAY = 0.25 # 250 ms of delay +This setting is also affected by the :setting:`RANDOMIZE_DOWNLOAD_DELAY` +setting (which is enabled by default). By default, Scrapy doesn't wait a fixed +amount of time between requests, but uses a random interval between 0.5 and 1.5 +* :setting:`DOWNLOAD_DELAY`. + +Another way to change the download delay (per spider, instead of globally) is +by using the ``download_delay`` spider attribute, which takes more precedence +than this setting. + .. setting:: DOWNLOAD_TIMEOUT DOWNLOAD_TIMEOUT @@ -677,6 +686,27 @@ Example:: NEWSPIDER_MODULE = 'mybot.spiders_dev' +.. setting:: RANDOMIZE_DOWNLOAD_DELAY + +RANDOMIZE_DOWNLOAD_DELAY +------------------------ + +Default: ``True`` + +If enabled, Scrapy will wait a random amount of time (between 0.5 and 1.5 +* :setting:`DOWNLOAD_DELAY`) while fetching requests from the same +spider. + +This randomization decreases the chance of the crawler being detected (and +subsequently blocked) by sites which analyze requests looking for statistically +significant similarities in the time between their times. + +The randomization policy is the same used by `wget`_ ``--random-wait`` option. + +If :setting:`DOWNLOAD_DELAY` is zero (default) this option has no effect. + +.. _wget: http://www.gnu.org/software/wget/manual/wget.html + .. setting:: REDIRECT_MAX_TIMES REDIRECT_MAX_TIMES @@ -773,7 +803,7 @@ The scheduler to use for crawling. SCHEDULER_ORDER --------------- -Default: ``'BFO'`` +Default: ``'DFO'`` Scope: ``scrapy.core.scheduler`` diff --git a/examples/experimental/googledir/googledir/__init__.py b/examples/experimental/googledir/googledir/__init__.py new file mode 100644 index 000000000..3104ef709 --- /dev/null +++ b/examples/experimental/googledir/googledir/__init__.py @@ -0,0 +1 @@ +# googledir project diff --git a/examples/experimental/googledir/googledir/items.py b/examples/experimental/googledir/googledir/items.py new file mode 100644 index 000000000..decc2c9ba --- /dev/null +++ b/examples/experimental/googledir/googledir/items.py @@ -0,0 +1,16 @@ +# Define here the models for your scraped items +# +# See documentation in: +# http://doc.scrapy.org/topics/items.html + +from scrapy.item import Item, Field + +class GoogledirItem(Item): + + name = Field(default='') + url = Field(default='') + description = Field(default='') + + def __str__(self): + return "Google Category: name=%s url=%s" \ + % (self['name'], self['url']) diff --git a/examples/experimental/googledir/googledir/pipelines.py b/examples/experimental/googledir/googledir/pipelines.py new file mode 100644 index 000000000..f775b254c --- /dev/null +++ b/examples/experimental/googledir/googledir/pipelines.py @@ -0,0 +1,22 @@ +# Define your item pipelines here +# +# Don't forget to add your pipeline to the ITEM_PIPELINES setting +# See: http://doc.scrapy.org/topics/item-pipeline.html + +from scrapy.core.exceptions import DropItem + +class FilterWordsPipeline(object): + """ + A pipeline for filtering out items which contain certain + words in their description + """ + + # put all words in lowercase + words_to_filter = ['politics', 'religion'] + + def process_item(self, spider, item): + for word in self.words_to_filter: + if word in unicode(item['description']).lower(): + raise DropItem("Contains forbidden word: %s" % word) + else: + return item diff --git a/examples/experimental/googledir/googledir/settings.py b/examples/experimental/googledir/googledir/settings.py new file mode 100644 index 000000000..4e3c11163 --- /dev/null +++ b/examples/experimental/googledir/googledir/settings.py @@ -0,0 +1,21 @@ +# Scrapy settings for googledir project +# +# For simplicity, this file contains only the most important settings by +# default. All the other settings are documented here: +# +# http://doc.scrapy.org/topics/settings.html +# +# Or you can copy and paste them from where they're defined in Scrapy: +# +# scrapy/conf/default_settings.py +# + +BOT_NAME = 'googledir' +BOT_VERSION = '1.0' + +SPIDER_MODULES = ['googledir.spiders'] +NEWSPIDER_MODULE = 'googledir.spiders' +DEFAULT_ITEM_CLASS = 'googledir.items.GoogledirItem' +USER_AGENT = '%s/%s' % (BOT_NAME, BOT_VERSION) + +ITEM_PIPELINES = ['googledir.pipelines.FilterWordsPipeline'] diff --git a/examples/experimental/googledir/googledir/spiders/__init__.py b/examples/experimental/googledir/googledir/spiders/__init__.py new file mode 100644 index 000000000..5065ccba5 --- /dev/null +++ b/examples/experimental/googledir/googledir/spiders/__init__.py @@ -0,0 +1,8 @@ +# This package will contain the spiders of your Scrapy project +# +# To create the first spider for your project use this command: +# +# scrapy-ctl.py genspider myspider myspider-domain.com +# +# For more info see: +# http://doc.scrapy.org/topics/spiders.html diff --git a/examples/experimental/googledir/googledir/spiders/google_directory.py b/examples/experimental/googledir/googledir/spiders/google_directory.py new file mode 100644 index 000000000..fdbf3f24a --- /dev/null +++ b/examples/experimental/googledir/googledir/spiders/google_directory.py @@ -0,0 +1,40 @@ +from scrapy.selector import HtmlXPathSelector +from scrapy.contrib.loader import XPathItemLoader +from scrapy.contrib_exp.crawlspider import CrawlSpider, Rule + +from googledir.items import GoogledirItem + +class GoogleDirectorySpider(CrawlSpider): + + domain_name = 'directory.google.com' + start_urls = ['http://directory.google.com/'] + + rules = ( + # search for categories pattern and follow links + Rule(r'/[A-Z][a-zA-Z_/]+$', 'parse_category', follow=True), + ) + + def parse_category(self, response): + # The main selector we're using to extract data from the page + main_selector = HtmlXPathSelector(response) + + # The XPath to website links in the directory page + xpath = '//td[descendant::a[contains(@href, "#pagerank")]]/following-sibling::td/font' + + # Get a list of (sub) selectors to each website node pointed by the XPath + sub_selectors = main_selector.select(xpath) + + # Iterate over the sub-selectors to extract data for each website + for selector in sub_selectors: + item = GoogledirItem() + + l = XPathItemLoader(item=item, selector=selector) + l.add_xpath('name', 'a/text()') + l.add_xpath('url', 'a/@href') + l.add_xpath('description', 'font[2]/text()') + + # Here we populate the item and yield it + yield l.load_item() + +SPIDER = GoogleDirectorySpider() + diff --git a/examples/experimental/googledir/scrapy-ctl.py b/examples/experimental/googledir/scrapy-ctl.py new file mode 100644 index 000000000..552421ac3 --- /dev/null +++ b/examples/experimental/googledir/scrapy-ctl.py @@ -0,0 +1,7 @@ +#!/usr/bin/env python + +import os +os.environ.setdefault('SCRAPY_SETTINGS_MODULE', 'googledir.settings') + +from scrapy.command.cmdline import execute +execute() diff --git a/examples/experimental/imdb/imdb/__init__.py b/examples/experimental/imdb/imdb/__init__.py new file mode 100644 index 000000000..5bb534f79 --- /dev/null +++ b/examples/experimental/imdb/imdb/__init__.py @@ -0,0 +1 @@ +# package diff --git a/examples/experimental/imdb/imdb/items.py b/examples/experimental/imdb/imdb/items.py new file mode 100644 index 000000000..03bb5c2c3 --- /dev/null +++ b/examples/experimental/imdb/imdb/items.py @@ -0,0 +1,12 @@ +# Define here the models for your scraped items +# +# See documentation in: +# http://doc.scrapy.org/topics/items.html + +from scrapy.item import Item, Field + +class ImdbItem(Item): + # define the fields for your item here like: + # name = Field() + title = Field() + url = Field() diff --git a/examples/experimental/imdb/imdb/pipelines.py b/examples/experimental/imdb/imdb/pipelines.py new file mode 100644 index 000000000..e60714159 --- /dev/null +++ b/examples/experimental/imdb/imdb/pipelines.py @@ -0,0 +1,8 @@ +# Define your item pipelines here +# +# Don't forget to add your pipeline to the ITEM_PIPELINES setting +# See: http://doc.scrapy.org/topics/item-pipeline.html + +class ImdbPipeline(object): + def process_item(self, spider, item): + return item diff --git a/examples/experimental/imdb/imdb/settings.py b/examples/experimental/imdb/imdb/settings.py new file mode 100644 index 000000000..de026dc14 --- /dev/null +++ b/examples/experimental/imdb/imdb/settings.py @@ -0,0 +1,20 @@ +# Scrapy settings for imdb project +# +# For simplicity, this file contains only the most important settings by +# default. All the other settings are documented here: +# +# http://doc.scrapy.org/topics/settings.html +# +# Or you can copy and paste them from where they're defined in Scrapy: +# +# scrapy/conf/default_settings.py +# + +BOT_NAME = 'imdb' +BOT_VERSION = '1.0' + +SPIDER_MODULES = ['imdb.spiders'] +NEWSPIDER_MODULE = 'imdb.spiders' +DEFAULT_ITEM_CLASS = 'imdb.items.ImdbItem' +USER_AGENT = '%s/%s' % (BOT_NAME, BOT_VERSION) + diff --git a/examples/experimental/imdb/imdb/spiders/__init__.py b/examples/experimental/imdb/imdb/spiders/__init__.py new file mode 100644 index 000000000..5065ccba5 --- /dev/null +++ b/examples/experimental/imdb/imdb/spiders/__init__.py @@ -0,0 +1,8 @@ +# This package will contain the spiders of your Scrapy project +# +# To create the first spider for your project use this command: +# +# scrapy-ctl.py genspider myspider myspider-domain.com +# +# For more info see: +# http://doc.scrapy.org/topics/spiders.html diff --git a/examples/experimental/imdb/imdb/spiders/imdb_site.py b/examples/experimental/imdb/imdb/spiders/imdb_site.py new file mode 100644 index 000000000..5c3f3c06b --- /dev/null +++ b/examples/experimental/imdb/imdb/spiders/imdb_site.py @@ -0,0 +1,140 @@ +from scrapy.http import Request +from scrapy.selector import HtmlXPathSelector +from scrapy.contrib.loader import XPathItemLoader +from scrapy.contrib_exp.crawlspider import CrawlSpider, Rule +from scrapy.contrib_exp.crawlspider.reqext import SgmlRequestExtractor +from scrapy.contrib_exp.crawlspider.reqproc import Canonicalize, \ + FilterDupes, FilterUrl +from scrapy.utils.url import urljoin_rfc + +from imdb.items import ImdbItem, Field + +from itertools import chain, imap, izip + +class UsaOpeningWeekMovie(ImdbItem): + pass + +class UsaTopWeekMovie(ImdbItem): + pass + +class Top250Movie(ImdbItem): + rank = Field() + rating = Field() + year = Field() + votes = Field() + +class MovieItem(ImdbItem): + release_date = Field() + tagline = Field() + + +class ImdbSiteSpider(CrawlSpider): + domain_name = 'imdb.com' + start_urls = ['http://www.imdb.com/'] + + # extract requests using this classes from urls matching 'follow' flag + request_extractors = [ + SgmlRequestExtractor(tags=['a'], attrs=['href']), + ] + + # process requests using this classes from urls matching 'follow' flag + request_processors = [ + Canonicalize(), + FilterDupes(), + FilterUrl(deny=r'/tt\d+/$'), # deny movie url as we will dispatch + # manually the movie requests + ] + + # include domain bit for demo purposes + rules = ( + # these two rules expects requests from start url + Rule(r'imdb.com/nowplaying/$', 'parse_now_playing'), + Rule(r'imdb.com/chart/top$', 'parse_top_250'), + # this rule will parse requests manually dispatched + Rule(r'imdb.com/title/tt\d+/$', 'parse_movie_info'), + ) + + def parse_now_playing(self, response): + """Scrapes USA openings this week and top 10 in week""" + self.log("Parsing USA Top Week") + hxs = HtmlXPathSelector(response) + + _urljoin = lambda url: self._urljoin(response, url) + + # + # openings this week + # + openings = hxs.select('//table[@class="movies"]//a[@class="title"]') + boxoffice = hxs.select('//table[@class="boxoffice movies"]//a[@class="title"]') + + opening_titles = openings.select('text()').extract() + opening_urls = imap(_urljoin, openings.select('@href').extract()) + + box_titles = boxoffice.select('text()').extract() + box_urls = imap(_urljoin, boxoffice.select('@href').extract()) + + # items + opening_items = (UsaOpeningWeekMovie(title=title, url=url) + for (title, url) + in izip(opening_titles, opening_urls)) + + box_items = (UsaTopWeekMovie(title=title, url=url) + for (title, url) + in izip(box_titles, box_urls)) + + # movie requests + requests = imap(self.make_requests_from_url, + chain(opening_urls, box_urls)) + + return chain(opening_items, box_items, requests) + + def parse_top_250(self, response): + """Scrapes movies from top 250 list""" + self.log("Parsing Top 250") + hxs = HtmlXPathSelector(response) + + # scrap each row in the table + rows = hxs.select('//div[@id="main"]/table/tr//a/ancestor::tr') + for row in rows: + fields = row.select('td//text()').extract() + url, = row.select('td//a/@href').extract() + url = self._urljoin(response, url) + + item = Top250Movie() + item['title'] = fields[2] + item['url'] = url + item['rank'] = fields[0] + item['rating'] = fields[1] + item['year'] = fields[3] + item['votes'] = fields[4] + + # scrapped top250 item + yield item + # fetch movie + yield self.make_requests_from_url(url) + + def parse_movie_info(self, response): + """Scrapes movie information""" + self.log("Parsing Movie Info") + hxs = HtmlXPathSelector(response) + selector = hxs.select('//div[@class="maindetails"]') + + item = MovieItem() + # set url + item['url'] = response.url + + # use item loader for other attributes + l = XPathItemLoader(item=item, selector=selector) + l.add_xpath('title', './/h1/text()') + l.add_xpath('release_date', './/h5[text()="Release Date:"]' + '/following-sibling::div/text()') + l.add_xpath('tagline', './/h5[text()="Tagline:"]' + '/following-sibling::div/text()') + + yield l.load_item() + + def _urljoin(self, response, url): + """Helper to convert relative urls to absolute""" + return urljoin_rfc(response.url, url, response.encoding) + +SPIDER = ImdbSiteSpider() diff --git a/examples/experimental/imdb/scrapy-ctl.py b/examples/experimental/imdb/scrapy-ctl.py new file mode 100644 index 000000000..df57621b3 --- /dev/null +++ b/examples/experimental/imdb/scrapy-ctl.py @@ -0,0 +1,7 @@ +#!/usr/bin/env python + +import os +os.environ.setdefault('SCRAPY_SETTINGS_MODULE', 'imdb.settings') + +from scrapy.command.cmdline import execute +execute() diff --git a/scrapy/__init__.py b/scrapy/__init__.py index 0051aa75e..6f8ca15e6 100644 --- a/scrapy/__init__.py +++ b/scrapy/__init__.py @@ -2,8 +2,8 @@ Scrapy - a screen scraping framework written in Python """ -version_info = (0, 8, 0, '', 0) -__version__ = "0.8" +version_info = (0, 9, 0, 'dev') +__version__ = "0.9-dev" import sys, os diff --git a/scrapy/command/cmdline.py b/scrapy/command/cmdline.py index 25f531cea..1a190549d 100644 --- a/scrapy/command/cmdline.py +++ b/scrapy/command/cmdline.py @@ -7,20 +7,14 @@ import cProfile import scrapy from scrapy import log -from scrapy.spider import spiders from scrapy.xlib import lsprofcalltree from scrapy.conf import settings from scrapy.command.models import ScrapyCommand +from scrapy.utils.signal import send_catch_log -# This dict holds information about the executed command for later use -command_executed = {} - -def _save_command_executed(cmdname, cmd, args, opts): - """Save command executed info for later reference""" - command_executed['name'] = cmdname - command_executed['class'] = cmd - command_executed['args'] = args[:] - command_executed['opts'] = opts.__dict__.copy() +# Signal that carries information about the command which was executed +# args: cmdname, cmdobj, args, opts +command_executed = object() def _find_commands(dir): try: @@ -127,7 +121,8 @@ def execute(argv=None): sys.exit(2) del args[0] # remove command name from args - _save_command_executed(cmdname, cmd, args, opts) + send_catch_log(signal=command_executed, cmdname=cmdname, cmdobj=cmd, \ + args=args, opts=opts) from scrapy.core.manager import scrapymanager scrapymanager.configure(control_reactor=True) ret = _run_command(cmd, args, opts) @@ -136,23 +131,25 @@ def execute(argv=None): def _run_command(cmd, args, opts): if opts.profile or opts.lsprof: - if opts.profile: - log.msg("writing cProfile stats to %r" % opts.profile) - if opts.lsprof: - log.msg("writing lsprof stats to %r" % opts.lsprof) - loc = locals() - p = cProfile.Profile() - p.runctx('ret = cmd.run(args, opts)', globals(), loc) - if opts.profile: - p.dump_stats(opts.profile) - k = lsprofcalltree.KCacheGrind(p) - if opts.lsprof: - with open(opts.lsprof, 'w') as f: - k.output(f) - ret = loc['ret'] + return _run_command_profiled(cmd, args, opts) else: - ret = cmd.run(args, opts) - return ret + return cmd.run(args, opts) + +def _run_command_profiled(cmd, args, opts): + if opts.profile: + log.msg("writing cProfile stats to %r" % opts.profile) + if opts.lsprof: + log.msg("writing lsprof stats to %r" % opts.lsprof) + loc = locals() + p = cProfile.Profile() + p.runctx('ret = cmd.run(args, opts)', globals(), loc) + if opts.profile: + p.dump_stats(opts.profile) + k = lsprofcalltree.KCacheGrind(p) + if opts.lsprof: + with open(opts.lsprof, 'w') as f: + k.output(f) + return loc['ret'] if __name__ == '__main__': execute() diff --git a/scrapy/conf/default_settings.py b/scrapy/conf/default_settings.py index 8892e41b4..3ee377a44 100644 --- a/scrapy/conf/default_settings.py +++ b/scrapy/conf/default_settings.py @@ -122,6 +122,8 @@ MYSQL_CONNECTION_SETTINGS = {} NEWSPIDER_MODULE = '' +RANDOMIZE_DOWNLOAD_DELAY = True + REDIRECT_MAX_METAREFRESH_DELAY = 100 REDIRECT_MAX_TIMES = 20 # uses Firefox default setting REDIRECT_PRIORITY_ADJUST = +2 @@ -150,7 +152,7 @@ SCHEDULER_MIDDLEWARES_BASE = { 'scrapy.contrib.schedulermiddleware.duplicatesfilter.DuplicatesFilterMiddleware': 500, } -SCHEDULER_ORDER = 'BFO' # available orders: BFO (default), DFO +SCHEDULER_ORDER = 'DFO' SPIDER_MANAGER_CLASS = 'scrapy.contrib.spidermanager.TwistedPluginSpiderManager' diff --git a/scrapy/contrib/groupsettings.py b/scrapy/contrib/groupsettings.py deleted file mode 100644 index 4bad4100b..000000000 --- a/scrapy/contrib/groupsettings.py +++ /dev/null @@ -1,26 +0,0 @@ -""" -Extensions to override scrapy settings with per-group settings according to the -group the spider belongs to. It only overrides the settings when running the -crawl command with *only one domain as argument*. -""" - -from scrapy.conf import settings -from scrapy.core.exceptions import NotConfigured -from scrapy.command.cmdline import command_executed - -class GroupSettings(object): - - def __init__(self): - if not settings.getbool("GROUPSETTINGS_ENABLED"): - raise NotConfigured - - if command_executed and command_executed['name'] == 'crawl': - mod = __import__(settings['GROUPSETTINGS_MODULE'], {}, {}, ['']) - args = command_executed['args'] - if len(args) == 1 and not args[0].startswith('http://'): - domain = args[0] - settings.overrides.update(mod.default_settings) - for group, domains in mod.group_spiders.iteritems(): - if domain in domains: - settings.overrides.update(mod.group_settings.get(group, {})) - diff --git a/scrapy/contrib_exp/crawlspider/__init__.py b/scrapy/contrib_exp/crawlspider/__init__.py new file mode 100644 index 000000000..03173eb38 --- /dev/null +++ b/scrapy/contrib_exp/crawlspider/__init__.py @@ -0,0 +1,4 @@ +"""CrawlSpider v2""" + +from .rules import Rule +from .spider import CrawlSpider diff --git a/scrapy/contrib_exp/crawlspider/matchers.py b/scrapy/contrib_exp/crawlspider/matchers.py new file mode 100644 index 000000000..3ef259c67 --- /dev/null +++ b/scrapy/contrib_exp/crawlspider/matchers.py @@ -0,0 +1,61 @@ +""" +Request/Response Matchers + +Perform evaluation to Request or Response attributes +""" + +import re + +class BaseMatcher(object): + """Base matcher. Returns True by default.""" + + def matches_request(self, request): + """Performs Request Matching""" + return True + + def matches_response(self, response): + """Performs Response Matching""" + return True + + +class UrlMatcher(BaseMatcher): + """Matches URL attribute""" + + def __init__(self, url): + """Initialize url attribute""" + self._url = url + + def matches_url(self, url): + """Returns True if given url is equal to matcher's url""" + return self._url == url + + def matches_request(self, request): + """Returns True if Request's url matches initial url""" + return self.matches_url(request.url) + + def matches_response(self, response): + """Returns True if Response's url matches initial url""" + return self.matches_url(response.url) + + +class UrlRegexMatcher(UrlMatcher): + """Matches URL using regular expression""" + + def __init__(self, regex, flags=0): + """Initialize regular expression""" + self._regex = re.compile(regex, flags) + + def matches_url(self, url): + """Returns True if url matches regular expression""" + return self._regex.search(url) is not None + + +class UrlListMatcher(UrlMatcher): + """Matches if URL is in List""" + + def __init__(self, urls): + self._urls = urls + + def matches_url(self, url): + """Returns True if url is in urls list""" + return url in self._urls diff --git a/scrapy/contrib_exp/crawlspider/reqext.py b/scrapy/contrib_exp/crawlspider/reqext.py new file mode 100644 index 000000000..bb7318f79 --- /dev/null +++ b/scrapy/contrib_exp/crawlspider/reqext.py @@ -0,0 +1,117 @@ +"""Request Extractors""" +from scrapy.http import Request +from scrapy.selector import HtmlXPathSelector +from scrapy.utils.misc import arg_to_iter +from scrapy.utils.python import FixedSGMLParser, str_to_unicode +from scrapy.utils.url import safe_url_string, urljoin_rfc + +from itertools import ifilter + + +class BaseSgmlRequestExtractor(FixedSGMLParser): + """Base SGML Request Extractor""" + + def __init__(self, tag='a', attr='href'): + """Initialize attributes""" + FixedSGMLParser.__init__(self) + + self.scan_tag = tag if callable(tag) else lambda t: t == tag + self.scan_attr = attr if callable(attr) else lambda a: a == attr + self.current_request = None + + def extract_requests(self, response): + """Returns list of requests extracted from response""" + return self._extract_requests(response.body, response.url, + response.encoding) + + def _extract_requests(self, response_text, response_url, response_encoding): + """Extract requests with absolute urls""" + self.reset() + self.feed(response_text) + self.close() + + base_url = self.base_url if self.base_url else response_url + self._make_absolute_urls(base_url, response_encoding) + self._fix_link_text_encoding(response_encoding) + + return self.requests + + def _make_absolute_urls(self, base_url, encoding): + """Makes all request's urls absolute""" + for req in self.requests: + url = req.url + # make absolute url + url = urljoin_rfc(base_url, url, encoding) + url = safe_url_string(url, encoding) + # replace in-place request's url + req.url = url + + def _fix_link_text_encoding(self, encoding): + """Convert link_text to unicode for each request""" + for req in self.requests: + req.meta.setdefault('link_text', '') + req.meta['link_text'] = str_to_unicode(req.meta['link_text'], + encoding) + + def reset(self): + """Reset state""" + FixedSGMLParser.reset(self) + self.requests = [] + self.base_url = None + + def unknown_starttag(self, tag, attrs): + """Process unknown start tag""" + if 'base' == tag: + self.base_url = dict(attrs).get('href') + + _matches = lambda (attr, value): self.scan_attr(attr) \ + and value is not None + if self.scan_tag(tag): + for attr, value in ifilter(_matches, attrs): + req = Request(url=value) + self.requests.append(req) + self.current_request = req + + def unknown_endtag(self, tag): + """Process unknown end tag""" + self.current_request = None + + def handle_data(self, data): + """Process data""" + current = self.current_request + if current and not 'link_text' in current.meta: + current.meta['link_text'] = data.strip() + + +class SgmlRequestExtractor(BaseSgmlRequestExtractor): + """SGML Request Extractor""" + + def __init__(self, tags=None, attrs=None): + """Initialize with custom tag & attribute function checkers""" + # defaults + tags = tuple(tags) if tags else ('a', 'area') + attrs = tuple(attrs) if attrs else ('href', ) + + tag_func = lambda x: x in tags + attr_func = lambda x: x in attrs + BaseSgmlRequestExtractor.__init__(self, tag=tag_func, attr=attr_func) + +# TODO: move to own file +class XPathRequestExtractor(SgmlRequestExtractor): + """SGML Request Extractor with XPath restriction""" + + def __init__(self, restrict_xpaths, tags=None, attrs=None): + """Initialize XPath restrictions""" + self.restrict_xpaths = tuple(arg_to_iter(restrict_xpaths)) + SgmlRequestExtractor.__init__(self, tags, attrs) + + def extract_requests(self, response): + """Restrict to XPath regions""" + hxs = HtmlXPathSelector(response) + fragments = (''.join( + html_frag for html_frag in hxs.select(xpath).extract() + ) for xpath in self.restrict_xpaths) + html_slice = ''.join(html_frag for html_frag in fragments) + return self._extract_requests(html_slice, response.url, + response.encoding) + diff --git a/scrapy/contrib_exp/crawlspider/reqgen.py b/scrapy/contrib_exp/crawlspider/reqgen.py new file mode 100644 index 000000000..3858fbcf7 --- /dev/null +++ b/scrapy/contrib_exp/crawlspider/reqgen.py @@ -0,0 +1,27 @@ +"""Request Generator""" +from itertools import imap + +class RequestGenerator(object): + """Extracto and process requests from response""" + + def __init__(self, req_extractors, req_processors, callback, spider=None): + """Initialize attributes""" + self._request_extractors = req_extractors + self._request_processors = req_processors + #TODO: resolve callback? + self._callback = callback + + def generate_requests(self, response): + """Extract and process new requests from response. + Attach callback to each request as default callback.""" + requests = [] + for ext in self._request_extractors: + requests.extend(ext.extract_requests(response)) + + for proc in self._request_processors: + requests = proc(requests) + + # return iterator + # @@@ creates new Request object with callback + return imap(lambda r: r.replace(callback=self._callback), requests) + diff --git a/scrapy/contrib_exp/crawlspider/reqproc.py b/scrapy/contrib_exp/crawlspider/reqproc.py new file mode 100644 index 000000000..d39399b20 --- /dev/null +++ b/scrapy/contrib_exp/crawlspider/reqproc.py @@ -0,0 +1,111 @@ +"""Request Processors""" +from scrapy.utils.misc import arg_to_iter +from scrapy.utils.url import canonicalize_url, url_is_from_any_domain + +from itertools import ifilter, imap + +import re + +class Canonicalize(object): + """Canonicalize Request Processor""" + def _replace_url(self, req): + # replace in-place + req.url = canonicalize_url(req.url) + return req + + def __call__(self, requests): + """Canonicalize all requests' urls""" + return imap(self._replace_url, requests) + + +class FilterDupes(object): + """Filter duplicate Requests""" + + def __init__(self, *attributes): + """Initialize comparison attributes""" + self._attributes = tuple(attributes) if attributes \ + else tuple(['url']) + + def _equal_attr(self, obj1, obj2, attr): + return getattr(obj1, attr) == getattr(obj2, attr) + + def _requests_equal(self, req1, req2): + """Attribute comparison helper""" + # look for not equal attribute + _not_equal = lambda attr: not self._equal_attr(req1, req2, attr) + for attr in ifilter(_not_equal, self._attributes): + return False + # all attributes equal + return True + + def _request_in(self, request, requests_seen): + """Check if request is in given requests seen list""" + _req_seen = lambda r: self._requests_equal(r, request) + for seen in ifilter(_req_seen, requests_seen): + return True + # request not seen + return False + + def __call__(self, requests): + """Filter seen requests""" + # per-call duplicates filter + self.requests_seen = set() + _not_seen = lambda r: not self._request_in(r, self.requests_seen) + for req in ifilter(_not_seen, requests): + yield req + # registry seen request + self.requests_seen.add(req) + + +class FilterDomain(object): + """Filter request's domain""" + + def __init__(self, allow=(), deny=()): + """Initialize allow/deny attributes""" + self.allow = tuple(arg_to_iter(allow)) + self.deny = tuple(arg_to_iter(deny)) + + def __call__(self, requests): + """Filter domains""" + processed = (req for req in requests) + + if self.allow: + processed = (req for req in requests + if url_is_from_any_domain(req.url, self.allow)) + if self.deny: + processed = (req for req in requests + if not url_is_from_any_domain(req.url, self.deny)) + + return processed + + +class FilterUrl(object): + """Filter request's url""" + + def __init__(self, allow=(), deny=()): + """Initialize allow/deny attributes""" + _re_type = type(re.compile('', 0)) + + self.allow_res = [x if isinstance(x, _re_type) else re.compile(x) + for x in arg_to_iter(allow)] + self.deny_res = [x if isinstance(x, _re_type) else re.compile(x) + for x in arg_to_iter(deny)] + + def __call__(self, requests): + """Filter request's url based on allow/deny rules""" + #TODO: filter valid urls here? + processed = (req for req in requests) + + if self.allow_res: + processed = (req for req in requests + if self._matches(req.url, self.allow_res)) + if self.deny_res: + processed = (req for req in requests + if not self._matches(req.url, self.deny_res)) + + return processed + + def _matches(self, url, regexs): + """Returns True if url matches any regex in given list""" + return any(r.search(url) for r in regexs) + diff --git a/scrapy/contrib_exp/crawlspider/rules.py b/scrapy/contrib_exp/crawlspider/rules.py new file mode 100644 index 000000000..ff1691ad0 --- /dev/null +++ b/scrapy/contrib_exp/crawlspider/rules.py @@ -0,0 +1,100 @@ +"""Crawler Rules""" +from scrapy.http import Request +from scrapy.http import Response + +from functools import partial +from itertools import ifilter + +from .matchers import BaseMatcher +# default strint-to-matcher class +from .matchers import UrlRegexMatcher + +class CompiledRule(object): + """Compiled version of Rule""" + def __init__(self, matcher, callback=None, follow=False): + """Initialize attributes checking type""" + assert isinstance(matcher, BaseMatcher) + assert callback is None or callable(callback) + assert isinstance(follow, bool) + + self.matcher = matcher + self.callback = callback + self.follow = follow + + +class Rule(object): + """Crawler Rule""" + def __init__(self, matcher=None, callback=None, follow=False, **kwargs): + """Store attributes""" + self.matcher = matcher + self.callback = callback + self.cb_kwargs = kwargs if kwargs else {} + self.follow = True if follow else False + + if self.callback is None and self.follow is False: + raise ValueError("Rule must either have a callback or " + "follow=True: %r" % self) + + def __repr__(self): + return "Rule(matcher=%r, callback=%r, follow=%r, **%r)" \ + % (self.matcher, self.callback, self.follow, self.cb_kwargs) + + +class RulesManager(object): + """Rules Manager""" + def __init__(self, rules, spider, default_matcher=UrlRegexMatcher): + """Initialize rules using spider and default matcher""" + self._rules = tuple() + + # compile absolute/relative-to-spider callbacks""" + for rule in rules: + # prepare matcher + if rule.matcher is None: + # instance BaseMatcher by default + matcher = BaseMatcher() + elif isinstance(rule.matcher, BaseMatcher): + matcher = rule.matcher + else: + # matcher not BaseMatcher, check for string + if isinstance(rule.matcher, basestring): + # instance default matcher + matcher = default_matcher(rule.matcher) + else: + raise ValueError('Not valid matcher given %r in %r' \ + % (rule.matcher, rule)) + + # prepare callback + if callable(rule.callback): + callback = rule.callback + elif not rule.callback is None: + # callback from spider + callback = getattr(spider, rule.callback) + + if not callable(callback): + raise AttributeError('Invalid callback %r can not be resolved' \ + % callback) + else: + callback = None + + if rule.cb_kwargs: + # build partial callback + callback = partial(callback, **rule.cb_kwargs) + + # append compiled rule to rules list + crule = CompiledRule(matcher, callback, follow=rule.follow) + self._rules += (crule, ) + + def get_rule_from_request(self, request): + """Returns first rule that matches given Request""" + _matches = lambda r: r.matcher.matches_request(request) + for rule in ifilter(_matches, self._rules): + # return first match of iterator + return rule + + def get_rule_from_response(self, response): + """Returns first rule that matches given Response""" + _matches = lambda r: r.matcher.matches_response(response) + for rule in ifilter(_matches, self._rules): + # return first match of iterator + return rule + diff --git a/scrapy/contrib_exp/crawlspider/spider.py b/scrapy/contrib_exp/crawlspider/spider.py new file mode 100644 index 000000000..730ad0e8d --- /dev/null +++ b/scrapy/contrib_exp/crawlspider/spider.py @@ -0,0 +1,69 @@ +"""CrawlSpider v2""" +from scrapy.spider import BaseSpider +from scrapy.utils.spider import iterate_spider_output + +from .matchers import UrlListMatcher +from .rules import Rule, RulesManager +from .reqext import SgmlRequestExtractor +from .reqgen import RequestGenerator +from .reqproc import Canonicalize, FilterDupes + +class CrawlSpider(BaseSpider): + """CrawlSpider v2""" + + request_extractors = None + request_processors = None + rules = [] + + def __init__(self): + """Initialize dispatcher""" + super(CrawlSpider, self).__init__() + + # auto follow start urls + if self.start_urls: + _matcher = UrlListMatcher(self.start_urls) + # append new rule using type from current self.rules + rules = self.rules + type(self.rules)([ + Rule(_matcher, follow=True) + ]) + else: + rules = self.rules + + # set defaults if not set + if self.request_extractors is None: + # default link extractor. Extracts all links from response + self.request_extractors = [ SgmlRequestExtractor() ] + + if self.request_processors is None: + # default proccessor. Filter duplicates requests + self.request_processors = [ FilterDupes() ] + + + # wrap rules + self._rulesman = RulesManager(rules, spider=self) + # generates new requests with given callback + self._reqgen = RequestGenerator(self.request_extractors, + self.request_processors, + callback=self.parse) + + def parse(self, response): + """Dispatch callback and generate requests""" + # get rule for response + rule = self._rulesman.get_rule_from_response(response) + + if rule: + # dispatch callback if set + if rule.callback: + output = iterate_spider_output(rule.callback(response)) + for req_or_item in output: + yield req_or_item + + if rule.follow: + for req in self._reqgen.generate_requests(response): + # only dispatch request if has matching rule + if self._rulesman.get_rule_from_request(req): + yield req + else: + self.log("No rule for response %s" % response, level=log.WARNING) + + diff --git a/scrapy/contrib_exp/pipeline/shoveitem.py b/scrapy/contrib_exp/pipeline/shoveitem.py deleted file mode 100644 index 869c62c58..000000000 --- a/scrapy/contrib_exp/pipeline/shoveitem.py +++ /dev/null @@ -1,55 +0,0 @@ -""" -A pipeline to persist objects using shove. - -Shove is a "new generation" shelve. For more information see: -http://pypi.python.org/pypi/shove -""" - -from string import Template - -from shove import Shove -from scrapy.xlib.pydispatch import dispatcher - -from scrapy import log -from scrapy.core import signals -from scrapy.conf import settings -from scrapy.core.exceptions import NotConfigured - -class ShoveItemPipeline(object): - - def __init__(self): - self.uritpl = settings['SHOVEITEM_STORE_URI'] - if not self.uritpl: - raise NotConfigured - self.opts = settings['SHOVEITEM_STORE_OPT'] or {} - self.stores = {} - - dispatcher.connect(self.spider_opened, signal=signals.spider_opened) - dispatcher.connect(self.spider_closed, signal=signals.spider_closed) - - def process_item(self, spider, item): - guid = str(item.guid) - - if guid in self.stores[spider]: - if self.stores[spider][guid] == item: - status = 'old' - else: - status = 'upd' - else: - status = 'new' - - if not status == 'old': - self.stores[spider][guid] = item - self.log(spider, item, status) - return item - - def spider_opened(self, spider): - uri = Template(self.uritpl).substitute(domain=spider.domain_name) - self.stores[spider] = Shove(uri, **self.opts) - - def spider_closed(self, spider): - self.stores[spider].sync() - - def log(self, spider, item, status): - log.msg("Shove (%s): Item guid=%s" % (status, item.guid), level=log.DEBUG, \ - spider=spider) diff --git a/scrapy/core/downloader/manager.py b/scrapy/core/downloader/manager.py index b53db5811..aec17b05e 100644 --- a/scrapy/core/downloader/manager.py +++ b/scrapy/core/downloader/manager.py @@ -2,6 +2,7 @@ Download web pages using asynchronous IO """ +import random from time import time from twisted.internet import reactor, defer @@ -20,15 +21,21 @@ class SpiderInfo(object): def __init__(self, download_delay=None, max_concurrent_requests=None): if download_delay is None: - self.download_delay = settings.getfloat('DOWNLOAD_DELAY') + self._download_delay = settings.getfloat('DOWNLOAD_DELAY') else: - self.download_delay = download_delay - if self.download_delay: + self._download_delay = float(download_delay) + if self._download_delay: self.max_concurrent_requests = 1 elif max_concurrent_requests is None: self.max_concurrent_requests = settings.getint('CONCURRENT_REQUESTS_PER_SPIDER') else: self.max_concurrent_requests = max_concurrent_requests + if self._download_delay and settings.getbool('RANDOMIZE_DOWNLOAD_DELAY'): + # same policy as wget --random-wait + self.random_delay_interval = (0.5*self._download_delay, \ + 1.5*self._download_delay) + else: + self.random_delay_interval = None self.active = set() self.queue = [] @@ -44,6 +51,12 @@ class SpiderInfo(object): # use self.active to include requests in the downloader middleware return len(self.active) > 2 * self.max_concurrent_requests + def download_delay(self): + if self.random_delay_interval: + return random.uniform(*self.random_delay_interval) + else: + return self._download_delay + def cancel_request_calls(self): for call in self.next_request_calls: call.cancel() @@ -99,8 +112,9 @@ class Downloader(object): # Delay queue processing if a download_delay is configured now = time() - if site.download_delay: - penalty = site.download_delay - now + site.lastseen + delay = site.download_delay() + if delay: + penalty = delay - now + site.lastseen if penalty > 0: d = defer.Deferred() d.addCallback(self._process_queue) diff --git a/scrapy/http/response/text.py b/scrapy/http/response/text.py index 1c13e729c..d42a89197 100644 --- a/scrapy/http/response/text.py +++ b/scrapy/http/response/text.py @@ -5,6 +5,7 @@ discovering (through HTTP headers) to base Response class. See documentation in docs/topics/request-response.rst """ +import codecs import re from scrapy.xlib.BeautifulSoup import UnicodeDammit @@ -64,7 +65,12 @@ class TextResponse(Response): if content_type: encoding = self._ENCODING_RE.search(content_type) if encoding: - return encoding.group(1) + enc = encoding.group(1) + try: + codecs.lookup(enc) # check if the encoding is valid + return enc + except LookupError: + pass @memoizemethod_noargs def body_as_unicode(self): diff --git a/scrapy/templates/project/module/pipelines.py.tmpl b/scrapy/templates/project/module/pipelines.py.tmpl index fa6f5ea6f..e3f89342d 100644 --- a/scrapy/templates/project/module/pipelines.py.tmpl +++ b/scrapy/templates/project/module/pipelines.py.tmpl @@ -4,5 +4,5 @@ # See: http://doc.scrapy.org/topics/item-pipeline.html class ${ProjectName}Pipeline(object): - def process_item(self, domain, item): + def process_item(self, spider, item): return item diff --git a/scrapy/templates/spiders/crawl.tmpl b/scrapy/templates/spiders/crawl.tmpl index 2d4f7fce6..18fd0ec42 100644 --- a/scrapy/templates/spiders/crawl.tmpl +++ b/scrapy/templates/spiders/crawl.tmpl @@ -10,15 +10,15 @@ class $classname(CrawlSpider): start_urls = ['http://www.$site/'] rules = ( - Rule(SgmlLinkExtractor(allow=(r'Items/', )), 'parse_item', follow=True), + Rule(SgmlLinkExtractor(allow=r'Items/'), callback='parse_item', follow=True), ) def parse_item(self, response): - xs = HtmlXPathSelector(response) + hxs = HtmlXPathSelector(response) i = ${ProjectName}Item() - #i['site_id'] = xs.select('//input[@id="sid"]/@value').extract() - #i['name'] = xs.select('//div[@id="name"]').extract() - #i['description'] = xs.select('//div[@id="description"]').extract() + #i['site_id'] = hxs.select('//input[@id="sid"]/@value').extract() + #i['name'] = hxs.select('//div[@id="name"]').extract() + #i['description'] = hxs.select('//div[@id="description"]').extract() return i SPIDER = $classname() diff --git a/scrapy/tests/test_contrib_exp_crawlspider_matchers.py b/scrapy/tests/test_contrib_exp_crawlspider_matchers.py new file mode 100644 index 000000000..4cb832aa1 --- /dev/null +++ b/scrapy/tests/test_contrib_exp_crawlspider_matchers.py @@ -0,0 +1,94 @@ +from twisted.trial import unittest + +from scrapy.http import Request +from scrapy.http import Response + +from scrapy.contrib_exp.crawlspider.matchers import BaseMatcher +from scrapy.contrib_exp.crawlspider.matchers import UrlMatcher +from scrapy.contrib_exp.crawlspider.matchers import UrlRegexMatcher +from scrapy.contrib_exp.crawlspider.matchers import UrlListMatcher + +import re + +class MatchersTest(unittest.TestCase): + + def setUp(self): + pass + + def test_base_matcher(self): + matcher = BaseMatcher() + + request = Request('http://example.com') + response = Response('http://example.com') + + self.assertTrue(matcher.matches_request(request)) + self.assertTrue(matcher.matches_response(response)) + + def test_url_matcher(self): + matcher = UrlMatcher('http://example.com') + + request = Request('http://example.com') + response = Response('http://example.com') + + self.failUnless(matcher.matches_request(request)) + self.failUnless(matcher.matches_request(response)) + + request = Request('http://example2.com') + response = Response('http://example2.com') + + self.failIf(matcher.matches_request(request)) + self.failIf(matcher.matches_request(response)) + + def test_url_regex_matcher(self): + matcher = UrlRegexMatcher(r'sample') + urls = ( + 'http://example.com/sample1.html', + 'http://example.com/sample2.html', + 'http://example.com/sample3.html', + 'http://example.com/sample4.html', + ) + for url in urls: + request, response = Request(url), Response(url) + self.failUnless(matcher.matches_request(request)) + self.failUnless(matcher.matches_response(response)) + + matcher = UrlRegexMatcher(r'sample_fail') + for url in urls: + request, response = Request(url), Response(url) + self.failIf(matcher.matches_request(request)) + self.failIf(matcher.matches_response(response)) + + matcher = UrlRegexMatcher(r'SAMPLE\d+', re.IGNORECASE) + for url in urls: + request, response = Request(url), Response(url) + self.failUnless(matcher.matches_request(request)) + self.failUnless(matcher.matches_response(response)) + + def test_url_list_matcher(self): + urls = ( + 'http://example.com/sample1.html', + 'http://example.com/sample2.html', + 'http://example.com/sample3.html', + 'http://example.com/sample4.html', + ) + urls2 = ( + 'http://example.com/sample5.html', + 'http://example.com/sample6.html', + 'http://example.com/sample7.html', + 'http://example.com/sample8.html', + 'http://example.com/', + ) + matcher = UrlListMatcher(urls) + + # match urls + for url in urls: + request, response = Request(url), Response(url) + self.failUnless(matcher.matches_request(request)) + self.failUnless(matcher.matches_response(response)) + + # non-match urls + for url in urls2: + request, response = Request(url), Response(url) + self.failIf(matcher.matches_request(request)) + self.failIf(matcher.matches_response(response)) + diff --git a/scrapy/tests/test_contrib_exp_crawlspider_reqext.py b/scrapy/tests/test_contrib_exp_crawlspider_reqext.py new file mode 100644 index 000000000..d0d0d1b5f --- /dev/null +++ b/scrapy/tests/test_contrib_exp_crawlspider_reqext.py @@ -0,0 +1,137 @@ +from twisted.trial import unittest + +from scrapy.http import Request +from scrapy.http import HtmlResponse +from scrapy.tests import get_testdata + +from scrapy.contrib_exp.crawlspider.reqext import BaseSgmlRequestExtractor +from scrapy.contrib_exp.crawlspider.reqext import SgmlRequestExtractor +from scrapy.contrib_exp.crawlspider.reqext import XPathRequestExtractor + +class AbstractRequestExtractorTest(unittest.TestCase): + + def _requests_equals(self, list1, list2): + """Compares request's urls and link_text""" + for (r1, r2) in zip(list1, list2): + if r1.url != r2.url: + return False + if r1.meta['link_text'] != r2.meta['link_text']: + return False + # all equal + return True + + +class RequestExtractorTest(AbstractRequestExtractorTest): + + def test_basic(self): + base_url = 'http://example.org/somepage/index.html' + html = """Page title<title> + <body><p><a href="item/12.html">Item 12</a></p> + <p><a href="/about.html">About us</a></p> + <img src="/logo.png" alt="Company logo (not a link)" /> + <p><a href="../othercat.html">Other category</a></p> + <p><a href="/" /></p></body></html>""" + requests = [ + Request('http://example.org/somepage/item/12.html', + meta={'link_text': 'Item 12'}), + Request('http://example.org/about.html', + meta={'link_text': 'About us'}), + Request('http://example.org/othercat.html', + meta={'link_text': 'Other category'}), + Request('http://example.org/', + meta={'link_text': ''}), + ] + + response = HtmlResponse(base_url, body=html) + reqx = BaseSgmlRequestExtractor() # default: tag=a, attr=href + + self.failUnless( + self._requests_equals(requests, reqx.extract_requests(response)) + ) + + def test_base_url(self): + html = """<html><head><title>Page title<title> + <base href="http://otherdomain.com/base/" /> + <body><p><a href="item/12.html">Item 12</a></p> + </body></html>""" + response = HtmlResponse("http://example.org/somepage/index.html", + body=html) + reqx = BaseSgmlRequestExtractor() + + self.failUnless( + self._requests_equals(reqx.extract_requests(response), + [ Request('http://otherdomain.com/base/item/12.html', + meta={'link_text': 'Item 12'}) ] + ) + ) + + def test_extraction_encoding(self): + #TODO: use own fixtures + body = get_testdata('link_extractor', 'linkextractor_noenc.html') + response_utf8 = HtmlResponse(url='http://example.com/utf8', body=body, + headers={'Content-Type': ['text/html; charset=utf-8']}) + response_noenc = HtmlResponse(url='http://example.com/noenc', + body=body) + body = get_testdata('link_extractor', 'linkextractor_latin1.html') + response_latin1 = HtmlResponse(url='http://example.com/latin1', + body=body) + + reqx = BaseSgmlRequestExtractor() + self.failUnless( + self._requests_equals( + reqx.extract_requests(response_utf8), + [ Request(url='http://example.com/sample_%C3%B1.html', + meta={'link_text': ''}), + Request(url='http://example.com/sample_%E2%82%AC.html', + meta={'link_text': + 'sample \xe2\x82\xac text'.decode('utf-8')}) ] + ) + ) + + self.failUnless( + self._requests_equals( + reqx.extract_requests(response_noenc), + [ Request(url='http://example.com/sample_%C3%B1.html', + meta={'link_text': ''}), + Request(url='http://example.com/sample_%E2%82%AC.html', + meta={'link_text': + 'sample \xe2\x82\xac text'.decode('utf-8')}) ] + ) + ) + + self.failUnless( + self._requests_equals( + reqx.extract_requests(response_latin1), + [ Request(url='http://example.com/sample_%F1.html', + meta={'link_text': ''}), + Request(url='http://example.com/sample_%E1.html', + meta={'link_text': + 'sample \xe1 text'.decode('latin1')}) ] + ) + ) + + +class SgmlRequestExtractorTest(AbstractRequestExtractorTest): + pass + + +class XPathRequestExtractorTest(AbstractRequestExtractorTest): + + def setUp(self): + # TODO: use own fixtures + body = get_testdata('link_extractor', 'sgml_linkextractor.html') + self.response = HtmlResponse(url='http://example.com/index', body=body) + + + def test_restrict_xpaths(self): + reqx = XPathRequestExtractor('//div[@id="subwrapper"]') + self.failUnless( + self._requests_equals( + reqx.extract_requests(self.response), + [ Request(url='http://example.com/sample1.html', + meta={'link_text': ''}), + Request(url='http://example.com/sample2.html', + meta={'link_text': 'sample 2'}) ] + ) + ) + diff --git a/scrapy/tests/test_contrib_exp_crawlspider_reqgen.py b/scrapy/tests/test_contrib_exp_crawlspider_reqgen.py new file mode 100644 index 000000000..67aca2387 --- /dev/null +++ b/scrapy/tests/test_contrib_exp_crawlspider_reqgen.py @@ -0,0 +1,128 @@ +from twisted.internet import defer +from twisted.trial import unittest + +from scrapy.http import Request +from scrapy.http import HtmlResponse +from scrapy.utils.python import equal_attributes + +from scrapy.contrib_exp.crawlspider.reqext import SgmlRequestExtractor +from scrapy.contrib_exp.crawlspider.reqgen import RequestGenerator +from scrapy.contrib_exp.crawlspider.reqproc import Canonicalize +from scrapy.contrib_exp.crawlspider.reqproc import FilterDomain +from scrapy.contrib_exp.crawlspider.reqproc import FilterUrl +from scrapy.contrib_exp.crawlspider.reqproc import FilterDupes + +class RequestGeneratorTest(unittest.TestCase): + + def setUp(self): + url = 'http://example.org/somepage/index.html' + html = """<html><head><title>Page title<title> + <body><p><a href="item/12.html">Item 12</a></p> + <p><a href="/about.html">About us</a></p> + <img src="/logo.png" alt="Company logo (not a link)" /> + <p><a href="../othercat.html">Other category</a></p> + <p><a href="/" /></p></body></html>""" + + self.response = HtmlResponse(url, body=html) + self.deferred = defer.Deferred() + self.requests = [ + Request('http://example.org/somepage/item/12.html', + meta={'link_text': 'Item 12'}), + Request('http://example.org/about.html', + meta={'link_text': 'About us'}), + Request('http://example.org/othercat.html', + meta={'link_text': 'Other category'}), + Request('http://example.org/', + meta={'link_text': ''}), + ] + + def _equal_requests_list(self, list1, list2): + list1 = list(list1) + list2 = list(list2) + if not len(list1) == len(list2): + return False + + for (req1, req2) in zip(list1, list2): + if not equal_attributes(req1, req2, ['url']): + return False + return True + + def test_basic(self): + reqgen = RequestGenerator([], [], callback=self.deferred) + # returns generator + requests = reqgen.generate_requests(self.response) + self.failUnlessEqual(list(requests), []) + + def test_request_extractor(self): + extractors = [ + SgmlRequestExtractor() + ] + + # extract all requests + reqgen = RequestGenerator(extractors, [], callback=self.deferred) + requests = reqgen.generate_requests(self.response) + self.failUnless(self._equal_requests_list(requests, self.requests)) + + for req in requests: + # check callback + self.failUnlessEqual(req.deferred, self.deferred) + + def test_request_processor(self): + extractors = [ + SgmlRequestExtractor() + ] + + processors = [ + Canonicalize(), + FilterDupes(), + ] + + reqgen = RequestGenerator(extractors, processors, callback=self.deferred) + requests = reqgen.generate_requests(self.response) + self.failUnless(self._equal_requests_list(requests, self.requests)) + + # filter domain + processors = [ + Canonicalize(), + FilterDupes(), + FilterDomain(deny='example.org'), + ] + + reqgen = RequestGenerator(extractors, processors, callback=self.deferred) + requests = reqgen.generate_requests(self.response) + self.failUnlessEqual(list(requests), []) + + # filter url + processors = [ + Canonicalize(), + FilterDupes(), + FilterUrl(deny=(r'about', r'othercat')), + ] + + reqgen = RequestGenerator(extractors, processors, callback=self.deferred) + requests = reqgen.generate_requests(self.response) + + self.failUnless(self._equal_requests_list(requests, [ + Request('http://example.org/somepage/item/12.html', + meta={'link_text': 'Item 12'}), + Request('http://example.org/', + meta={'link_text': ''}), + ])) + + processors = [ + Canonicalize(), + FilterDupes(), + FilterUrl(allow=r'/somepage/'), + ] + + reqgen = RequestGenerator(extractors, processors, callback=self.deferred) + requests = reqgen.generate_requests(self.response) + + self.failUnless(self._equal_requests_list(requests, [ + Request('http://example.org/somepage/item/12.html', + meta={'link_text': 'Item 12'}), + ])) + + + + diff --git a/scrapy/tests/test_contrib_exp_crawlspider_reqproc.py b/scrapy/tests/test_contrib_exp_crawlspider_reqproc.py new file mode 100644 index 000000000..da5db67b2 --- /dev/null +++ b/scrapy/tests/test_contrib_exp_crawlspider_reqproc.py @@ -0,0 +1,144 @@ +from twisted.trial import unittest + +from scrapy.http import Request + +from scrapy.contrib_exp.crawlspider.reqproc import Canonicalize +from scrapy.contrib_exp.crawlspider.reqproc import FilterDomain +from scrapy.contrib_exp.crawlspider.reqproc import FilterUrl +from scrapy.contrib_exp.crawlspider.reqproc import FilterDupes + +import copy + +class RequestProcessorsTest(unittest.TestCase): + + def test_canonicalize_requests(self): + urls = [ + 'http://example.com/do?&b=1&a=2&c=3', + 'http://example.com/do?123,&q=a space', + ] + urls_after = [ + 'http://example.com/do?a=2&b=1&c=3', + 'http://example.com/do?123%2C=&q=a+space', + ] + + proc = Canonicalize() + results = [req.url for req in proc(Request(url) for url in urls)] + self.failUnlessEquals(results, urls_after) + + def test_unique_requests(self): + urls = [ + 'http://example.com/sample1.html', + 'http://example.com/sample2.html', + 'http://example.com/sample3.html', + 'http://example.com/sample1.html', + 'http://example.com/sample2.html', + ] + urls_unique = [ + 'http://example.com/sample1.html', + 'http://example.com/sample2.html', + 'http://example.com/sample3.html', + ] + + proc = FilterDupes() + results = [req.url for req in proc(Request(url) for url in urls)] + self.failUnlessEquals(results, urls_unique) + + # Check custom attributes + requests = [ + Request('http://example.com', method='GET'), + Request('http://example.com', method='POST'), + ] + proc = FilterDupes('url', 'method') + self.failUnlessEqual(len(list(proc(requests))), 2) + + proc = FilterDupes('url') + self.failUnlessEqual(len(list(proc(requests))), 1) + + def test_filter_domain(self): + urls = [ + 'http://blah1.com/index', + 'http://blah2.com/index', + 'http://blah1.com/section', + 'http://blah2.com/section', + ] + + proc = FilterDomain(allow=('blah1.com'), deny=('blah2.com')) + filtered = [req.url for req in proc(Request(url) for url in urls)] + self.failUnlessEquals(filtered, [ + 'http://blah1.com/index', + 'http://blah1.com/section', + ]) + + proc = FilterDomain(deny=('blah1.com', 'blah2.com')) + filtered = [req.url for req in proc(Request(url) for url in urls)] + self.failUnlessEquals(filtered, []) + + proc = FilterDomain(allow=('blah1.com', 'blah2.com')) + filtered = [req.url for req in proc(Request(url) for url in urls)] + self.failUnlessEquals(filtered, urls) + + def test_filter_url(self): + urls = [ + 'http://blah1.com/index', + 'http://blah2.com/index', + 'http://blah1.com/section', + 'http://blah2.com/section', + ] + + proc = FilterUrl(allow=(r'blah1'), deny=(r'blah2')) + filtered = [req.url for req in proc(Request(url) for url in urls)] + self.failUnlessEquals(filtered, [ + 'http://blah1.com/index', + 'http://blah1.com/section', + ]) + + proc = FilterUrl(deny=('blah1', 'blah2')) + filtered = [req.url for req in proc(Request(url) for url in urls)] + self.failUnlessEquals(filtered, []) + + proc = FilterUrl(allow=('index$', 'section$')) + filtered = [req.url for req in proc(Request(url) for url in urls)] + self.failUnlessEquals(filtered, urls) + + + + def test_all_processors(self): + urls = [ + 'http://example.com/sample1.html', + 'http://example.com/sample2.html', + 'http://example.com/sample3.html', + 'http://example.com/sample1.html', + 'http://example.com/sample2.html', + 'http://example.com/do?&b=1&a=2&c=3', + 'http://example.com/do?123,&q=a space', + ] + urls_processed = [ + 'http://example.com/sample1.html', + 'http://example.com/sample2.html', + 'http://example.com/sample3.html', + 'http://example.com/do?a=2&b=1&c=3', + 'http://example.com/do?123%2C=&q=a+space', + ] + + processors = [ + Canonicalize(), + FilterDupes(), + ] + + def _process(requests): + """Apply all processors""" + # copy list + processed = [copy.copy(req) for req in requests] + for proc in processors: + processed = proc(processed) + return processed + + # empty requests + results1 = [r.url for r in _process([])] + self.failUnlessEquals(results1, []) + + # try urls + requests = (Request(url) for url in urls) + results2 = [r.url for r in _process(requests)] + self.failUnlessEquals(results2, urls_processed) + diff --git a/scrapy/tests/test_contrib_exp_crawlspider_rules.py b/scrapy/tests/test_contrib_exp_crawlspider_rules.py new file mode 100644 index 000000000..0fbe52415 --- /dev/null +++ b/scrapy/tests/test_contrib_exp_crawlspider_rules.py @@ -0,0 +1,262 @@ +from twisted.trial import unittest + +from scrapy.http import HtmlResponse +from scrapy.spider import BaseSpider +from scrapy.contrib_exp.crawlspider.matchers import BaseMatcher +from scrapy.contrib_exp.crawlspider.matchers import UrlMatcher +from scrapy.contrib_exp.crawlspider.matchers import UrlRegexMatcher + +from scrapy.contrib_exp.crawlspider.rules import CompiledRule +from scrapy.contrib_exp.crawlspider.rules import Rule +from scrapy.contrib_exp.crawlspider.rules import RulesManager + +from functools import partial + +class RuleInitializationTest(unittest.TestCase): + + def test_fail_if_rule_null(self): + # fail on empty rule + self.failUnlessRaises(ValueError, Rule) + self.failUnlessRaises(ValueError, Rule, + **dict(callback=None, follow=None)) + self.failUnlessRaises(ValueError, Rule, + **dict(callback=None, follow=False)) + + def test_minimal_arguments_to_instantiation(self): + # not fail if callback set + self.failUnless(Rule(callback=lambda: True)) + # not fail if follow set + self.failUnless(Rule(follow=True)) + + def test_validate_default_attributes(self): + # test null Rule + rule = Rule(follow=True) + self.failUnlessEqual(None, rule.matcher) + self.failUnlessEqual(None, rule.callback) + self.failUnlessEqual({}, rule.cb_kwargs) + # follow default False + self.failUnlessEqual(True, rule.follow) + + def test_validate_attributes_set(self): + matcher = BaseMatcher() + callback = lambda: True + rule = Rule(matcher, callback, True, a=1) + # test attributes + self.failUnlessEqual(matcher, rule.matcher) + self.failUnlessEqual(callback, rule.callback) + self.failUnlessEqual({'a': 1}, rule.cb_kwargs) + self.failUnlessEqual(True, rule.follow) + +class CompiledRuleInitializationTest(unittest.TestCase): + + def test_fail_on_invalid_matcher(self): + # pass with valid matcher + self.failUnless(CompiledRule(BaseMatcher()), + "Failed CompiledRule instantiation") + + # at least needs valid matcher + self.assertRaises(AssertionError, CompiledRule, None) + self.assertRaises(AssertionError, CompiledRule, False) + self.assertRaises(AssertionError, CompiledRule, True) + + def test_fail_on_invalid_callback(self): + # pass with valid callback + callback = lambda: True + self.failUnless(CompiledRule(BaseMatcher(), callback)) + # pass with callback none + self.failUnless(CompiledRule(BaseMatcher(), None)) + + # assert on invalid callback + self.assertRaises(AssertionError, CompiledRule, BaseMatcher(), + 'myfunc') + + # numeric variable + var = 123 + self.assertRaises(AssertionError, CompiledRule, BaseMatcher(), + var) + + class A: + pass + + # random instance + self.assertRaises(AssertionError, CompiledRule, BaseMatcher(), + A()) + + + def test_fail_on_invalid_follow_value(self): + callback = lambda: True + matcher = BaseMatcher() + # pass bool + self.failUnless(CompiledRule(matcher, callback, True)) + self.failUnless(CompiledRule(matcher, callback, False)) + + # assert with non-bool + self.assertRaises(AssertionError, CompiledRule, matcher, + callback, None) + self.assertRaises(AssertionError, CompiledRule, matcher, + callback, 1) + + def test_validate_default_attributes(self): + callback = lambda: True + matcher = BaseMatcher() + rule = CompiledRule(matcher, callback, True) + + # test attributes + self.failUnlessEqual(matcher, rule.matcher) + self.failUnlessEqual(callback, rule.callback) + self.failUnlessEqual(True, rule.follow) + + +class RulesTest(unittest.TestCase): + def test_rules_manager_basic(self): + spider = BaseSpider() + response1 = HtmlResponse('http://example.org') + response2 = HtmlResponse('http://othersite.org') + rulesman = RulesManager([], spider) + + # should return none + self.failIf(rulesman.get_rule_from_response(response1)) + self.failIf(rulesman.get_rule_from_response(response2)) + + # rules manager with match-all rule + rulesman = RulesManager([ + Rule(BaseMatcher(), follow=True), + ], spider) + + # returns CompiledRule + rule1 = rulesman.get_rule_from_response(response1) + rule2 = rulesman.get_rule_from_response(response2) + + self.failUnless(isinstance(rule1, CompiledRule)) + self.failUnless(isinstance(rule2, CompiledRule)) + self.assert_(rule1 is rule2) + self.failUnlessEqual(rule1.callback, None) + self.failUnlessEqual(rule1.follow, True) + + def test_rules_manager_empty_rule(self): + spider = BaseSpider() + response = HtmlResponse('http://example.org') + + rulesman = RulesManager([Rule(follow=True)], spider) + + rule = rulesman.get_rule_from_response(response) + # default matcher if None: BaseMatcher + self.failUnless(isinstance(rule.matcher, BaseMatcher)) + + def test_rules_manager_default_matcher(self): + spider = BaseSpider() + response = HtmlResponse('http://example.org') + callback = lambda x: None + + rulesman = RulesManager([ + Rule('http://example.org', callback), + ], spider, default_matcher=UrlMatcher) + + rule = rulesman.get_rule_from_response(response) + self.failUnless(isinstance(rule.matcher, UrlMatcher)) + + def test_rules_manager_matchers(self): + spider = BaseSpider() + response1 = HtmlResponse('http://example.org') + response2 = HtmlResponse('http://othersite.org') + + urlmatcher = UrlMatcher('http://example.org') + basematcher = BaseMatcher() + # callback needed for Rule + callback = lambda x: None + + # test fail matcher resolve + self.assertRaises(ValueError, RulesManager, + [Rule(False, callback)], spider) + self.assertRaises(ValueError, RulesManager, + [Rule(spider, callback)], spider) + + rulesman = RulesManager([ + Rule(urlmatcher, callback), + Rule(basematcher, callback), + ], spider) + + # response1 matches example.org + rule1 = rulesman.get_rule_from_response(response1) + # response2 is catch by BaseMatcher() + rule2 = rulesman.get_rule_from_response(response2) + + self.failUnlessEqual(rule1.matcher, urlmatcher) + self.failUnlessEqual(rule2.matcher, basematcher) + + # reverse order. BaseMatcher should match all + rulesman = RulesManager([ + Rule(basematcher, callback), + Rule(urlmatcher, callback), + ], spider) + + rule1 = rulesman.get_rule_from_response(response1) + rule2 = rulesman.get_rule_from_response(response2) + + self.failUnlessEqual(rule1.matcher, basematcher) + self.failUnlessEqual(rule2.matcher, basematcher) + self.failUnless(rule1 is rule2) + + def test_rules_manager_callbacks(self): + mycallback = lambda: True + + spider = BaseSpider() + spider.parse_item = lambda: True + + response1 = HtmlResponse('http://example.org') + response2 = HtmlResponse('http://othersite.org') + + rulesman = RulesManager([ + Rule('example', mycallback), + Rule('othersite', 'parse_item'), + ], spider, default_matcher=UrlRegexMatcher) + + rule1 = rulesman.get_rule_from_response(response1) + rule2 = rulesman.get_rule_from_response(response2) + + self.failUnlessEqual(rule1.callback, mycallback) + self.failUnlessEqual(rule2.callback, spider.parse_item) + + # fail unknown callback + self.assertRaises(AttributeError, RulesManager, [ + Rule(BaseMatcher(), 'mycallback') + ], spider) + # fail not callable + spider.not_callable = True + self.assertRaises(AttributeError, RulesManager, [ + Rule(BaseMatcher(), 'not_callable') + ], spider) + + + def test_rules_manager_callback_with_arguments(self): + spider = BaseSpider() + response = HtmlResponse('http://example.org') + + kwargs = {'a': 1} + + def myfunc(**mykwargs): + return mykwargs + + # verify return validation + self.failUnlessEquals(kwargs, myfunc(**kwargs)) + + # test callback w/o arguments + rulesman = RulesManager([ + Rule(BaseMatcher(), myfunc), + ], spider) + rule = rulesman.get_rule_from_response(response) + + # without arguments should return same callback + self.failUnlessEqual(rule.callback, myfunc) + + # test callback w/ arguments + rulesman = RulesManager([ + Rule(BaseMatcher(), myfunc, **kwargs), + ], spider) + rule = rulesman.get_rule_from_response(response) + + # with argument should return partial applied callback + self.failUnless(isinstance(rule.callback, partial)) + self.failUnlessEquals(kwargs, rule.callback()) + + diff --git a/scrapy/tests/test_contrib_exp_crawlspider_spider.py b/scrapy/tests/test_contrib_exp_crawlspider_spider.py new file mode 100644 index 000000000..5e067508a --- /dev/null +++ b/scrapy/tests/test_contrib_exp_crawlspider_spider.py @@ -0,0 +1,222 @@ +from twisted.trial import unittest + +from scrapy.http import Request +from scrapy.http import HtmlResponse +from scrapy.item import BaseItem +from scrapy.utils.spider import iterate_spider_output + +# basics +from scrapy.contrib_exp.crawlspider import CrawlSpider +from scrapy.contrib_exp.crawlspider import Rule + +# matchers +from scrapy.contrib_exp.crawlspider.matchers import BaseMatcher +from scrapy.contrib_exp.crawlspider.matchers import UrlRegexMatcher +from scrapy.contrib_exp.crawlspider.matchers import UrlListMatcher + +# extractors +from scrapy.contrib_exp.crawlspider.reqext import SgmlRequestExtractor + +# processors +from scrapy.contrib_exp.crawlspider.reqproc import Canonicalize +from scrapy.contrib_exp.crawlspider.reqproc import FilterDupes + + +# mock items +class Item1(BaseItem): + pass + +class Item2(BaseItem): + pass + +class Item3(BaseItem): + pass + + +class CrawlSpiderTest(unittest.TestCase): + + def spider_factory(self, rules=[], + extractors=[], processors=[], + start_urls=[]): + # mock spider + class Spider(CrawlSpider): + def parse_item1(self, response): + return Item1() + + def parse_item2(self, response): + return Item2() + + def parse_item3(self, response): + return Item3() + + def parse_request1(self, response): + return Request('http://example.org/request1') + + def parse_request2(self, response): + return Request('http://example.org/request2') + + Spider.start_urls = start_urls + Spider.rules = rules + Spider.request_extractors = extractors + Spider.request_processors = processors + + return Spider() + + def test_start_url_auto_rule(self): + spider = self.spider_factory() + # zero spider rules + self.failUnlessEqual(len(spider.rules), 0) + self.failUnlessEqual(len(spider._rulesman._rules), 0) + + spider = self.spider_factory(start_urls=['http://example.org']) + + self.failUnlessEqual(len(spider.rules), 0) + self.failUnlessEqual(len(spider._rulesman._rules), 1) + + def test_start_url_matcher(self): + url = 'http://example.org' + spider = self.spider_factory(start_urls=[url]) + + response = HtmlResponse(url) + + rule = spider._rulesman.get_rule_from_response(response) + self.failUnless(isinstance(rule.matcher, UrlListMatcher)) + + response = HtmlResponse(url + '/item.html') + + rule = spider._rulesman.get_rule_from_response(response) + self.failUnless(rule is None) + + # TODO: remove this block + # in previous version get_rule returns rule from response.request + response.request = Request(url) + rule = spider._rulesman.get_rule_from_response(response.request) + self.failUnless(isinstance(rule.matcher, UrlListMatcher)) + self.failUnlessEqual(rule.follow, True) + + def test_parse_callback(self): + response = HtmlResponse('http://example.org') + rules = ( + Rule(BaseMatcher(), 'parse_item1'), + ) + spider = self.spider_factory(rules) + + result = list(spider.parse(response)) + self.failUnlessEqual(len(result), 1) + self.failUnless(isinstance(result[0], Item1)) + + def test_crawling_start_url(self): + url = 'http://example.org/' + html = """<html><head><title>Page title<title> + <body><p><a href="item/12.html">Item 12</a></p> + <p><a href="/about.html">About us</a></p> + <img src="/logo.png" alt="Company logo (not a link)" /> + <p><a href="../othercat.html">Other category</a></p> + <p><a href="/" /></p></body></html>""" + response = HtmlResponse(url, body=html) + + extractors = (SgmlRequestExtractor(), ) + spider = self.spider_factory(start_urls=[url], + extractors=extractors) + result = list(spider.parse(response)) + + # 1 request extracted: example.org/ + # because requests returns only matching + self.failUnlessEqual(len(result), 1) + + # we will add catch-all rule to extract all + callback = lambda x: None + rules = [Rule(r'\.html$', callback=callback)] + spider = self.spider_factory(rules, start_urls=[url], + extractors=extractors) + result = list(spider.parse(response)) + + # 4 requests extracted + # 3 of .html pattern + # 1 of start url patter + self.failUnlessEqual(len(result), 4) + + def test_crawling_simple_rule(self): + url = 'http://example.org/somepage/index.html' + html = """<html><head><title>Page title<title> + <body><p><a href="item/12.html">Item 12</a></p> + <p><a href="/about.html">About us</a></p> + <img src="/logo.png" alt="Company logo (not a link)" /> + <p><a href="../othercat.html">Other category</a></p> + <p><a href="/" /></p></body></html>""" + + response = HtmlResponse(url, body=html) + + rules = ( + # first response callback + Rule(r'index\.html', 'parse_item1'), + ) + spider = self.spider_factory(rules) + result = list(spider.parse(response)) + + # should return Item1 + self.failUnlessEqual(len(result), 1) + self.failUnless(isinstance(result[0], Item1)) + + # test request generation + rules = ( + # first response without callback and follow flag + Rule(r'index\.html', follow=True), + Rule(r'(\.html|/)$', 'parse_item1'), + ) + spider = self.spider_factory(rules) + result = list(spider.parse(response)) + + # 0 because spider does not have extractors + self.failUnlessEqual(len(result), 0) + + extractors = (SgmlRequestExtractor(), ) + + # instance spider with extractor + spider = self.spider_factory(rules, extractors) + result = list(spider.parse(response)) + # 4 requests extracted + self.failUnlessEqual(len(result), 4) + + def test_crawling_multiple_rules(self): + html = """<html><head><title>Page title<title> + <body><p><a href="item/12.html">Item 12</a></p> + <p><a href="/about.html">About us</a></p> + <img src="/logo.png" alt="Company logo (not a link)" /> + <p><a href="../othercat.html">Other category</a></p> + <p><a href="/" /></p></body></html>""" + + response = HtmlResponse('http://example.org/index.html', body=html) + response1 = HtmlResponse('http://example.org/1.html') + response2 = HtmlResponse('http://example.org/othercat.html') + + rules = ( + Rule(r'\d+\.html$', 'parse_item1'), + Rule(r'othercat\.html$', 'parse_item2'), + # follow-only rules + Rule(r'index\.html', 'parse_item3', follow=True) + ) + extractors = [SgmlRequestExtractor()] + spider = self.spider_factory(rules, extractors) + + result = list(spider.parse(response)) + # 1 Item 2 Requests + self.failUnlessEqual(len(result), 3) + # parse_item3 + self.failUnless(isinstance(result[0], Item3)) + only_requests = lambda r: isinstance(r, Request) + requests = filter(only_requests, result[1:]) + self.failUnlessEqual(len(requests), 2) + self.failUnless(all(requests)) + + result1 = list(spider.parse(response1)) + # parse_item1 + self.failUnlessEqual(len(result1), 1) + self.failUnless(isinstance(result1[0], Item1)) + + result2 = list(spider.parse(response2)) + # parse_item2 + self.failUnlessEqual(len(result2), 1) + self.failUnless(isinstance(result2[0], Item2)) + + diff --git a/scrapy/tests/test_http_response.py b/scrapy/tests/test_http_response.py index 3b0a144d2..5851572ac 100644 --- a/scrapy/tests/test_http_response.py +++ b/scrapy/tests/test_http_response.py @@ -175,6 +175,8 @@ class TextResponseTest(BaseResponseTest): r2 = self.response_class("http://www.example.com", encoding='utf-8', body=u"\xa3") r3 = self.response_class("http://www.example.com", headers={"Content-type": ["text/html; charset=iso-8859-1"]}, body="\xa3") r4 = self.response_class("http://www.example.com", body="\xa2\xa3") + r5 = self.response_class("http://www.example.com", + headers={"Content-type": ["text/html; charset=None"]}, body="\xc2\xa3") self.assertEqual(r1.headers_encoding(), "utf-8") self.assertEqual(r2.headers_encoding(), None) @@ -182,6 +184,8 @@ class TextResponseTest(BaseResponseTest): self.assertEqual(r3.headers_encoding(), "iso-8859-1") self.assertEqual(r3.encoding, 'iso-8859-1') self.assertEqual(r4.headers_encoding(), None) + self.assertEqual(r5.headers_encoding(), None) + self.assertEqual(r5.encoding, "utf-8") assert r4.body_encoding() is not None and r4.body_encoding() != 'ascii' self._assert_response_values(r1, 'utf-8', u"\xa3") self._assert_response_values(r2, 'utf-8', u"\xa3") diff --git a/scrapy/tests/test_utils_python.py b/scrapy/tests/test_utils_python.py index 13d8be30c..dc46e1f91 100644 --- a/scrapy/tests/test_utils_python.py +++ b/scrapy/tests/test_utils_python.py @@ -1,7 +1,8 @@ +import operator import unittest from scrapy.utils.python import str_to_unicode, unicode_to_str, \ - memoizemethod_noargs, isbinarytext + memoizemethod_noargs, isbinarytext, equal_attributes class UtilsPythonTestCase(unittest.TestCase): def test_str_to_unicode(self): @@ -61,5 +62,52 @@ class UtilsPythonTestCase(unittest.TestCase): # finally some real binary bytes assert isbinarytext("\x02\xa3") + def test_equal_attributes(self): + class Obj: + pass + + a = Obj() + b = Obj() + # no attributes given return False + self.failIf(equal_attributes(a, b, [])) + # not existent attributes + self.failIf(equal_attributes(a, b, ['x', 'y'])) + + a.x = 1 + b.x = 1 + # equal attribute + self.failUnless(equal_attributes(a, b, ['x'])) + + b.y = 2 + # obj1 has no attribute y + self.failIf(equal_attributes(a, b, ['x', 'y'])) + + a.y = 2 + # equal attributes + self.failUnless(equal_attributes(a, b, ['x', 'y'])) + + a.y = 1 + # differente attributes + self.failIf(equal_attributes(a, b, ['x', 'y'])) + + # test callable + a.meta = {} + b.meta = {} + self.failUnless(equal_attributes(a, b, ['meta'])) + + # compare ['meta']['a'] + a.meta['z'] = 1 + b.meta['z'] = 1 + + get_z = operator.itemgetter('z') + get_meta = operator.attrgetter('meta') + compare_z = lambda obj: get_z(get_meta(obj)) + + self.failUnless(equal_attributes(a, b, [compare_z, 'x'])) + # fail z equality + a.meta['z'] = 2 + self.failIf(equal_attributes(a, b, [compare_z, 'x'])) + + if __name__ == "__main__": unittest.main() diff --git a/scrapy/utils/python.py b/scrapy/utils/python.py index 7d43c4566..319acdc05 100644 --- a/scrapy/utils/python.py +++ b/scrapy/utils/python.py @@ -216,3 +216,28 @@ def get_func_args(func): else: raise TypeError('%s is not callable' % type(func)) return func_args + + +def equal_attributes(obj1, obj2, attributes): + """Compare two objects attributes""" + # not attributes given return False by default + if not attributes: + return False + + for attr in attributes: + # support callables like itemgetter + if callable(attr): + if not attr(obj1) == attr(obj2): + return False + else: + # check that objects has attribute + if not hasattr(obj1, attr): + return False + if not hasattr(obj2, attr): + return False + # compare object attributes + if not getattr(obj1, attr) == getattr(obj2, attr): + return False + # all attributes equal + return True +